diff --git a/.claude/settings.json b/.claude/settings.json new file mode 100644 index 000000000..7cb404c71 --- /dev/null +++ b/.claude/settings.json @@ -0,0 +1 @@ +{"worktree": {"bgIsolation": "none"}} diff --git a/.cspell-crd-redesign.txt b/.cspell-crd-redesign.txt new file mode 100644 index 000000000..8356e30c7 --- /dev/null +++ b/.cspell-crd-redesign.txt @@ -0,0 +1,159 @@ +# Spell-check terms for operator/docs/designs/crd-redesign and its test plans. +# +# Produced by running cspell with the repository's .cspell.json over the ten design +# documents, their assets, and the nine test plans. Every entry below is a word that +# run reported and the dictionary does not carry. +# +# This file is deliberately untracked and is a staging list, not a dictionary the +# build reads. Nothing consumes it yet: MegaLinter reads .cspell.json's `words`, so +# the groups below are the candidates for that list once each has been decided. +# +# Four groups, and only the last one is a judgment call. +# +# kinds Kind names, short names, and generated file names. Add as they are. +# code Go, kubebuilder, and package identifiers appearing in prose. Add as they are. +# product Product, protocol, and platform names not this repository's to spell. +# Add as they are, and check the casing variants are the ones actually used. +# prose Words the design prose coins. Each is either added here or the prose is +# reworded; that decision has not been taken. + +# --- kinds (35) ----------------------------------------------- +autoplacement +backupops +backuppolicy +backuprestore +clusterdeploymentconfig +controlplane +controlplaneops +cpops +nodeset +operatorops +persistentvolume +persistentvolumeclaims +persistentvolumeops +persistentvolumes +pvops +rollingrestart +sbops +sdops +simplyblockdriver +Snode +snode +spops +storagebackup +storagebackupops +storagebackuppolicy +storagebackups +storageclass +storageclasses +storagecluster +storageclusters +storagedevice +storagedeviceops +storagepool +storagepoolops +storagepools + +# --- code (26) ------------------------------------------------ +corev +cpinformer +customresources +envtest +foundationdb +kubebuilder +livenessprobe +metav +metricsapi +mongodbcommunity +omitempty +pernodeconfig +printcolumn +resizer +sbcli +SBTLS +simplybk +statemachine +subresource +subtest +# the two below are British spellings, and they are shipped Go test function +# names the test plans quote verbatim, not prose this corpus can reword. +# operator/internal/controller: TestSanitiseDNSLabel_*, *_NodeRollingRestart_Initialises +Initialises +Sanitise +WEBAPI +webapi +webapimock +webappapi + +# --- product (42) --------------------------------------------- +AXYZ +dhchap +DHCHAP +elon +finalizers +HCLS +hostnames +hostpath +hugepages +Ifname +kubelet +Kubelet +lvol +Lvol +Lvols +mbps +Mbytes +mbytes +mgmt +Mgmt +MZQL +NDCS +ndcs +npcs +NPCS +nvme +NVME +Nvme +nvmeof +Nvmf +nvmf +schedulable +smartctl +spdk +Spdk +SPDK +subsys +SUBSYS +tolerations +Tolerations +VCPU +vcpu + +# --- prose (26) ----------------------------------------------- +alertable +antipattern +catchable +cutover +deadlineless +graphable +metacharacter +parameterizes +provisionable +Recasing +regroupings +Repointing +sourceable +streamability +streamable +unaddressable +unasserted +unbuilt +undrainable +unexempted +unpartitioned +Unprovisioned +unreworked +upserted +upserts +waivable + diff --git a/atlas-lib/controlplane/consistencygroups.go b/atlas-lib/controlplane/consistencygroups.go new file mode 100644 index 000000000..35fa05edb --- /dev/null +++ b/atlas-lib/controlplane/consistencygroups.go @@ -0,0 +1,152 @@ +// Consistency-group reads the csi-addons VolumeGroup service needs: resolving the +// backend consistency group a set of member volumes already belongs to (design +// design-csi-addons-replication.md §14.3). A group is formed at provisioning by +// the storage.simplyblock.io/consistency-group label; this only reads it back. +package controlplane + +import ( + "context" + "errors" + "fmt" + + "github.com/google/uuid" + + "github.com/simplyblock/atlas/errs" + "github.com/simplyblock/atlas/internal/cpapi" + "github.com/simplyblock/atlas/lvol" +) + +// ConsistencyGroupForLvols returns the id of the backend consistency group in +// clusterID whose current (open-epoch) membership is exactly lvolIDs. +// +// The members provisioned under one storage.simplyblock.io/consistency-group +// label share one backend group, so the VolumeGroup service resolves the group +// by its members rather than by the name it is handed (which is the caller's own +// generated name, not the group's). Returns an error when no group's membership +// matches exactly, so a partial or mixed selection is never silently grouped. +func (c *Client) ConsistencyGroupForLvols(ctx context.Context, clusterID string, lvolIDs []string) (string, error) { + cluster, err := parseUUID("cluster id", clusterID) + if err != nil { + return "", err + } + if len(lvolIDs) == 0 { + return "", fmt.Errorf("no volumes given to resolve a consistency group") + } + want := make(map[string]bool, len(lvolIDs)) + for _, id := range lvolIDs { + want[id] = true + } + + listResp, err := c.api.ClustersConsistencyGroupsListApiV2ClustersClusterIdConsistencyGroupsGetWithResponse( + ctx, cluster, &cpapi.ClustersConsistencyGroupsListApiV2ClustersClusterIdConsistencyGroupsGetParams{}) + if err != nil { + return "", fmt.Errorf("list consistency groups in cluster %s: %w", clusterID, err) + } + groups, err := payload("consistency groups in cluster "+clusterID, + listResp.JSON200, listResp.StatusCode(), listResp.Body) + if err != nil { + return "", err + } + + for _, g := range *groups { + members, err := c.consistencyGroupMembers(ctx, cluster, g.Id) + if err != nil { + return "", err + } + if sameStringSet(members, want) { + return g.Id.String(), nil + } + } + return "", fmt.Errorf("no consistency group in cluster %s has exactly the %d requested member(s)", + clusterID, len(lvolIDs)) +} + +// consistencyGroupMembers returns a group's current open-epoch member lvol ids. +func (c *Client) consistencyGroupMembers(ctx context.Context, cluster, group uuid.UUID) ([]string, error) { + resp, err := c.api.ClustersConsistencyGroupsMembersApiV2ClustersClusterIdConsistencyGroupsGroupIdMembersGetWithResponse( + ctx, cluster, group) + if err != nil { + return nil, fmt.Errorf("members of consistency group %s: %w", group, err) + } + dtos, err := payload("members of consistency group "+group.String(), + resp.JSON200, resp.StatusCode(), resp.Body) + if err != nil { + return nil, err + } + open := make([]string, 0, len(*dtos)) + for _, m := range *dtos { + if m.RemovedSeq == 0 { + open = append(open, m.LvolId) + } + } + return open, nil +} + +// sameStringSet reports whether ids is exactly the set want (same length, same +// elements), so a resolved group is neither a superset nor a subset of the +// requested members. +func sameStringSet(ids []string, want map[string]bool) bool { + if len(ids) != len(want) { + return false + } + for _, id := range ids { + if !want[id] { + return false + } + } + return true +} + +// ConsistencyGroupMemberHandles returns the volume handles +// ("::") of a consistency group's current (open-epoch) +// members, in the order the control plane lists them. +// +// The member listing carries bare lvol ids. Every member of one group lives in +// one storage pool (the backend refuses a cross-pool join), so the pool is +// resolved once, from the first member, by probing the cluster's pools. An +// empty group is an error: a caller mapping members (csi-addons destination +// info) must never return an empty map as if it were complete. +func (c *Client) ConsistencyGroupMemberHandles( + ctx context.Context, gh lvol.GroupHandle, +) ([]lvol.VolumeHandle, error) { + cluster, err := parseUUID("cluster id", gh.ClusterID) + if err != nil { + return nil, err + } + group, err := parseUUID("consistency group id", gh.GroupID) + if err != nil { + return nil, err + } + members, err := c.consistencyGroupMembers(ctx, cluster, group) + if err != nil { + return nil, err + } + if len(members) == 0 { + return nil, fmt.Errorf("consistency group %s has no current members", gh) + } + pools, err := c.ListStoragePools(ctx, gh.ClusterID) + if err != nil { + return nil, err + } + poolID := "" + for _, p := range pools { + h := lvol.Handle{ClusterID: gh.ClusterID, PoolRef: p.ID, VolumeID: members[0]} + _, err := c.Volume(ctx, h.Handle()) + if err == nil { + poolID = p.ID + break + } + if !errors.Is(err, errs.ErrNotFound) { + return nil, err + } + } + if poolID == "" { + return nil, fmt.Errorf("member %s of consistency group %s is in none of the %d pools of cluster %s: %w", + members[0], gh, len(pools), gh.ClusterID, errs.ErrNotFound) + } + out := make([]lvol.VolumeHandle, 0, len(members)) + for _, m := range members { + out = append(out, lvol.Handle{ClusterID: gh.ClusterID, PoolRef: poolID, VolumeID: m}.Handle()) + } + return out, nil +} diff --git a/atlas-lib/controlplane/consistencygroups_members_test.go b/atlas-lib/controlplane/consistencygroups_members_test.go new file mode 100644 index 000000000..f57b42a42 --- /dev/null +++ b/atlas-lib/controlplane/consistencygroups_members_test.go @@ -0,0 +1,72 @@ +package controlplane + +import ( + "context" + "net/http" + "strings" + "testing" + + "github.com/simplyblock/atlas/lvol" +) + +// The member listing carries bare lvol ids; the handles need the pool, which +// the group's first member is probed for across the cluster's pools. Here it +// lives in the SECOND pool, and a removed member is left out. +func TestConsistencyGroupMemberHandlesResolvesThePool(t *testing.T) { + const ( + group = "c9c9c9c9-c9c9-4c9c-8c9c-c9c9c9c9c9c9" + m1 = "a1111111-1111-4111-8111-111111111111" + m2 = "b2222222-2222-4222-8222-222222222222" + gone = "d4444444-4444-4444-8444-444444444444" + otherPool = "55555555-5555-5555-5555-555555555555" + ) + pool := func(id, name string) string { + return `{"id":"` + id + `","cluster_id":"` + testCluster + `","name":"` + name + `",` + + `"max_size":1000,"capacity":{},"max_r_mbytes":0,"max_rw_iops":0,"max_rw_mbytes":0,` + + `"max_w_mbytes":0,"volume_max_size":0,"status":"active"}` + } + member := func(id string, removed int) string { + return `{"lvol_id":"` + id + `","joined_seq":1,"removed_seq":` + string(rune('0'+removed)) + + `,"node_id":"n","lvs_name":"l","online":true}` + } + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch { + case strings.HasSuffix(r.URL.Path, "/consistency-groups/"+group+"/members"): + _, _ = w.Write([]byte("[" + member(m1, 0) + "," + member(gone, 3) + "," + member(m2, 0) + "]")) + case strings.HasSuffix(r.URL.Path, "/storage-pools/"): + _, _ = w.Write([]byte("[" + pool(otherPool, "other") + "," + pool(testPool, "pool1") + "]")) + case strings.Contains(r.URL.Path, "/storage-pools/"+testPool+"/volumes/"+m1): + _, _ = w.Write([]byte(`{"id":"` + m1 + `","name":"pvc-1","pool_name":"pool1",` + + `"size":20971520,"ns_id":1,"nqn":"nqn.2023-02.io.simplyblock:c:lvol:v"}`)) + default: + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte(`{"detail":"not found"}`)) + } + }) + + got, err := c.ConsistencyGroupMemberHandles(context.Background(), + lvol.GroupHandle{ClusterID: testCluster, GroupID: group}) + if err != nil { + t.Fatal(err) + } + want := []lvol.VolumeHandle{ + lvol.VolumeHandle(testCluster + ":" + testPool + ":" + m1), + lvol.VolumeHandle(testCluster + ":" + testPool + ":" + m2), + } + if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] { + t.Fatalf("handles = %v, want %v", got, want) + } +} + +func TestConsistencyGroupMemberHandlesOfAnEmptyGroupIsAnError(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`[]`)) + }) + _, err := c.ConsistencyGroupMemberHandles(context.Background(), + lvol.GroupHandle{ClusterID: testCluster, GroupID: "c9c9c9c9-c9c9-4c9c-8c9c-c9c9c9c9c9c9"}) + if err == nil { + t.Fatal("an empty group answered without an error") + } +} diff --git a/atlas-lib/controlplane/groupreplication.go b/atlas-lib/controlplane/groupreplication.go new file mode 100644 index 000000000..987358209 --- /dev/null +++ b/atlas-lib/controlplane/groupreplication.go @@ -0,0 +1,243 @@ +// Group replication verbs on a consistency-group handle (design +// design-csi-addons-replication.md §14.4): the whole group replicates as one +// unit through the cluster-scoped /consistency-groups/{id}/replication/* +// endpoints, the group twins of the per-volume verbs in replication.go. The +// driver's Replication verbs route a cg: handle here (§14.4). +package controlplane + +import ( + "bytes" + "context" + "fmt" + "net/http" + + "github.com/google/uuid" + + "github.com/simplyblock/atlas/internal/cpapi" + "github.com/simplyblock/atlas/lvol" +) + +// groupIDs parses a group handle into the cluster and group UUIDs the v2 path +// parameters require. +func groupIDs(gh lvol.GroupHandle) (cluster, group uuid.UUID, err error) { + cluster, err = parseUUID("cluster id", gh.ClusterID) + if err != nil { + return + } + group, err = parseUUID("consistency group id", gh.GroupID) + return +} + +// EnableGroupReplication attaches the consistency group to a group replication +// policy, so the whole group replicates as one unit. +func (c *Client) EnableGroupReplication(ctx context.Context, gh lvol.GroupHandle, policyID string) error { + cluster, group, err := groupIDs(gh) + if err != nil { + return err + } + policy, err := parseUUID("replication policy id", policyID) + if err != nil { + return err + } + resp, err := c.api.ClustersConsistencyGroupsReplicationConfigureApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationPutWithResponse( + ctx, cluster, group, cpapi.ConsistencyGroupReplicationIntentDTO{ReplicationPolicyId: &policy}) + if err != nil { + return fmt.Errorf("enable group replication %s: %w", gh.Handle(), err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("enable group replication "+string(gh.Handle()), code, resp.Body) + } + return nil +} + +// DisableGroupReplication detaches the consistency group from its policy, +// stopping replication without dissolving the group. +func (c *Client) DisableGroupReplication(ctx context.Context, gh lvol.GroupHandle) error { + cluster, group, err := groupIDs(gh) + if err != nil { + return err + } + // ReplicationPolicyId is not omitempty on the intent DTO, so a nil pointer + // marshals as an explicit null. The backend reads that null as a detach, + // distinct from an absent key. + resp, err := c.api.ClustersConsistencyGroupsReplicationConfigureApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationPutWithResponse( + ctx, cluster, group, cpapi.ConsistencyGroupReplicationIntentDTO{ReplicationPolicyId: nil}) + if err != nil { + return fmt.Errorf("disable group replication %s: %w", gh.Handle(), err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("disable group replication "+string(gh.Handle()), code, resp.Body) + } + return nil +} + +// PromoteGroup fails the whole group over as one unit: every member is cloned +// from the same group-snapshot generation on the target, atomically. +func (c *Client) PromoteGroup(ctx context.Context, gh lvol.GroupHandle) error { + cluster, group, err := groupIDs(gh) + if err != nil { + return err + } + resp, err := c.api.ClustersConsistencyGroupsReplicationFailoverApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationFailoverPostWithResponse( + ctx, cluster, group) + if err != nil { + return fmt.Errorf("promote group %s: %w", gh.Handle(), err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("promote group "+string(gh.Handle()), code, resp.Body) + } + return nil +} + +// DemoteGroup demotes the whole group: quiesce every member, ship one final +// group snapshot, confirm, fence all. Re-drivable, not queued: done=false while +// any member is still converging, so the caller calls again. +func (c *Client) DemoteGroup(ctx context.Context, gh lvol.GroupHandle) (bool, error) { + cluster, group, err := groupIDs(gh) + if err != nil { + return false, err + } + resp, err := c.api.ClustersConsistencyGroupsReplicationDemoteApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationDemotePostWithResponse( + ctx, cluster, group) + if err != nil { + return false, fmt.Errorf("demote group %s: %w", gh.Handle(), err) + } + switch code := resp.StatusCode(); code { + case http.StatusNoContent: + return true, nil + case http.StatusAccepted: + return false, nil + default: + return false, respError("demote group "+string(gh.Handle()), code, resp.Body) + } +} + +// ResyncGroup reverses the shipping direction for the whole group. It never cuts +// over, matching the per-volume ResyncVolume. sourceClusterID selects the source +// explicitly; "" leaves it unset. +func (c *Client) ResyncGroup(ctx context.Context, gh lvol.GroupHandle, sourceClusterID string) error { + cluster, group, err := groupIDs(gh) + if err != nil { + return err + } + var body cpapi.GroupFailbackParams + if sourceClusterID != "" { + id, err := parseUUID("source cluster id", sourceClusterID) + if err != nil { + return err + } + body.SourceClusterId = &id + } + resp, err := c.api.ClustersConsistencyGroupsReplicationFailbackApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationFailbackPostWithResponse( + ctx, cluster, group, body) + if err != nil { + return fmt.Errorf("resync group %s: %w", gh.Handle(), err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("resync group "+string(gh.Handle()), code, resp.Body) + } + return nil +} + +// GetGroupReplicationInfo returns the group's aggregate replication status +// (oldest recovery point, worst lag, and health; design §14.6), in the same +// shape as the per-volume status so the driver's Info verb maps it uniformly. +func (c *Client) GetGroupReplicationInfo(ctx context.Context, gh lvol.GroupHandle) (ReplicationStatus, error) { + cluster, group, err := groupIDs(gh) + if err != nil { + return ReplicationStatus{}, err + } + resp, err := c.api.ClustersConsistencyGroupsReplicationStatusApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationStatusGetWithResponse( + ctx, cluster, group) + if err != nil { + return ReplicationStatus{}, fmt.Errorf("group replication status %s: %w", gh.Handle(), err) + } + d, err := payload("group replication status "+string(gh.Handle()), resp.JSON200, resp.StatusCode(), resp.Body) + if err != nil { + return ReplicationStatus{}, err + } + return ReplicationStatus{ + Role: string(d.Role), + State: string(d.State), + LastReplicatedAt: d.LastReplicatedAt, + LagSeconds: d.LagSeconds, + OutstandingCount: derefInt(d.OutstandingCount), + OutstandingBytes: derefInt(d.OutstandingBytes), + Resyncing: d.Resyncing != nil && *d.Resyncing, + }, nil +} + +func derefInt(p *int) int { + if p == nil { + return 0 + } + return *p +} + +// GroupResolution is where a consistency group's data lives now, keyed by the +// handles its PersistentVolumes keep (sbcli GET +// /consistency-groups/{id}/replication/resolution). +type GroupResolution struct { + // Active is the group holding live members: the group itself while it has + // any, else its peer group of the same name on another cluster. Nil when no + // group holds a live member. + Active *lvol.GroupHandle + // Members has one entry per protected volume with a live volume at the end + // of its lineage: Origin is the handle its PV carries, Active the volume + // serving the data now. + Members []GroupMemberResolution + // Legacy reports a control plane without the resolution endpoint (404): the + // caller then treats the group as live where it is, as before. + Legacy bool +} + +// GroupMemberResolution maps one PV handle to the volume serving its data. +type GroupMemberResolution struct { + Origin lvol.VolumeHandle + Active lvol.VolumeHandle +} + +// ResolveGroup resolves a group handle to where the group's data lives now. +// +// A VolumeGroupReplication keeps its original group handle across a relocate, +// while the group it names is emptied by design (its demoted members are +// deleted so a relocate back stays possible) and the data moves to the peer +// group as clones. The group verbs resolve the handle here, the group analogue +// of the per-volume relationship chain (2026-10-04: WordPress's VRG waited for +// destination info for ever against the emptied source group). +func (c *Client) ResolveGroup(ctx context.Context, gh lvol.GroupHandle) (GroupResolution, error) { + cluster, group, err := groupIDs(gh) + if err != nil { + return GroupResolution{}, err + } + resp, err := c.api.ClustersConsistencyGroupsReplicationResolutionApiV2ClustersClusterIdConsistencyGroupsGroupIdReplicationResolutionGetWithResponse( + ctx, cluster, group) + if err != nil { + return GroupResolution{}, fmt.Errorf("resolve group %s: %w", gh.Handle(), err) + } + if resp.StatusCode() == http.StatusNotFound && resp.JSON200 == nil && !groupNotFound(resp.Body) { + return GroupResolution{Active: &gh, Legacy: true}, nil + } + d, err := payload("resolve group "+string(gh.Handle()), resp.JSON200, resp.StatusCode(), resp.Body) + if err != nil { + return GroupResolution{}, err + } + out := GroupResolution{} + if d.ActiveGroupId != nil && *d.ActiveGroupId != "" && d.ActiveClusterId != nil && *d.ActiveClusterId != "" { + out.Active = &lvol.GroupHandle{ClusterID: *d.ActiveClusterId, GroupID: *d.ActiveGroupId} + } + if d.Members != nil { + for _, m := range *d.Members { + out.Members = append(out.Members, GroupMemberResolution{ + Origin: lvol.VolumeHandle(m.OriginHandle), Active: lvol.VolumeHandle(m.ActiveHandle)}) + } + } + return out, nil +} + +// groupNotFound tells the group-level 404 ("ConsistencyGroup not found", +// the resource dependency's answer) from a route-level 404 (FastAPI's +// {"detail":"Not Found"} on a control plane that predates the endpoint). +func groupNotFound(body []byte) bool { + return bytes.Contains(bytes.ToLower(body), []byte("consistencygroup")) +} diff --git a/atlas-lib/controlplane/replication.go b/atlas-lib/controlplane/replication.go new file mode 100644 index 000000000..ca85cd07d --- /dev/null +++ b/atlas-lib/controlplane/replication.go @@ -0,0 +1,302 @@ +package controlplane + +import ( + "context" + "fmt" + "net/http" + "strings" + "time" + + openapi_types "github.com/oapi-codegen/runtime/types" + + "github.com/simplyblock/atlas/internal/cpapi" + "github.com/simplyblock/atlas/lvol" +) + +func uuidPtrString(u *openapi_types.UUID) string { + if u == nil { + return "" + } + return u.String() +} + +// ReplicationStatus is the typed steady-state replication status of one +// volume, for the volume's whole replicated life -- unlike a cutover-record +// relationship read, this is never a 404 for a volume that exists. +// +// The pointer fields mirror the API's own optionality: a volume that has +// never replicated reports every timing/lag field nil, which is a valid +// answer, not an error, and is a different thing from a genuine zero. +type ReplicationStatus struct { + Role string + State string + + LastReplicatedAt *time.Time + LagSeconds *int + LagBudgetSeconds *int + + OutstandingCount int + OutstandingBytes int + + FailingCount int + MaxRetryReached bool + + LastCycleBytes *int + LastCycleSeconds *int + + Resyncing bool +} + +func replicationStatusFromDTO(d cpapi.ReplicationStatusDTO) ReplicationStatus { + return ReplicationStatus{ + Role: string(d.Role), + State: string(d.State), + LastReplicatedAt: d.LastReplicatedAt, + LagSeconds: d.LagSeconds, + LagBudgetSeconds: d.LagBudgetSeconds, + OutstandingCount: intFrom(d.OutstandingCount), + OutstandingBytes: intFrom(d.OutstandingBytes), + FailingCount: intFrom(d.FailingCount), + MaxRetryReached: boolFrom(d.MaxRetryReached), + LastCycleBytes: d.LastCycleBytes, + LastCycleSeconds: d.LastCycleSeconds, + Resyncing: boolFrom(d.Resyncing), + } +} + +func intFrom(p *int) int { + if p == nil { + return 0 + } + return *p +} + +func boolFrom(p *bool) bool { + if p == nil { + return false + } + return *p +} + +// EnableVolumeReplication attaches the volume to the named replication +// policy, starting replication. Attaching a volume already following that +// same policy is success (the backend's own idempotency); attaching one that +// follows a different policy is refused (control-plane 412), because a +// silent re-attach would force a full re-sync. +func (c *Client) EnableVolumeReplication(ctx context.Context, h lvol.VolumeHandle, policyID string) error { + cluster, pool, volume, err := h.Split() + if err != nil { + return err + } + policy, err := parseUUID("replication policy id", policyID) + if err != nil { + return err + } + resp, err := c.api.ClustersStoragePoolsVolumesUpdateApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdPutWithResponse( + ctx, cluster, pool, volume, cpapi.UpdatableLVolParams{ReplicationPolicyId: &policy}) + if err != nil { + return fmt.Errorf("enable replication on volume %s: %w", h, err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("enable replication on volume "+string(h), code, resp.Body) + } + return nil +} + +// DisableVolumeReplication detaches the volume from whatever replication +// policy it follows, stopping replication. Detaching a volume that follows no +// policy is success (the backend's own idempotency); a 409 (a cutover in +// flight) is a retryable refusal. +func (c *Client) DisableVolumeReplication(ctx context.Context, h lvol.VolumeHandle) error { + cluster, pool, volume, err := h.Split() + if err != nil { + return err + } + // ReplicationPolicyId is `omitempty` on the generated request struct, so + // building it with a nil pointer would drop the key entirely rather than + // send it as an explicit null -- and the backend distinguishes "the key + // was absent" (leave the policy alone) from "the key was null" (detach) + // by which keys the request body carries, not by the decoded value. The + // raw-body variant is the only way to say "null" here. + resp, err := c.api.ClustersStoragePoolsVolumesUpdateApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdPutWithBodyWithResponse( + ctx, cluster, pool, volume, "application/json", strings.NewReader(`{"replication_policy_id":null}`)) + if err != nil { + return fmt.Errorf("disable replication on volume %s: %w", h, err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("disable replication on volume "+string(h), code, resp.Body) + } + return nil +} + +// GetVolumeReplicationInfo returns the volume's typed steady-state +// replication status. +func (c *Client) GetVolumeReplicationInfo(ctx context.Context, h lvol.VolumeHandle) (ReplicationStatus, error) { + cluster, pool, volume, err := h.Split() + if err != nil { + return ReplicationStatus{}, err + } + resp, err := c.api.ClustersStoragePoolsVolumesReplicationStatusApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdReplicationStatusGetWithResponse( + ctx, cluster, pool, volume) + if err != nil { + return ReplicationStatus{}, fmt.Errorf("replication status of volume %s: %w", h, err) + } + d, err := payload("replication status of volume "+string(h), resp.JSON200, resp.StatusCode(), resp.Body) + if err != nil { + return ReplicationStatus{}, err + } + return replicationStatusFromDTO(*d), nil +} + +// PromoteVolume brings the volume up as primary on this cluster. +// +// force=true is the unplanned path: it ignores demote state entirely, +// because its whole premise is that the peer may never have been reachable +// to demote. force=false is the planned path, gated on a completed demote +// (P0-3): a 409 (demote still converging, retryable) or 412 (no demote was +// ever requested) surfaces as a *StatusError the caller classifies -- the +// 409/412 split matters because the vendored csi-addons controller +// auto-escalates ANY FAILED_PRECONDITION from a force=false promote to +// force=true inline, with no wait-and-retry grace period of its own. +func (c *Client) PromoteVolume(ctx context.Context, h lvol.VolumeHandle, force bool) error { + cluster, pool, volume, err := h.Split() + if err != nil { + return err + } + planned := !force + params := &cpapi.ClustersStoragePoolsVolumesReplicationFailoverApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdReplicationFailoverPostParams{ + Planned: &planned, + } + resp, err := c.api.ClustersStoragePoolsVolumesReplicationFailoverApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdReplicationFailoverPostWithResponse( + ctx, cluster, pool, volume, params) + if err != nil { + return fmt.Errorf("promote volume %s: %w", h, err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("promote volume "+string(h), code, resp.Body) + } + return nil +} + +// DemoteVolume fences the source and confirms the last write replicated +// (P0-3) -- the lossless half of a planned swap. Synchronous and +// re-drivable, not queued: it returns done=false while the final snapshot is +// still converging, and the caller (the driver's DemoteVolume RPC) is +// expected to call this again rather than block, matching the backend's own +// call-repeatedly contract. +func (c *Client) DemoteVolume(ctx context.Context, h lvol.VolumeHandle) (bool, error) { + cluster, pool, volume, err := h.Split() + if err != nil { + return false, err + } + resp, err := c.api.ClustersStoragePoolsVolumesReplicationDemoteApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdReplicationDemotePostWithResponse( + ctx, cluster, pool, volume) + if err != nil { + return false, fmt.Errorf("demote volume %s: %w", h, err) + } + switch code := resp.StatusCode(); code { + case http.StatusNoContent: + return true, nil + case http.StatusAccepted: + return false, nil + default: + return false, respError("demote volume "+string(h), code, resp.Body) + } +} + +// ResyncVolume reconciles a diverged copy back onto the current primary's +// history. It configures the reverse direction only -- it never cuts over, +// matching the design's own "it never merges": cutover is PromoteVolume's job +// on a separate, later call. sourceClusterID selects the source explicitly +// when it isn't the cluster's configured default; "" leaves it unset. +func (c *Client) ResyncVolume(ctx context.Context, h lvol.VolumeHandle, sourceClusterID string) error { + cluster, pool, volume, err := h.Split() + if err != nil { + return err + } + var body cpapi.FailbackParams + if sourceClusterID != "" { + id, err := parseUUID("source cluster id", sourceClusterID) + if err != nil { + return err + } + body.SourceClusterId = &id + } + resp, err := c.api.ClustersStoragePoolsVolumesReplicationFailbackApiV2ClustersClusterIdStoragePoolsPoolIdVolumesVolumeIdReplicationFailbackPostWithResponse( + ctx, cluster, pool, volume, body) + if err != nil { + return fmt.Errorf("resync volume %s: %w", h, err) + } + if code := resp.StatusCode(); code != http.StatusOK && code != http.StatusNoContent { + return respError("resync volume "+string(h), code, resp.Body) + } + return nil +} + +// Relationship is the replication pairing h belongs to: which volume +// replicates to which, and which side h itself names (IsSource). TargetClusterID/ +// TargetPoolID/TargetLvolID always name the same, fixed target (replica) side +// of the pairing regardless of whether h names the source or the target -- +// querying by either volume's own id returns the identical target_* answer. +type Relationship struct { + IsSource bool + + SourceClusterID string + SourceLvolID string + + TargetClusterID string + TargetPoolID string + TargetLvolID string + + // ActiveLvolID names the volume currently serving the pairing's data, + // resolved transitively by the backend across chained fail-overs (a + // relocate round trip leaves the FIRST pairing's target pointing at a + // clone that a SECOND pairing has since superseded). A caller cleaning up + // retired copies keys off this: any member that is not the active volume + // is garbage, and the active one must never be touched through a stale + // handle. Empty when the backend predates the field. + ActiveLvolID string +} + +// GetVolumeReplicationRelationship resolves h to its replication pairing. +// This is what a caller handed a volume identity inherited from the OTHER +// side of a pairing (e.g., a destination PVC whose PV was restored carrying +// the source's own volumeHandle) uses to find the volume it should actually +// operate on locally: TargetClusterID/TargetPoolID/TargetLvolID name that +// volume regardless of which side h itself named. Returns an error +// unwrapping to errs.ErrNotFound when h has no replication relationship at +// all yet (e.g., a volume never enabled for replication) -- callers treat that +// as "use h unchanged," not a failure. +// GetVolumeReplicationRelationship reads the relationship through the +// cluster-scoped endpoint, not the pool-scoped one: the pool-scoped route +// requires the queried volume to still exist (sbcli's FastAPI Volume +// dependency 404s before the handler body runs), but a relationship must stay +// resolvable by SOURCE id after the source volume itself is deleted -- e.g., a +// demoted volume whose fail-over already completed and was reaped by +// lvol_monitor's deferred-removal hold, confirmed live 2026-09-24 (relocate +// M-02's round trip: resolveToLocalReplica needs exactly this to redirect +// DemoteVolume/DisableVolumeReplication on the SECOND hop of a relocate). +func (c *Client) GetVolumeReplicationRelationship(ctx context.Context, h lvol.VolumeHandle) (Relationship, error) { + cluster, _, volume, err := h.Split() + if err != nil { + return Relationship{}, err + } + resp, err := c.api.ClustersReplicationRelationshipsDetailApiV2ClustersClusterIdReplicationRelationshipsLvolIdGetWithResponse( + ctx, cluster, volume) + if err != nil { + return Relationship{}, fmt.Errorf("replication relationship of volume %s: %w", h, err) + } + d, err := payload("replication relationship of volume "+string(h), resp.JSON200, resp.StatusCode(), resp.Body) + if err != nil { + return Relationship{}, err + } + return Relationship{ + IsSource: d.IsSource, + SourceClusterID: uuidPtrString(d.SourceClusterId), + SourceLvolID: uuidPtrString(d.SourceLvolId), + TargetClusterID: uuidPtrString(d.TargetClusterId), + TargetPoolID: uuidPtrString(d.TargetPoolId), + TargetLvolID: uuidPtrString(d.TargetLvolId), + ActiveLvolID: uuidPtrString(d.ActiveLvolId), + }, nil +} diff --git a/atlas-lib/controlplane/replication_test.go b/atlas-lib/controlplane/replication_test.go new file mode 100644 index 000000000..e5a1ffa35 --- /dev/null +++ b/atlas-lib/controlplane/replication_test.go @@ -0,0 +1,436 @@ +package controlplane + +import ( + "context" + "errors" + "io" + "net/http" + "strings" + "testing" + + "github.com/simplyblock/atlas/errs" +) + +func TestClientEnableVolumeReplication(t *testing.T) { + var gotBody string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPut { + t.Errorf("method = %s, want PUT", r.Method) + } + b, _ := io.ReadAll(r.Body) + gotBody = string(b) + w.WriteHeader(http.StatusNoContent) + }) + if err := c.EnableVolumeReplication(context.Background(), testHandle, testPolicy); err != nil { + t.Fatal(err) + } + if !strings.Contains(gotBody, `"replication_policy_id":"`+testPolicy+`"`) { + t.Errorf("request body = %q, want it to carry replication_policy_id %s", gotBody, testPolicy) + } +} + +func TestClientEnableVolumeReplicationDifferentPolicyIsAnError(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusPreconditionFailed) + _, _ = w.Write([]byte("already attached to a different policy")) + }) + err := c.EnableVolumeReplication(context.Background(), testHandle, testPolicy) + var se *StatusError + if !errors.As(err, &se) || se.StatusCode != http.StatusPreconditionFailed { + t.Fatalf("err = %v, want a *StatusError carrying 412", err) + } +} + +// DisableVolumeReplication must send an EXPLICIT JSON null for +// replication_policy_id, not omit the key. The generated request struct's +// field is `omitempty`, so a nil pointer there would be dropped from the body +// entirely -- and the backend distinguishes "the key was absent" (leave the +// policy alone) from "the key was null" (detach) by which keys the request +// actually carries, not by the value. A body of `{}` would silently do +// nothing. +func TestClientDisableVolumeReplicationSendsExplicitNull(t *testing.T) { + var gotBody string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodPut { + t.Errorf("method = %s, want PUT", r.Method) + } + b, _ := io.ReadAll(r.Body) + gotBody = string(b) + w.WriteHeader(http.StatusNoContent) + }) + if err := c.DisableVolumeReplication(context.Background(), testHandle); err != nil { + t.Fatal(err) + } + if !strings.Contains(gotBody, `"replication_policy_id":null`) { + t.Errorf("request body = %q, want an explicit null for replication_policy_id", gotBody) + } +} + +func TestClientDisableVolumeReplicationNotAttachedIsSuccess(t *testing.T) { + // The backend's own idempotency: detaching a volume that follows no + // policy already returns success, so the client has nothing extra to do + // here beyond not treating any 2xx as an error. + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusOK) + }) + if err := c.DisableVolumeReplication(context.Background(), testHandle); err != nil { + t.Errorf("DisableVolumeReplication = %v, want nil", err) + } +} + +func TestClientDisableVolumeReplicationDuringCutoverIsAnError(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusConflict) + }) + err := c.DisableVolumeReplication(context.Background(), testHandle) + if !errors.Is(err, errs.ErrAlreadyExists) { + t.Fatalf("err = %v, want it to unwrap to a 409 sentinel so the caller can retry", err) + } +} + +func TestClientGetVolumeReplicationInfo(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if !strings.HasSuffix(r.URL.Path, "/replication/status") { + t.Errorf("unexpected path %q", r.URL.Path) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "role": "source", "state": "in_sync", + "last_replicated_at": "2026-09-17T12:00:00Z", + "lag_seconds": 42, "lag_budget_seconds": 900, + "outstanding_count": 1, "outstanding_bytes": 1048576, + "failing_count": 0, "max_retry_reached": false, + "last_cycle_bytes": 2097152, "last_cycle_seconds": 12, + "resyncing": false + }`)) + }) + + info, err := c.GetVolumeReplicationInfo(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if info.Role != "source" || info.State != "in_sync" { + t.Errorf("role/state = %q/%q, want source/in_sync", info.Role, info.State) + } + if info.LastReplicatedAt == nil || info.LastReplicatedAt.Unix() != 1789646400 { + t.Errorf("LastReplicatedAt = %v", info.LastReplicatedAt) + } + if info.LagSeconds == nil || *info.LagSeconds != 42 { + t.Errorf("LagSeconds = %v, want 42", info.LagSeconds) + } + if info.LagBudgetSeconds == nil || *info.LagBudgetSeconds != 900 { + t.Errorf("LagBudgetSeconds = %v, want 900", info.LagBudgetSeconds) + } + if info.OutstandingCount != 1 || info.OutstandingBytes != 1048576 { + t.Errorf("outstanding = %d/%d, want 1/1048576", info.OutstandingCount, info.OutstandingBytes) + } + if info.FailingCount != 0 || info.MaxRetryReached { + t.Errorf("failing/max-retry = %d/%v, want 0/false", info.FailingCount, info.MaxRetryReached) + } + if info.LastCycleBytes == nil || *info.LastCycleBytes != 2097152 { + t.Errorf("LastCycleBytes = %v, want 2097152", info.LastCycleBytes) + } + if info.LastCycleSeconds == nil || *info.LastCycleSeconds != 12 { + t.Errorf("LastCycleSeconds = %v, want 12", info.LastCycleSeconds) + } + if info.Resyncing { + t.Error("Resyncing = true, want false") + } +} + +// A volume that never replicated is a valid, non-error answer: role "none," +// state "not_replicating," and every timing/lag field null. +func TestClientGetVolumeReplicationInfoNeverReplicated(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "role": "none", "state": "not_replicating", + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": false + }`)) + }) + + info, err := c.GetVolumeReplicationInfo(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if info.Role != "none" || info.State != "not_replicating" { + t.Errorf("role/state = %q/%q", info.Role, info.State) + } + if info.LastReplicatedAt != nil || info.LagSeconds != nil || info.LagBudgetSeconds != nil { + t.Errorf("expected every timing field nil, got LastReplicatedAt=%v LagSeconds=%v LagBudgetSeconds=%v", + info.LastReplicatedAt, info.LagSeconds, info.LagBudgetSeconds) + } +} + +func TestClientGetVolumeReplicationInfoNotFound(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusNotFound) + }) + if _, err := c.GetVolumeReplicationInfo(context.Background(), testHandle); !errors.Is(err, errs.ErrNotFound) { + t.Errorf("err = %v, want ErrNotFound", err) + } +} + +func TestClientPromoteVolumeForcedIgnoresDemoteState(t *testing.T) { + var gotQuery string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if !strings.HasSuffix(r.URL.Path, "/replication/failover") { + t.Errorf("unexpected path %q", r.URL.Path) + } + gotQuery = r.URL.RawQuery + w.WriteHeader(http.StatusNoContent) + }) + if err := c.PromoteVolume(context.Background(), testHandle, true); err != nil { + t.Fatal(err) + } + if strings.Contains(gotQuery, "planned=true") { + t.Errorf("query = %q, forced promote must not ask for the planned gate", gotQuery) + } +} + +func TestClientPromoteVolumePlannedSendsThePlannedFlag(t *testing.T) { + var gotQuery string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + gotQuery = r.URL.RawQuery + w.WriteHeader(http.StatusNoContent) + }) + if err := c.PromoteVolume(context.Background(), testHandle, false); err != nil { + t.Fatal(err) + } + if !strings.Contains(gotQuery, "planned=true") { + t.Errorf("query = %q, want planned=true", gotQuery) + } +} + +// The whole point of the planned gate: a demote still converging must surface +// as something the driver can map to codes.Aborted (retryable), never +// codes.FailedPrecondition -- the vendored csi-addons controller escalates +// ANY FAILED_PRECONDITION from a force=false promote to force=true inline, +// with no wait-and-retry grace period of its own. +func TestClientPromoteVolumeWhileDemoteConvergingIsA409(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusConflict) + }) + err := c.PromoteVolume(context.Background(), testHandle, false) + var se *StatusError + if !errors.As(err, &se) || se.StatusCode != http.StatusConflict { + t.Fatalf("err = %v, want a *StatusError carrying 409", err) + } +} + +func TestClientPromoteVolumeWithNoDemoteIsA412(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusPreconditionFailed) + }) + err := c.PromoteVolume(context.Background(), testHandle, false) + var se *StatusError + if !errors.As(err, &se) || se.StatusCode != http.StatusPreconditionFailed { + t.Fatalf("err = %v, want a *StatusError carrying 412", err) + } +} + +func TestClientDemoteVolumeNotYetDone(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if !strings.HasSuffix(r.URL.Path, "/replication/demote") { + t.Errorf("unexpected path %q", r.URL.Path) + } + w.WriteHeader(http.StatusAccepted) + }) + done, err := c.DemoteVolume(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if done { + t.Error("done = true, want false while still converging") + } +} + +func TestClientDemoteVolumeDone(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusNoContent) + }) + done, err := c.DemoteVolume(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if !done { + t.Error("done = false, want true") + } +} + +func TestClientDemoteVolumeFailureIsAnError(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusInternalServerError) + }) + if _, err := c.DemoteVolume(context.Background(), testHandle); err == nil { + t.Error("want an error on a genuine backend failure") + } +} + +func TestClientResyncVolume(t *testing.T) { + var gotBody string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + if !strings.HasSuffix(r.URL.Path, "/replication/failback") { + t.Errorf("unexpected path %q", r.URL.Path) + } + b, _ := io.ReadAll(r.Body) + gotBody = string(b) + w.WriteHeader(http.StatusNoContent) + }) + if err := c.ResyncVolume(context.Background(), testHandle, testCluster); err != nil { + t.Fatal(err) + } + if !strings.Contains(gotBody, `"source_cluster_id":"`+testCluster+`"`) { + t.Errorf("request body = %q, want source_cluster_id %s", gotBody, testCluster) + } +} + +func TestClientResyncVolumeWithoutSourceCluster(t *testing.T) { + var gotBody string + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + b, _ := io.ReadAll(r.Body) + gotBody = string(b) + w.WriteHeader(http.StatusNoContent) + }) + if err := c.ResyncVolume(context.Background(), testHandle, ""); err != nil { + t.Fatal(err) + } + if strings.Contains(gotBody, "source_cluster_id") { + t.Errorf("request body = %q, want no source_cluster_id when none is given", gotBody) + } +} + +const ( + testTargetCluster = "44444444-4444-4444-4444-444444444444" + testTargetPool = "55555555-5555-5555-5555-555555555555" + testTargetVolume = "66666666-6666-6666-6666-666666666667" +) + +// GetVolumeReplicationRelationship must call the CLUSTER-scoped endpoint +// (GET /clusters/{cluster_id}/replication/relationships/{lvol_id}), never the +// pool-scoped one (.../storage-pools/{pool}/volumes/{volume}/replication/): +// the pool-scoped route requires the queried volume to still exist (sbcli's +// FastAPI Volume dependency 404s before the handler body even runs), but a +// relationship must stay resolvable by SOURCE id after the source volume +// itself is gone -- e.g., a demoted volume whose fail-over already completed +// and was reaped by lvol_monitor's deferred-removal hold (confirmed live +// 2026-09-24, relocate M-02's round trip: DemoteVolume/DisableVolumeReplication +// on the SECOND hop 404'd resolving through the pool-scoped endpoint against +// exactly this). sbcli's cluster-scoped endpoint is built for this case -- +// its own docstring: "resolvable even when the source volume has been +// deleted... The CSI driver uses this to redirect." +func TestClientGetVolumeReplicationRelationship(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + wantPath := "/api/v2/clusters/" + testCluster + "/replication/relationships/" + testVolume + if r.URL.Path != wantPath { + t.Errorf("path = %q, want %q", r.URL.Path, wantPath) + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "replication_id": "` + testCluster + `", + "direction": "to_target", + "mode": "migration", + "state": "replicating", + "is_source": true, + "source_cluster_id": "` + testCluster + `", + "source_lvol_id": "` + testVolume + `", + "target_cluster_id": "` + testTargetCluster + `", + "target_pool_id": "` + testTargetPool + `", + "target_lvol_id": "` + testTargetVolume + `", + "target_nqn": "nqn.test", + "target_ns_id": 1 + }`)) + }) + + rel, err := c.GetVolumeReplicationRelationship(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if !rel.IsSource { + t.Error("IsSource = false, want true (queried by the source volume)") + } + if rel.TargetClusterID != testTargetCluster || rel.TargetPoolID != testTargetPool || rel.TargetLvolID != testTargetVolume { + t.Errorf("target = %s/%s/%s, want %s/%s/%s", + rel.TargetClusterID, rel.TargetPoolID, rel.TargetLvolID, + testTargetCluster, testTargetPool, testTargetVolume) + } +} + +// Regression: 2026-09-24-delete-foreign-handle-leak — the backend resolves +// active_lvol_id transitively across chained fail-overs, and the driver's +// retired-copy cleanup keys off it: without it parsed, the cleanup cannot +// tell a stale clone from the volume the workload currently runs on. +func TestClientGetVolumeReplicationRelationshipParsesTheActiveVolume(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "replication_id": "` + testCluster + `", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": "` + testCluster + `", + "source_lvol_id": "` + testVolume + `", + "target_cluster_id": "` + testTargetCluster + `", + "target_pool_id": "` + testTargetPool + `", + "target_lvol_id": "` + testTargetVolume + `", + "target_nqn": "nqn.test", + "target_ns_id": 1, + "active": "target", + "active_lvol_id": "` + testTargetVolume + `" + }`)) + }) + + rel, err := c.GetVolumeReplicationRelationship(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if rel.ActiveLvolID != testTargetVolume { + t.Errorf("ActiveLvolID = %q, want %q", rel.ActiveLvolID, testTargetVolume) + } +} + +// Queried by the TARGET volume's own id, the relationship still reports the +// SAME fixed target_* fields -- this is what lets a caller always resolve to +// target_* regardless of which side of the pairing it was handed, without +// having to branch on IsSource first. +func TestClientGetVolumeReplicationRelationshipQueriedByTarget(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{ + "replication_id": "` + testCluster + `", + "direction": "to_target", + "mode": "migration", + "state": "replicating", + "is_source": false, + "source_cluster_id": "` + testCluster + `", + "source_lvol_id": "` + testVolume + `", + "target_cluster_id": "` + testTargetCluster + `", + "target_pool_id": "` + testTargetPool + `", + "target_lvol_id": "` + testTargetVolume + `", + "target_nqn": "nqn.test", + "target_ns_id": 1 + }`)) + }) + + rel, err := c.GetVolumeReplicationRelationship(context.Background(), testHandle) + if err != nil { + t.Fatal(err) + } + if rel.IsSource { + t.Error("IsSource = true, want false (queried by the target volume)") + } + if rel.TargetLvolID != testTargetVolume { + t.Errorf("TargetLvolID = %s, want %s (unchanged regardless of which side was queried)", rel.TargetLvolID, testTargetVolume) + } +} + +func TestClientGetVolumeReplicationRelationshipNotFound(t *testing.T) { + c := newTestClient(t, func(w http.ResponseWriter, r *http.Request) { + w.WriteHeader(http.StatusNotFound) + }) + if _, err := c.GetVolumeReplicationRelationship(context.Background(), testHandle); !errors.Is(err, errs.ErrNotFound) { + t.Errorf("err = %v, want ErrNotFound for a volume with no replication relationship yet", err) + } +} diff --git a/atlas-lib/devmapper/devmapper.go b/atlas-lib/devmapper/devmapper.go new file mode 100644 index 000000000..0baab0e25 --- /dev/null +++ b/atlas-lib/devmapper/devmapper.go @@ -0,0 +1,150 @@ +// Package devmapper manages single-segment dm-linear devices: the indirection +// between an NVMe-oF namespace and what a pod or a filesystem uses. +// +// A dm-linear device maps its whole range onto one underlying block device and +// writes nothing to it, so it can be put in front of a volume that already +// carries data. What it buys is a device whose identity does not change when +// the namespace behind it does: when a volume's namespace moves to another NVMe +// subsystem (consistency-group co-location), the new namespace shows up as a +// different block device, and Swap re-points the mapping at it -- suspend, load +// the new table, resume -- while the consumer keeps the device it opened. I/O +// issued during the suspend is queued by device-mapper, not failed. +package devmapper + +import ( + "context" + "errors" + "fmt" + "os" + "os/exec" + "strconv" + "strings" +) + +// Runner execs a command and returns its combined output. Tests substitute a +// recorder; production execs the host's dmsetup and blockdev. +type Runner func(ctx context.Context, args ...string) (string, error) + +// ErrNotFound reports that the named mapping does not exist. +var ErrNotFound = errors.New("device-mapper mapping not found") + +// Mapper creates, reads, re-points and removes dm-linear mappings. +type Mapper struct { + run Runner +} + +// New returns a Mapper over run; nil runs the host's commands. +func New(run Runner) *Mapper { + if run == nil { + run = runCommand + } + return &Mapper{run: run} +} + +func runCommand(ctx context.Context, args ...string) (string, error) { + //nolint:gosec // fixed set of binaries (dmsetup, blockdev), structured args + cmd := exec.CommandContext(ctx, args[0], args[1:]...) + // No udev in the node plugin's container to complete the handshake. + cmd.Env = append(os.Environ(), "DM_DISABLE_UDEV=1") + out, err := cmd.CombinedOutput() + if err != nil { + return string(out), fmt.Errorf("%v: %w: %s", args, err, strings.TrimSpace(string(out))) + } + return string(out), nil +} + +// Path is the device node of mapping name. +func Path(name string) string { return "/dev/mapper/" + name } + +// Target is what a mapping currently points at. +type Target struct { + // Sectors is the mapped length in 512-byte sectors. + Sectors uint64 + // Device is the underlying device as "major:minor". + Device string +} + +// table renders the one-segment linear table for device. +func table(sectors uint64, device string) string { + return fmt.Sprintf("0 %d linear %s 0", sectors, device) +} + +// Sectors is device's size in 512-byte sectors. +func (m *Mapper) Sectors(ctx context.Context, device string) (uint64, error) { + out, err := m.run(ctx, "blockdev", "--getsz", device) + if err != nil { + return 0, fmt.Errorf("size of %s: %w", device, err) + } + n, err := strconv.ParseUint(strings.TrimSpace(out), 10, 64) + if err != nil { + return 0, fmt.Errorf("size of %s: unexpected %q", device, strings.TrimSpace(out)) + } + return n, nil +} + +// Table reads mapping name's target, or ErrNotFound. +func (m *Mapper) Table(ctx context.Context, name string) (Target, error) { + out, err := m.run(ctx, "dmsetup", "table", name) + if err != nil { + if strings.Contains(out, "No such device") || strings.Contains(err.Error(), "No such device") { + return Target{}, ErrNotFound + } + return Target{}, fmt.Errorf("dmsetup table %s: %w", name, err) + } + return parseTable(name, out) +} + +func parseTable(name, out string) (Target, error) { + lines := strings.Split(strings.TrimSpace(out), "\n") + if len(lines) != 1 { + return Target{}, fmt.Errorf("mapping %s has %d segments; a linear indirection has one", name, len(lines)) + } + f := strings.Fields(lines[0]) + if len(f) != 5 || f[2] != "linear" || f[0] != "0" || f[4] != "0" { + return Target{}, fmt.Errorf("mapping %s is not a whole-device linear mapping: %q", name, lines[0]) + } + sectors, err := strconv.ParseUint(f[1], 10, 64) + if err != nil { + return Target{}, fmt.Errorf("mapping %s: length %q: %w", name, f[1], err) + } + return Target{Sectors: sectors, Device: f[3]}, nil +} + +// Create maps name onto device over sectors. +func (m *Mapper) Create(ctx context.Context, name, device string, sectors uint64) error { + if _, err := m.run(ctx, "dmsetup", "create", name, "--table", table(sectors, device)); err != nil { + return fmt.Errorf("dmsetup create %s: %w", name, err) + } + return nil +} + +// Swap re-points name at device: suspend (in-flight I/O drains, new I/O is +// queued), load the new table, resume. A failed load resumes on the old table, +// so a swap never leaves the mapping suspended. +func (m *Mapper) Swap(ctx context.Context, name, device string, sectors uint64) error { + if _, err := m.run(ctx, "dmsetup", "suspend", name); err != nil { + return fmt.Errorf("dmsetup suspend %s: %w", name, err) + } + if _, err := m.run(ctx, "dmsetup", "reload", name, "--table", table(sectors, device)); err != nil { + if _, rerr := m.run(ctx, "dmsetup", "resume", name); rerr != nil { + return fmt.Errorf("dmsetup reload %s: %w; resume on the old table also failed: %v", name, err, rerr) + } + return fmt.Errorf("dmsetup reload %s: %w (resumed on the old table)", name, err) + } + if _, err := m.run(ctx, "dmsetup", "resume", name); err != nil { + return fmt.Errorf("dmsetup resume %s: %w", name, err) + } + return nil +} + +// Remove deletes mapping name; a missing mapping is success. +func (m *Mapper) Remove(ctx context.Context, name string) error { + out, err := m.run(ctx, "dmsetup", "remove", "--retry", name) + if err != nil { + if strings.Contains(out, "No such device") || strings.Contains(err.Error(), "No such device") { + return nil + } + return fmt.Errorf("dmsetup remove %s: %w", name, err) + } + return nil +} diff --git a/atlas-lib/devmapper/devmapper_test.go b/atlas-lib/devmapper/devmapper_test.go new file mode 100644 index 000000000..2bf468a7a --- /dev/null +++ b/atlas-lib/devmapper/devmapper_test.go @@ -0,0 +1,106 @@ +package devmapper + +import ( + "context" + "errors" + "strings" + "testing" +) + +type recorder struct { + calls []string + out map[string]string + err map[string]error +} + +func (r *recorder) run(_ context.Context, args ...string) (string, error) { + key := strings.Join(args, " ") + r.calls = append(r.calls, key) + for prefix, err := range r.err { + if strings.HasPrefix(key, prefix) { + return r.out[prefix], err + } + } + for prefix, out := range r.out { + if strings.HasPrefix(key, prefix) { + return out, nil + } + } + return "", nil +} + +func TestTableParsesAWholeDeviceLinearMapping(t *testing.T) { + r := &recorder{out: map[string]string{"dmsetup table sb-v": "0 2097152 linear 259:3 0\n"}} + got, err := New(r.run).Table(context.Background(), "sb-v") + if err != nil || got != (Target{Sectors: 2097152, Device: "259:3"}) { + t.Fatalf("Table = %+v, %v", got, err) + } +} + +func TestTableOfAMissingMappingIsErrNotFound(t *testing.T) { + r := &recorder{ + out: map[string]string{"dmsetup table": "device-mapper: table ioctl on sb-v failed: No such device or address"}, + err: map[string]error{"dmsetup table": errors.New("exit status 1")}, + } + if _, err := New(r.run).Table(context.Background(), "sb-v"); !errors.Is(err, ErrNotFound) { + t.Fatalf("err = %v, want ErrNotFound", err) + } +} + +func TestTableRefusesAMappingThatIsNotOneLinearSegment(t *testing.T) { + for _, out := range []string{ + "0 100 striped 2 128 259:3 0 259:4 0", + "0 100 linear 259:3 0\n100 100 linear 259:4 0", + "0 100 linear 259:3 8", + } { + if _, err := parseTable("sb-v", out); err == nil { + t.Errorf("accepted %q", out) + } + } +} + +func TestSwapSuspendsReloadsAndResumes(t *testing.T) { + r := &recorder{} + if err := New(r.run).Swap(context.Background(), "sb-v", "259:11", 2048); err != nil { + t.Fatal(err) + } + want := []string{ + "dmsetup suspend sb-v", + "dmsetup reload sb-v --table 0 2048 linear 259:11 0", + "dmsetup resume sb-v", + } + if strings.Join(r.calls, "|") != strings.Join(want, "|") { + t.Fatalf("calls = %v, want %v", r.calls, want) + } +} + +// A refused reload must not leave the mapping suspended: I/O queued during the +// suspend would hang until someone resumed it by hand. +func TestAFailedReloadResumesOnTheOldTable(t *testing.T) { + r := &recorder{err: map[string]error{"dmsetup reload": errors.New("invalid table")}} + err := New(r.run).Swap(context.Background(), "sb-v", "259:11", 2048) + if err == nil || !strings.Contains(err.Error(), "resumed on the old table") { + t.Fatalf("err = %v", err) + } + if r.calls[len(r.calls)-1] != "dmsetup resume sb-v" { + t.Fatalf("last call %q, want a resume", r.calls[len(r.calls)-1]) + } +} + +func TestRemovingAMissingMappingSucceeds(t *testing.T) { + r := &recorder{ + out: map[string]string{"dmsetup remove": "No such device or address"}, + err: map[string]error{"dmsetup remove": errors.New("exit status 1")}, + } + if err := New(r.run).Remove(context.Background(), "sb-v"); err != nil { + t.Fatal(err) + } +} + +func TestSectorsReadsBlockdev(t *testing.T) { + r := &recorder{out: map[string]string{"blockdev --getsz /dev/nvme1n1": "2097152\n"}} + n, err := New(r.run).Sectors(context.Background(), "/dev/nvme1n1") + if err != nil || n != 2097152 { + t.Fatalf("Sectors = %d, %v", n, err) + } +} diff --git a/atlas-lib/lvol/grouphandle.go b/atlas-lib/lvol/grouphandle.go new file mode 100644 index 000000000..ced8902e5 --- /dev/null +++ b/atlas-lib/lvol/grouphandle.go @@ -0,0 +1,63 @@ +// Consistency-group volume handles: the identity a csi-addons VolumeGroup and +// its group-level replication verbs address, kept distinct from a per-volume +// handle so the driver's Replication verbs can route a whole group to the +// cluster-scoped group-replication endpoints instead of the per-volume ones +// (design-csi-addons-replication.md §14.4). It lives beside handle.go because it +// is the same handle grammar with a group sentinel. +package lvol + +import "strings" + +// groupHandlePrefix marks a handle as naming a consistency group rather than a +// single volume. A per-volume handle's first segment is a cluster UUID, so a +// non-UUID sentinel here keeps the two grammars unambiguous: ParseHandle rejects +// a group handle (its first segment is not a UUID), and ParseGroupHandle rejects +// a per-volume one (it lacks the sentinel). +const groupHandlePrefix = "cg" + +// GroupHandle is a consistency-group handle taken apart: the cluster and the +// group it names. The csi-addons VolumeGroup service returns one of these as a +// group's replication handle. The cluster-scoped group-replication endpoints +// need only the cluster and the group id, so a handle carries no pool segment. +type GroupHandle struct { + // ClusterID and GroupID are canonical UUIDs, spelled exactly as the handle + // spells them (not normalized), for the same string-comparison reason + // Handle keeps its segments verbatim. + ClusterID string + GroupID string +} + +// ParseGroupHandle splits a consistency-group handle (cg:{clusterID}:{groupID}) +// into the cluster and group it names, reporting whether it was well formed. +// Both ids must be canonical UUIDs, and the leading "cg" sentinel is what +// distinguishes it from a per-volume handle. Surrounding whitespace is trimmed, +// as ParseHandle trims it, because a handle is read back out of a YAML object. +func ParseGroupHandle(h VolumeHandle) (GroupHandle, bool) { + parts := strings.Split(strings.TrimSpace(string(h)), handleSeparator) + if len(parts) != 3 || parts[0] != groupHandlePrefix { + return GroupHandle{}, false + } + clusterID, groupID := parts[1], parts[2] + if !IsCanonicalUUID(clusterID) || !IsCanonicalUUID(groupID) { + return GroupHandle{}, false + } + return GroupHandle{ClusterID: clusterID, GroupID: groupID}, true +} + +// String renders the group handle back into the form ParseGroupHandle reads. +func (h GroupHandle) String() string { + return strings.Join([]string{groupHandlePrefix, h.ClusterID, h.GroupID}, handleSeparator) +} + +// Handle renders the parts into a VolumeHandle. +func (h GroupHandle) Handle() VolumeHandle { + return VolumeHandle(h.String()) +} + +// IsGroupHandle reports whether a handle names a consistency group rather than a +// single volume. The driver's Replication verbs branch on this to route a group +// handle to the group-replication endpoints (design §14.4). +func IsGroupHandle(h VolumeHandle) bool { + _, ok := ParseGroupHandle(h) + return ok +} diff --git a/atlas-lib/lvol/grouphandle_test.go b/atlas-lib/lvol/grouphandle_test.go new file mode 100644 index 000000000..0562ec629 --- /dev/null +++ b/atlas-lib/lvol/grouphandle_test.go @@ -0,0 +1,80 @@ +package lvol + +import "testing" + +func TestParseGroupHandle(t *testing.T) { + const ( + cluster = "8ffac363-0c46-4714-a71b-f9c0b58a1269" + group = "a1111111-1111-4111-8111-111111111111" + pool = "df34f16c-1a2b-3c4d-5e6f-7a8b9c0d1e2f" + volume = "b2222222-2222-4222-8222-222222222222" + ) + + t.Run("parses a well-formed group handle", func(t *testing.T) { + got, ok := ParseGroupHandle(VolumeHandle("cg:" + cluster + ":" + group)) + if !ok { + t.Fatalf("ParseGroupHandle rejected a well-formed handle") + } + if got.ClusterID != cluster || got.GroupID != group { + t.Fatalf("ParseGroupHandle = %+v, want cluster=%s group=%s", got, cluster, group) + } + }) + + t.Run("trims surrounding whitespace", func(t *testing.T) { + if _, ok := ParseGroupHandle(VolumeHandle(" cg:" + cluster + ":" + group + "\n")); !ok { + t.Fatalf("ParseGroupHandle did not trim whitespace") + } + }) + + t.Run("round-trips through Handle", func(t *testing.T) { + h := GroupHandle{ClusterID: cluster, GroupID: group} + got, ok := ParseGroupHandle(h.Handle()) + if !ok || got != h { + t.Fatalf("round-trip: got %+v ok=%v, want %+v", got, ok, h) + } + }) + + reject := []struct { + name, handle string + }{ + {"a per-volume handle", cluster + ":" + pool + ":" + volume}, + {"the wrong sentinel", "vg:" + cluster + ":" + group}, + {"a missing sentinel", cluster + ":" + group}, + {"a non-UUID cluster", "cg:not-a-uuid:" + group}, + {"a non-UUID group", "cg:" + cluster + ":not-a-uuid"}, + {"too few segments", "cg:" + cluster}, + {"too many segments", "cg:" + cluster + ":" + group + ":extra"}, + {"empty", ""}, + } + for _, tc := range reject { + t.Run("rejects "+tc.name, func(t *testing.T) { + if _, ok := ParseGroupHandle(VolumeHandle(tc.handle)); ok { + t.Fatalf("ParseGroupHandle accepted %q", tc.handle) + } + }) + } +} + +// The two grammars must not overlap: a per-volume handle is never a group +// handle, and a group handle is never a per-volume handle, so the routing branch +// can tell them apart. +func TestGroupAndVolumeHandlesAreDisjoint(t *testing.T) { + const ( + cluster = "8ffac363-0c46-4714-a71b-f9c0b58a1269" + group = "a1111111-1111-4111-8111-111111111111" + pool = "df34f16c-1a2b-3c4d-5e6f-7a8b9c0d1e2f" + volume = "b2222222-2222-4222-8222-222222222222" + ) + groupHandle := VolumeHandle("cg:" + cluster + ":" + group) + volumeHandle := VolumeHandle(cluster + ":" + pool + ":" + volume) + + if _, ok := ParseHandle(groupHandle); ok { + t.Errorf("ParseHandle accepted a group handle %q", groupHandle) + } + if !IsGroupHandle(groupHandle) { + t.Errorf("IsGroupHandle(%q) = false, want true", groupHandle) + } + if IsGroupHandle(volumeHandle) { + t.Errorf("IsGroupHandle(%q) = true, want false", volumeHandle) + } +} diff --git a/atlas-lib/volstack/layers/contract_test.go b/atlas-lib/volstack/layers/contract_test.go index 5d23b055f..6cdd86853 100644 --- a/atlas-lib/volstack/layers/contract_test.go +++ b/atlas-lib/volstack/layers/contract_test.go @@ -68,6 +68,13 @@ func shippedLayers() map[string]volstack.Layer { Ops: newFakeFS(), Content: fakeReader{reading: blockdev.Reading{Content: blockdev.ContentBlank}}, }), + "dmLinear": NewDMLinear(DMLinearConfig{ + Name: DMLinearName("vol-x"), + Mapper: newFakeMapper(), + Resolve: func(path string) (blockdev.Device, error) { + return blockdev.Device{Path: path, Major: 253, Minor: 1}, nil + }, + }), } } diff --git a/atlas-lib/volstack/layers/dmlinear.go b/atlas-lib/volstack/layers/dmlinear.go new file mode 100644 index 000000000..6e3c40b77 --- /dev/null +++ b/atlas-lib/volstack/layers/dmlinear.go @@ -0,0 +1,184 @@ +// The dmLinear layer: a dm-linear device between the fabric and whatever uses +// the volume, so the device the consumer holds survives a change of the +// namespace behind it. +// +// A volume's namespace can move to another NVMe subsystem (consistency-group +// co-location, sbcli docs/consistency-group-colocation.md §6). The new +// namespace is a different block device: without an indirection the pod's +// raw block device or the mounted filesystem would have to be torn down and +// set up again. With it, the fabric layer below brings up the new namespace, +// and this layer's Heal re-points its mapping there (suspend, load, resume) +// while the device above stays the same node with the same data. +package layers + +import ( + "context" + "errors" + "fmt" + + "github.com/simplyblock/atlas/blockdev" + "github.com/simplyblock/atlas/devmapper" + "github.com/simplyblock/atlas/volstack" +) + +// DMMapper is the device-mapper surface the layer uses; *devmapper.Mapper +// satisfies it, and tests substitute a recorder. +type DMMapper interface { + Table(ctx context.Context, name string) (devmapper.Target, error) + Sectors(ctx context.Context, device string) (uint64, error) + Create(ctx context.Context, name, device string, sectors uint64) error + Swap(ctx context.Context, name, device string, sectors uint64) error + Remove(ctx context.Context, name string) error +} + +// DMLinearConfig is what a dmLinear layer is built with. +type DMLinearConfig struct { + // Name is the mapping's name, derived from the volume's identity so a + // restage finds the mapping it made (sb-). + Name string + // Mapper runs the device-mapper commands. + Mapper DMMapper + // Resolve describes the mapping's device node upward; defaults to + // blockdev.ResolveDevice. + Resolve DeviceResolver +} + +// DMLinear is the indirection layer. +type DMLinear struct { + cfg DMLinearConfig + // stale is what the last Observe found: the mapping points somewhere else + // than the device below now is. The heal loop asks Healthy without the + // layer below, so the comparison is made where both are known. + stale bool +} + +// NewDMLinear returns the layer. +func NewDMLinear(cfg DMLinearConfig) *DMLinear { + if cfg.Resolve == nil { + cfg.Resolve = blockdev.ResolveDevice + } + return &DMLinear{cfg: cfg} +} + +// Name is what the record calls this layer. +func (d *DMLinear) Name() string { return "dmLinear" } + +// DMLinearName is the mapping name of the volume with UUID uuid. +func DMLinearName(uuid string) string { return "sb-" + uuid } + +func deviceNumber(dev blockdev.Device) string { return fmt.Sprintf("%d:%d", dev.Major, dev.Minor) } + +func belowDevice(below volstack.Artifact) (blockdev.Device, bool) { + if len(below.Devices) != 1 { + return blockdev.Device{}, false + } + return below.Devices[0], true +} + +// Observe reports the mapping: absent, ready when it points at the device +// below, and partial when it points elsewhere (the namespace moved and the +// fabric below brought up the new one) or there is nothing below to point at. +func (d *DMLinear) Observe(ctx context.Context, below volstack.Artifact) (volstack.State, volstack.Artifact, error) { + target, err := d.cfg.Mapper.Table(ctx, d.cfg.Name) + if errors.Is(err, devmapper.ErrNotFound) { + d.stale = false + return volstack.StateAbsent, volstack.Artifact{}, nil + } + if err != nil { + return volstack.StateAbsent, volstack.Artifact{}, fmt.Errorf("dmLinear: read %s: %w", d.cfg.Name, err) + } + own, err := d.own() + if err != nil { + return volstack.StateAbsent, volstack.Artifact{}, err + } + dev, ok := belowDevice(below) + d.stale = !ok || target.Device != deviceNumber(dev) + if d.stale { + return volstack.StatePartial, own, nil + } + return volstack.StateReady, own, nil +} + +func (d *DMLinear) own() (volstack.Artifact, error) { + dev, err := d.cfg.Resolve(devmapper.Path(d.cfg.Name)) + if err != nil { + return volstack.Artifact{}, fmt.Errorf("dmLinear: resolve %s: %w", devmapper.Path(d.cfg.Name), err) + } + return volstack.Artifact{Devices: []blockdev.Device{dev}}, nil +} + +// Ensure creates the mapping over the device below, or re-points it there. +func (d *DMLinear) Ensure(ctx context.Context, below volstack.Artifact) (volstack.Artifact, error) { + state, own, err := d.Observe(ctx, below) + if err != nil { + return volstack.Artifact{}, err + } + switch state { + case volstack.StateReady: + return own, nil + case volstack.StatePartial: + if err := d.swap(ctx, below); err != nil { + return volstack.Artifact{}, err + } + return d.own() + } + dev, ok := belowDevice(below) + if !ok { + return volstack.Artifact{}, fmt.Errorf("dmLinear: %s needs exactly one device below, got %d", + d.cfg.Name, len(below.Devices)) + } + sectors, err := d.cfg.Mapper.Sectors(ctx, dev.Path) + if err != nil { + return volstack.Artifact{}, fmt.Errorf("dmLinear: %w", err) + } + if err := d.cfg.Mapper.Create(ctx, d.cfg.Name, deviceNumber(dev), sectors); err != nil { + return volstack.Artifact{}, fmt.Errorf("dmLinear: %w", err) + } + return d.own() +} + +func (d *DMLinear) swap(ctx context.Context, below volstack.Artifact) error { + dev, ok := belowDevice(below) + if !ok { + return fmt.Errorf("dmLinear: cannot re-point %s: no device below", d.cfg.Name) + } + sectors, err := d.cfg.Mapper.Sectors(ctx, dev.Path) + if err != nil { + return fmt.Errorf("dmLinear: %w", err) + } + if err := d.cfg.Mapper.Swap(ctx, d.cfg.Name, deviceNumber(dev), sectors); err != nil { + return fmt.Errorf("dmLinear: %w", err) + } + d.stale = false + return nil +} + +// Release removes the mapping and keeps the data, which lives below it. +func (d *DMLinear) Release(ctx context.Context, _ volstack.Artifact) error { + return d.cfg.Mapper.Remove(ctx, d.cfg.Name) +} + +// Destroy has nothing durable to remove: the mapping writes no metadata. +func (d *DMLinear) Destroy(context.Context, volstack.Artifact) error { return nil } + +// Healthy reports whether the mapping points at the device below, as the last +// Observe found it. +func (d *DMLinear) Healthy(context.Context, volstack.Artifact) (bool, error) { return !d.stale, nil } + +// Heal re-points the mapping at the device below, which the fabric layer may +// just have brought up on another subsystem. +func (d *DMLinear) Heal(ctx context.Context, below, _ volstack.Artifact) error { + if _, err := d.cfg.Mapper.Table(ctx, d.cfg.Name); errors.Is(err, devmapper.ErrNotFound) { + _, err := d.Ensure(ctx, below) + return err + } + return d.swap(ctx, below) +} + +// DMLinearParams is what the record keeps of this layer. +type DMLinearParams struct { + Name string `json:"name"` +} + +// Params is the record's view of the layer. +func (d *DMLinear) Params() any { return DMLinearParams{Name: d.cfg.Name} } diff --git a/atlas-lib/volstack/layers/dmlinear_test.go b/atlas-lib/volstack/layers/dmlinear_test.go new file mode 100644 index 000000000..452a17706 --- /dev/null +++ b/atlas-lib/volstack/layers/dmlinear_test.go @@ -0,0 +1,171 @@ +package layers + +import ( + "context" + "errors" + "testing" + + "github.com/simplyblock/atlas/blockdev" + "github.com/simplyblock/atlas/devmapper" + "github.com/simplyblock/atlas/volstack" +) + +// fakeMapper keeps mappings in memory and records the verbs run against them. +type fakeMapper struct { + tables map[string]devmapper.Target + sectors map[string]uint64 + calls []string + swapErr error +} + +func newFakeMapper() *fakeMapper { + return &fakeMapper{tables: map[string]devmapper.Target{}, sectors: map[string]uint64{}} +} + +func (f *fakeMapper) Table(_ context.Context, name string) (devmapper.Target, error) { + t, ok := f.tables[name] + if !ok { + return devmapper.Target{}, devmapper.ErrNotFound + } + return t, nil +} + +func (f *fakeMapper) Sectors(_ context.Context, device string) (uint64, error) { + if n, ok := f.sectors[device]; ok { + return n, nil + } + return 2048, nil +} + +func (f *fakeMapper) Create(_ context.Context, name, device string, sectors uint64) error { + f.calls = append(f.calls, "create "+name+" "+device) + f.tables[name] = devmapper.Target{Sectors: sectors, Device: device} + return nil +} + +func (f *fakeMapper) Swap(_ context.Context, name, device string, sectors uint64) error { + f.calls = append(f.calls, "swap "+name+" "+device) + if f.swapErr != nil { + return f.swapErr + } + f.tables[name] = devmapper.Target{Sectors: sectors, Device: device} + return nil +} + +func (f *fakeMapper) Remove(_ context.Context, name string) error { + f.calls = append(f.calls, "remove "+name) + delete(f.tables, name) + return nil +} + +func nvmeBelow(path string, major, minor uint32) volstack.Artifact { + return volstack.Artifact{Devices: []blockdev.Device{{Path: path, Major: major, Minor: minor}}} +} + +func dmLayer(m *fakeMapper) *DMLinear { + return NewDMLinear(DMLinearConfig{ + Name: DMLinearName("vol-1"), + Mapper: m, + Resolve: func(path string) (blockdev.Device, error) { + return blockdev.Device{Path: path, Name: "dm-7", Major: 253, Minor: 7}, nil + }, + }) +} + +func TestDMLinearEnsureMapsTheDeviceBelowAndExposesTheMapping(t *testing.T) { + m := newFakeMapper() + layer := dmLayer(m) + own, err := layer.Ensure(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)) + if err != nil { + t.Fatalf("ensure: %v", err) + } + if len(m.calls) != 1 || m.calls[0] != "create sb-vol-1 259:3" { + t.Fatalf("calls = %v, want one create over 259:3", m.calls) + } + if own.Devices[0].Path != "/dev/mapper/sb-vol-1" { + t.Fatalf("exposes %s, want the mapping", own.Devices[0].Path) + } + state, _, _ := layer.Observe(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)) + if state != volstack.StateReady { + t.Fatalf("state after ensure = %v, want Ready", state) + } +} + +// The namespace moved to another subsystem: the fabric below now exposes a +// different device. The mapping is reported stale and the heal re-points it +// at the new device; the device above keeps its identity. +func TestDMLinearHealRepointsAMappingWhoseNamespaceMoved(t *testing.T) { + m := newFakeMapper() + layer := dmLayer(m) + if _, err := layer.Ensure(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)); err != nil { + t.Fatal(err) + } + moved := nvmeBelow("/dev/nvme4n2", 259, 11) + state, own, err := layer.Observe(context.Background(), moved) + if err != nil || state != volstack.StatePartial { + t.Fatalf("observe after the move = %v %v, want Partial", state, err) + } + if healthy, _ := layer.Healthy(context.Background(), own); healthy { + t.Fatal("a mapping pointing at the old namespace is not healthy") + } + if err := layer.Heal(context.Background(), moved, own); err != nil { + t.Fatalf("heal: %v", err) + } + if got := m.tables["sb-vol-1"].Device; got != "259:11" { + t.Fatalf("mapping points at %s, want the new namespace 259:11", got) + } + if healthy, _ := layer.Healthy(context.Background(), own); !healthy { + t.Fatal("healthy after the swap") + } + if own.Devices[0].Path != "/dev/mapper/sb-vol-1" { + t.Fatalf("the device above changed: %s", own.Devices[0].Path) + } +} + +func TestDMLinearHealWithNothingBelowFailsAndLeavesTheMapping(t *testing.T) { + m := newFakeMapper() + layer := dmLayer(m) + if _, err := layer.Ensure(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)); err != nil { + t.Fatal(err) + } + if err := layer.Heal(context.Background(), volstack.Artifact{}, volstack.Artifact{}); err == nil { + t.Fatal("re-pointing at nothing must fail") + } + if got := m.tables["sb-vol-1"].Device; got != "259:3" { + t.Fatalf("a failed heal must leave the mapping, now %s", got) + } +} + +func TestDMLinearAFailedSwapKeepsTheLayerUnhealthy(t *testing.T) { + m := newFakeMapper() + layer := dmLayer(m) + if _, err := layer.Ensure(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)); err != nil { + t.Fatal(err) + } + moved := nvmeBelow("/dev/nvme4n2", 259, 11) + _, own, _ := layer.Observe(context.Background(), moved) + m.swapErr = errors.New("reload refused") + if err := layer.Heal(context.Background(), moved, own); err == nil { + t.Fatal("expected the swap error") + } + if healthy, _ := layer.Healthy(context.Background(), own); healthy { + t.Fatal("still stale after a failed swap") + } +} + +func TestDMLinearReleaseRemovesTheMappingAndDestroyKeepsNothing(t *testing.T) { + m := newFakeMapper() + layer := dmLayer(m) + if _, err := layer.Ensure(context.Background(), nvmeBelow("/dev/nvme1n1", 259, 3)); err != nil { + t.Fatal(err) + } + if err := layer.Release(context.Background(), volstack.Artifact{}); err != nil { + t.Fatal(err) + } + if _, ok := m.tables["sb-vol-1"]; ok { + t.Fatal("release must remove the mapping") + } + if err := layer.Destroy(context.Background(), volstack.Artifact{}); err != nil { + t.Fatal(err) + } +} diff --git a/atlas-lib/volstack/layers/filesystem.go b/atlas-lib/volstack/layers/filesystem.go index 7d799c903..10d8a472f 100644 --- a/atlas-lib/volstack/layers/filesystem.go +++ b/atlas-lib/volstack/layers/filesystem.go @@ -63,8 +63,19 @@ type FilesystemConfig struct { // formatted as, and it is also the only filesystem the layer will mount: a // device carrying another is refused, because neither reformatting it nor // serving what is on it is safe. + // + // Empty, the plan expresses no opinion: a device carrying a filesystem is + // mounted as what it carries, and a blank one is formatted as DefaultFsType. + // That is what a PersistentVolume without fsType means -- a static PV that + // adopts an existing volume (a test fail-over's clone, 2026-10-03), or one + // Ramen restored without the field -- and turning it into "ext4" before the + // device was looked at made the layer refuse every such XFS volume. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. Empty means ext4. + DefaultFsType string + // StagingPath is where the filesystem is mounted. StagingPath string @@ -107,6 +118,43 @@ type FilesystemConfig struct { // the volume is, and refuses every other device. type Filesystem struct { cfg FilesystemConfig + // detected is the filesystem found on the device when the plan named none. + detected string +} + +// effective is the filesystem this layer acts with: the one the plan named, +// else the one the device carries, else the default for a blank device. +func (f *Filesystem) effective(reading blockdev.Reading) string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if reading.Content == blockdev.ContentFilesystem && reading.Type != "" { + f.detected = reading.Type + return reading.Type + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +// known is the filesystem this layer stands for once it has acted: named, +// detected, or the default it formats with. +func (f *Filesystem) known() string { + if f.cfg.FsType != "" { + return f.cfg.FsType + } + if f.detected != "" { + return f.detected + } + return f.defaultFsType() +} + +func (f *Filesystem) defaultFsType() string { + if f.cfg.DefaultFsType != "" { + return f.cfg.DefaultFsType + } + return "ext4" } // NewFilesystem returns the filesystem layer for one volume. @@ -206,11 +254,12 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // decide the state and again to act on it. The reading itself is not needed // past that, because the filesystem to act on is the one the plan named and // observe has already refused every device carrying another. - state, _, own, err := f.observe(ctx, below) + state, reading, own, err := f.observe(ctx, below) if err != nil { return volstack.Artifact{}, err } if state == volstack.StateReady { + f.effective(reading) return own, nil } @@ -219,7 +268,7 @@ func (f *Filesystem) Ensure(ctx context.Context, below volstack.Artifact) (volst // disagreement, which is the point: the only two ways to reconcile one are to // reformat, which destroys the volume, and to serve the other filesystem, // which hides the misconfiguration until something else acts on it. - fsType := f.cfg.FsType + fsType := f.effective(reading) if state == volstack.StateAbsent { if err := f.cfg.Ops.Format(ctx, dev.Path, fsType, f.formatOptions(below)); err != nil { return volstack.Artifact{}, fmt.Errorf("filesystem: format %s as %s: %w", dev.Path, fsType, err) @@ -321,7 +370,7 @@ func (f *Filesystem) Heal(ctx context.Context, below, _ volstack.Artifact) error return err } - if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.cfg.FsType, f.mountFlags()); err != nil { + if err := f.cfg.Ops.Mount(ctx, dev.Path, f.cfg.StagingPath, f.effective(reading), f.mountFlags()); err != nil { return fmt.Errorf("filesystem: remount %s at %s: %w", dev.Path, f.cfg.StagingPath, err) } return nil @@ -379,7 +428,7 @@ type FilesystemParams struct { // recorded is the one the volume asked for, and a teardown needs no more than // that: what is actually on the device is read from the device. func (f *Filesystem) Params() any { - return FilesystemParams{FsType: f.cfg.FsType} + return FilesystemParams{FsType: f.known()} } // agrees reports whether the filesystem on the device is the one the plan asked @@ -436,13 +485,16 @@ func (f *Filesystem) blank( switch { case prior == "": return volstack.StateAbsent, reading, volstack.Artifact{}, nil - case prior != f.cfg.FsType: + case f.cfg.FsType != "" && prior != f.cfg.FsType: return volstack.StateAbsent, reading, volstack.Artifact{}, fmt.Errorf( "filesystem: refusing to stage %s, which is recorded as carrying %s where the plan "+ "asks for %s: reformatting would destroy the volume, and mounting it as %s would "+ "serve a filesystem the plan does not declare", deviceOf(below), prior, f.cfg.FsType, prior) default: + if f.cfg.FsType == "" { + f.detected = prior + } // Recorded as formatted while nothing was found on it: the reading is a // failed probe rather than an empty device, so the filesystem is treated as // present and unmounted. Mounting it is the honest next step, and a mount @@ -486,7 +538,7 @@ func (f *Filesystem) mountFlags() []string { // asked for. That is also the only one the layer acts on, since a device // carrying another is refused rather than reconciled. func (f *Filesystem) strategy() FilesystemLayerStrategy { - return FilesystemStrategyFor(f.cfg.FsType) + return FilesystemStrategyFor(f.known()) } // deviceOf names the device below for an error message, without asserting there diff --git a/atlas-lib/volstack/layers/filesystem_test.go b/atlas-lib/volstack/layers/filesystem_test.go index aa107980b..742791b16 100644 --- a/atlas-lib/volstack/layers/filesystem_test.go +++ b/atlas-lib/volstack/layers/filesystem_test.go @@ -592,3 +592,41 @@ func TestReleaseForcesWhenAPlainUnmountRefuses(t *testing.T) { t.Fatal("a plain unmount refused and the release did not fall back to its force path") } } + +// A plan that names no filesystem -- a static PV without fsType, such as a +// test fail-over's clone or a PV Ramen restored without the field -- mounts +// what the device carries instead of refusing it for not being ext4 +// (2026-10-03), and formats a blank device as the default. +func TestAPlanNamingNoFilesystemMountsWhatTheDeviceCarries(t *testing.T) { + fs := newFakeFS() + l := newFSAsking(t, fs, "", blockdev.Reading{Content: blockdev.ContentFilesystem, Type: "xfs"}, nil) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 0 { + t.Fatalf("formatted a device that carries a filesystem: %+v", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "xfs" { + t.Fatalf("mounted %+v, want once as xfs", fs.mounted) + } + if p, _ := l.Params().(FilesystemParams); p.FsType != "xfs" { + t.Fatalf("params %+v, want the detected xfs recorded", p) + } +} + +func TestAPlanNamingNoFilesystemFormatsABlankDeviceAsTheDefault(t *testing.T) { + fs := newFakeFS() + l := NewFilesystem(FilesystemConfig{ + FsType: "", DefaultFsType: "ext4", StagingPath: stagingPath, Ops: fs, + Content: fakeReader{reading: blockdev.Reading{Content: blockdev.ContentBlank}}, + }) + if _, err := l.Ensure(context.Background(), belowArtifact()); err != nil { + t.Fatal(err) + } + if len(fs.formatted) != 1 || fs.formatted[0].fsType != "ext4" { + t.Fatalf("formatted %+v, want once as ext4", fs.formatted) + } + if len(fs.mounted) != 1 || fs.mounted[0].fsType != "ext4" { + t.Fatalf("mounted %+v, want once as ext4", fs.mounted) + } +} diff --git a/atlas-lib/volstack/plans/node.go b/atlas-lib/volstack/plans/node.go index 5aa5c02d8..6e1ea8156 100644 --- a/atlas-lib/volstack/plans/node.go +++ b/atlas-lib/volstack/plans/node.go @@ -13,6 +13,7 @@ import ( "context" "github.com/simplyblock/atlas/blockdev" + "github.com/simplyblock/atlas/devmapper" "github.com/simplyblock/atlas/lvm" "github.com/simplyblock/atlas/lvol" "github.com/simplyblock/atlas/nvme" @@ -55,6 +56,10 @@ type NodeConfig struct { // it nil, and the reading decides alone. PriorFormat func(ctx context.Context, volume Volume) (string, error) + // Mapper runs the device-mapper commands of the dmLinear indirection, and + // defaults to the host's dmsetup. + Mapper layers.DMMapper + // Resolve answers what the kernel says about a device path, and defaults to // blockdev.ResolveDevice. It is a seam only because the logical-volume layer // creates a device-mapper node and has to describe it upward, which a test @@ -95,6 +100,21 @@ func NewNode(cfg NodeConfig) *Node { } // fabric is the bottom layer of every plan: one namespace, attached. +// dmLinear is the indirection between the fabric and what uses the volume +// (docs/consistency-group-colocation.md §6 in sbcli): its device survives a +// move of the volume's namespace to another subsystem. +func (n *Node) dmLinear(volume Volume) volstack.Layer { + mapper := n.cfg.Mapper + if mapper == nil { + mapper = devmapper.New(nil) + } + return layers.NewDMLinear(layers.DMLinearConfig{ + Name: layers.DMLinearName(volume.UUID), + Mapper: mapper, + Resolve: n.cfg.Resolve, + }) +} + func (n *Node) fabric(connection lvol.Connection) volstack.Layer { return layers.NewFabric(layers.FabricConfig{ Connection: connection, @@ -109,6 +129,7 @@ func (n *Node) fabric(connection lvol.Connection) volstack.Layer { func (n *Node) filesystem(volume Volume) volstack.Layer { return layers.NewFilesystem(layers.FilesystemConfig{ FsType: volume.FsType, + DefaultFsType: volume.DefaultFsType, StagingPath: volume.StagingPath, MountFlags: volume.MountFlags, FormatOptions: volume.FormatOptions, diff --git a/atlas-lib/volstack/plans/plans.go b/atlas-lib/volstack/plans/plans.go index c514dae4a..481236621 100644 --- a/atlas-lib/volstack/plans/plans.go +++ b/atlas-lib/volstack/plans/plans.go @@ -52,9 +52,14 @@ type Volume struct { // FsType is the filesystem this volume is. It decides what a blank device is // formatted as, and it is also the only filesystem that will be mounted: a - // device carrying another is refused. + // device carrying another is refused. Empty: whatever the device carries, + // and DefaultFsType for a blank one. FsType string + // DefaultFsType is what a blank device is formatted as when FsType names + // nothing. + DefaultFsType string + // MountFlags are the flags the volume asked for, ahead of the ones the // filesystem layer derives from the filesystem itself. MountFlags []string @@ -178,6 +183,20 @@ func (n *Node) Plain(connection lvol.Connection, volume Volume) volstack.Plan { return volstack.Plan{n.fabric(connection), n.filesystem(volume)} } +// IndirectRawBlock is `fabric` → `dmLinear`: a raw block volume behind the +// device-mapper indirection, so the device the pod holds survives a move of +// the volume's namespace to another subsystem (the indirection's Heal re-points +// it at the namespace the fabric brought up). +func (n *Node) IndirectRawBlock(connection lvol.Connection, volume Volume) volstack.Plan { + return volstack.Plan{n.fabric(connection), n.dmLinear(volume)} +} + +// IndirectPlain is `fabric` → `dmLinear` → `filesystem`: Plain with the +// indirection under the filesystem, so a namespace move does not unmount it. +func (n *Node) IndirectPlain(connection lvol.Connection, volume Volume) volstack.Plan { + return volstack.Plan{n.fabric(connection), n.dmLinear(volume), n.filesystem(volume)} +} + // LVM is `fabric` → `lvmPhysicalVolume` → `lvmVolumeGroup` → `lvmLogicalVolume` // → `filesystem`, the shape a volume with client-side deduplication or // compression takes. What the logical volume is to be lives in options, so this diff --git a/csi-driver/go.mod b/csi-driver/go.mod index f5821c3b8..b574f9b3b 100644 --- a/csi-driver/go.mod +++ b/csi-driver/go.mod @@ -4,6 +4,7 @@ go 1.26.2 require ( github.com/container-storage-interface/spec v1.12.0 + github.com/csi-addons/spec v0.2.1-0.20260515055340-d4a373713b9a github.com/kubernetes-csi/csi-lib-utils v0.24.0 github.com/kubernetes-csi/csi-test/v5 v5.5.0 github.com/onsi/gomega v1.42.1 @@ -23,11 +24,18 @@ require ( cel.dev/expr v0.25.2 // indirect github.com/Masterminds/semver/v3 v3.4.0 // indirect github.com/antlr4-go/antlr/v4 v4.13.0 // indirect + github.com/apapsch/go-jsonmerge/v2 v2.0.0 // indirect github.com/cenkalti/backoff/v5 v5.0.3 // indirect github.com/distribution/reference v0.6.0 // indirect + github.com/dprotaso/go-yit v0.0.0-20220510233725-9ba8df137936 // indirect github.com/fsnotify/fsnotify v1.9.0 // indirect github.com/fxamacker/cbor/v2 v2.9.0 // indirect + github.com/gabriel-vasile/mimetype v1.4.13 // indirect + github.com/getkin/kin-openapi v0.142.0 // indirect github.com/go-openapi/swag/jsonname v0.26.0 // indirect + github.com/go-playground/locales v0.14.1 // indirect + github.com/go-playground/universal-translator v0.18.1 // indirect + github.com/go-playground/validator/v10 v10.30.3 // indirect github.com/go-task/slim-sprig/v3 v3.0.0 // indirect github.com/google/cel-go v0.26.0 // indirect github.com/google/gnostic-models v0.7.0 // indirect @@ -35,15 +43,25 @@ require ( github.com/gorilla/websocket v1.5.4-0.20250319132907-e064f32e3674 // indirect github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 // indirect github.com/hashicorp/yamux v0.1.2 // indirect + github.com/leodido/go-urn v1.4.0 // indirect + github.com/oapi-codegen/oapi-codegen/v2 v2.8.0 // indirect + github.com/oapi-codegen/runtime v1.6.0 // indirect + github.com/oasdiff/yaml v0.1.1 // indirect + github.com/oasdiff/yaml3 v0.0.14 // indirect github.com/pmezard/go-difflib v1.0.1-0.20181226105442-5d4384ee4fb2 // indirect github.com/robfig/cron/v3 v3.0.1 // indirect + github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 // indirect + github.com/speakeasy-api/jsonpath v0.6.3 // indirect + github.com/speakeasy-api/openapi v1.24.0 // indirect github.com/stoewer/go-strcase v1.3.0 // indirect + github.com/vmware-labs/yaml-jsonpath v0.3.2 // indirect github.com/x448/float16 v0.8.4 // indirect go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.40.0 // indirect go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.40.0 // indirect go.uber.org/mock v0.5.2 // indirect go.yaml.in/yaml/v2 v2.4.3 // indirect go.yaml.in/yaml/v3 v3.0.4 // indirect + golang.org/x/crypto v0.56.0 // indirect golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93 // indirect golang.org/x/mod v0.38.0 // indirect golang.org/x/sync v0.22.0 // indirect diff --git a/csi-driver/go.sum b/csi-driver/go.sum index 36567e4df..40a2cc779 100644 --- a/csi-driver/go.sum +++ b/csi-driver/go.sum @@ -2,35 +2,55 @@ cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs= cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4= github.com/Masterminds/semver/v3 v3.4.0 h1:Zog+i5UMtVoCU8oKka5P7i9q9HgrJeGzI9SA1Xbatp0= github.com/Masterminds/semver/v3 v3.4.0/go.mod h1:4V+yj/TJE1HU9XfppCwVMZq3I84lprf4nC11bSS5beM= +github.com/RaveNoX/go-jsoncommentstrip v1.0.0/go.mod h1:78ihd09MekBnJnxpICcwzCMzGrKSKYe4AqU6PDYYpjk= github.com/antlr4-go/antlr/v4 v4.13.0 h1:lxCg3LAv+EUK6t1i0y1V6/SLeUi0eKEKdhQAlS8TVTI= github.com/antlr4-go/antlr/v4 v4.13.0/go.mod h1:pfChB/xh/Unjila75QW7+VU4TSnWnnk9UTnmpPaOR2g= +github.com/apapsch/go-jsonmerge/v2 v2.0.0 h1:axGnT1gRIfimI7gJifB699GoE/oq+F2MU7Dml6nw9rQ= +github.com/apapsch/go-jsonmerge/v2 v2.0.0/go.mod h1:lvDnEdqiQrp0O42VQGgmlKpxL1AP2+08jFMw88y4klk= github.com/armon/go-socks5 v0.0.0-20160902184237-e75332964ef5 h1:0CwZNZbxp69SHPdPJAN/hZIm0C4OItdklCFmMRWYpio= github.com/armon/go-socks5 v0.0.0-20160902184237-e75332964ef5/go.mod h1:wHh0iHkYZB8zMSxRWpUBQtwG5a7fFgvEO+odwuTv2gs= github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM= github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw= github.com/blang/semver/v4 v4.0.0 h1:1PFHFE6yCCTv8C1TeyNNarDzntLi7wMI5i/pzqYIsAM= github.com/blang/semver/v4 v4.0.0/go.mod h1:IbckMUScFkM3pff0VJDNKRiT6TG/YpiHIM2yvyW5YoQ= +github.com/bmatcuk/doublestar v1.1.1/go.mod h1:UD6OnuiIn0yFxxA2le/rnRU1G4RaI4UvFv1sNto9p6w= github.com/cenkalti/backoff/v5 v5.0.3 h1:ZN+IMa753KfX5hd8vVaMixjnqRZ3y8CuJKRKj1xcsSM= github.com/cenkalti/backoff/v5 v5.0.3/go.mod h1:rkhZdG3JZukswDf7f0cwqPNk4K0sa+F97BxZthm/crw= github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs= github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs= +github.com/chzyer/logex v1.1.10/go.mod h1:+Ywpsq7O8HXn0nuIou7OrIPyXbp3wmkHB+jjWRnGsAI= +github.com/chzyer/readline v0.0.0-20180603132655-2972be24d48e/go.mod h1:nSuG5e5PlCu98SY8svDHJxuZscDgtXS6KTTbou5AhLI= +github.com/chzyer/test v0.0.0-20180213035817-a1ea475d72b1/go.mod h1:Q3SI9o4m/ZMnBNeIyt5eFwwo7qiLfzFZmjNmxjkiQlU= github.com/container-storage-interface/spec v1.12.0 h1:zrFOEqpR5AghNaaDG4qyedwPBqU2fU0dWjLQMP/azK0= github.com/container-storage-interface/spec v1.12.0/go.mod h1:txsm+MA2B2WDa5kW69jNbqPnvTtfvZma7T/zsAZ9qX8= github.com/cpuguy83/go-md2man/v2 v2.0.6/go.mod h1:oOW0eioCTA6cOiMLiUPZOpcVxMig6NIQQ7OS05n1F4g= +github.com/csi-addons/spec v0.2.1-0.20260515055340-d4a373713b9a h1:AxvSsTN8TxwpeP6X/ScNK5i+6eXu3APRfuIXSEMENVQ= +github.com/csi-addons/spec v0.2.1-0.20260515055340-d4a373713b9a/go.mod h1:Mwq4iLiUV4s+K1bszcWU6aMsR5KPsbIYzzszJ6+56vI= github.com/davecgh/go-spew v1.1.0/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc h1:U9qPSI2PIWSS1VwoXQT9A3Wy9MM3WgvqSxFWenqJduM= github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/distribution/reference v0.6.0 h1:0IXCQ5g4/QMHHkarYzh5l+u8T3t73zM5QvfrDyIgxBk= github.com/distribution/reference v0.6.0/go.mod h1:BbU0aIcezP1/5jX/8MP0YiH4SdvB5Y4f/wlDRiLyi3E= +github.com/dlclark/regexp2 v1.11.4 h1:rPYF9/LECdNymJufQKmri9gV604RvvABwgOA8un7yAo= +github.com/dlclark/regexp2 v1.11.4/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8= +github.com/dprotaso/go-yit v0.0.0-20191028211022-135eb7262960/go.mod h1:9HQzr9D/0PGwMEbC3d5AB7oi67+h4TsQqItC1GVYG58= +github.com/dprotaso/go-yit v0.0.0-20220510233725-9ba8df137936 h1:PRxIJD8XjimM5aTknUK9w6DHLDox2r2M3DI4i2pnd3w= +github.com/dprotaso/go-yit v0.0.0-20220510233725-9ba8df137936/go.mod h1:ttYvX5qlB+mlV1okblJqcSMtR4c52UKxDiX9GRBS8+Q= github.com/emicklei/go-restful/v3 v3.13.0 h1:C4Bl2xDndpU6nJ4bc1jXd+uTmYPVUwkD6bFY/oTyCes= github.com/emicklei/go-restful/v3 v3.13.0/go.mod h1:6n3XBCmQQb25CM2LCACGz8ukIrRry+4bhvbpWn3mrbc= github.com/felixge/httpsnoop v1.0.4 h1:NFTV2Zj1bL4mc9sqWACXbQFVBBg2W3GPvqp8/ESS2Wg= github.com/felixge/httpsnoop v1.0.4/go.mod h1:m8KPJKqk1gH5J9DgRY2ASl2lWCfGKXixSwevea8zH2U= +github.com/fsnotify/fsnotify v1.4.7/go.mod h1:jwhsz4b93w/PPRr/qN1Yymfu8t87LnFCMoQvtojpjFo= +github.com/fsnotify/fsnotify v1.4.9/go.mod h1:znqG4EE+3YCdAaPaxE2ZRY/06pZUdp0tY4IgpuI1SZQ= github.com/fsnotify/fsnotify v1.9.0 h1:2Ml+OJNzbYCTzsxtv8vKSFD9PbJjmhYF14k/jKC7S9k= github.com/fsnotify/fsnotify v1.9.0/go.mod h1:8jBTzvmWwFyi3Pb8djgCCO5IBqzKJ/Jwo8TRcHyHii0= github.com/fxamacker/cbor/v2 v2.9.0 h1:NpKPmjDBgUfBms6tr6JZkTHtfFGcMKsw3eGcmD/sapM= github.com/fxamacker/cbor/v2 v2.9.0/go.mod h1:vM4b+DJCtHn+zz7h3FFp/hDAI9WNWCsZj23V5ytsSxQ= +github.com/gabriel-vasile/mimetype v1.4.13 h1:46nXokslUBsAJE/wMsp5gtO500a4F3Nkz9Ufpk2AcUM= +github.com/gabriel-vasile/mimetype v1.4.13/go.mod h1:d+9Oxyo1wTzWdyVUPMmXFvp4F9tea18J8ufA774AB3s= +github.com/getkin/kin-openapi v0.142.0 h1:izj0vBdFprMhitfzaX8sTqztsEQyvwhssBoB6n8NO7w= +github.com/getkin/kin-openapi v0.142.0/go.mod h1:3BH9M9XDe/y9M5DSvEocVYAYq1w0qrhJHjC/vZi0AaY= github.com/gkampitakis/ciinfo v0.3.2 h1:JcuOPk8ZU7nZQjdUhctuhQofk7BGHuIy0c9Ez8BNhXs= github.com/gkampitakis/ciinfo v0.3.2/go.mod h1:1NIwaOcFChN4fa/B0hEBdAb6npDlFL8Bwx4dfRLRqAo= github.com/gkampitakis/go-diff v1.3.2 h1:Qyn0J9XJSDTgnsgHRdz9Zp24RaJeKMUHg2+PDZZdC4M= @@ -55,19 +75,42 @@ github.com/go-openapi/swag/jsonname v0.26.0 h1:gV1NFX9M8avo0YSpmWogqfQISigCmpaiN github.com/go-openapi/swag/jsonname v0.26.0/go.mod h1:urBBR8bZNoDYGr653ynhIx+gTeIz0ARZxHkAPktJK2M= github.com/go-openapi/testify/v2 v2.4.2 h1:tiByHpvE9uHrrKjOszax7ZvKB7QOgizBWGBLuq0ePx4= github.com/go-openapi/testify/v2 v2.4.2/go.mod h1:SgsVHtfooshd0tublTtJ50FPKhujf47YRqauXXOUxfw= +github.com/go-playground/assert/v2 v2.2.0 h1:JvknZsQTYeFEAhQwI4qEt9cyV5ONwRHC+lYKSsYSR8s= +github.com/go-playground/assert/v2 v2.2.0/go.mod h1:VDjEfimB/XKnb+ZQfWdccd7VUvScMdVu0Titje2rxJ4= +github.com/go-playground/locales v0.14.1 h1:EWaQ/wswjilfKLTECiXz7Rh+3BjFhfDFKv/oXslEjJA= +github.com/go-playground/locales v0.14.1/go.mod h1:hxrqLVvrK65+Rwrd5Fc6F2O76J/NuW9t0sjnWqG1slY= +github.com/go-playground/universal-translator v0.18.1 h1:Bcnm0ZwsGyWbCzImXv+pAJnYK9S473LQFuzCbDbfSFY= +github.com/go-playground/universal-translator v0.18.1/go.mod h1:xekY+UJKNuX9WP91TpwSH2VMlDf28Uj24BCp08ZFTUY= +github.com/go-playground/validator/v10 v10.30.3 h1:4MU6YkEwx7GbcPJOZxrtbu+QfF3pJLJuaYTeAH0DYy8= +github.com/go-playground/validator/v10 v10.30.3/go.mod h1:4Axh7oCNGcoGkqLoE4YWt6n20mcEIsPRlB7vPk3lpyc= +github.com/go-task/slim-sprig v0.0.0-20210107165309-348f09dbbbc0/go.mod h1:fyg7847qk6SyHyPtNmDHnmrv/HOrqktSC+C9fM+CJOE= github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI= github.com/go-task/slim-sprig/v3 v3.0.0/go.mod h1:W848ghGpv3Qj3dhTPRyJypKRiqCdHZiAzKg9hl15HA8= github.com/goccy/go-yaml v1.18.0 h1:8W7wMFS12Pcas7KU+VVkaiCng+kG8QiFeFwzFb+rwuw= github.com/goccy/go-yaml v1.18.0/go.mod h1:XBurs7gK8ATbW4ZPGKgcbrY1Br56PdM69F7LkFRi1kA= +github.com/golang/protobuf v1.2.0/go.mod h1:6lQm79b+lXiMfvg/cZm0SGofjICqVBUtrP5yJMmIC1U= +github.com/golang/protobuf v1.4.0-rc.1/go.mod h1:ceaxUfeHdC40wWswd/P6IGgMaK3YpKi5j83Wpe3EHw8= +github.com/golang/protobuf v1.4.0-rc.1.0.20200221234624-67d41d38c208/go.mod h1:xKAWHe0F5eneWXFV3EuXVDTCmh+JuBKY0li0aMyXATA= +github.com/golang/protobuf v1.4.0-rc.2/go.mod h1:LlEzMj4AhA7rCAGe4KMBDvJI+AwstrUpVNzEA03Pprs= +github.com/golang/protobuf v1.4.0-rc.4.0.20200313231945-b860323f09d0/go.mod h1:WU3c8KckQ9AFe+yFwt9sWVRKCVIyN9cPHBJSNnbL67w= +github.com/golang/protobuf v1.4.0/go.mod h1:jodUvKwWbYaEsadDk5Fwe5c77LiNKVO9IDvqG2KuDX0= +github.com/golang/protobuf v1.4.2/go.mod h1:oDoupMAO8OvCJWAcko0GGGIgR6R6ocIYbsSw735rRwI= +github.com/golang/protobuf v1.5.0/go.mod h1:FsONVRAS9T7sI+LIUmWTfcYkHO4aIWwzhcaSAoJOfIk= +github.com/golang/protobuf v1.5.2/go.mod h1:XVQd3VNwM+JqD3oG2Ue2ip4fOMUkwXdXDdiuN0vRsmY= github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek= github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps= github.com/google/cel-go v0.26.0 h1:DPGjXackMpJWH680oGY4lZhYjIameYmR+/6RBdDGmaI= github.com/google/cel-go v0.26.0/go.mod h1:A9O8OU9rdvrK5MQyrqfIxo1a0u4g3sF8KB6PUIaryMM= github.com/google/gnostic-models v0.7.0 h1:qwTtogB15McXDaNqTZdzPJRHvaVJlAl+HVQnLmJEJxo= github.com/google/gnostic-models v0.7.0/go.mod h1:whL5G0m6dmc5cPxKc5bdKdEN3UjI7OUGxBlw57miDrQ= +github.com/google/go-cmp v0.3.0/go.mod h1:8QqcDgzrUqlUb/G2PQTWiueGozuR1884gddMywk6iLU= +github.com/google/go-cmp v0.3.1/go.mod h1:8QqcDgzrUqlUb/G2PQTWiueGozuR1884gddMywk6iLU= +github.com/google/go-cmp v0.4.0/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE= +github.com/google/go-cmp v0.5.5/go.mod h1:v8dTdLbMG2kIc/vJvl+f65V22dbkXbowE6jgT/gNBxE= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/gofuzz v1.0.0/go.mod h1:dBl0BpW6vV/+mYPU4Po3pmUjxk6FQPldtuIdl/M65Eg= +github.com/google/pprof v0.0.0-20210407192527-94a9f03dee38/go.mod h1:kpwsk12EmLew5upagYY7GY0pfYCcupk39gWOCRROcvE= github.com/google/pprof v0.0.0-20260402051712-545e8a4df936 h1:EwtI+Al+DeppwYX2oXJCETMO23COyaKGP6fHVpkpWpg= github.com/google/pprof v0.0.0-20260402051712-545e8a4df936/go.mod h1:MxpfABSjhmINe3F1It9d+8exIHFvUqtLIRCdOGNXqiI= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= @@ -78,6 +121,8 @@ github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7 h1:X+2YciYSxvMQK0UZ7sg45ZVabVZ github.com/grpc-ecosystem/grpc-gateway/v2 v2.27.7/go.mod h1:lW34nIZuQ8UDPdkon5fmfp2l3+ZkQ2me/+oecHYLOII= github.com/hashicorp/yamux v0.1.2 h1:XtB8kyFOyHXYVFnwT5C3+Bdo8gArse7j2AQ0DA0Uey8= github.com/hashicorp/yamux v0.1.2/go.mod h1:C+zze2n6e/7wshOZep2A70/aQU6QBRWJO/G6FT1wIns= +github.com/hpcloud/tail v1.0.0/go.mod h1:ab1qPbhIpdTxEkNHXyeSf5vhxWSCs/tWer42PpOxQnU= +github.com/ianlancetaylor/demangle v0.0.0-20200824232613-28f6c0f3b639/go.mod h1:aSSvb/t6k1mPoxDqO4vJh6VOCGPwU4O0C2/Eqndh1Sc= github.com/inconshreveable/mousetrap v1.1.0 h1:wN+x4NVGpMsO7ErUn/mUI3vEoE6Jt13X2s0bqwp9tc8= github.com/inconshreveable/mousetrap v1.1.0/go.mod h1:vpF70FUmC8bwa3OWnCshd2FqLfsEA9PFc4w1p2J65bw= github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY= @@ -86,10 +131,14 @@ github.com/joshdk/go-junit v1.0.0 h1:S86cUKIdwBHWwA6xCmFlf3RTLfVXYQfvanM5Uh+K6GE github.com/joshdk/go-junit v1.0.0/go.mod h1:TiiV0PqkaNfFXjEiyjWM3XXrhVyCa1K4Zfga6W52ung= github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM= github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo= +github.com/juju/gnuflag v0.0.0-20171113085948-2ce1bb71843d/go.mod h1:2PavIy+JPciBPrBUjwbNvtwB6RQlve+hkpll6QSNmOE= github.com/klauspost/compress v1.18.0 h1:c/Cqfb0r+Yi+JtIEq73FWXVkRonBlf0CRNYc8Zttxdo= github.com/klauspost/compress v1.18.0/go.mod h1:2Pp+KzxcywXVXMr50+X0Q/Lsb43OQHYWRCY2AiWywWQ= +github.com/kr/pretty v0.1.0/go.mod h1:dAy3ld7l9f0ibDNOQOHHMYYIIbhfbHSm3C4ZsoJORNo= github.com/kr/pretty v0.3.1 h1:flRD4NNwYAUpkphVc1HcthR4KEIFJ65n8Mw5qdRn3LE= github.com/kr/pretty v0.3.1/go.mod h1:hoEshYVHaxMs3cyo3Yncou5ZscifuDolrwPKZanG3xk= +github.com/kr/pty v1.1.1/go.mod h1:pFQYn66WHrOpPYNljwOMqo10TkYh1fy3cYio2l3bCsQ= +github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= github.com/kubernetes-csi/csi-lib-utils v0.24.0 h1:hpL5ecxtr07/DNIF9Qn/gbNG/ZlgMSMZxfRtgfVm9tY= @@ -98,6 +147,8 @@ github.com/kubernetes-csi/csi-test/v5 v5.5.0 h1:21NYP33XXfzsAGwFuFHJUIf60hY08B4A github.com/kubernetes-csi/csi-test/v5 v5.5.0/go.mod h1:5ZyneETi47SniZuPA9e8fIL6TTkkKv8/+jkaF0IHqKY= github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc= github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw= +github.com/leodido/go-urn v1.4.0 h1:WT9HwE9SGECu3lg4d/dIA+jxlljEa1/ffXKmRjqdmIQ= +github.com/leodido/go-urn v1.4.0/go.mod h1:bvxc+MVxLKB4z00jd1z+Dvzr47oO32F/QSNjSBOlFxI= github.com/mailru/easyjson v0.9.0 h1:PrnmzHw7262yW8sTBwxi1PdJA3Iw/EKBa8psRf7d9a4= github.com/mailru/easyjson v0.9.0/go.mod h1:1+xMtQp2MRNVL/V1bOzuP3aP8VNwRW55fQUto+XFtTU= github.com/maruel/natural v1.1.1 h1:Hja7XhhmvEFhcByqDoHz9QZbkWey+COd9xWfCfn1ioo= @@ -116,8 +167,32 @@ github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee h1:W5t00kpgFd github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= +github.com/nxadm/tail v1.4.4/go.mod h1:kenIhsEOeOJmVchQTgglprH7qJGnHDVpk1VPCcaMI8A= +github.com/nxadm/tail v1.4.8 h1:nPr65rt6Y5JFSKQO7qToXr7pePgD6Gwiw05lkbyAQTE= +github.com/nxadm/tail v1.4.8/go.mod h1:+ncqLTQzXmGhMZNUePPaPqPvBxHAIsmXswZKocGu+AU= +github.com/oapi-codegen/nullable v1.1.0 h1:eAh8JVc5430VtYVnq00Hrbpag9PFRGWLjxR1/3KntMs= +github.com/oapi-codegen/nullable v1.1.0/go.mod h1:KUZ3vUzkmEKY90ksAmit2+5juDIhIZhfDl+0PwOQlFY= +github.com/oapi-codegen/oapi-codegen/v2 v2.8.0 h1:s4hxMxuqtR8jPzXkBTtFwY/SBuj3gEAYikmbBSdtLMM= +github.com/oapi-codegen/oapi-codegen/v2 v2.8.0/go.mod h1:yae2TI9IYB5vxQ35gFrpXh9L5H1eJv4MAUK1jumGMTo= +github.com/oapi-codegen/runtime v1.6.0 h1:7Xx+GlueD6nRuyKoCPzL434Jfi3BetbiJOrzCHp/VPU= +github.com/oapi-codegen/runtime v1.6.0/go.mod h1:GwV7hC2hviaMzj+ITfHVRESK5J2W/GefVwIND/bMGvU= +github.com/oasdiff/yaml v0.1.1 h1:6nHx+pn9gBRM6YpBlFZFQGCCd1nuvqOBtTD3KKTgGxY= +github.com/oasdiff/yaml v0.1.1/go.mod h1:EYJNoyktvWMJ0Hmhx+6qTaqMOsalUaRGT8Sj1hNcegU= +github.com/oasdiff/yaml3 v0.0.14 h1:aLJee3hxBK2H5wdXd9iPcIXb93Nty1Ge0pT171eHtkw= +github.com/oasdiff/yaml3 v0.0.14/go.mod h1:csto2xfDjYccdUn/yw/bPjj/cYTdp6HtFA0J4TWG+gg= +github.com/onsi/ginkgo v1.6.0/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE= +github.com/onsi/ginkgo v1.10.2/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE= +github.com/onsi/ginkgo v1.12.1/go.mod h1:zj2OWP4+oCPe1qIXoGWkgMRwljMUYCdkwsT2108oapk= +github.com/onsi/ginkgo v1.16.4 h1:29JGrr5oVBm5ulCWet69zQkzWipVXIol6ygQUe/EzNc= +github.com/onsi/ginkgo v1.16.4/go.mod h1:dX+/inL/fNMqNlz0e9LfyB9TswhZpCVdJM/Z6Vvnwo0= +github.com/onsi/ginkgo/v2 v2.1.3/go.mod h1:vw5CSIxN1JObi/U8gcbwft7ZxR2dgaR70JSE3/PpL4c= github.com/onsi/ginkgo/v2 v2.32.0 h1:Hw7s2pVrQo/8Yz5N77qdnpHaoc+c6cC9WIV1Jce+J6E= github.com/onsi/ginkgo/v2 v2.32.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= +github.com/onsi/gomega v1.7.0/go.mod h1:ex+gbHU/CVuBBDIJjb2X0qEXbFg53c61hWP/1CpauHY= +github.com/onsi/gomega v1.7.1/go.mod h1:XdKZgCCFLUoM/7CFJVPcG8C1xQ1AJ0vpAezJrB7JYyY= +github.com/onsi/gomega v1.10.1/go.mod h1:iN09h71vgCQne3DLsj+A5owkum+a2tYe+TOCB1ybHNo= +github.com/onsi/gomega v1.17.0/go.mod h1:HnhC7FXeEQY45zxNK3PPoIUhzk/80Xly9PcubAlGdZY= +github.com/onsi/gomega v1.19.0/go.mod h1:LY+I3pBVzYsTBU1AnDwOSxaYi9WoWiqgwooUqq9yPro= github.com/onsi/gomega v1.42.1 h1:iN1rCUX+44NZ1Dc97MPoeFYbFR0vh8zxoxMFwKdyZ6I= github.com/onsi/gomega v1.42.1/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg= github.com/opencontainers/go-digest v1.0.0 h1:apOUWs51W5PlhuyGyz9FCeeBIOUDA/6nW8Oi/yOhh5U= @@ -138,10 +213,19 @@ github.com/robfig/cron/v3 v3.0.1/go.mod h1:eQICP3HwyT7UooqI/z+Ov+PtYAWygg1TEWWzG github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= +github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ= +github.com/santhosh-tekuri/jsonschema/v6 v6.0.2/go.mod h1:JXeL+ps8p7/KNMjDQk3TCwPpBy0wYklyWTfbkIzdIFU= +github.com/sergi/go-diff v1.1.0 h1:we8PVUC3FE2uYfodKH/nBHMSetSfHDR6scGdBi+erh0= +github.com/sergi/go-diff v1.1.0/go.mod h1:STckp+ISIX8hZLjrqAeVduY0gWCT9IjLuqbuNXdaHfM= +github.com/speakeasy-api/jsonpath v0.6.3 h1:c+QPwzAOdrWvzycuc9HFsIZcxKIaWcNpC+xhOW9rJxU= +github.com/speakeasy-api/jsonpath v0.6.3/go.mod h1:2cXloNuQ+RSXi5HTRaeBh7JEmjRXTiaKpFTdZiL7URI= +github.com/speakeasy-api/openapi v1.24.0 h1:opoD27rupX7zBVPq1HkIGLeMOzNNA7JalhYP8q34i04= +github.com/speakeasy-api/openapi v1.24.0/go.mod h1:g3+dIMe0AYgbbGvnlQZqesmjAVWSm9BmsjLevnefQrg= github.com/spf13/cobra v1.10.2 h1:DMTTonx5m65Ic0GOoRY2c16WCbHxOOw6xxezuLaBpcU= github.com/spf13/cobra v1.10.2/go.mod h1:7C1pvHqHw5A4vrJfjNwvOdzYu0Gml16OCs2GRiTUUS4= github.com/spf13/pflag v1.0.9 h1:9exaQaMOCwffKiiiYk6/BndUBv+iRViNW+4lEMi0PvY= github.com/spf13/pflag v1.0.9/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= +github.com/spkg/bom v0.0.0-20160624110644-59b7046e48ad/go.mod h1:qLr4V1qq6nMqFKkMo8ZTx3f+BZEkzsRUY10Xsm2mwU0= github.com/stoewer/go-strcase v1.3.0 h1:g0eASXYtp+yvN9fK8sH94oCIk0fau9uV1/ZdJ0AVEzs= github.com/stoewer/go-strcase v1.3.0/go.mod h1:fAH5hQ5pehh+j3nZfvwdk2RgEgQjAoM8wodgtPmh1xo= github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= @@ -150,6 +234,8 @@ github.com/stretchr/objx v0.5.0/go.mod h1:Yh+to48EsGEfYuaHDzXPcE3xhTkx73EhmCGUpE github.com/stretchr/objx v0.5.2 h1:xuMeJ0Sdp5ZMRXx/aWO6RZxdr3beISkG5/G/aIRr3pY= github.com/stretchr/objx v0.5.2/go.mod h1:FRsXN1f5AsAjCGJKqEizvkpNtU+EGNCLh3NxZ/8L+MA= github.com/stretchr/testify v1.3.0/go.mod h1:M5WIy9Dh21IEIfnGCwXGc5bZfKNJtfHm1UVUgZn+9EI= +github.com/stretchr/testify v1.4.0/go.mod h1:j7eGeouHqKxXV5pUuKE4zz7dFj8WfuZ+81PSLYec5m4= +github.com/stretchr/testify v1.5.1/go.mod h1:5W2xD1RspED5o8YsWQXVCued0rvSQ+mT+I5cxcmMvtA= github.com/stretchr/testify v1.7.1/go.mod h1:6Fq8oRcR53rry900zMqJjRRixrwX3KX962/h/Wwjteg= github.com/stretchr/testify v1.8.0/go.mod h1:yNjHg4UonilssWZ8iaSj1OCr/vHnekPRkoO+kdMU+MU= github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4= @@ -163,8 +249,11 @@ github.com/tidwall/pretty v1.2.1 h1:qjsOFOWWQl+N3RsoF5/ssm1pHmJJwhjlSbZ51I6wMl4= github.com/tidwall/pretty v1.2.1/go.mod h1:ITEVvHYasfjBbM0u2Pg8T2nJnzm8xPwvNhhsoaGGjNU= github.com/tidwall/sjson v1.2.5 h1:kLy8mja+1c9jlljvWTlSazM7cKDRfJuR/bOJhcY5NcY= github.com/tidwall/sjson v1.2.5/go.mod h1:Fvgq9kS/6ociJEDnK0Fk1cpYF4FIW6ZF7LAe+6jwd28= +github.com/vmware-labs/yaml-jsonpath v0.3.2 h1:/5QKeCBGdsInyDCyVNLbXyilb61MXGi9NP674f9Hobk= +github.com/vmware-labs/yaml-jsonpath v0.3.2/go.mod h1:U6whw1z03QyqgWdgXxvVnQ90zN1BWz5V+51Ewf8k+rQ= github.com/x448/float16 v0.8.4 h1:qLwI1I70+NjRFUR3zs1JPUCgaCXSh3SW62uAKT1mSBM= github.com/x448/float16 v0.8.4/go.mod h1:14CWIYCyZA/cWjXOioeEpHeN/83MdbZDRQHoFcYsOfg= +github.com/yuin/goldmark v1.2.1/go.mod h1:3hX8gzYuyVAZsxl0MRgGTJEmQBFcNTphYh9decYSb74= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.65.0 h1:7iP2uCb7sGddAr30RRS6xjKy7AZ2JtTOPA3oolgVSw8= @@ -197,26 +286,69 @@ go.yaml.in/yaml/v2 v2.4.3 h1:6gvOSjQoTB3vt1l+CU+tSyi/HOjfOjRLJ4YwYZGwRO0= go.yaml.in/yaml/v2 v2.4.3/go.mod h1:zSxWcmIDjOzPXpjlTTbAsKokqkDNAVtZO0WOMiT90s8= go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w= +golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= +golang.org/x/crypto v0.0.0-20200622213623-75b288015ac9/go.mod h1:LzIPMQfyMNhhGPhUkYOs5KpL4U8rLKemX1yGLhDgUto= +golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y= +golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I= golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93 h1:fQsdNF2N+/YewlRZiricy4P1iimyPKZ/xwniHj8Q2a0= golang.org/x/exp v0.0.0-20251219203646-944ab1f22d93/go.mod h1:EPRbTFwzwjXj9NpYyyrvenVh9Y+GFeEvMNh7Xuz7xgU= +golang.org/x/mod v0.3.0/go.mod h1:s0Qsj1ACt9ePp/hMypM3fl4fZqREWJwdYDEqhRiZZUA= golang.org/x/mod v0.38.0 h1:MECBjubtXD7yj4HrhIUcywNaGeNVUdfVnxmPajOk4yk= golang.org/x/mod v0.38.0/go.mod h1:V6Xz0pq8TQ3dGqVQ1FVHuelZpAL0uNhSkk9ogYP3c40= +golang.org/x/net v0.0.0-20180906233101-161cd47e91fd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4= +golang.org/x/net v0.0.0-20190404232315-eb5bcb51f2a3/go.mod h1:t9HGtf8HONx5eT2rtn7q6eTqICYqUVnKs3thJo3Qplg= +golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= +golang.org/x/net v0.0.0-20200520004742-59133d7f0dd7/go.mod h1:qpuaurCH72eLCgpAm/N6yyVIVM9cpaDIP3A8BGJEC5A= +golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU= +golang.org/x/net v0.0.0-20210428140749-89ef3d95e781/go.mod h1:OJAsFXCWl8Ukc7SiCT/9KSuxbyM7479/AVlXFRxuMCk= +golang.org/x/net v0.0.0-20220225172249-27dd8689420f/go.mod h1:CfG3xpIq0wQ8r1q4Su4UZFWDARRcnwPjda9FqA0JpMk= golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To= golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU= golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs= golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q= +golang.org/x/sync v0.0.0-20180314180146-1d60e4601c6f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= +golang.org/x/sync v0.0.0-20201020160332-67f06af15bc9/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek= golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0= +golang.org/x/sys v0.0.0-20180909124046-d0be0721c37e/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= +golang.org/x/sys v0.0.0-20190215142949-d0b11bdaac8a/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY= +golang.org/x/sys v0.0.0-20190412213103-97732733099d/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20190904154756-749cb33beabd/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20191005200804-aed5e4c7ecf9/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20191120155948-bd437916bb0e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20191204072324-ce4227a45e2e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20200323222414-85ca7c5b95cd/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210112080510-489259a85091/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20211216021012-1d35b9e2eb4e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs= golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8= golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0= golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w= +golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ= +golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ= golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8= golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M= golang.org/x/time v0.14.0 h1:MRx4UaLrDotUKUdCIqzPC48t1Y9hANFKIRpNx+Te8PI= golang.org/x/time v0.14.0/go.mod h1:eL/Oa2bBBK0TkX57Fyni+NgnyQQN4LitPmob2Hjnqw4= +golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ= +golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo= +golang.org/x/tools v0.0.0-20201224043029-2b0845dc783e/go.mod h1:emZCQorbCU4vsT4fOWvOPXz4eW1wZW4PmDk9uLelYpA= golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE= golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk= +golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191011141410-1b5146add898/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20191204190536-9bdfabe68543/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= +golang.org/x/xerrors v0.0.0-20200804184101-5ec99f83aff1/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= google.golang.org/genproto/googleapis/api v0.0.0-20260526163538-3dc84a4a5aaa h1:Kjn0N0tCrDgiAFW+lGO4JZ3ck44CehvJQMAwj9QF0G8= @@ -225,17 +357,34 @@ google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa h1: google.golang.org/genproto/googleapis/rpc v0.0.0-20260526163538-3dc84a4a5aaa/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= google.golang.org/grpc v1.83.2 h1:EManeRomTObA0BU7I8vXgg/78uE5MJ9M8B39EX2WscU= google.golang.org/grpc v1.83.2/go.mod h1:YPI1hK3kDked6iHvgX3tR0y+nX/qpMFKhPgFsokw1S8= +google.golang.org/protobuf v0.0.0-20200109180630-ec00e32a8dfd/go.mod h1:DFci5gLYBciE7Vtevhsrf46CRTquxDuWsQurQQe4oz8= +google.golang.org/protobuf v0.0.0-20200221191635-4d8936d0db64/go.mod h1:kwYJMbMJ01Woi6D6+Kah6886xMZcty6N08ah7+eCXa0= +google.golang.org/protobuf v0.0.0-20200228230310-ab0ca4ff8a60/go.mod h1:cfTl7dwQJ+fmap5saPgwCLgHXTUD7jkjRqWcaiX5VyM= +google.golang.org/protobuf v1.20.1-0.20200309200217-e05f789c0967/go.mod h1:A+miEFZTKqfCUM6K7xSMQL9OKL/b6hQv+e19PK+JZNE= +google.golang.org/protobuf v1.21.0/go.mod h1:47Nbq4nVaFHyn7ilMalzfO3qCViNmqZ2kzikPIcrTAo= +google.golang.org/protobuf v1.23.0/go.mod h1:EGpADcykh3NcUnDUJcl1+ZksZNG86OlYog2l/sGQquU= +google.golang.org/protobuf v1.26.0-rc.1/go.mod h1:jlhhOSvTdKEhbULTjvd4ARK9grFBp09yW+WbY/TyQbw= +google.golang.org/protobuf v1.26.0/go.mod h1:9q0QmTI4eRPtz6boOQmLYwt+qCgq0jsYwAQnmE0givc= google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af h1:+5/Sw3GsDNlEmu7TfklWKPdQ0Ykja5VEmq2i817+jbI= google.golang.org/protobuf v1.36.12-0.20260120151049-f2248ac996af/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= +gopkg.in/check.v1 v1.0.0-20190902080502-41f04d3bba15/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= gopkg.in/evanphx/json-patch.v4 v4.13.0 h1:czT3CmqEaQ1aanPc5SdlgQrrEIb8w/wwCvWWnfEbYzo= gopkg.in/evanphx/json-patch.v4 v4.13.0/go.mod h1:p8EYWUEYMpynmqDbY58zCKCFZw8pRWMG4EsWvDvM72M= +gopkg.in/fsnotify.v1 v1.4.7/go.mod h1:Tz8NjZHkW78fSQdbUxIjBTcgA1z1m8ZHf0WmKUhAMys= gopkg.in/inf.v0 v0.9.1 h1:73M5CoZyi3ZLMOyDlQh031Cx6N9NDJ2Vvfl76EDAgDc= gopkg.in/inf.v0 v0.9.1/go.mod h1:cWUDdTG/fYaXco+Dcufb5Vnc6Gp2YChqWtbxRZE0mXw= +gopkg.in/tomb.v1 v1.0.0-20141024135613-dd632973f1e7 h1:uRGJdciOHaEIrze2W8Q3AKkepLTh2hOroT7a+7czfdQ= +gopkg.in/tomb.v1 v1.0.0-20141024135613-dd632973f1e7/go.mod h1:dt/ZhP58zS4L8KSrWDmTeBkI65Dw0HsyUHuEVlX15mw= +gopkg.in/yaml.v2 v2.2.1/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= +gopkg.in/yaml.v2 v2.2.2/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= +gopkg.in/yaml.v2 v2.2.4/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= +gopkg.in/yaml.v2 v2.3.0/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI= gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY= gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ= +gopkg.in/yaml.v3 v3.0.0-20191026110619-0b21df46bc1d/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= diff --git a/csi-driver/internal/clusters/clusters.go b/csi-driver/internal/clusters/clusters.go index 49276c17b..daadbacf6 100644 --- a/csi-driver/internal/clusters/clusters.go +++ b/csi-driver/internal/clusters/clusters.go @@ -15,6 +15,7 @@ import ( "os" "strings" + atlascp "github.com/simplyblock/atlas/controlplane" "github.com/simplyblock/atlas/errs/deferrers" "github.com/simplyblock/atlas/lvol" "k8s.io/klog" @@ -38,6 +39,10 @@ type Config struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster of the site this driver runs on (written by the + // site's operator); an entry without it belongs to another site, kept so + // that a failed-over volume's handle still resolves. + Local bool `json:"local,omitempty"` } // Info is the secret file as a whole. @@ -85,6 +90,24 @@ func Load() (Info, error) { return clusters, nil } +// Local returns the ids of the clusters the secret marks local, and whether +// the secret marks any: a secret written by an operator that predates the +// flag marks none, and callers then fall back to treating every cluster as +// local. +func Local() (map[string]bool, bool, error) { + clusters, err := Load() + if err != nil { + return nil, false, err + } + local := map[string]bool{} + for _, cluster := range clusters.Clusters { + if cluster.Local { + local[cluster.ClusterID] = true + } + } + return local, len(local) > 0, nil +} + // List returns the ID of every cluster in the secret. func List() ([]string, error) { clusters, err := Load() @@ -102,35 +125,11 @@ func List() ([]string, error) { // pool. poolIDOrName may be a pool UUID (used as-is), a pool name (resolved via // the API), or empty (no pool context, so only cluster-level operations work). func Client(ctx context.Context, clusterID, poolIDOrName string) (*controlplane.ClusterClient, error) { - clusters, err := Load() + clusterConfig, credential, err := resolve(clusterID) if err != nil { return nil, err } - var clusterConfig *Config - for _, cluster := range clusters.Clusters { - if cluster.ClusterID == clusterID { - clusterConfig = &cluster - break - } - } - - if clusterConfig == nil { - return nil, fmt.Errorf("failed to find secret for clusterID %s: %w", clusterID, controlplane.ErrClusterNotFound) - } - - if clusterConfig.ClusterEndpoint == "" { - return nil, fmt.Errorf("invalid cluster configuration for clusterID %s: missing endpoint", clusterID) - } - - credential := credentialFor(clusterConfig) - if credential == "" { - return nil, fmt.Errorf( - "invalid cluster configuration for clusterID %s: no cluster_secret and no API token available", - clusterID, - ) - } - klog.Infof("Simplyblock client created for ClusterID:%s, Endpoint:%s", clusterConfig.ClusterID, clusterConfig.ClusterEndpoint, @@ -152,6 +151,54 @@ func Client(ctx context.Context, clusterID, poolIDOrName string) (*controlplane. return client, nil } +// ReplicationClient creates the generated, atlas-lib control-plane client for +// a cluster, sharing this package's secret.json resolution and credential +// precedence with Client. It is a sibling rather than a Client return-type +// change: Client's hand-rolled controlplane.ClusterClient is what every +// existing Volume/Snapshot/Clone RPC already depends on, and only the +// Replication service (csi-addons) is built against the generated client. +func ReplicationClient(_ context.Context, clusterID string) (*atlascp.Client, error) { + clusterConfig, credential, err := resolve(clusterID) + if err != nil { + return nil, err + } + return atlascp.New(atlascp.Config{Endpoint: clusterConfig.ClusterEndpoint, Token: credential}) +} + +// resolve looks up the named cluster's endpoint and credential in the driver's +// secret.json. It is the lookup Client and ReplicationClient share, so a +// cluster missing from the secret is invisible to neither or both, never one. +func resolve(clusterID string) (*Config, string, error) { + clusters, err := Load() + if err != nil { + return nil, "", err + } + + var clusterConfig *Config + for _, cluster := range clusters.Clusters { + if cluster.ClusterID == clusterID { + clusterConfig = &cluster + break + } + } + + if clusterConfig == nil { + return nil, "", fmt.Errorf("failed to find secret for clusterID %s: %w", clusterID, controlplane.ErrClusterNotFound) + } + if clusterConfig.ClusterEndpoint == "" { + return nil, "", fmt.Errorf("invalid cluster configuration for clusterID %s: missing endpoint", clusterID) + } + + credential := credentialFor(clusterConfig) + if credential == "" { + return nil, "", fmt.Errorf( + "invalid cluster configuration for clusterID %s: no cluster_secret and no API token available", + clusterID, + ) + } + return clusterConfig, credential, nil +} + // credentialFor returns the bearer credential for a cluster: the API token when // SPDKCSI_API_TOKEN_PATH is set and names a readable, non-empty file, otherwise // the cluster_secret from the secret entry. diff --git a/csi-driver/internal/controlplane/consistency_group.go b/csi-driver/internal/controlplane/consistency_group.go index 72e7b14f0..eebedeaea 100644 --- a/csi-driver/internal/controlplane/consistency_group.go +++ b/csi-driver/internal/controlplane/consistency_group.go @@ -283,3 +283,79 @@ func (c *ClusterClient) JoinConsistencyGroupMember(ctx context.Context, groupID, func (c *ClusterClient) DetachConsistencyGroupMember(ctx context.Context, groupID, lvolID string) error { return c.API.detachConsistencyGroupMember(ctx, groupUUID(groupID), lvolID) } + +// Join plan step names (sbcli cg_colocation, docs/consistency-group-colocation.md). +const ( + JoinStepMigrate = "migrate" + JoinStepJoin = "join" + JoinStepColocate = "colocate" +) + +// ErrColocationRefused wraps a backend 409 on a co-location step: namespace +// moves are disabled, or a host is connected and the client cannot swap +// paths. Like a membership refusal it is a standing state, not a fault. +var ErrColocationRefused = errors.New("consistency-group co-location refused") + +// JoinPlan is what joining an existing volume to a group takes: a live +// migration of MigrateLvolIDs to TargetNodeID first (the volume is off the +// group's pinned node/LVS), the join, and a namespace move into TargetNQN. +type JoinPlan struct { + Steps []string `json:"steps"` + TargetNodeID string `json:"target_node_id"` + MigrateLvolIDs []string `json:"migrate_lvol_ids"` + TargetNQN string `json:"target_nqn"` +} + +// Has reports whether the plan contains step. +func (p *JoinPlan) Has(step string) bool { + for _, s := range p.Steps { + if s == step { + return true + } + } + return false +} + +func (client APIClient) planConsistencyGroupJoin(ctx context.Context, gid, lvolID string) (*JoinPlan, error) { + body := map[string]string{"lvol_id": lvolID} + raw, err := client.do(ctx, http.MethodPost, client.v2consistencyGroupMembers(gid)+"/plan", body) + if err != nil { + if isHTTPStatus(err, http.StatusConflict) { + return nil, fmt.Errorf("%w: %s", ErrMembershipRefused, err.Error()) + } + return nil, err + } + var plan JoinPlan + if err := json.Unmarshal(raw, &plan); err != nil { + return nil, fmt.Errorf("unexpected response for join plan: %w", err) + } + return &plan, nil +} + +func (client APIClient) colocateConsistencyGroupMember( + ctx context.Context, gid, lvolID string, clientSwapReady bool, +) error { + body := map[string]bool{"client_swap_ready": clientSwapReady} + _, err := client.do(ctx, http.MethodPost, client.v2consistencyGroupMember(gid, lvolID)+"/colocate", body) + if err != nil && isHTTPStatus(err, http.StatusConflict) { + return fmt.Errorf("%w: %s", ErrColocationRefused, err.Error()) + } + return err +} + +// PlanConsistencyGroupJoin asks the backend which steps joining lvolID to the +// group takes, without taking any. A join that can never succeed (another +// group, another pool, another group's subsystem siblings) is +// ErrMembershipRefused. +func (c *ClusterClient) PlanConsistencyGroupJoin(ctx context.Context, groupID, lvolID string) (*JoinPlan, error) { + return c.API.planConsistencyGroupJoin(ctx, groupUUID(groupID), lvolID) +} + +// ColocateConsistencyGroupMember moves a member's namespace into its group's +// subsystem. clientSwapReady asserts the client stages the volume behind the +// device-mapper indirection and swaps paths itself. +func (c *ClusterClient) ColocateConsistencyGroupMember( + ctx context.Context, groupID, lvolID string, clientSwapReady bool, +) error { + return c.API.colocateConsistencyGroupMember(ctx, groupUUID(groupID), lvolID, clientSwapReady) +} diff --git a/csi-driver/internal/csi/common/server.go b/csi-driver/internal/csi/common/server.go index 8ab1cde6b..a83acf61b 100644 --- a/csi-driver/internal/csi/common/server.go +++ b/csi-driver/internal/csi/common/server.go @@ -13,7 +13,17 @@ import ( ) type NonBlockingGRPCServer interface { - Start(endpoint string, ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer) + // register is called once per extra service (csi-addons Identity and + // Replication) with the same *grpc.Server the CSI services register on, + // so csicommon never has to import csi-addons: the caller builds the + // closures and this package only invokes them. + Start( + endpoint string, + ids csi.IdentityServer, + cs csi.ControllerServer, + ns csi.NodeServer, + register ...func(*grpc.Server), + ) Wait() Stop() ForceStop() @@ -33,10 +43,11 @@ func (s *nonBlockingGRPCServer) Start( ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer, + register ...func(*grpc.Server), ) { s.wg.Add(1) - go s.serve(endpoint, ids, cs, ns) + go s.serve(endpoint, ids, cs, ns, register) } func (s *nonBlockingGRPCServer) Wait() { @@ -56,6 +67,7 @@ func (s *nonBlockingGRPCServer) serve( ids csi.IdentityServer, cs csi.ControllerServer, ns csi.NodeServer, + register []func(*grpc.Server), ) { var err error @@ -112,6 +124,9 @@ func (s *nonBlockingGRPCServer) serve( if ns != nil { csi.RegisterNodeServer(server, ns) } + for _, r := range register { + r(server) + } klog.Infof("Listening for connections on address: %#v", listener.Addr()) diff --git a/csi-driver/internal/csi/controller/cg_membership_watcher.go b/csi-driver/internal/csi/controller/cg_membership_watcher.go index 96b8aaba2..28d8d50f5 100644 --- a/csi-driver/internal/csi/controller/cg_membership_watcher.go +++ b/csi-driver/internal/csi/controller/cg_membership_watcher.go @@ -16,6 +16,7 @@ import ( corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/client-go/dynamic" "k8s.io/client-go/informers" "k8s.io/client-go/kubernetes" "k8s.io/client-go/kubernetes/scheme" @@ -42,6 +43,17 @@ type membershipClient interface { GetConsistencyGroup(ctx context.Context, groupID string) (*controlplane.ConsistencyGroupSummary, error) JoinConsistencyGroupMember(ctx context.Context, groupID, lvolID string) error DetachConsistencyGroupMember(ctx context.Context, groupID, lvolID string) error + PlanConsistencyGroupJoin(ctx context.Context, groupID, lvolID string) (*controlplane.JoinPlan, error) + ColocateConsistencyGroupMember(ctx context.Context, groupID, lvolID string, clientSwapReady bool) error +} + +// migrationRequester asks the operator to live-migrate a PersistentVolume's +// backing volume to a storage node, through a VolumeMigration: the operator +// attaches the target paths on the consumer host before the backend moves the +// data, which only it can do. Ensure creates the request once (by name) and +// reports its phase; "" while the operator has not picked it up. +type migrationRequester interface { + Ensure(ctx context.Context, name, pvName, targetNodeUUID string) (string, error) } // cgMembershipWatcher reconciles PVC label state to backend group membership. @@ -52,12 +64,29 @@ type cgMembershipWatcher struct { // pool; production wires clusters.Client, tests substitute a stub. clientFor func(ctx context.Context, clusterID, poolRef string) (membershipClient, error) recorder record.EventRecorder + // migrations runs the pre-join live migration of a volume off its group's + // pinned node (co-location design §5); nil disables it, and an off-pin + // join stays refused as before. + migrations migrationRequester + // colocate moves a member into its group's subsystem after the join; off + // by default, because the backend refuses the move for an attached volume + // until the node plugin swaps paths (design §6). + colocate bool + // clientSwapReady asserts every node stages volumes behind the + // device-mapper indirection and swaps paths itself (design §6). + clientSwapReady bool } // StartConsistencyGroupLabelWatcher runs the membership watcher until ctx is // canceled. It is a no-op (with a log line) when kube is nil, mirroring how // the annotation helpers degrade without an in-cluster config. -func StartConsistencyGroupLabelWatcher(ctx context.Context, kube kubernetes.Interface, driverName string) { +// +// dyn, when non-nil, lets the watcher request the pre-join live migration of a +// volume off its group's pinned node (a VolumeMigration); without it an +// off-pin join stays refused. +func StartConsistencyGroupLabelWatcher( + ctx context.Context, kube kubernetes.Interface, dyn dynamic.Interface, driverName string, +) { if kube == nil { klog.Warning("consistency-group label watcher disabled: no Kubernetes client") return @@ -72,6 +101,12 @@ func StartConsistencyGroupLabelWatcher(ctx context.Context, kube kubernetes.Inte }, recorder: broadcaster.NewRecorder(scheme.Scheme, corev1.EventSource{Component: "spdkcsi-cg-membership"}), } + preJoin, colocate, swapReady := watcherOptions() + if preJoin && dyn != nil { + watcher.migrations = newVolumeMigrations(dyn) + } + watcher.colocate = colocate + watcher.clientSwapReady = swapReady factory := informers.NewSharedInformerFactory(kube, membershipResync) informer := factory.Core().V1().PersistentVolumeClaims().Informer() @@ -136,7 +171,7 @@ func (w *cgMembershipWatcher) reconcile(ctx context.Context, pvc *corev1.Persist switch { case label != "" && groupID == "": - return w.join(ctx, client, pvc, label, handle.VolumeID) + return w.join(ctx, client, pvc, pv.Name, label, handle.VolumeID) case label == "" && groupID != "": return w.detach(ctx, client, pvc, groupID, handle.VolumeID) case label != "" && groupID != "": @@ -152,14 +187,20 @@ func (w *cgMembershipWatcher) reconcile(ctx context.Context, pvc *corev1.Persist "volume is a member of consistency group %q but the PVC is labeled %q; "+ "membership is one-way, so relabeling cannot move a volume between groups", group.Name, label) + return nil } + return w.colocateMember(ctx, client, pvc, groupID, label, handle.VolumeID) } return nil } +// preJoinMigrationName is the VolumeMigration a late join requests for a +// volume: one per volume, so a resync finds the request it already made. +func preJoinMigrationName(lvolID string) string { return "cg-join-" + lvolID } + func (w *cgMembershipWatcher) join( ctx context.Context, client membershipClient, - pvc *corev1.PersistentVolumeClaim, label, lvolID string, + pvc *corev1.PersistentVolumeClaim, pvName, label, lvolID string, ) error { group, err := client.ResolveConsistencyGroupByName(ctx, label) if err != nil { @@ -174,17 +215,97 @@ func (w *cgMembershipWatcher) join( } if err := client.JoinConsistencyGroupMember(ctx, group.ID, lvolID); err != nil { if errors.Is(err, controlplane.ErrMembershipRefused) { - w.recorder.Eventf(pvc, corev1.EventTypeWarning, "ConsistencyGroupJoinRefused", - "volume cannot join consistency group %q: %v", label, err) - return nil + return w.joinRefused(ctx, client, pvc, pvName, label, group.ID, lvolID, err) } return fmt.Errorf("join consistency group %q: %w", label, err) } w.recorder.Eventf(pvc, corev1.EventTypeNormal, "ConsistencyGroupJoined", "volume joined consistency group %q; it is included from the next generation", label) + return w.colocateMember(ctx, client, pvc, group.ID, label, lvolID) +} + +// joinRefused turns a refused join into the pre-join migration when the only +// obstacle is placement: the backend's plan says the volume (with its +// subsystem siblings) must first move to the group's pinned node. Anything +// else stays a refusal on the PVC. +func (w *cgMembershipWatcher) joinRefused( + ctx context.Context, client membershipClient, pvc *corev1.PersistentVolumeClaim, + pvName, label, groupID, lvolID string, refusal error, +) error { + refuse := func(err error) error { + w.recorder.Eventf(pvc, corev1.EventTypeWarning, "ConsistencyGroupJoinRefused", + "volume cannot join consistency group %q: %v", label, err) + return nil + } + if w.migrations == nil { + return refuse(refusal) + } + plan, err := client.PlanConsistencyGroupJoin(ctx, groupID, lvolID) + if err != nil { + if errors.Is(err, controlplane.ErrMembershipRefused) { + return refuse(err) + } + return fmt.Errorf("plan the join to consistency group %q: %w", label, err) + } + if !plan.Has(controlplane.JoinStepMigrate) || plan.TargetNodeID == "" { + return refuse(refusal) + } + name := preJoinMigrationName(lvolID) + phase, err := w.migrations.Ensure(ctx, name, pvName, plan.TargetNodeID) + if err != nil { + return fmt.Errorf("request the pre-join migration of %s: %w", pvName, err) + } + switch phase { + case migrationCompleted: + // Moved: the placement precondition holds now. The resync or the next + // PVC event joins; joining here would race the backend's record switch + // of the last subsystem sibling. + w.recorder.Eventf(pvc, corev1.EventTypeNormal, "ConsistencyGroupMigrated", + "volume moved to the group's node %s; joining consistency group %q on the next pass", + plan.TargetNodeID, label) + case migrationFailed, migrationAborted: + w.recorder.Eventf(pvc, corev1.EventTypeWarning, "ConsistencyGroupMigrationFailed", + "the migration to consistency group %q's node %s ended %s; delete VolumeMigration %s to retry", + label, plan.TargetNodeID, phase, name) + default: + w.recorder.Eventf(pvc, corev1.EventTypeNormal, "ConsistencyGroupMigrating", + "volume is off consistency group %q's node: migrating it (with %d volume(s) of its "+ + "subsystem) to node %s through VolumeMigration %s before the join", label, + len(plan.MigrateLvolIDs), plan.TargetNodeID, name) + } return nil } +// The VolumeMigration phases the watcher acts on (operator api/v1alpha1). +const ( + migrationCompleted = "Completed" + migrationFailed = "Failed" + migrationAborted = "Aborted" +) + +// colocateMember moves a current member into its group's subsystem when the +// watcher is configured to; a refusal (moves disabled, or a connected host and +// no client swap) is a Normal event, since the member is valid where it is. +func (w *cgMembershipWatcher) colocateMember( + ctx context.Context, client membershipClient, pvc *corev1.PersistentVolumeClaim, + groupID, label, lvolID string, +) error { + if !w.colocate { + return nil + } + err := client.ColocateConsistencyGroupMember(ctx, groupID, lvolID, w.clientSwapReady) + switch { + case err == nil: + return nil + case errors.Is(err, controlplane.ErrColocationRefused): + w.recorder.Eventf(pvc, corev1.EventTypeNormal, "ConsistencyGroupColocationDeferred", + "volume stays in its own subsystem for now (consistency group %q): %v", label, err) + return nil + default: + return fmt.Errorf("co-locate with consistency group %q: %w", label, err) + } +} + func (w *cgMembershipWatcher) detach( ctx context.Context, client membershipClient, pvc *corev1.PersistentVolumeClaim, groupID, lvolID string, diff --git a/csi-driver/internal/csi/controller/cg_membership_watcher_test.go b/csi-driver/internal/csi/controller/cg_membership_watcher_test.go index ba9f958eb..850daa973 100644 --- a/csi-driver/internal/csi/controller/cg_membership_watcher_test.go +++ b/csi-driver/internal/csi/controller/cg_membership_watcher_test.go @@ -32,6 +32,45 @@ type fakeMembership struct { joined [][2]string // {groupID, lvolID} detached [][2]string + + plan *controlplane.JoinPlan + planErr error + colocateErr error + colocated [][2]string + swapReady []bool +} + +func (f *fakeMembership) PlanConsistencyGroupJoin( + _ context.Context, _, _ string, +) (*controlplane.JoinPlan, error) { + if f.planErr != nil { + return nil, f.planErr + } + if f.plan == nil { + return &controlplane.JoinPlan{Steps: []string{controlplane.JoinStepJoin}}, nil + } + return f.plan, nil +} + +func (f *fakeMembership) ColocateConsistencyGroupMember( + _ context.Context, groupID, lvolID string, clientSwapReady bool, +) error { + f.colocated = append(f.colocated, [2]string{groupID, lvolID}) + f.swapReady = append(f.swapReady, clientSwapReady) + return f.colocateErr +} + +// fakeMigrations records the VolumeMigrations the watcher requests and +// reports a scripted phase for them. +type fakeMigrations struct { + phase string + err error + requests [][3]string // {name, pvName, target} +} + +func (f *fakeMigrations) Ensure(_ context.Context, name, pvName, target string) (string, error) { + f.requests = append(f.requests, [3]string{name, pvName, target}) + return f.phase, f.err } func (f *fakeMembership) GetVolumeGroupID(_ context.Context, lvolID string) (string, error) { @@ -253,3 +292,127 @@ func TestConflictingLabelIsSurfacedNotActedOn(t *testing.T) { } requireEvent(t, recorder, "ConsistencyGroupConflict") } + +func offPinFixture(t *testing.T, phase string) ( + *cgMembershipWatcher, *corev1.PersistentVolumeClaim, *record.FakeRecorder, + *fakeMembership, *fakeMigrations, +) { + t.Helper() + membership := &fakeMembership{ + groupIDByLvol: map[string]string{}, + groupsByName: map[string]*controlplane.ConsistencyGroupSummary{ + "db-group": {ID: "gid-1", Name: "db-group"}, + }, + joinErr: controlplane.ErrMembershipRefused, + plan: &controlplane.JoinPlan{ + Steps: []string{controlplane.JoinStepMigrate, controlplane.JoinStepJoin}, + TargetNodeID: "node-pin", + MigrateLvolIDs: []string{testLvol, "sibling"}, + }, + } + migrations := &fakeMigrations{phase: phase} + watcher, pvc, recorder := watcherFixture(t, + map[string]string{consistencyGroupLabel: "db-group"}, testDriver, membership) + watcher.migrations = migrations + return watcher, pvc, recorder, membership, migrations +} + +func TestOffPinJoinRequestsThePreJoinMigration(t *testing.T) { + watcher, pvc, recorder, membership, migrations := offPinFixture(t, "") + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + if len(migrations.requests) != 1 || + migrations.requests[0] != [3]string{"cg-join-" + testLvol, "pv-data", "node-pin"} { + t.Fatalf("expected one VolumeMigration of pv-data to node-pin, got %v", migrations.requests) + } + if len(membership.joined) != 0 { + t.Fatalf("no join may happen before the volume is on the pin, got %v", membership.joined) + } + requireEvent(t, recorder, "ConsistencyGroupMigrating") +} + +func TestACompletedPreJoinMigrationIsReportedAndTheJoinFollowsOnTheNextPass(t *testing.T) { + watcher, pvc, recorder, _, migrations := offPinFixture(t, migrationCompleted) + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + requireEvent(t, recorder, "ConsistencyGroupMigrated") + if len(migrations.requests) != 1 { + t.Fatalf("the existing request is read, not duplicated: %v", migrations.requests) + } +} + +func TestAFailedPreJoinMigrationIsAWarning(t *testing.T) { + watcher, pvc, recorder, _, _ := offPinFixture(t, migrationFailed) + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + requireEvent(t, recorder, "ConsistencyGroupMigrationFailed") +} + +func TestAJoinThatCanNeverSucceedIsNotMigrated(t *testing.T) { + watcher, pvc, recorder, membership, migrations := offPinFixture(t, "") + membership.planErr = controlplane.ErrMembershipRefused + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + if len(migrations.requests) != 0 { + t.Fatalf("a refused plan must not request a migration: %v", migrations.requests) + } + requireEvent(t, recorder, "ConsistencyGroupJoinRefused") +} + +func TestWithoutTheMigrationRequesterAnOffPinJoinStaysRefused(t *testing.T) { + watcher, pvc, recorder, _, _ := offPinFixture(t, "") + watcher.migrations = nil + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + requireEvent(t, recorder, "ConsistencyGroupJoinRefused") +} + +func TestColocationRunsAfterAJoinOnlyWhenEnabled(t *testing.T) { + membership := &fakeMembership{ + groupIDByLvol: map[string]string{}, + groupsByName: map[string]*controlplane.ConsistencyGroupSummary{ + "db-group": {ID: "gid-1", Name: "db-group"}, + }, + } + watcher, pvc, recorder := watcherFixture(t, + map[string]string{consistencyGroupLabel: "db-group"}, testDriver, membership) + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + if len(membership.colocated) != 0 { + t.Fatalf("co-location is off by default, got %v", membership.colocated) + } + requireEvent(t, recorder, "ConsistencyGroupJoined") + + membership.joined = nil + watcher.colocate, watcher.clientSwapReady = true, true + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("reconcile: %v", err) + } + if len(membership.colocated) != 1 || !membership.swapReady[0] { + t.Fatalf("expected one co-location with the client swap asserted, got %v %v", + membership.colocated, membership.swapReady) + } +} + +func TestARefusedColocationIsANormalEvent(t *testing.T) { + membership := &fakeMembership{ + groupIDByLvol: map[string]string{testLvol: "gid-1"}, + groupsByID: map[string]*controlplane.ConsistencyGroupSummary{ + "gid-1": {ID: "gid-1", Name: "db-group"}, + }, + colocateErr: controlplane.ErrColocationRefused, + } + watcher, pvc, recorder := watcherFixture(t, + map[string]string{consistencyGroupLabel: "db-group"}, testDriver, membership) + watcher.colocate = true + if err := watcher.reconcile(context.Background(), pvc); err != nil { + t.Fatalf("a refused co-location is not an error: %v", err) + } + requireEvent(t, recorder, "ConsistencyGroupColocationDeferred") +} diff --git a/csi-driver/internal/csi/controller/cg_prejoin_migration.go b/csi-driver/internal/csi/controller/cg_prejoin_migration.go new file mode 100644 index 000000000..156cb687d --- /dev/null +++ b/csi-driver/internal/csi/controller/cg_prejoin_migration.go @@ -0,0 +1,104 @@ +// The pre-join live migration (docs/consistency-group-colocation.md §5, in +// sbcli): a volume whose PVC gets the consistency-group label while it lives +// off the group's pinned node/LVS is moved there first, then joined. +// +// The move is requested as a VolumeMigration, not made against the control +// plane directly: a live migration needs the consumer host attached to the +// target's paths before the backend moves the data, and attaching them is the +// operator's VolumeMigration controller's job. The CSI driver only asks for +// the move and reads how it went, through a dynamic client, so it does not +// depend on the operator's API types. +package controller + +import ( + "context" + "fmt" + "os" + "strings" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/client-go/dynamic" +) + +var volumeMigrationGVR = schema.GroupVersionResource{ + Group: "storage.simplyblock.io", Version: "v1alpha1", Resource: "volumemigrations", +} + +// preJoinPurposeLabel marks the VolumeMigrations the watcher creates, so they +// are told apart from the rebalancer's and an operator's own. +const preJoinPurposeLabel = "storage.simplyblock.io/purpose" + +// volumeMigrations creates VolumeMigrations in the driver's namespace. +type volumeMigrations struct { + client dynamic.Interface + namespace string +} + +func newVolumeMigrations(client dynamic.Interface) *volumeMigrations { + ns := strings.TrimSpace(os.Getenv("POD_NAMESPACE")) + if ns == "" { + ns = "simplyblock" + } + return &volumeMigrations{client: client, namespace: ns} +} + +// Ensure creates the named VolumeMigration when it does not exist and returns +// its status.phase. An existing one is never rewritten: a request for another +// target (the group re-pinned meanwhile) waits until the old one is deleted. +func (m *volumeMigrations) Ensure(ctx context.Context, name, pvName, targetNodeUUID string) (string, error) { + res := m.client.Resource(volumeMigrationGVR).Namespace(m.namespace) + obj, err := res.Get(ctx, name, metav1.GetOptions{}) + if apierrors.IsNotFound(err) { + obj = &unstructured.Unstructured{Object: map[string]any{ + "apiVersion": "storage.simplyblock.io/v1alpha1", + "kind": "VolumeMigration", + "metadata": map[string]any{ + "name": name, + "namespace": m.namespace, + "labels": map[string]any{ + preJoinPurposeLabel: "consistency-group-join", + "app.kubernetes.io/managed-by": "spdkcsi", + }, + }, + "spec": map[string]any{ + "pvName": pvName, + "targetNodeUUID": targetNodeUUID, + }, + }} + created, cerr := res.Create(ctx, obj, metav1.CreateOptions{}) + if cerr != nil && !apierrors.IsAlreadyExists(cerr) { + return "", fmt.Errorf("create VolumeMigration %s/%s: %w", m.namespace, name, cerr) + } + if cerr == nil { + obj = created + } else if obj, err = res.Get(ctx, name, metav1.GetOptions{}); err != nil { + return "", fmt.Errorf("read VolumeMigration %s/%s: %w", m.namespace, name, err) + } + } else if err != nil { + return "", fmt.Errorf("read VolumeMigration %s/%s: %w", m.namespace, name, err) + } + phase, _, _ := unstructured.NestedString(obj.Object, "status", "phase") + return phase, nil +} + +// watcherOptions reads the co-location switches from the environment: +// SPDKCSI_CG_PREJOIN_MIGRATION (default true), SPDKCSI_CG_COLOCATE (default +// false) and SPDKCSI_CG_CLIENT_SWAP_READY (default false). +func watcherOptions() (preJoin, colocate, swapReady bool) { + flag := func(name string, def bool) bool { + switch strings.ToLower(strings.TrimSpace(os.Getenv(name))) { + case "1", "true", "yes", "on": + return true + case "0", "false", "no", "off": + return false + default: + return def + } + } + return flag("SPDKCSI_CG_PREJOIN_MIGRATION", true), + flag("SPDKCSI_CG_COLOCATE", false), + flag("SPDKCSI_CG_CLIENT_SWAP_READY", false) +} diff --git a/csi-driver/internal/csi/controller/cg_prejoin_migration_test.go b/csi-driver/internal/csi/controller/cg_prejoin_migration_test.go new file mode 100644 index 000000000..ce1a98e61 --- /dev/null +++ b/csi-driver/internal/csi/controller/cg_prejoin_migration_test.go @@ -0,0 +1,72 @@ +package controller + +import ( + "context" + "testing" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + dynamicfake "k8s.io/client-go/dynamic/fake" +) + +func newFakeVolumeMigrations(objects ...runtime.Object) (*volumeMigrations, *dynamicfake.FakeDynamicClient) { + client := dynamicfake.NewSimpleDynamicClientWithCustomListKinds(runtime.NewScheme(), + map[schema.GroupVersionResource]string{volumeMigrationGVR: "VolumeMigrationList"}, objects...) + return &volumeMigrations{client: client, namespace: "simplyblock"}, client +} + +func TestEnsureCreatesTheRequestOnceAndReportsItsPhase(t *testing.T) { + m, client := newFakeVolumeMigrations() + phase, err := m.Ensure(context.Background(), "cg-join-v1", "pv-1", "node-pin") + if err != nil || phase != "" { + t.Fatalf("first Ensure: phase %q err %v", phase, err) + } + got, err := client.Resource(volumeMigrationGVR).Namespace("simplyblock"). + Get(context.Background(), "cg-join-v1", metav1.GetOptions{}) + if err != nil { + t.Fatalf("the VolumeMigration was not created: %v", err) + } + pv, _, _ := unstructured.NestedString(got.Object, "spec", "pvName") + target, _, _ := unstructured.NestedString(got.Object, "spec", "targetNodeUUID") + if pv != "pv-1" || target != "node-pin" { + t.Fatalf("spec = %s -> %s, want pv-1 -> node-pin", pv, target) + } + if got.GetLabels()[preJoinPurposeLabel] != "consistency-group-join" { + t.Fatalf("missing the purpose label: %v", got.GetLabels()) + } + + if err := unstructured.SetNestedField(got.Object, "Completed", "status", "phase"); err != nil { + t.Fatal(err) + } + if _, err := client.Resource(volumeMigrationGVR).Namespace("simplyblock"). + Update(context.Background(), got, metav1.UpdateOptions{}); err != nil { + t.Fatal(err) + } + phase, err = m.Ensure(context.Background(), "cg-join-v1", "pv-1", "other-node") + if err != nil || phase != "Completed" { + t.Fatalf("second Ensure: phase %q err %v", phase, err) + } + again, _ := client.Resource(volumeMigrationGVR).Namespace("simplyblock"). + Get(context.Background(), "cg-join-v1", metav1.GetOptions{}) + if target, _, _ := unstructured.NestedString(again.Object, "spec", "targetNodeUUID"); target != "node-pin" { + t.Fatalf("an existing request must never be rewritten, target is now %s", target) + } +} + +func TestWatcherOptionsDefaults(t *testing.T) { + t.Setenv("SPDKCSI_CG_PREJOIN_MIGRATION", "") + t.Setenv("SPDKCSI_CG_COLOCATE", "") + t.Setenv("SPDKCSI_CG_CLIENT_SWAP_READY", "") + preJoin, colocate, swap := watcherOptions() + if !preJoin || colocate || swap { + t.Fatalf("defaults = %v %v %v, want true false false", preJoin, colocate, swap) + } + t.Setenv("SPDKCSI_CG_PREJOIN_MIGRATION", "off") + t.Setenv("SPDKCSI_CG_COLOCATE", "true") + preJoin, colocate, _ = watcherOptions() + if preJoin || !colocate { + t.Fatalf("overrides = %v %v, want false true", preJoin, colocate) + } +} diff --git a/csi-driver/internal/csi/controller/errorclass.go b/csi-driver/internal/csi/controller/errorclass.go index a5b9f0a8c..1e29731f6 100644 --- a/csi-driver/internal/csi/controller/errorclass.go +++ b/csi-driver/internal/csi/controller/errorclass.go @@ -8,6 +8,7 @@ import ( "google.golang.org/grpc/codes" + atlascp "github.com/simplyblock/atlas/controlplane" "github.com/simplyblock/csi-driver/internal/controlplane" ) @@ -100,6 +101,13 @@ func httpStatusOf(err error) int { if errors.As(err, &httpErr) { return httpErr.StatusCode } + // The Replication service talks to the control plane through the + // generated atlas-lib client instead, whose error type reports its + // status through a method rather than a public field. + var statusErr *atlascp.StatusError + if errors.As(err, &statusErr) { + return statusErr.HTTPStatus() + } return 0 } diff --git a/csi-driver/internal/csi/controller/errorclass_rpc.go b/csi-driver/internal/csi/controller/errorclass_rpc.go index 16dd58c2c..c1355b92d 100644 --- a/csi-driver/internal/csi/controller/errorclass_rpc.go +++ b/csi-driver/internal/csi/controller/errorclass_rpc.go @@ -108,6 +108,24 @@ func classifyValidateVolumeCapabilitiesError(err error) classifiedError { func classifyListSnapshotsError(err error) classifiedError { return classifiedError{ListSnapshotsErrorClassifier.Classify(err), err} } +func classifyEnableVolumeReplicationError(err error) classifiedError { + return classifiedError{EnableVolumeReplicationErrorClassifier.Classify(err), err} +} +func classifyDisableVolumeReplicationError(err error) classifiedError { + return classifiedError{DisableVolumeReplicationErrorClassifier.Classify(err), err} +} +func classifyGetVolumeReplicationInfoError(err error) classifiedError { + return classifiedError{GetVolumeReplicationInfoErrorClassifier.Classify(err), err} +} +func classifyPromoteVolumeError(err error) classifiedError { + return classifiedError{PromoteVolumeErrorClassifier.Classify(err), err} +} +func classifyDemoteVolumeError(err error) classifiedError { + return classifiedError{DemoteVolumeErrorClassifier.Classify(err), err} +} +func classifyResyncVolumeError(err error) classifiedError { + return classifiedError{ResyncVolumeErrorClassifier.Classify(err), err} +} // Dispositions reused across RPCs. var ( @@ -120,6 +138,10 @@ var ( // resolveConflict: a 409 must be resolved by looking up the existing object // (same source and params → return it as success, otherwise AlreadyExists). resolveConflict = controlPlaneErrorClass{Idempotent: true} + // cutoverInFlight: a 409 on a replication verb means a cutover is + // currently running, not a conflicting object to resolve → ABORTED, + // retryable, since Ramen re-drives every reconcile until it clears. + cutoverInFlight = controlPlaneErrorClass{Code: codes.Aborted, Retryable: true} ) // Preconfigured per-RPC classifiers. Every RPC that talks to the control plane @@ -154,4 +176,40 @@ var ( // ListSnapshotsErrorClassifier has no operation-specific statuses: every // status is handled generically. ListSnapshotsErrorClassifier = errorClassifier{} + + // EnableVolumeReplicationErrorClassifier: a different-policy attach is a + // 412 (design §10), already generic (FailedPrecondition); a 404 means the + // volume itself does not exist. + EnableVolumeReplicationErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + }} + // DisableVolumeReplicationErrorClassifier: a 409 means a cutover is in + // flight (design §10), not a conflicting object to resolve. + DisableVolumeReplicationErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + http.StatusConflict: cutoverInFlight, + }} + GetVolumeReplicationInfoErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + }} + // PromoteVolumeErrorClassifier: a 409 on a planned promote means demote is + // still converging (design §5.2) -- reuses cutoverInFlight's ABORTED, + // retryable disposition, since the vendored csi-addons controller + // auto-escalates ANY FAILED_PRECONDITION to force=true inline with no + // wait-and-retry grace period of its own. A 412 falls through to the + // generic classifier's FailedPrecondition unchanged: that is the one case + // meant to let the controller's own escalation take over. + PromoteVolumeErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + http.StatusConflict: cutoverInFlight, + }} + // DemoteVolumeErrorClassifier classifies only genuine backend failures. + // "Not yet done" (202) is not an error at atlas-lib's DemoteVolume, so it + // never reaches this classifier -- the RPC handler checks it directly. + DemoteVolumeErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + }} + ResyncVolumeErrorClassifier = errorClassifier{overrides: map[int]controlPlaneErrorClass{ + http.StatusNotFound: sourceNotFound, + }} ) diff --git a/csi-driver/internal/csi/controller/mock_controlplane_test.go b/csi-driver/internal/csi/controller/mock_controlplane_test.go index a33407973..99fe69220 100644 --- a/csi-driver/internal/csi/controller/mock_controlplane_test.go +++ b/csi-driver/internal/csi/controller/mock_controlplane_test.go @@ -27,6 +27,12 @@ type mockVolume struct { Size int64 Status string // defaults to "online" when empty GroupID string // consistency group id, "" for a non-member + + // ReplicationPolicyID is the policy this volume currently follows, "" + // when none. Set by a PUT carrying replication_policy_id (a string + // attaches, an explicit JSON null detaches; the key's absence, as an + // ordinary resize PUT sends, leaves it untouched). + ReplicationPolicyID string } // status returns the volume's reported status, defaulting to `online`. @@ -103,13 +109,75 @@ type mockSBCLI struct { // It lets a test drive an RPC through every control-plane response and assert // the resulting gRPC code. injectStatus func(r *http.Request) int + + // replicationStatus, keyed by volume id, is the raw JSON body GET + // .../replication/status serves for that volume. A test sets it directly + // rather than the mock deriving it, since the derivation itself + // (get_replication_info) is sbcli's, already covered there; this mock + // only has to prove the driver maps whatever shape the endpoint returns. + replicationStatus map[string]map[string]any + + // replicationRelationship, keyed by volume id (either side of the + // pairing), is the raw JSON body GET .../replication/ serves for that + // volume. Absent means no relationship exists yet (404, matching + // sbcli's get_relationship returning None for a volume never enabled for + // replication) -- this look-up deliberately does not require the id to + // exist in m.volumes, matching the real backend, whose relationship + // records outlive a deleted source volume. + replicationRelationship map[string]map[string]any + + // replicationPUTStatus, when set, makes every PUT carrying + // replication_policy_id respond with this HTTP status instead of the + // normal idempotent update, modeling a backend refusal (e.g., a policy + // that is not active) or a transient failure. + replicationPUTStatus int + + // failoverStatus, when set, is the HTTP status POST .../failover answers + // with instead of its default success (204) -- modeling the planned + // gate's 409 (demote still converging) and 412 (no demote requested). + failoverStatus int + // lastFailoverQuery captures the raw query string of the last failover + // call, so a test can assert the driver actually sent planned=true/false + // rather than only checking the resulting gRPC code. + lastFailoverQuery string + // lastFailoverVolumeID captures which volume's path the last failover + // call landed on, so a test can assert a relationship-resolved call + // reached the TARGET volume rather than the one it was originally given. + lastFailoverVolumeID string + + // demoteStatus, when set, is the HTTP status POST .../demote answers with + // instead of its default success (204). 202 models "still converging." + demoteStatus int + // lastDemoteVolumeID captures which volume's path the last demote call + // landed on, so a test can assert a relationship-resolved call reached + // the TARGET volume rather than the one it was originally given. + lastDemoteVolumeID string + + // failbackStatus, when set, is the HTTP status POST .../failback answers + // with instead of its default success (204). + failbackStatus int + // lastFailbackBody captures the raw JSON body of the last failback call. + lastFailbackBody []byte + // lastFailbackVolumeID captures which volume's path the last failback + // call landed on, so a test can assert the relationship resolution + // redirected it -- the same capture handleFailover keeps for promote. + lastFailbackVolumeID string + + // groupResolution, keyed by group id, is the body GET + // .../consistency-groups/{id}/replication/resolution serves. Absent means + // a control plane that predates the endpoint (a route-level 404), so the + // driver keeps its pre-resolution behaviour. + groupResolution map[string]map[string]any } func newMockSBCLI() *mockSBCLI { m := &mockSBCLI{ - volumes: make(map[string]*mockVolume), - snapshots: make(map[string]*mockSnapshot), - groups: make(map[string]*mockGroup), + volumes: make(map[string]*mockVolume), + snapshots: make(map[string]*mockSnapshot), + groups: make(map[string]*mockGroup), + replicationStatus: make(map[string]map[string]any), + replicationRelationship: make(map[string]map[string]any), + groupResolution: make(map[string]map[string]any), } mux := http.NewServeMux() @@ -128,6 +196,31 @@ func newMockSBCLI() *mockSBCLI { "PUT /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/", m.locked(m.handleResizeVolume), ) + mux.HandleFunc( + "GET /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/replication/status", + m.locked(m.handleReplicationStatus), + ) + mux.HandleFunc( + // Cluster-scoped, not pool-scoped: GetVolumeReplicationRelationship + // must stay resolvable by source id after the source volume itself is + // deleted, which the pool-scoped route (requiring the volume to still + // exist) cannot do -- see TestClientGetVolumeReplicationRelationship + // in atlas-lib/controlplane for the live-confirmed reason. + "GET /api/v2/clusters/{clusterID}/replication/relationships/{volumeID}", + m.locked(m.handleReplicationRelationship), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/replication/failover", + m.locked(m.handleFailover), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/replication/demote", + m.locked(m.handleDemote), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/replication/failback", + m.locked(m.handleFailback), + ) mux.HandleFunc( "GET /api/v2/clusters/{clusterID}/storage-pools/{poolID}/volumes/{volumeID}/connect", m.locked(m.handleVolumeConnect), @@ -148,10 +241,38 @@ func newMockSBCLI() *mockSBCLI { "DELETE /api/v2/clusters/{clusterID}/storage-pools/{poolID}/snapshots/{snapshotID}/", m.locked(m.handleDeleteSnapshot), ) + mux.HandleFunc( + "GET /api/v2/clusters/{clusterID}/consistency-groups/{$}", + m.locked(m.handleListGroups), + ) mux.HandleFunc( "GET /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/members", m.locked(m.handleGroupMembers), ) + mux.HandleFunc( + "PUT /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication", + m.locked(m.handleConfigureGroupReplication), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication/failover", + m.locked(m.handleGroupFailover), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication/demote", + m.locked(m.handleGroupDemote), + ) + mux.HandleFunc( + "POST /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication/failback", + m.locked(m.handleGroupFailback), + ) + mux.HandleFunc( + "GET /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication/status", + m.locked(m.handleGroupReplicationStatus), + ) + mux.HandleFunc( + "GET /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/replication/resolution", + m.locked(m.handleGroupResolution), + ) mux.HandleFunc( "POST /api/v2/clusters/{clusterID}/consistency-groups/{groupID}/snapshots", m.locked(m.handleTakeGroupSnapshot), @@ -218,9 +339,11 @@ func (m *mockSBCLI) lookupSnapshot(w http.ResponseWriter, snapshotID string) *mo } func (m *mockSBCLI) handleListPools(w http.ResponseWriter, _ *http.Request) { - writeJSON(w, http.StatusOK, []map[string]string{ - {"name": sanityPoolName, "id": sanityPoolUUID}, - }) + writeJSON(w, http.StatusOK, []map[string]any{{ + "name": sanityPoolName, "id": sanityPoolUUID, "cluster_id": sanityClusterID, + "max_size": 0, "capacity": map[string]any{}, "max_r_mbytes": 0, "max_rw_iops": 0, + "max_rw_mbytes": 0, "max_w_mbytes": 0, "volume_max_size": 0, "status": "active", + }}) } func (m *mockSBCLI) handleListVolumes(w http.ResponseWriter, _ *http.Request) { @@ -244,7 +367,8 @@ func (m *mockSBCLI) handleGetVolume(w http.ResponseWriter, r *http.Request) { } writeJSON(w, http.StatusOK, map[string]any{ "id": volume.UUID, "name": volume.Name, "size": volume.Size, "status": volume.status(), - "group_id": volume.GroupID, + "group_id": volume.GroupID, "pool_name": sanityPoolName, "ns_id": 1, + "nqn": "nqn.2023-02.io.simplyblock:" + sanityClusterID + ":lvol:" + volume.UUID, }) } @@ -262,12 +386,104 @@ func (m *mockSBCLI) handleResizeVolume(w http.ResponseWriter, r *http.Request) { if volume == nil { return } - var body struct { - Size int64 `json:"size"` + if m.replicationPUTStatus != 0 { + writeJSON(w, m.replicationPUTStatus, map[string]string{"detail": "injected status"}) + return } - _ = json.NewDecoder(r.Body).Decode(&body) - if body.Size > 0 { - volume.Size = body.Size + raw, _ := io.ReadAll(r.Body) + var fields map[string]json.RawMessage + _ = json.Unmarshal(raw, &fields) + + if sizeRaw, ok := fields["size"]; ok { + var size int64 + if err := json.Unmarshal(sizeRaw, &size); err == nil && size > 0 { + volume.Size = size + } + } + // A key present with a JSON null attaches nothing (detach); a key present + // with a string attaches that policy; the key's absence (an ordinary + // resize PUT) leaves the volume's policy untouched -- omitted and null + // are different requests, which is exactly the distinction this mock + // exists to exercise. + if policyRaw, ok := fields["replication_policy_id"]; ok { + if string(policyRaw) == "null" { + volume.ReplicationPolicyID = "" + } else { + var policyID string + _ = json.Unmarshal(policyRaw, &policyID) + volume.ReplicationPolicyID = policyID + } + } + w.WriteHeader(http.StatusNoContent) +} + +// handleReplicationStatus serves the typed steady-state status a test +// configured via replicationStatus, or a default "not_replicating" body for +// a volume nothing has configured -- the same "never a 404" contract P0-1 +// promises for a volume that exists but never replicated. +func (m *mockSBCLI) handleReplicationStatus(w http.ResponseWriter, r *http.Request) { + volumeID := r.PathValue("volumeID") + if m.lookupVolume(w, volumeID) == nil { + return + } + body, ok := m.replicationStatus[volumeID] + if !ok { + body = map[string]any{ + "role": "none", "state": "not_replicating", + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": false, + } + } + writeJSON(w, http.StatusOK, body) +} + +func (m *mockSBCLI) handleReplicationRelationship(w http.ResponseWriter, r *http.Request) { + volumeID := r.PathValue("volumeID") + body, ok := m.replicationRelationship[volumeID] + if !ok { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "Volume has no replication relationship"}) + return + } + writeJSON(w, http.StatusOK, body) +} + +func (m *mockSBCLI) handleFailover(w http.ResponseWriter, r *http.Request) { + volumeID := r.PathValue("volumeID") + if m.lookupVolume(w, volumeID) == nil { + return + } + m.lastFailoverQuery = r.URL.RawQuery + m.lastFailoverVolumeID = volumeID + if m.failoverStatus != 0 { + writeJSON(w, m.failoverStatus, map[string]string{"detail": "injected status"}) + return + } + w.WriteHeader(http.StatusNoContent) +} + +func (m *mockSBCLI) handleDemote(w http.ResponseWriter, r *http.Request) { + volumeID := r.PathValue("volumeID") + if m.lookupVolume(w, volumeID) == nil { + return + } + m.lastDemoteVolumeID = volumeID + if m.demoteStatus != 0 { + writeJSON(w, m.demoteStatus, map[string]bool{"demoted": m.demoteStatus == http.StatusNoContent}) + return + } + w.WriteHeader(http.StatusNoContent) +} + +func (m *mockSBCLI) handleFailback(w http.ResponseWriter, r *http.Request) { + volumeID := r.PathValue("volumeID") + if m.lookupVolume(w, volumeID) == nil { + return + } + m.lastFailbackBody, _ = io.ReadAll(r.Body) + m.lastFailbackVolumeID = volumeID + if m.failbackStatus != 0 { + writeJSON(w, m.failbackStatus, map[string]string{"detail": "injected status"}) + return } w.WriteHeader(http.StatusNoContent) } @@ -461,6 +677,13 @@ type mockGroup struct { Members []string // lvol UUIDs with an open epoch LastSeq int Gens map[int][]mockGenMember + // Group-replication state the driver's group-handle routing drives. + PolicyID string + Promoted bool + Demoted bool + DemoteConverging bool // when set, /demote answers 202 (still converging) + FailbackSource string + LastReplicatedAt int64 // unix seconds surfaced by /replication/status } // seedGroup registers a group with the given member lvol UUIDs and stamps each @@ -476,6 +699,102 @@ func (m *mockSBCLI) seedGroup(groupID string, memberUUIDs ...string) { } } +func (m *mockSBCLI) handleListGroups(w http.ResponseWriter, r *http.Request) { + name := r.URL.Query().Get("name") + rows := make([]map[string]any, 0, len(m.groups)) + for id, g := range m.groups { + gname := "cg-" + id + if name != "" && name != gname { + continue + } + rows = append(rows, map[string]any{ + "id": id, "cluster_id": r.PathValue("clusterID"), "name": gname, + "member_count": len(g.Members), "last_group_seq": g.LastSeq, + }) + } + writeJSON(w, http.StatusOK, rows) +} + +func (m *mockSBCLI) handleConfigureGroupReplication(w http.ResponseWriter, r *http.Request) { + g := m.groups[r.PathValue("groupID")] + if g == nil { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "group not found"}) + return + } + var body struct { + ReplicationPolicyID *string `json:"replication_policy_id"` + } + _ = json.NewDecoder(r.Body).Decode(&body) + if body.ReplicationPolicyID == nil { + g.PolicyID = "" + } else { + g.PolicyID = *body.ReplicationPolicyID + } + w.WriteHeader(http.StatusNoContent) +} + +func (m *mockSBCLI) handleGroupFailover(w http.ResponseWriter, r *http.Request) { + g := m.groups[r.PathValue("groupID")] + if g == nil { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "group not found"}) + return + } + g.Promoted = true + writeJSON(w, http.StatusOK, map[string]any{"members": []any{}}) +} + +func (m *mockSBCLI) handleGroupDemote(w http.ResponseWriter, r *http.Request) { + g := m.groups[r.PathValue("groupID")] + if g == nil { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "group not found"}) + return + } + if g.DemoteConverging { + writeJSON(w, http.StatusAccepted, map[string]any{"demoted": false}) + return + } + g.Demoted = true + w.WriteHeader(http.StatusNoContent) +} + +func (m *mockSBCLI) handleGroupFailback(w http.ResponseWriter, r *http.Request) { + g := m.groups[r.PathValue("groupID")] + if g == nil { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "group not found"}) + return + } + var body struct { + SourceClusterID *string `json:"source_cluster_id"` + } + _ = json.NewDecoder(r.Body).Decode(&body) + if body.SourceClusterID != nil { + g.FailbackSource = *body.SourceClusterID + } + w.WriteHeader(http.StatusNoContent) +} + +func (m *mockSBCLI) handleGroupReplicationStatus(w http.ResponseWriter, r *http.Request) { + g := m.groups[r.PathValue("groupID")] + if g == nil { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "group not found"}) + return + } + out := map[string]any{"role": "source", "state": "in_sync", "member_count": len(g.Members)} + if g.LastReplicatedAt != 0 { + out["last_replicated_at"] = time.Unix(g.LastReplicatedAt, 0).UTC().Format(time.RFC3339) + } + writeJSON(w, http.StatusOK, out) +} + +func (m *mockSBCLI) handleGroupResolution(w http.ResponseWriter, r *http.Request) { + body, ok := m.groupResolution[r.PathValue("groupID")] + if !ok { + writeJSON(w, http.StatusNotFound, map[string]string{"detail": "Not Found"}) + return + } + writeJSON(w, http.StatusOK, body) +} + func (m *mockSBCLI) handleGroupMembers(w http.ResponseWriter, r *http.Request) { g := m.groups[r.PathValue("groupID")] if g == nil { diff --git a/csi-driver/internal/csi/controller/replication.go b/csi-driver/internal/csi/controller/replication.go new file mode 100644 index 000000000..5bc76275f --- /dev/null +++ b/csi-driver/internal/csi/controller/replication.go @@ -0,0 +1,611 @@ +// The csi-addons Replication service: EnableVolumeReplication, +// DisableVolumeReplication, GetVolumeReplicationInfo (design §5.1), and the +// Phase 2 lifecycle verbs PromoteVolume, DemoteVolume, and ResyncVolume +// (design §5.2). Each verb is a thin adapter onto the atlas-lib control-plane +// client's replication calls, resolved through the same +// {clusterID}:{poolID}:{lvolID} handle every other RPC uses. +package controller + +import ( + "context" + "errors" + "fmt" + "sort" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + "google.golang.org/protobuf/types/known/timestamppb" + + atlascp "github.com/simplyblock/atlas/controlplane" + "github.com/simplyblock/atlas/errs" + "github.com/simplyblock/atlas/lvol" + "github.com/simplyblock/csi-driver/internal/clusters" + csicommon "github.com/simplyblock/csi-driver/internal/csi/common" +) + +// replicationPolicyParam is the VolumeReplicationClass parameter key naming +// the backend policy to attach. It is spelled as an id rather than the +// design's own `replicationPolicy` (a name, §7.1), because resolving a name +// to an id needs a policy-list-and-match call this phase does not yet wrap +// in atlas-lib. A future change adds that resolution and accepts either. +const replicationPolicyParam = "replicationPolicyID" + +// sourceClusterIDParam is the VolumeReplicationClass parameter naming the +// cluster to resync from, when it isn't the one the backend already has on +// record for this volume's relationship. Optional: sbcli's replication_failback +// resolves it from the existing relationship when omitted (the common case, +// design §5.2's "Recovered source"). +const sourceClusterIDParam = "sourceClusterID" + +// volumeIDCarrier is every Replication request type: each exposes the legacy +// flat VolumeId field and the ReplicationSource oneof. +type volumeIDCarrier interface { + GetVolumeId() string + GetReplicationSource() *replication.ReplicationSource +} + +// volumeIDFrom resolves a request's volume ID. The real upstream sidecar +// (kubernetes-csi-addons v0.15.0, internal/sidecar/service.ReplicationServer) +// proxies every Replication RPC through ReplicationSource and never sets the +// legacy flat VolumeId, so that field is checked first; the flat field is +// kept as a fallback for any caller that still sends it. +// +// A VolumeGroupReplication drives the SAME Replication verbs, but the sidecar +// sets the group source oneof (ReplicationSource.volumegroup.volume_group_id) +// rather than the per-volume one, carrying the "cg:{cluster}:{group}" handle. +// Reading only the volume oneof left the handle empty and every group verb +// failed with `invalid volume handle ""` (VGR promote, confirmed live +// 2026-09-26). The group handle is returned as-is so ParseGroupHandle routes it +// to the group-replication path. +func volumeIDFrom(req volumeIDCarrier) string { + if v := req.GetReplicationSource().GetVolume().GetVolumeId(); v != "" { + return v + } + if g := req.GetReplicationSource().GetVolumegroup().GetVolumeGroupId(); g != "" { + return g + } + return req.GetVolumeId() +} + +// resolveToLocalReplica rewrites h to the volume that actually replicates +// data on this side of an existing pairing, when h names the OTHER (foreign) +// side of it instead. Ramen's S3-restore recreates a destination PV carrying +// the ORIGINAL source's own volumeHandle verbatim (confirmed live +// 2026-09-23, relocate M-02), and every Replication RPC parses its target +// straight from the handle it's given -- without this resolution, "promote" +// or "enable replication" would operate on the foreign, original volume +// instead of the local replica that has actually been receiving replicated +// data. The returned handle and client change together, since the target +// side may live on a different cluster with its own secret.json entry. +// +// Returns h and client unchanged when h has no replication relationship yet +// (errs.ErrNotFound -- the ordinary case for a volume never enabled for +// replication, e.g., M-01's first-ever protect) or when h already names the +// target side. +// +// The resolution WALKS chained pairings rather than taking one step: a +// relocate round trip leaves original -> hop-1 clone -> hop-2 clone, and the +// volume actually serving the workload is the LAST hop (each record names it +// as active_lvol_id, resolved transitively by the backend). Stopping at the +// first pairing's target landed every post-round-trip verb -- including the +// policy attach Enable performs -- on the retired middle clone, on the wrong +// cluster (confirmed live 2026-09-24). Each hop uses that record's own +// target triple, whose cluster/pool/lvol are consistent with each other; +// combining active_lvol_id with ANOTHER record's cluster is exactly the bug +// this walk exists to avoid. An empty ActiveLvolID (backend predating the +// field) stops after the first hop, the old single-step behavior. +// +// The chain behind a handle alternates between the sites: every fail-over +// adds a hop to the other side. The volume this driver must act on is the +// chain's last member on a LOCAL cluster (the secret marks the site's own +// clusters, clusters.Local), not the chain's end: after an unplanned +// fail-over A->B, Ramen makes the old primary on A secondary, and the +// chain's end is the NEW primary on B. Resolving to the end demoted -- and +// on VR deletion detached -- the live production volume on the other site +// (2026-10-02, WordPress: the demote fenced the live primary's paths and +// took demote snapshots of it; the fail-back never got PeerReady). A secret +// that marks no cluster local (an operator predating the flag) keeps the +// previous behaviour, the chain's active end. +func resolveToLocalReplica( + ctx context.Context, h *lvol.Handle, client *atlascp.Client, +) (*lvol.Handle, *atlascp.Client, error) { + h, client, _, err := resolveReplica(ctx, h, client) + return h, client, err +} + +// resolveReplica is resolveToLocalReplica reporting also whether h has a +// replication relationship at all (known): the member it resolves to is then +// one of a chain the backend records, and a volume of that chain that no +// longer exists is a superseded, reaped old primary -- nothing left to demote +// or detach -- rather than an unknown handle. +func resolveReplica( + ctx context.Context, h *lvol.Handle, client *atlascp.Client, +) (*lvol.Handle, *atlascp.Client, bool, error) { + hops, known, err := resolveChain(ctx, h, client) + if err != nil { + return nil, nil, false, err + } + local, flagged, err := clusters.Local() + if err != nil { + return nil, nil, false, err + } + pick := chooseReplica(hops, local, flagged) + return pick.h, pick.client, known, nil +} + +// resolveChain walks the replication chain behind h: h itself, then every +// IsSource->target hop to the active end. known is whether h has a +// relationship at all. +func resolveChain(ctx context.Context, h *lvol.Handle, client *atlascp.Client) ([]chainHop, bool, error) { + known := false + hops := []chainHop{{h: h, client: client}} + // One hop per past fail-over, never compacted (the PV keeps the original + // handle): a cap of 8 ended the walk one hop short of a ninth move's + // clone (2026-10-03). The bound is a cycle guard, not a length estimate. + visited := map[lvol.VolumeHandle]bool{} + for range maxChainHops { + if visited[h.Handle()] { + return nil, false, fmt.Errorf("replication chain of %s loops at %s", hops[0].h.Handle(), h.Handle()) + } + visited[h.Handle()] = true + rel, err := client.GetVolumeReplicationRelationship(ctx, h.Handle()) + if err != nil { + if errors.Is(err, errs.ErrNotFound) { + break + } + return nil, false, err + } + known = true + if !rel.IsSource { + break + } + target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} + targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) + if err != nil { + return nil, false, err + } + h, client = target, targetClient + hops = append(hops, chainHop{h: h, client: client}) + if rel.ActiveLvolID == "" || rel.ActiveLvolID == rel.TargetLvolID { + break + } + } + if len(hops) > maxChainHops { + return nil, false, fmt.Errorf("replication chain of %s did not converge within %d hops", + hops[0].h.Handle(), maxChainHops) + } + return hops, known, nil +} + +// maxChainHops bounds a replication-chain walk: a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 + +// activeEndFallback is where a Resync or a status read goes when the local +// member of the chain is reaped: the chain's active end -- the live primary +// on the other site -- and, as the cluster to fail back to, the local one. +// sbcli's replication_failback is addressed to the failed-over clone and +// re-aims its replication at the original site's node (the recovered-source +// case: only the delta ships), which is exactly the fail-back of a site that +// lost its primary (live 2026-10-02, site A after the unplanned fail-over of +// WordPress: the old primary 80e3e748 was reaped, the clone e3d439ca on B +// holds the data). +func activeEndFallback(hops []chainHop, sourceClusterID string) (chainHop, string) { + end := hops[len(hops)-1] + if local, flagged, err := clusters.Local(); err == nil && flagged { + ids := make([]string, 0, len(local)) + for id := range local { + ids = append(ids, id) + } + sort.Strings(ids) + for _, hop := range hops { + if local[hop.h.ClusterID] { + return end, hop.h.ClusterID + } + } + if len(ids) > 0 { + return end, ids[0] + } + } + return end, sourceClusterID +} + +// reapedChainMember is whether a Replication verb on a resolved chain member +// found the volume gone (404): a superseded old primary the control plane +// has reaped after its fail-over completed (deferred removal; live +// 2026-10-02 on site A). Demoting or detaching it is a no-op that succeeds; +// a 404 on a handle with no relationship stays NotFound. +func reapedChainMember(known bool, ce classifiedError) bool { + return known && status.Code(ce) == codes.NotFound +} + +// chainHop is one member of a replication chain, with the client of its +// cluster. +type chainHop struct { + h *lvol.Handle + client *atlascp.Client +} + +// chooseReplica picks the chain member a Replication RPC acts on: the last +// member on a local cluster when the secret marks local clusters (and the +// chain's end when none of the members is local, e.g. a volume that only +// ever lived elsewhere), else the chain's end. +func chooseReplica(hops []chainHop, local map[string]bool, flagged bool) chainHop { + end := hops[len(hops)-1] + if !flagged { + return end + } + for i := len(hops) - 1; i >= 0; i-- { + if local[hops[i].h.ClusterID] { + return hops[i] + } + } + return end +} + +// EnableVolumeReplication attaches the volume to the policy named by the +// VolumeReplicationClass. Attaching to the policy the volume already follows +// is success (the backend's own idempotency, P0-2). +// +// The design (§5.1, §10) wants a different-policy attach refused with +// FAILED_PRECONDITION, because a silent re-attach forces a full re-sync. That +// refusal is NOT implemented here: it needs the volume's CURRENTLY attached +// policy id to compare against, and neither the P0-1 status read nor any +// other backend endpoint exposes it today (attach_policy in sbcli's +// replication_policy_controller.py detaches and re-attaches on a policy +// change without refusing). Until the backend adds that field, a +// different-policy attach silently re-syncs, exactly as it does through +// every other existing caller of this same endpoint. +func (cs *Server) EnableVolumeReplication( + ctx context.Context, + req *replication.EnableVolumeReplicationRequest, +) (*replication.EnableVolumeReplicationResponse, error) { + policyID := req.GetParameters()[replicationPolicyParam] + // An empty replicationPolicyID is the FAIL-OVER TARGET: the side becoming + // primary carries no reverse-direction policy yet, because the reverse + // direction is a fail-back-time concern (design-ramen-integration.md §6.2) -- + // nothing on that side replicates until it becomes primary. csi-addons always + // calls Enable before Promote for whichever side is becoming Primary, so this + // Enable is a legitimate no-op there: there is nothing to attach, and + // Promote/PromoteGroup is what clones the replicated snapshot (and, for a + // group, reconstitutes it) and does the real work. Rejecting it as a hard + // "required" error blocked every fail-over whose target class had no policy, + // both per-volume and group. Mirrors the per-volume ErrNotFound tolerance + // below (Enable handed a handle that names nothing yet). + if policyID == "" { + return &replication.EnableVolumeReplicationResponse{}, nil + } + // A group handle drives the whole consistency group as one unit through the + // group-replication endpoints (design §14.4); a per-volume handle takes the + // §5 path below unchanged. + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + // The live group: after a relocate the named group is empty and the + // re-protection attaches the group serving the data. + gh, client, err := resolveGroupTarget(ctx, gh, groupActiveEnd) + if err != nil { + return nil, err + } + if err := client.EnableGroupReplication(ctx, gh, policyID); err != nil { + return nil, classifyEnableVolumeReplicationError(err) + } + return &replication.EnableVolumeReplicationResponse{}, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + h, client, err = resolveToLocalReplica(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + if err := client.EnableVolumeReplication(ctx, h.Handle(), policyID); err != nil { + if errors.Is(err, errs.ErrNotFound) { + // This backend's replication is one-way: the destination never + // carries a persistent, independently provisioned LVol of its + // own -- the writable clone only comes into existence when + // PromoteVolume clones the last replicated snapshot. csi-addons + // always calls Enable before Promote, unconditionally, for + // whichever side is becoming Primary, so on a first-ever + // relocate (nothing for resolveToLocalReplica to redirect + // through either, since no relationship exists until a promote + // has actually happened) Enable is legitimately handed a handle + // that names nothing yet. There is nothing to attach a policy + // to, and nothing wrong either -- PromoteVolume is what actually + // creates and validates the volume, and is what surfaces a real + // error if there truly is nothing to clone from. + return &replication.EnableVolumeReplicationResponse{}, nil + } + return nil, classifyEnableVolumeReplicationError(err) + } + return &replication.EnableVolumeReplicationResponse{}, nil +} + +// DisableVolumeReplication detaches the volume from whatever policy it +// follows. Detaching an already-detached volume is success; a cutover in +// flight (409) is a retryable ABORTED, since Ramen re-drives every reconcile. +func (cs *Server) DisableVolumeReplication( + ctx context.Context, + req *replication.DisableVolumeReplicationRequest, +) (*replication.DisableVolumeReplicationResponse, error) { + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + // The local group: this site detaches what it holds, never the live + // group on the other site. + gh, client, err := resolveGroupTarget(ctx, gh, groupLocalSite) + if err != nil { + return nil, err + } + if err := client.DisableGroupReplication(ctx, gh); err != nil { + return nil, classifyDisableVolumeReplicationError(err) + } + return &replication.DisableVolumeReplicationResponse{}, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + h, client, known, err := resolveReplica(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + if err := client.DisableVolumeReplication(ctx, h.Handle()); err != nil { + if ce := classifyDisableVolumeReplicationError(err); !reapedChainMember(known, ce) { + return nil, ce + } + } + return &replication.DisableVolumeReplicationResponse{}, nil +} + +// GetVolumeReplicationInfo returns the volume's replicated-life status, +// resolved to the local replica first for the same reason as every other +// verb: Ramen polls this with the S3-restored, foreign volumeHandle, whose +// own lvol record may already be reaped (see ResyncVolume). +// +// The spec's response carries only LastSyncTime in this version +// (github.com/csi-addons/spec v0.2.0); lastSyncBytes/lastSyncDuration are not +// yet part of the wire contract this driver builds against, so they cannot +// be set even though the backend status read already computes them (design +// §6.1). Ramen's condition derivation (§6.2) does not depend on them. +func (cs *Server) GetVolumeReplicationInfo( + ctx context.Context, + req *replication.GetVolumeReplicationInfoRequest, +) (*replication.GetVolumeReplicationInfoResponse, error) { + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + gh, client, err := resolveGroupTarget(ctx, gh, groupActiveEnd) + if err != nil { + return nil, err + } + info, err := client.GetGroupReplicationInfo(ctx, gh) + if err != nil { + return nil, classifyGetVolumeReplicationInfoError(err) + } + resp := &replication.GetVolumeReplicationInfoResponse{} + if info.LastReplicatedAt != nil { + resp.LastSyncTime = timestamppb.New(*info.LastReplicatedAt) + } + return resp, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + // The pairing's status is the ACTIVE END's: the volume that holds the + // data and replicates. On the primary site that is the local volume; on + // the secondary site the local member is the demoted or reaped old + // primary, whose status says nothing about the pipe back to this site. + hops, _, err := resolveChain(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + end := hops[len(hops)-1] + h, client = end.h, end.client + info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) + if err != nil { + return nil, classifyGetVolumeReplicationInfoError(err) + } + resp := &replication.GetVolumeReplicationInfoResponse{} + if info.LastReplicatedAt != nil { + resp.LastSyncTime = timestamppb.New(*info.LastReplicatedAt) + } + return resp, nil +} + +// PromoteVolume brings the volume up as primary on this cluster (design +// §5.2). Force=true is the unplanned path: it clones the last fully +// replicated generation and ignores demote state entirely, because its whole +// premise is that the peer may never have been reachable to demote. +// Force=false is the planned path, refused with ABORTED (retryable) while a +// demote is still converging and FAILED_PRECONDITION when no demote was ever +// requested -- the split matters because the vendored csi-addons controller +// auto-escalates ANY FAILED_PRECONDITION from a force=false promote to +// force=true inline, with no wait-and-retry grace period of its own. +func (cs *Server) PromoteVolume( + ctx context.Context, + req *replication.PromoteVolumeRequest, +) (*replication.PromoteVolumeResponse, error) { + // A group handle promotes the whole consistency group atomically (design + // §14.4): every member is cloned from the same group generation. The + // planned/forced split is the backend group failover's own concern, so + // force is not forwarded here. + // + // The named group, unresolved: the control plane's group fail-over resolves + // the peer itself and tells apart "already promoted there" (a no-op), a + // fail-back (clone the peer's members home) and a fail-over from the newest + // replicated generation (sbcli failover_group). Promoting the live group + // instead would turn a fail-back into a no-op on the group being left. + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + client, err := clusters.ReplicationClient(ctx, gh.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + if err := client.PromoteGroup(ctx, gh); err != nil { + return nil, classifyPromoteVolumeError(err) + } + return &replication.PromoteVolumeResponse{}, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + // Promote is addressed to the chain's ACTIVE END, never to the local + // member: the control plane's failover endpoint takes the volume that + // currently holds the data (the source of the pairing) and creates the + // clone on the replication target -- this site. The local member here + // is the volume being replaced: a demoted old primary, or one the + // control plane already reaped (2026-10-02: promote on site A hit the + // reaped 80e3e748 and 404ed while the live primary e3d439ca on B held + // the data). + hops, _, err := resolveChain(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + end := hops[len(hops)-1] + h, client = end.h, end.client + if err := client.PromoteVolume(ctx, h.Handle(), req.GetForce()); err != nil { + return nil, classifyPromoteVolumeError(err) + } + // Deliberately NOT attaching the class's policy to the promoted volume + // here, even though the new primary comes up unprotected (policy=NONE) + // until the next VR creation re-runs Enable. Attaching at promote time + // was tried and confirmed harmful live (2026-09-24, relocate M-02): the + // attach starts policy-driven replication of the fresh clone immediately, + // and when the relocate's fail-back then re-points the SAME volume's + // replication at the original source node, the two chains collide -- the + // fail-over clone build picked a snapshot from the policy's chain and + // died on "Failed to create BDev" on the wrong LVS, wedging the whole + // relocate. Re-protecting the new primary belongs to the planned-cutover + // (replication_commit) flow, where it can be sequenced strictly after the + // cutover completes instead of racing the fail-back. + return &replication.PromoteVolumeResponse{}, nil +} + +// DemoteVolume fences the source and confirms the last write replicated +// (P0-3) -- the lossless half of a planned swap. Synchronous and +// non-blocking: it never waits out the backend's own convergence loop. +// While still converging it returns ABORTED (retryable), matching the actual +// upstream reconciler's requeue-until-ready behavior for a Secondary +// transition that has not yet settled, rather than holding the RPC open. +func (cs *Server) DemoteVolume( + ctx context.Context, + req *replication.DemoteVolumeRequest, +) (*replication.DemoteVolumeResponse, error) { + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + // The local group: a relocate back demotes the group this site serves, + // which after the first move is the peer of the group the VGR names. + gh, client, err := resolveGroupTarget(ctx, gh, groupLocalSite) + if err != nil { + return nil, err + } + done, err := client.DemoteGroup(ctx, gh) + if err != nil { + return nil, classifyDemoteVolumeError(err) + } + if !done { + return nil, status.Error(codes.Aborted, "group demote is still converging") + } + return &replication.DemoteVolumeResponse{}, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + h, client, known, err := resolveReplica(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + done, err := client.DemoteVolume(ctx, h.Handle()) + if err != nil { + ce := classifyDemoteVolumeError(err) + if !reapedChainMember(known, ce) { + return nil, ce + } + done = true + } + if !done { + return nil, status.Error(codes.Aborted, "demote is still converging") + } + return &replication.DemoteVolumeResponse{}, nil +} + +// ResyncVolume reconciles a diverged copy back onto the current primary's +// history. It configures the reverse direction only and reports readiness +// off the ordinary lag read -- it never cuts over, matching the design's own +// "it never merges" (§5.2): cutover is PromoteVolume's job, on a separate, +// later call. +// +// The resolveToLocalReplica step is what keeps a relocate's round trip alive: +// the Secondary side's VR carries the ORIGINAL source's volumeHandle +// (S3-restored verbatim), and by the second hop that source lvol record has +// been reaped by lvol_monitor's post-failover hold -- confirmed live +// 2026-09-24, when resync (then the only verb without the resolution) 404ed +// against the dead handle on every reconcile and stalled the relocate back. +func (cs *Server) ResyncVolume( + ctx context.Context, + req *replication.ResyncVolumeRequest, +) (*replication.ResyncVolumeResponse, error) { + if gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(volumeIDFrom(req))); ok { + gh, client, err := resolveGroupTarget(ctx, gh, groupActiveEnd) + if err != nil { + return nil, err + } + if err := client.ResyncGroup(ctx, gh, req.GetParameters()[sourceClusterIDParam]); err != nil { + return nil, classifyResyncVolumeError(err) + } + info, err := client.GetGroupReplicationInfo(ctx, gh) + if err != nil { + return nil, classifyGetVolumeReplicationInfoError(err) + } + ready := info.LagSeconds == nil || info.LagBudgetSeconds == nil || *info.LagSeconds <= *info.LagBudgetSeconds + return &replication.ResyncVolumeResponse{Ready: ready}, nil + } + h, err := csicommon.ParseVolumeHandle(volumeIDFrom(req)) + if err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + // Resync is addressed to the chain's ACTIVE END with this site as the + // cluster to fail back to: sbcli's replication_failback takes the volume + // that holds the data (the failed-over clone) and re-aims its replication + // at the recovered site's node, shipping only the delta. The local member + // is the demoted or reaped old primary; re-aiming IT configured nothing + // for the live clone (2026-10-02, Gitea after the unplanned fail-over: + // the clones on A had no replication, lastGroupSyncTime stayed empty). + hops, _, err := resolveChain(ctx, h, client) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + end, sourceClusterID := activeEndFallback(hops, req.GetParameters()[sourceClusterIDParam]) + h, client = end.h, end.client + if err := client.ResyncVolume(ctx, h.Handle(), sourceClusterID); err != nil { + return nil, classifyResyncVolumeError(err) + } + info, err := client.GetVolumeReplicationInfo(ctx, h.Handle()) + if err != nil { + return nil, classifyGetVolumeReplicationInfoError(err) + } + ready := info.LagSeconds == nil || info.LagBudgetSeconds == nil || *info.LagSeconds <= *info.LagBudgetSeconds + return &replication.ResyncVolumeResponse{Ready: ready}, nil +} diff --git a/csi-driver/internal/csi/controller/replication_destination.go b/csi-driver/internal/csi/controller/replication_destination.go new file mode 100644 index 000000000..190297954 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_destination.go @@ -0,0 +1,134 @@ +package controller + +import ( + "context" + "fmt" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + + atlascp "github.com/simplyblock/atlas/controlplane" + "github.com/simplyblock/atlas/lvol" + "github.com/simplyblock/csi-driver/internal/clusters" + csicommon "github.com/simplyblock/csi-driver/internal/csi/common" +) + +// GetReplicationDestinationInfo answers csi-addons' destination-info call +// (capability GET_REPLICATION_DESTINATION_INFO, csi-addons/spec replication.proto, +// kubernetes-csi-addons >= v0.15.0). +// +// The csi-addons v0.15 controller lists every member of a +// VolumeGroupReplicationContent in status.persistentVolumeMappingList and fills +// each destinationVolumeHandle (and status.destinationVolumeGroupID) only from +// this call. Without it the destinations stay empty, and Ramen's restore on the +// target refuses the whole group: "destination volume ID is empty for VGRC …" +// (vrg_volgrouprep.go updateVGRCVolumeHandlesForRestore). That stopped the +// first consistency-group relocate on the DR test bed with the application +// demoted on the source and not started on the target (2026-10-03, WordPress +// site-a -> site-b). A single volume did not need it: Ramen skips a +// VolumeReplication without the destination-info condition. +// +// The destination of a simplyblock volume or group is addressed by its own +// handle. One control plane manages both clusters of a replication pair; every +// Replication verb resolves a handle through the replication relationship to +// the member it must act on (resolveChain, the group endpoints), and the PV +// keeps the original handle across every move -- the behaviour all relocate and +// fail-over cases were proven with. Answering with the replica's raw volume +// (the landing copy on the peer) would make Ramen rewrite the restored PV to a +// volume that a promote replaces with a clone, bypassing that resolution. So +// the answer is the source handle itself. +// +// A single volume's answer needs no lookup and never fails for a valid handle: +// once the capability is advertised csi-addons asks for every +// VolumeReplication too, and a failure sets DestinationInfoAvailable=False, +// which makes Ramen refuse the VRG instead of skipping it (vrg_volrep.go +// destinationInfoAvailableOrSkip) -- an error here would block protecting a +// volume that has not replicated yet. A group needs its current members for the +// complete map; a group the control plane cannot list is UNAVAILABLE +// (retryable). +func (cs *Server) GetReplicationDestinationInfo( + ctx context.Context, + req *replication.GetReplicationDestinationInfoRequest, +) (*replication.GetReplicationDestinationInfoResponse, error) { + if g := req.GetReplicationSource().GetVolumegroup().GetVolumeGroupId(); g != "" { + return groupDestinationInfo(ctx, g) + } + volumeID := req.GetReplicationSource().GetVolume().GetVolumeId() + if volumeID == "" { + return nil, status.Error(codes.InvalidArgument, "replication source names no volume or volume group") + } + if _, err := csicommon.ParseVolumeHandle(volumeID); err != nil { + return nil, status.Error(codes.InvalidArgument, err.Error()) + } + return &replication.GetReplicationDestinationInfoResponse{ + ReplicationDestination: &replication.ReplicationDestination{ + Type: &replication.ReplicationDestination_Volume{ + Volume: &replication.ReplicationDestination_VolumeDestination{VolumeId: volumeID}, + }, + }, + }, nil +} + +// groupPVHandles is every handle a PersistentVolume of the group carries. It +// is resolved by the control plane (ResolveGroup): after a relocate the group a +// VGR names is empty -- its demoted members were deleted so a relocate back +// stays possible -- while its PVs keep their original handles, and listing the +// empty group left Ramen's VRG waiting for destination info for ever +// (2026-10-04, WordPress A -> B). The handles are the lineages' origins, the +// keys csi-addons matches; each maps to itself, the contract above. A control +// plane without the resolution endpoint is asked for the current members as +// before. +func groupPVHandles(ctx context.Context, client *atlascp.Client, gh lvol.GroupHandle) ([]lvol.VolumeHandle, error) { + res, err := client.ResolveGroup(ctx, gh) + if err != nil { + return nil, err + } + if res.Legacy { + return client.ConsistencyGroupMemberHandles(ctx, gh) + } + if len(res.Members) == 0 { + return nil, fmt.Errorf("consistency group %s has no live member", gh.Handle()) + } + handles := make([]lvol.VolumeHandle, 0, len(res.Members)) + for _, m := range res.Members { + handles = append(handles, m.Origin) + } + return handles, nil +} + +// groupDestinationInfo is the group branch: the group's own handle, and a +// complete source -> destination map over the group's current members (the spec +// forbids a partial map). The keys are the members' volume handles exactly as +// their PersistentVolumes carry them, which is what csi-addons matches them +// against. +func groupDestinationInfo( + ctx context.Context, groupID string, +) (*replication.GetReplicationDestinationInfoResponse, error) { + gh, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(groupID)) + if !ok { + return nil, status.Errorf(codes.InvalidArgument, "invalid volume group handle %q", groupID) + } + client, err := clusters.ReplicationClient(ctx, gh.ClusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + members, err := groupPVHandles(ctx, client, gh) + if err != nil { + return nil, status.Errorf(codes.Unavailable, "members of %s: %v", groupID, err) + } + ids := make(map[string]string, len(members)) + for _, m := range members { + ids[string(m)] = string(m) + } + return &replication.GetReplicationDestinationInfoResponse{ + ReplicationDestination: &replication.ReplicationDestination{ + Type: &replication.ReplicationDestination_Volumegroup{ + Volumegroup: &replication.ReplicationDestination_VolumeGroupDestination{ + VolumeGroupId: groupID, + VolumeIds: ids, + }, + }, + }, + }, nil +} diff --git a/csi-driver/internal/csi/controller/replication_destination_test.go b/csi-driver/internal/csi/controller/replication_destination_test.go new file mode 100644 index 000000000..e8db93fb5 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_destination_test.go @@ -0,0 +1,109 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +func volumeSource(id string) *replication.ReplicationSource { + return &replication.ReplicationSource{ + Type: &replication.ReplicationSource_Volume{ + Volume: &replication.ReplicationSource_VolumeSource{VolumeId: id}, + }, + } +} + +// A single volume's destination is its own handle: the PV keeps it across +// moves and every Replication verb resolves it to the replica. The answer needs +// no relationship yet -- an error would make Ramen refuse the VRG. +func TestReplicationDestinationOfAVolumeIsItsOwnHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + resp, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: volumeSource(testReplVolID)}) + if err != nil { + t.Fatalf("GetReplicationDestinationInfo: %v", err) + } + if got := resp.GetReplicationDestination().GetVolume().GetVolumeId(); got != testReplVolID { + t.Fatalf("destination volume = %q, want the source handle %q", got, testReplVolID) + } + if resp.GetReplicationDestination().GetVolumegroup() != nil { + t.Fatal("a volume source answered with a group destination") + } +} + +// A group's answer is its own handle and a COMPLETE map over its current +// members, keyed by the members' volume handles exactly as their PVs carry +// them: csi-addons matches persistentVolumeMappingList[].volumeHandle against +// the keys and leaves the destination empty for any miss, which is what made +// Ramen refuse the restore (2026-10-03). +func TestReplicationDestinationOfAGroupMapsEveryMember(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + resp, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: groupSource()}) + if err != nil { + t.Fatalf("GetReplicationDestinationInfo: %v", err) + } + vg := resp.GetReplicationDestination().GetVolumegroup() + if vg == nil { + t.Fatal("a group source answered without a group destination") + } + if vg.GetVolumeGroupId() != vgGroupHandle { + t.Fatalf("destination group = %q, want %q", vg.GetVolumeGroupId(), vgGroupHandle) + } + want := map[string]string{ + sanityClusterID + ":" + sanityPoolUUID + ":" + vgMember1: sanityClusterID + ":" + sanityPoolUUID + ":" + vgMember1, + sanityClusterID + ":" + sanityPoolUUID + ":" + vgMember2: sanityClusterID + ":" + sanityPoolUUID + ":" + vgMember2, + } + got := vg.GetVolumeIds() + if len(got) != len(want) { + t.Fatalf("volume_ids = %v, want %v", got, want) + } + for k, v := range want { + if got[k] != v { + t.Fatalf("volume_ids[%s] = %q, want %q (all: %v)", k, got[k], v, got) + } + } +} + +func TestReplicationDestinationRejectsAnEmptyOrInvalidSource(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + for name, src := range map[string]*replication.ReplicationSource{ + "empty": nil, + "not-handle": volumeSource("not-a-handle"), + "bad-group": {Type: &replication.ReplicationSource_Volumegroup{ + Volumegroup: &replication.ReplicationSource_VolumeGroupSource{VolumeGroupId: "cg:x:y"}, + }}, + } { + _, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: src}) + if status.Code(err) != codes.InvalidArgument { + t.Errorf("%s: code %v (%v), want InvalidArgument", name, status.Code(err), err) + } + } +} + +// A group the control plane does not know is retryable, never an empty map. +func TestReplicationDestinationOfAnUnknownGroupIsUnavailable(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + _, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: groupSource()}) + if status.Code(err) != codes.Unavailable { + t.Fatalf("code %v (%v), want Unavailable", status.Code(err), err) + } +} diff --git a/csi-driver/internal/csi/controller/replication_group_test.go b/csi-driver/internal/csi/controller/replication_group_test.go new file mode 100644 index 000000000..3414b6d47 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_group_test.go @@ -0,0 +1,174 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// The Replication verbs route a cg: group handle to the group-replication +// endpoints, driving the whole consistency group as one unit (design §14.4). A +// per-volume handle still takes the §5 path (covered by replication_test.go). + +const ( + vgGroupHandle = "cg:" + sanityClusterID + ":" + vgGroupID + vgPolicyID = "dddddddd-dddd-4ddd-8ddd-dddddddddddd" +) + +// groupSource builds the ReplicationSource the csi-addons sidecar actually sends +// when driving a VolumeGroupReplication: the group handle rides the volumegroup +// oneof, NOT the per-volume one. Using the volume oneof here (as this helper +// once did) exercised the same wrong field the code read, so every group-routing +// test passed while live VGR promote failed with `invalid volume handle ""` +// (2026-09-26). Regression: 2026-09-26-vgr-source-oneof. +func groupSource() *replication.ReplicationSource { + return &replication.ReplicationSource{ + Type: &replication.ReplicationSource_Volumegroup{ + Volumegroup: &replication.ReplicationSource_VolumeGroupSource{VolumeGroupId: vgGroupHandle}, + }, + } +} + +func newGroupReplTestServer(t *testing.T, mock *mockSBCLI) *Server { + t.Helper() + mock.seedGroup(vgGroupID, vgMember1, vgMember2) + return newTestControllerServer(t, mock) +} + +func TestEnableVolumeReplicationRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + ReplicationSource: groupSource(), + Parameters: map[string]string{replicationPolicyParam: vgPolicyID}, + }) + if err != nil { + t.Fatalf("EnableVolumeReplication: %v", err) + } + if got := mock.groups[vgGroupID].PolicyID; got != vgPolicyID { + t.Fatalf("group policy = %q, want %q", got, vgPolicyID) + } +} + +// The group fail-over target has no reverse policy on its VolumeGroupReplication +// class either, so Enable on the cg: handle must be a no-op there; PromoteGroup +// clones and reconstitutes the group. Regression: 2026-09-27-failover-empty-policy. +func TestEnableVolumeReplicationEmptyPolicyIsNoOpForGroup(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + ReplicationSource: groupSource(), + Parameters: map[string]string{}, + }) + if err != nil { + t.Fatalf("empty policy should be a no-op on the group fail-over target, got: %v", err) + } + if got := mock.groups[vgGroupID].PolicyID; got != "" { + t.Fatalf("group policy = %q, want empty (nothing attached when no policy)", got) + } +} + +func TestDisableVolumeReplicationRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + mock.groups[vgGroupID].PolicyID = vgPolicyID + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + ReplicationSource: groupSource(), + }) + if err != nil { + t.Fatalf("DisableVolumeReplication: %v", err) + } + if got := mock.groups[vgGroupID].PolicyID; got != "" { + t.Fatalf("group policy = %q after disable, want empty", got) + } +} + +func TestPromoteVolumeRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + if _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + ReplicationSource: groupSource(), Force: true, + }); err != nil { + t.Fatalf("PromoteVolume: %v", err) + } + if !mock.groups[vgGroupID].Promoted { + t.Fatal("group was not promoted") + } +} + +func TestDemoteVolumeRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + if _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + ReplicationSource: groupSource(), + }); err != nil { + t.Fatalf("DemoteVolume: %v", err) + } + if !mock.groups[vgGroupID].Demoted { + t.Fatal("group was not demoted") + } +} + +func TestDemoteVolumeGroupStillConvergingIsAborted(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + mock.groups[vgGroupID].DemoteConverging = true + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + ReplicationSource: groupSource(), + }) + if status.Code(err) != codes.Aborted { + t.Fatalf("err = %v, want Aborted", err) + } +} + +func TestResyncVolumeRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + + resp, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + ReplicationSource: groupSource(), + Parameters: map[string]string{sourceClusterIDParam: sanityClusterID}, + }) + if err != nil { + t.Fatalf("ResyncVolume: %v", err) + } + if !resp.GetReady() { + t.Error("expected Ready (group status has no lag budget, so ready)") + } + if got := mock.groups[vgGroupID].FailbackSource; got != sanityClusterID { + t.Fatalf("failback source = %q, want %q", got, sanityClusterID) + } +} + +func TestGetVolumeReplicationInfoRoutesAGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + mock.groups[vgGroupID].LastReplicatedAt = 1_700_000_000 + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + ReplicationSource: groupSource(), + }) + if err != nil { + t.Fatalf("GetVolumeReplicationInfo: %v", err) + } + if resp.GetLastSyncTime() == nil { + t.Fatal("expected a LastSyncTime for the group") + } +} diff --git a/csi-driver/internal/csi/controller/replication_groupresolve.go b/csi-driver/internal/csi/controller/replication_groupresolve.go new file mode 100644 index 000000000..48851404c --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_groupresolve.go @@ -0,0 +1,98 @@ +package controller + +import ( + "context" + + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + "k8s.io/klog" + + atlascp "github.com/simplyblock/atlas/controlplane" + "github.com/simplyblock/atlas/lvol" + "github.com/simplyblock/csi-driver/internal/clusters" +) + +// groupSide says which group of a moved consistency group a verb acts on, the +// group analogue of chooseReplica's local member and the chain's active end. +type groupSide int + +const ( + // groupActiveEnd is the group serving the data now: Info, Resync and + // Enable (re-protection attaches the live group). + groupActiveEnd groupSide = iota + // groupLocalSite is the group on this driver's own cluster: Demote and + // Disable act on what this site holds, never on the live primary elsewhere + // (the per-volume rule of 2026-10-02, PR #618). + groupLocalSite +) + +// resolveGroupTarget resolves a VolumeGroupReplication's group handle to the +// group a verb must act on, and a client for that group's cluster. +// +// A VGR keeps its original group handle across a relocate, while the group it +// names is emptied by design -- its demoted members are deleted so a relocate +// back stays possible -- and the data lives in the peer group of the same name +// as clones (2026-10-04, WordPress A -> B: every group verb on the original +// handle reached the empty source group). The control plane resolves the +// handle (ResolveGroup); the two candidates are the named group and the group +// holding live members. groupActiveEnd picks the latter; groupLocalSite picks +// the one on a cluster flagged local in the driver's secret, preferring the +// live one, and falls back to the live one when no cluster is flagged, as +// chooseReplica does. +// +// A control plane without the resolution endpoint, or a resolution that fails, +// leaves the handle as it is: the verbs then behave as before this resolution. +func resolveGroupTarget( + ctx context.Context, gh lvol.GroupHandle, side groupSide, +) (lvol.GroupHandle, *atlascp.Client, error) { + client, err := clusters.ReplicationClient(ctx, gh.ClusterID) + if err != nil { + return gh, nil, status.Error(codes.Unavailable, err.Error()) + } + res, err := client.ResolveGroup(ctx, gh) + if err != nil { + klog.Warningf("resolve group %s: %v; acting on the handle as named", gh.Handle(), err) + return gh, client, nil + } + target := chooseGroup(gh, res, side, localClusters()) + if target == gh { + return gh, client, nil + } + targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) + if err != nil { + return gh, nil, status.Error(codes.Unavailable, err.Error()) + } + klog.Infof("group %s resolves to %s", gh.Handle(), target.Handle()) + return target, targetClient, nil +} + +// chooseGroup is resolveGroupTarget's pure choice: the named group or the live +// one, by side and the local clusters (nil when none is flagged). +func chooseGroup( + gh lvol.GroupHandle, res atlascp.GroupResolution, side groupSide, local map[string]bool, +) lvol.GroupHandle { + if res.Active == nil || *res.Active == gh { + return gh + } + live := *res.Active + if side == groupActiveEnd || local == nil { + return live + } + if local[live.ClusterID] { + return live + } + if local[gh.ClusterID] { + return gh + } + return live +} + +// localClusters is the set of clusters the driver's secret flags local, nil +// when none is flagged (an older operator) or the secret cannot be read. +func localClusters() map[string]bool { + local, flagged, err := clusters.Local() + if err != nil || !flagged { + return nil + } + return local +} diff --git a/csi-driver/internal/csi/controller/replication_groupresolve_test.go b/csi-driver/internal/csi/controller/replication_groupresolve_test.go new file mode 100644 index 000000000..7f8c6e771 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_groupresolve_test.go @@ -0,0 +1,214 @@ +package controller + +import ( + "context" + "encoding/json" + "os" + "path/filepath" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + + atlascp "github.com/simplyblock/atlas/controlplane" + "github.com/simplyblock/atlas/lvol" + "github.com/simplyblock/csi-driver/internal/clusters" +) + +// A VolumeGroupReplication keeps its original group handle across a relocate, +// while the group it names is emptied by design (its demoted members are +// deleted so a relocate back stays possible) and the data lives in the peer +// group of the same name on the other site. The group verbs resolve the handle +// through the control plane (2026-10-04, WordPress A -> B: Ramen's VRG waited +// for destination info for ever against the emptied source group). + +const ( + peerClusterID = "bbbbbbbb-bbbb-4bbb-8bbb-bbbbbbbbbbbb" + peerGroupID = "e5e5e5e5-e5e5-4e5e-8e5e-e5e5e5e5e5e5" + peerClone = "f6666666-6666-4666-8666-666666666666" + origPV = sanityClusterID + ":p-a:" + vgMember1 +) + +// writeTwoSiteSecret registers the named cluster (site A) and its peer (site +// B), both answered by the one mock, with site B flagged local: the driver runs +// on site B, where the relocate landed. +func writeTwoSiteSecret(t *testing.T, mock *mockSBCLI, localCluster string) { + t.Helper() + data, _ := json.Marshal(clusters.Info{Clusters: []clusters.Config{ + {ClusterID: sanityClusterID, ClusterEndpoint: mock.URL(), ClusterSecret: sanitySecret, + Local: localCluster == sanityClusterID}, + {ClusterID: peerClusterID, ClusterEndpoint: mock.URL(), ClusterSecret: sanitySecret, + Local: localCluster == peerClusterID}, + }}) + f := filepath.Join(t.TempDir(), "secret.json") + if err := os.WriteFile(f, data, 0o600); err != nil { + t.Fatal(err) + } + t.Setenv("SPDKCSI_SECRET", f) +} + +// movedGroup is the incident's state: the named group on A is empty, the data +// lives in the peer group on B as a clone of the original volume. +func movedGroup(t *testing.T, localCluster string) (*Server, *mockSBCLI) { + t.Helper() + mock := newMockSBCLI() + t.Cleanup(mock.Close) + mock.seedGroup(vgGroupID) + mock.seedGroup(peerGroupID, peerClone) + mock.groupResolution[vgGroupID] = map[string]any{ + "cluster_id": sanityClusterID, "group_id": vgGroupID, + "active_cluster_id": peerClusterID, "active_group_id": peerGroupID, + "members": []map[string]string{{ + "origin_handle": origPV, "active_handle": peerClusterID + ":p-b:" + peerClone}}, + } + cs := newTestControllerServer(t, mock) + writeTwoSiteSecret(t, mock, localCluster) + return cs, mock +} + +func TestDestinationInfoOfAGroupEmptiedByARelocateMapsItsPVs(t *testing.T) { + cs, _ := movedGroup(t, peerClusterID) + + resp, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: groupSource()}) + if err != nil { + t.Fatalf("GetReplicationDestinationInfo: %v", err) + } + vg := resp.GetReplicationDestination().GetVolumegroup() + if vg.GetVolumeGroupId() != vgGroupHandle { + t.Fatalf("destination group = %q, want the VGR's own handle %q", vg.GetVolumeGroupId(), vgGroupHandle) + } + if got := vg.GetVolumeIds(); len(got) != 1 || got[origPV] != origPV { + t.Fatalf("map = %v, want the PV's original handle mapped to itself", got) + } +} + +func TestDestinationInfoOfAGroupWithNoLiveMemberIsUnavailable(t *testing.T) { + cs, mock := movedGroup(t, peerClusterID) + mock.groupResolution[vgGroupID] = map[string]any{"cluster_id": sanityClusterID, "group_id": vgGroupID} + + _, err := cs.GetReplicationDestinationInfo(context.Background(), + &replication.GetReplicationDestinationInfoRequest{ReplicationSource: groupSource()}) + if status.Code(err) != codes.Unavailable { + t.Fatalf("code = %v, want Unavailable (retryable, never a partial map)", status.Code(err)) + } +} + +// The relocate back B -> A: Ramen demotes the VGR on B, which names the group +// on A. Site B must demote the group it serves -- the peer group -- not the +// empty one on A. +func TestDemoteOfAMovedGroupActsOnTheLocalGroup(t *testing.T) { + cs, mock := movedGroup(t, peerClusterID) + + if _, err := cs.DemoteVolume(context.Background(), + &replication.DemoteVolumeRequest{ReplicationSource: groupSource()}); err != nil { + t.Fatalf("DemoteVolume: %v", err) + } + if !mock.groups[peerGroupID].Demoted || mock.groups[vgGroupID].Demoted { + t.Fatalf("demoted: peer %v, named %v; want the peer group on site B only", + mock.groups[peerGroupID].Demoted, mock.groups[vgGroupID].Demoted) + } +} + +// Site A, the old primary, demoting its side must stay on its own group. +func TestDemoteOnTheOldPrimarysSiteStaysOnItsOwnGroup(t *testing.T) { + cs, mock := movedGroup(t, sanityClusterID) + + if _, err := cs.DemoteVolume(context.Background(), + &replication.DemoteVolumeRequest{ReplicationSource: groupSource()}); err != nil { + t.Fatalf("DemoteVolume: %v", err) + } + if mock.groups[peerGroupID].Demoted || !mock.groups[vgGroupID].Demoted { + t.Fatal("site A demoted the live group on site B") + } +} + +// Re-protection B -> A: Enable on the VGR's handle attaches the live group. +func TestEnableOfAMovedGroupAttachesTheLiveGroup(t *testing.T) { + cs, mock := movedGroup(t, peerClusterID) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + ReplicationSource: groupSource(), + Parameters: map[string]string{replicationPolicyParam: vgPolicyID}, + }) + if err != nil { + t.Fatalf("EnableVolumeReplication: %v", err) + } + if mock.groups[peerGroupID].PolicyID != vgPolicyID || mock.groups[vgGroupID].PolicyID != "" { + t.Fatalf("policy: peer %q, named %q; want it on the live group only", + mock.groups[peerGroupID].PolicyID, mock.groups[vgGroupID].PolicyID) + } +} + +func TestInfoOfAMovedGroupReadsTheLiveGroup(t *testing.T) { + cs, mock := movedGroup(t, peerClusterID) + mock.groups[peerGroupID].LastReplicatedAt = 1791105000 + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), + &replication.GetVolumeReplicationInfoRequest{ReplicationSource: groupSource()}) + if err != nil { + t.Fatalf("GetVolumeReplicationInfo: %v", err) + } + if resp.GetLastSyncTime().GetSeconds() != 1791105000 { + t.Fatalf("last sync = %v, want the live group's", resp.GetLastSyncTime()) + } +} + +// Promote stays on the named group: the control plane's group fail-over +// resolves the peer itself, and redirecting it would turn a fail-back into a +// no-op on the group being left. +func TestPromoteOfAMovedGroupStaysOnTheNamedGroup(t *testing.T) { + cs, mock := movedGroup(t, peerClusterID) + + if _, err := cs.PromoteVolume(context.Background(), + &replication.PromoteVolumeRequest{ReplicationSource: groupSource()}); err != nil { + t.Fatalf("PromoteVolume: %v", err) + } + if !mock.groups[vgGroupID].Promoted || mock.groups[peerGroupID].Promoted { + t.Fatal("promote did not reach the named group") + } +} + +// A group still live where it was created resolves to itself. +func TestAGroupLiveAtItsSourceIsActedOnAsNamed(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newGroupReplTestServer(t, mock) + mock.groupResolution[vgGroupID] = map[string]any{ + "cluster_id": sanityClusterID, "group_id": vgGroupID, + "active_cluster_id": sanityClusterID, "active_group_id": vgGroupID, + "members": []map[string]string{{"origin_handle": origPV, "active_handle": origPV}}, + } + if _, err := cs.DemoteVolume(context.Background(), + &replication.DemoteVolumeRequest{ReplicationSource: groupSource()}); err != nil { + t.Fatalf("DemoteVolume: %v", err) + } + if !mock.groups[vgGroupID].Demoted { + t.Fatal("the named, live group was not demoted") + } +} + +func TestChooseGroup(t *testing.T) { + named := lvol.GroupHandle{ClusterID: sanityClusterID, GroupID: vgGroupID} + live := lvol.GroupHandle{ClusterID: peerClusterID, GroupID: peerGroupID} + moved := atlascp.GroupResolution{Active: &live} + for _, tc := range []struct { + name string + res atlascp.GroupResolution + side groupSide + local map[string]bool + want lvol.GroupHandle + }{ + {"not moved", atlascp.GroupResolution{Active: &named}, groupLocalSite, map[string]bool{peerClusterID: true}, named}, + {"nothing live", atlascp.GroupResolution{}, groupActiveEnd, nil, named}, + {"active end", moved, groupActiveEnd, map[string]bool{sanityClusterID: true}, live}, + {"local on the live site", moved, groupLocalSite, map[string]bool{peerClusterID: true}, live}, + {"local on the named site", moved, groupLocalSite, map[string]bool{sanityClusterID: true}, named}, + {"no local flags", moved, groupLocalSite, nil, live}, + } { + if got := chooseGroup(named, tc.res, tc.side, tc.local); got != tc.want { + t.Errorf("%s: got %s, want %s", tc.name, got.Handle(), tc.want.Handle()) + } + } +} diff --git a/csi-driver/internal/csi/controller/replication_lifecycle_test.go b/csi-driver/internal/csi/controller/replication_lifecycle_test.go new file mode 100644 index 000000000..4ebf8a687 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_lifecycle_test.go @@ -0,0 +1,588 @@ +package controller + +import ( + "context" + "net/http" + "strings" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +func TestPromoteVolumeForced(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: true, + }) + if err != nil { + t.Fatal(err) + } + if strings.Contains(mock.lastFailoverQuery, "planned=true") { + t.Errorf("query = %q, forced promote must not ask for the planned gate", mock.lastFailoverQuery) + } +} + +// A Ramen-restored destination PV inherits the ORIGINAL source's own +// volumeHandle verbatim (confirmed live 2026-09-23, relocate M-02): Ramen's +// S3-restore recreates the exact PV/PVC object it archived at protect time, +// including the source cluster+lvol identity, on a cluster that never +// provisioned that volume at all. Every Replication RPC parses its target +// straight from the given handle, so without resolving through the backend's +// own source->target relationship first, "promote" would be asking the +// ORIGINAL, foreign volume to fail over -- not the local replica that has +// actually been receiving replicated data. PromoteVolume must resolve a +// handle whose relationship says IsSource and redirect to TargetLvolId +// (on TargetClusterId/TargetPoolId) before calling failover. +func TestPromoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "migration", + "state": "replicating", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: true, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastFailoverVolumeID != testReplTargetVolumeID { + t.Errorf("failover landed on volume %q, want the resolved target %q", + mock.lastFailoverVolumeID, testReplTargetVolumeID) + } +} + +// A volume already naming the target side of its own relationship (IsSource +// false) is promoted directly, unchanged -- resolving again would be a +// harmless no-op, but this proves it takes that path rather than one that +// happens to work only by coincidence. +func TestPromoteVolumeAlreadyNamingTheTargetIsUnchanged(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "migration", + "state": "replicating", + "is_source": false, + "source_cluster_id": sanityClusterID, "source_lvol_id": "99999999-9999-9999-9999-999999999998", + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: true, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastFailoverVolumeID != testReplVolumeID { + t.Errorf("failover landed on volume %q, want %q unchanged", mock.lastFailoverVolumeID, testReplVolumeID) + } +} + +// The ordinary case, and by far the most common: a volume never enabled for +// replication (M-01's own first-ever protect) has no relationship at all yet. +// This must promote the given handle directly rather than fail the whole +// call over a 404 that just means "nothing to resolve." +func TestPromoteVolumeWithNoRelationshipYetUsesTheGivenHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: true, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastFailoverVolumeID != testReplVolumeID { + t.Errorf("failover landed on volume %q, want %q", mock.lastFailoverVolumeID, testReplVolumeID) + } +} + +func TestPromoteVolumePlannedSendsThePlannedFlag(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: false, + }) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(mock.lastFailoverQuery, "planned=true") { + t.Errorf("query = %q, want planned=true", mock.lastFailoverQuery) + } +} + +// The whole point of the planned gate: a demote still converging must map to +// ABORTED (retryable), never FAILED_PRECONDITION -- the vendored csi-addons +// controller auto-escalates ANY FAILED_PRECONDITION from a force=false +// promote to force=true inline, in the same reconcile, with no +// wait-and-retry grace period of its own. Mapping this to FailedPrecondition +// would silently force through a promote while demote is still converging. +func TestPromoteVolumePlannedWhileDemoteConvergingIsAborted(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.failoverStatus = http.StatusConflict + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: false, + }) + st, _ := status.FromError(err) + if st.Code() != codes.Aborted { + t.Errorf("code = %v, want Aborted (retryable, never auto-forced by the vendored controller)", st.Code()) + } +} + +// The case that SHOULD let the vendored controller's own force-escalation +// take over: no demote was ever requested, so this planned attempt only +// makes sense as a genuinely unplanned failover the caller mislabeled. +func TestPromoteVolumePlannedWithNoDemoteIsFailedPrecondition(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.failoverStatus = http.StatusPreconditionFailed + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: false, + }) + st, _ := status.FromError(err) + if st.Code() != codes.FailedPrecondition { + t.Errorf("code = %v, want FailedPrecondition", st.Code()) + } +} + +// Same real-world shape as TestEnableVolumeReplicationUsesReplicationSourceWhenVolumeIdIsEmpty: +// the vendored controller-manager/sidecar chain sends every Replication RPC, +// Promote included, via ReplicationSource with the legacy flat VolumeId left +// empty. +func TestPromoteVolumeUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + ReplicationSource: replicationSourceFor(), Force: true, + }) + if err != nil { + t.Fatal(err) + } +} + +// Same relationship-resolution requirement as PromoteVolume/EnableVolumeReplication +// (see TestPromoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship): +// a second (or later) relocate demotes a volume that itself came into +// existence via an earlier promote, so its PV/PVC still carries the +// ORIGINAL, foreign source's volumeHandle. Confirmed live 2026-09-24 (relocate +// M-02's round trip, B -> A): DemoteVolume issued the RPC against that +// foreign, pre-promote lvol id and got a 404 from the control plane, well +// before ever reaching the actual local replica that had been serving as +// primary. DemoteVolume must resolve a handle whose relationship says +// IsSource and redirect to TargetLvolId first, exactly like Promote and +// Enable already do. +func TestDemoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastDemoteVolumeID != testReplTargetVolumeID { + t.Errorf("demote landed on volume %q, want the resolved target %q", + mock.lastDemoteVolumeID, testReplTargetVolumeID) + } +} + +func TestDemoteVolumeDone(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } +} + +// Non-blocking and re-driven: a still-converging demote must not hang the +// RPC or report success, it must fail with a retryable code so the +// controller-manager's own reconcile loop re-invokes DemoteVolume later -- +// matching the actual upstream reconciler's requeue-until-ready behavior. +func TestDemoteVolumeNotYetDoneIsAborted(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.demoteStatus = http.StatusAccepted + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + VolumeId: testReplVolID, + }) + st, _ := status.FromError(err) + if st.Code() != codes.Aborted { + t.Errorf("code = %v, want Aborted (retryable)", st.Code()) + } +} + +func TestDemoteVolumeBackendFailureIsUnavailable(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.demoteStatus = http.StatusInternalServerError + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + VolumeId: testReplVolID, + }) + if err == nil { + t.Fatal("want an error on a genuine backend failure") + } +} + +func TestDemoteVolumeUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.DemoteVolume(context.Background(), &replication.DemoteVolumeRequest{ + ReplicationSource: replicationSourceFor(), + }) + if err != nil { + t.Fatal(err) + } +} + +func TestResyncVolume(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } +} + +func TestResyncVolumeUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + ReplicationSource: replicationSourceFor(), + }) + if err != nil { + t.Fatal(err) + } +} + +func TestResyncVolumeSendsTheSourceClusterParameter(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + Parameters: map[string]string{sourceClusterIDParam: sanityClusterID}, + }) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(mock.lastFailbackBody), `"source_cluster_id":"`+sanityClusterID+`"`) { + t.Errorf("failback body = %q, want source_cluster_id %s", mock.lastFailbackBody, sanityClusterID) + } +} + +func TestResyncVolumeReadyReflectsLag(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationStatus[testReplVolumeID] = map[string]any{ + "role": "source", "state": "in_sync", + "lag_seconds": 120, "lag_budget_seconds": 900, + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": true, + } + + resp, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if !resp.Ready { + t.Error("Ready = false, want true: lag (120s) is inside the budget (900s)") + } +} + +func TestResyncVolumeNotReadyWhileLagExceedsBudget(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationStatus[testReplVolumeID] = map[string]any{ + "role": "source", "state": "in_sync", + "lag_seconds": 1800, "lag_budget_seconds": 900, + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": true, + } + + resp, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if resp.Ready { + t.Error("Ready = true, want false: lag (1800s) exceeds the budget (900s)") + } +} + +// Regression: 2026-09-24-resync-foreign-handle-404 — on relocate M-02's round +// trip (B -> A), cluster B's Secondary-role VolumeReplication still carries the +// ORIGINAL cluster-A volumeHandle (Ramen's S3-restore preserves it verbatim), +// and by then the A-side lvol record has been reaped by lvol_monitor's +// LVOL_DEMOTE_FAILOVER_HOLD_SEC deferred removal. ResyncVolume was the only +// Replication verb that skipped resolveToLocalReplica, so it fired the +// failback call at that dead, foreign lvol and got a permanent 404 ("LVol +// 00660ccf... not found") on every reconcile -- the VR never finished becoming +// Secondary and the whole relocate-back stalled. The demote in the very same +// reconcile succeeded, because DemoteVolume resolves. Resync must redirect a +// handle whose relationship says IsSource to the live local replica +// (TargetLvolId), and the relationship must carry it there even though the +// source volume itself no longer exists (the cluster-scoped relationship +// endpoint stays resolvable by a deleted source id, confirmed live +// 2026-09-24). +func TestResyncVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + // The source lvol record is gone -- reaped after the failover hold -- so + // any call landing on it 404s, exactly as the live control plane did. + delete(mock.volumes, testReplVolumeID) + + resp, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastFailbackVolumeID != testReplTargetVolumeID { + t.Errorf("failback landed on volume %q, want the resolved target %q", + mock.lastFailbackVolumeID, testReplTargetVolumeID) + } + if !resp.Ready { + t.Error("Ready = false, want true: the resolved target reports no lag at all") + } +} + +// Regression: 2026-09-24-resync-foreign-handle-404 — the same missing +// resolution as TestResyncVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship, +// on the standalone info read: Ramen polls GetVolumeReplicationInfo for +// lastSyncTime against the same S3-restored, foreign volumeHandle, so once the +// source record is reaped the read 404s instead of reporting the local +// replica's status. +func TestGetVolumeReplicationInfoResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + mock.replicationStatus[testReplTargetVolumeID] = map[string]any{ + "role": "target", "state": "in_sync", + "last_replicated_at": "2026-09-24T13:00:00Z", + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": false, + } + delete(mock.volumes, testReplVolumeID) + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if resp.LastSyncTime == nil { + t.Fatal("LastSyncTime = nil, want the resolved target's last_replicated_at") + } + if got := resp.LastSyncTime.AsTime().UTC().Format("2006-01-02T15:04:05Z"); got != "2026-09-24T13:00:00Z" { + t.Errorf("LastSyncTime = %s, want the resolved target's 2026-09-24T13:00:00Z", got) + } +} + +// Regression: 2026-09-24-chained-relationship-resolves-one-hop-short — a +// relocate ROUND TRIP leaves two chained pairings: original -> hop-1 clone +// (cluster B), and hop-1 clone -> hop-2 clone (cluster A, the volume actually +// serving the workload, named by active_lvol_id on every record in the +// chain). Single-step resolution stopped at the FIRST pairing's target -- the +// retired hop-1 clone -- so post-round-trip Replication verbs (and the policy +// attach that Enable performs) landed on a superseded volume on the wrong +// cluster (confirmed live 2026-09-24: the policy stuck to B's clone while A's +// new primary ran unprotected). Resolution must walk hop by hop, using each +// record's own consistent target triple, until the hop whose target IS the +// active volume. +func TestPromoteVolumeResolvesAcrossAChainedRelationshipToTheActiveVolume(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-hop1-clone", Size: 1 << 30, + } + mock.volumes[testReplActiveVolumeID] = &mockVolume{ + UUID: testReplActiveVolumeID, Name: "repl-vol-active", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeee1", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + "active": "target", "active_lvol_id": testReplActiveVolumeID, + } + mock.replicationRelationship[testReplTargetVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeee2", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplTargetVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplActiveVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + "active": "target", "active_lvol_id": testReplActiveVolumeID, + } + delete(mock.volumes, testReplVolumeID) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, Force: true, + }) + if err != nil { + t.Fatal(err) + } + if mock.lastFailoverVolumeID != testReplActiveVolumeID { + t.Errorf("failover landed on volume %q, want the chain's ACTIVE volume %q, not the retired middle hop", + mock.lastFailoverVolumeID, testReplActiveVolumeID) + } +} + +// Regression: 2026-09-24-promote-must-not-attach-the-policy — attaching the +// class's policy to the promoted volume inside PromoteVolume was tried (to +// close the "new primary comes up with policy=NONE" gap) and confirmed +// harmful live the same day: the attach starts policy-driven replication of +// the fresh clone immediately, the relocate's fail-back then re-points the +// SAME volume's replication at the original source node, and the two chains +// collide -- the fail-over clone build died on "Failed to create BDev" and +// the whole relocate wedged at WaitForReadiness. Promote must promote and +// nothing else, even when the class parameters (which ride on every RPC) +// name a policy; re-protection is the planned-cutover flow's job. +func TestPromoteVolumeDoesNotAttachThePolicyItWasHanded(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + "active": "target", "active_lvol_id": testReplTargetVolumeID, + } + delete(mock.volumes, testReplVolumeID) + + _, err := cs.PromoteVolume(context.Background(), &replication.PromoteVolumeRequest{ + VolumeId: testReplVolID, + Force: true, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplTargetVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("promote attached policy %q to the promoted volume; it must attach nothing", got) + } +} + +func TestResyncVolumeBackendFailureIsUnavailable(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.failbackStatus = http.StatusInternalServerError + + _, err := cs.ResyncVolume(context.Background(), &replication.ResyncVolumeRequest{ + VolumeId: testReplVolID, + }) + if err == nil { + t.Fatal("want an error on a genuine backend failure") + } +} diff --git a/csi-driver/internal/csi/controller/replication_local_test.go b/csi-driver/internal/csi/controller/replication_local_test.go new file mode 100644 index 000000000..83f93c855 --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_local_test.go @@ -0,0 +1,98 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "github.com/simplyblock/atlas/lvol" +) + +// The chain of 2026-10-02 (realbed, WordPress): created on A (0aea, gone), +// failed over to B (6e83), relocated back to A (80e3), failed over to B +// (e3d4, the live primary). Ramen then made the old primary on A secondary. +const ( + siteA, siteB = "A", "B" + oldPrimaryOnA = "80e3" + livePrimaryOnB = "e3d4" +) + +func liveChain() []chainHop { + mk := func(cluster, id string) chainHop { + return chainHop{h: &lvol.Handle{ClusterID: cluster, PoolRef: "p", VolumeID: id}} + } + return []chainHop{mk(siteA, "0aea"), mk(siteB, "6e83"), mk(siteA, oldPrimaryOnA), mk(siteB, livePrimaryOnB)} +} + +func TestChooseReplicaOnTheOldPrimarysSiteIsTheOldPrimaryNotTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{siteA: true}, true) + if got.h.VolumeID != oldPrimaryOnA { + t.Fatalf("site A acts on %s, want 80e3 (its own, superseded primary); e3d4 is the live primary on B", got.h.VolumeID) + } +} + +func TestChooseReplicaOnTheNewPrimarysSiteIsTheLivePrimary(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{siteB: true}, true) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("site B acts on %s, want e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithoutLocalFlagsKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), nil, false) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("unflagged secret: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaWithNoLocalMemberKeepsTheChainsEnd(t *testing.T) { + got := chooseReplica(liveChain(), map[string]bool{"C": true}, true) + if got.h.VolumeID != livePrimaryOnB { + t.Fatalf("no member on C: %s, want the chain's end e3d4", got.h.VolumeID) + } +} + +func TestChooseReplicaOfAVolumeWithoutARelationshipIsTheVolume(t *testing.T) { + one := liveChain()[:1] + if got := chooseReplica(one, map[string]bool{siteB: true}, true); got.h.VolumeID != "0aea" { + t.Fatalf("got %s", got.h.VolumeID) + } +} + +// The old primary of an unplanned fail-over is reaped by the control plane +// once its fail-over completed; Ramen still demotes it (and deletes its VR) +// when the site returns. Nothing is left to demote: success, not NotFound +// (live 2026-10-02: the VR on site A stayed Degraded on a 404). +func TestDemoteAndDisableOfAReapedChainMemberSucceed(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + gone := "99999999-aaaa-bbbb-cccc-dddddddddddd" + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": gone, + "active_lvol_id": gone, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + ctx := context.Background() + if _, err := cs.DemoteVolume(ctx, &replication.DemoteVolumeRequest{VolumeId: testReplVolID}); err != nil { + t.Fatalf("demote of a reaped chain member: %v", err) + } + disable := &replication.DisableVolumeReplicationRequest{VolumeId: testReplVolID} + if _, err := cs.DisableVolumeReplication(ctx, disable); err != nil { + t.Fatalf("disable of a reaped chain member: %v", err) + } +} + +func TestActiveEndFallbackAimsAtTheChainsEndFromTheLocalSite(t *testing.T) { + // Not flagged in the test secret: the class parameter stays. + end, src := activeEndFallback(liveChain(), "param") + if end.h.VolumeID != livePrimaryOnB || src != "param" { + t.Fatalf("end %s source %s", end.h.VolumeID, src) + } +} diff --git a/csi-driver/internal/csi/controller/replication_test.go b/csi-driver/internal/csi/controller/replication_test.go new file mode 100644 index 000000000..433ceca5a --- /dev/null +++ b/csi-driver/internal/csi/controller/replication_test.go @@ -0,0 +1,411 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/replication" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +const ( + testReplVolumeID = "88888888-8888-8888-8888-888888888888" + testReplVolID = sanityClusterID + ":" + sanityPoolUUID + ":" + testReplVolumeID + testReplPolicyID = "77777777-7777-7777-7777-777777777777" + testReplTargetVolumeID = "88888888-8888-8888-8888-888888888889" +) + +func newReplicationTestServer(t *testing.T, mock *mockSBCLI) *Server { + t.Helper() + mock.volumes[testReplVolumeID] = &mockVolume{UUID: testReplVolumeID, Name: "repl-vol", Size: 1 << 30} + return newTestControllerServer(t, mock) +} + +// replicationSourceFor builds the ReplicationSource the real +// kubernetes-csi-addons v0.15.0 sidecar sends on every Replication RPC +// instead of the legacy flat VolumeId field (internal/sidecar/service's +// ReplicationServer proxy never sets it). +func replicationSourceFor() *replication.ReplicationSource { + return &replication.ReplicationSource{ + Type: &replication.ReplicationSource_Volume{ + Volume: &replication.ReplicationSource_VolumeSource{VolumeId: testReplVolID}, + }, + } +} + +func TestEnableVolumeReplication(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: testReplVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != testReplPolicyID { + t.Errorf("ReplicationPolicyID = %q, want %q", got, testReplPolicyID) + } +} + +// Repeating an enable already in effect is success without a second +// meaningful change (P0-2's own idempotency; the mock does not distinguish, +// it just re-applies the same value). +func TestEnableVolumeReplicationRepeatedIsIdempotent(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + req := &replication.EnableVolumeReplicationRequest{ + VolumeId: testReplVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + } + if _, err := cs.EnableVolumeReplication(context.Background(), req); err != nil { + t.Fatal(err) + } + if _, err := cs.EnableVolumeReplication(context.Background(), req); err != nil { + t.Errorf("second EnableVolumeReplication = %v, want nil (idempotent)", err) + } +} + +// The real controller-manager/sidecar chain (v0.15.0) sends the volume +// identity via ReplicationSource, leaving the legacy flat VolumeId field +// empty -- confirmed against a live cluster, where every Replication RPC +// failed with "invalid volume handle \"\"" despite the controller-manager's +// own log showing it resolved a correct handle. +func TestEnableVolumeReplicationUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + ReplicationSource: replicationSourceFor(), + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != testReplPolicyID { + t.Errorf("ReplicationPolicyID = %q, want %q", got, testReplPolicyID) + } +} + +// Same relationship-resolution requirement as PromoteVolume (see its own +// TestPromoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship): +// csi-addons calls EnableVolumeReplication as a promote precondition too, so +// a handle inherited from the foreign source side must resolve to the local +// target volume before the policy attach is attempted against it. +func TestEnableVolumeReplicationResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "migration", + "state": "replicating", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: testReplVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplTargetVolumeID].ReplicationPolicyID; got != testReplPolicyID { + t.Errorf("target volume's ReplicationPolicyID = %q, want %q", got, testReplPolicyID) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("source volume's ReplicationPolicyID = %q, want untouched", got) + } +} + +// simplyblock's replication is one-way and the destination never carries a +// persistent, independently provisioned LVol of its own (confirmed live +// 2026-09-23, relocate M-02): the writable clone only comes into existence +// when PromoteVolume clones the last replicated snapshot. csi-addons always +// calls EnableVolumeReplication before PromoteVolume, unconditionally, for +// the side that's becoming Primary -- so on a first-ever relocate (no prior +// relationship for resolveToLocalReplica to redirect through either), Enable +// is handed a handle that legitimately names nothing yet. That must no-op +// rather than fail: PromoteVolume is what actually creates and validates the +// volume, and is what surfaces a real error if there's genuinely nothing to +// clone from. +func TestEnableVolumeReplicationNoOpsWhenVolumeDoesNotExistYet(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + notYetClonedVolumeID := "99999999-8888-8888-8888-888888888888" + notYetClonedVolID := sanityClusterID + ":" + sanityPoolUUID + ":" + notYetClonedVolumeID + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: notYetClonedVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + if err != nil { + t.Errorf("EnableVolumeReplication = %v, want nil (no-op: nothing to attach a policy to yet)", err) + } +} + +// An empty replicationPolicyID is the FAIL-OVER TARGET: the side becoming +// primary carries no reverse-direction policy yet (the reverse direction is a +// fail-back-time concern, design-ramen-integration.md §6.2). csi-addons always +// calls Enable before Promote for the side becoming Primary, so Enable must be a +// no-op here -- there is nothing to attach, and PromoteVolume clones from the +// replicated snapshot and does the real work. Rejecting it as "required" blocked +// every fail-over whose target VolumeReplicationClass had no policy. +// Regression: 2026-09-27-failover-empty-policy. +func TestEnableVolumeReplicationEmptyPolicyIsNoOp(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatalf("empty policy should be a no-op on the fail-over target, got: %v", err) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("ReplicationPolicyID = %q, want empty (nothing attached when no policy)", got) + } +} + +func TestEnableVolumeReplicationBackendRefusal(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationPUTStatus = 412 + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: testReplVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + st, _ := status.FromError(err) + if st.Code() != codes.FailedPrecondition { + t.Errorf("code = %v, want FailedPrecondition", st.Code()) + } +} + +func TestEnableVolumeReplicationMalformedVolumeHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: "not-a-valid-handle", + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + st, _ := status.FromError(err) + if st.Code() != codes.InvalidArgument { + t.Errorf("code = %v, want InvalidArgument", st.Code()) + } +} + +// A cluster this deployment's secret.json has no entry for is unreachable in +// the same sense a network partition would be: there is no client to make the +// call with, so the RPC must not be confused with a backend-side refusal. +func TestEnableVolumeReplicationUnknownCluster(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + unregisteredClusterID := "99999999-9999-9999-9999-999999999999" + unknownClusterVolID := unregisteredClusterID + ":" + sanityPoolUUID + ":" + testReplVolumeID + + _, err := cs.EnableVolumeReplication(context.Background(), &replication.EnableVolumeReplicationRequest{ + VolumeId: unknownClusterVolID, + Parameters: map[string]string{replicationPolicyParam: testReplPolicyID}, + }) + st, _ := status.FromError(err) + if st.Code() != codes.Unavailable { + t.Errorf("code = %v, want Unavailable", st.Code()) + } +} + +func TestDisableVolumeReplication(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplVolumeID].ReplicationPolicyID = testReplPolicyID + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("ReplicationPolicyID = %q, want cleared", got) + } +} + +// Disabling a volume that follows no policy is success, matching the +// backend's own idempotency (P0-2): there is nothing to detach. +func TestDisableVolumeReplicationNotAttachedIsSuccess(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Errorf("DisableVolumeReplication = %v, want nil", err) + } +} + +// Same relationship-resolution requirement as DemoteVolume/PromoteVolume/ +// EnableVolumeReplication: Ramen's own teardown sequence calls Disable right +// after a successful demote, on the SAME volume -- so it inherits the SAME +// foreign, pre-promote handle and must resolve to the local replica before +// the policy detach is attempted against it. +func TestDisableVolumeReplicationResolvesToTargetWhenGivenTheSourceSideOfARelationship(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-target", Size: 1 << 30, ReplicationPolicyID: testReplPolicyID, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + } + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplTargetVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("target volume's ReplicationPolicyID = %q, want cleared", got) + } +} + +func TestDisableVolumeReplicationUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplVolumeID].ReplicationPolicyID = testReplPolicyID + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + ReplicationSource: replicationSourceFor(), + }) + if err != nil { + t.Fatal(err) + } + if got := mock.volumes[testReplVolumeID].ReplicationPolicyID; got != "" { + t.Errorf("ReplicationPolicyID = %q, want cleared", got) + } +} + +func TestDisableVolumeReplicationDuringCutoverIsAborted(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationPUTStatus = 409 + + _, err := cs.DisableVolumeReplication(context.Background(), &replication.DisableVolumeReplicationRequest{ + VolumeId: testReplVolID, + }) + st, _ := status.FromError(err) + if st.Code() != codes.Aborted { + t.Errorf("code = %v, want Aborted (retryable)", st.Code()) + } +} + +func TestGetVolumeReplicationInfo(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationStatus[testReplVolumeID] = map[string]any{ + "role": "source", "state": "in_sync", + "last_replicated_at": "2026-09-17T12:00:00Z", + "lag_seconds": 42, + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": false, + } + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if resp.LastSyncTime == nil || resp.LastSyncTime.AsTime().Unix() != 1789646400 { + t.Errorf("LastSyncTime = %v, want 2026-09-17T12:00:00Z", resp.LastSyncTime) + } +} + +// A volume that never replicated is a valid answer, never a 404: the mock's +// default GET .../replication/status body for a volume nothing configured. +func TestGetVolumeReplicationInfoNeverReplicated(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + VolumeId: testReplVolID, + }) + if err != nil { + t.Fatal(err) + } + if resp.LastSyncTime != nil { + t.Errorf("LastSyncTime = %v, want nil", resp.LastSyncTime) + } +} + +func TestGetVolumeReplicationInfoUsesReplicationSourceWhenVolumeIdIsEmpty(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.replicationStatus[testReplVolumeID] = map[string]any{ + "role": "source", "state": "in_sync", + "last_replicated_at": "2026-09-17T12:00:00Z", + "lag_seconds": 42, + "outstanding_count": 0, "outstanding_bytes": 0, + "failing_count": 0, "max_retry_reached": false, "resyncing": false, + } + + resp, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + ReplicationSource: replicationSourceFor(), + }) + if err != nil { + t.Fatal(err) + } + if resp.LastSyncTime == nil || resp.LastSyncTime.AsTime().Unix() != 1789646400 { + t.Errorf("LastSyncTime = %v, want 2026-09-17T12:00:00Z", resp.LastSyncTime) + } +} + +func TestGetVolumeReplicationInfoUnknownVolume(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + delete(mock.volumes, testReplVolumeID) + + _, err := cs.GetVolumeReplicationInfo(context.Background(), &replication.GetVolumeReplicationInfoRequest{ + VolumeId: testReplVolID, + }) + st, _ := status.FromError(err) + if st.Code() != codes.NotFound { + t.Errorf("code = %v, want NotFound", st.Code()) + } +} diff --git a/csi-driver/internal/csi/controller/server.go b/csi-driver/internal/csi/controller/server.go index ca9b630d8..ffe66d755 100644 --- a/csi-driver/internal/csi/controller/server.go +++ b/csi-driver/internal/csi/controller/server.go @@ -4,6 +4,7 @@ package controller import ( "github.com/container-storage-interface/spec/lib/go/csi" + "github.com/csi-addons/spec/lib/go/replication" "k8s.io/client-go/kubernetes" csicommon "github.com/simplyblock/csi-driver/internal/csi/common" @@ -15,6 +16,16 @@ type Server struct { // groupsnapshot.go; embedding the unimplemented server satisfies the // interface's forward-compat guard for any method not overridden. csi.UnimplementedGroupControllerServer + // The csi-addons Replication service (design §5) is implemented in + // replication.go, for the three Phase 1 verbs; the rest + // (PromoteVolume, DemoteVolume, ResyncVolume) fall through to this + // embedded default until Phase 2. + replication.UnimplementedControllerServer + // The csi-addons VolumeGroup (GroupController) service (design §14.3) is + // implemented in volumegroup.go. Its unimplemented base is embedded through + // a named wrapper (volumeGroupUnimplemented) because the replication base + // above already occupies the UnimplementedControllerServer embed name. + volumeGroupUnimplemented volumeLocks *csicommon.VolumeLocks // kubeClient reads/patches PVC annotations (host_id resolution, placement-hint // cleanup). Built once at construction and reused, and nil when no in-cluster diff --git a/csi-driver/internal/csi/controller/volume.go b/csi-driver/internal/csi/controller/volume.go index da802d331..872b73525 100644 --- a/csi-driver/internal/csi/controller/volume.go +++ b/csi-driver/internal/csi/controller/volume.go @@ -11,7 +11,9 @@ import ( "strings" "github.com/container-storage-interface/spec/lib/go/csi" + "github.com/simplyblock/atlas/errs" "github.com/simplyblock/atlas/kube" + "github.com/simplyblock/atlas/lvol" "google.golang.org/grpc/codes" "google.golang.org/grpc/status" "k8s.io/klog" @@ -199,9 +201,62 @@ func (cs *Server) DeleteVolume( return nil, classifyDeleteVolumeError(err) } + if err := cs.deleteRetiredReplicaChain(ctx, volumeID); err != nil { + klog.Errorf("failed to delete retired replication copies of volume %s: %v", volumeID, err) + return nil, classifyDeleteVolumeError(err) + } + return &csi.DeleteVolumeResponse{}, nil } +// deleteRetiredReplicaChain removes the RETIRED members of the volume's +// replication pairing chain. Ramen's S3-restore keeps every PV on the +// ORIGINAL volumeHandle across fail-overs, so the DeleteVolume that cleans up +// a retired side arrives carrying an identity whose own lvol record is +// already reaped, while the actual local copy -- the pairing's superseded +// target clone -- lives on untouched (confirmed live 2026-09-24, relocate +// M-02's round trip: cluster B kept clone 6102a48e, policy still attached, +// after its PV was deleted). Walking source->target and deleting every member +// that is NOT the pairing's active volume removes exactly those leftovers. +// +// The active volume is never deleted through a stale handle: the workload is +// running on it, and its own deletion arrives through this same path once no +// pairing supersedes it. A missing ActiveLvolID (backend predating the field) +// deletes nothing, erring toward leaking a clone over destroying live data. +func (cs *Server) deleteRetiredReplicaChain(ctx context.Context, volumeID string) error { + h, err := csicommon.ParseVolumeHandle(volumeID) + if err != nil { + return nil // unparsable handles were already tolerated as deleted above + } + cur := h + for range 8 { // one hop per past fail-over; capped far above any real chain + client, err := clusters.ReplicationClient(ctx, cur.ClusterID) + if err != nil { + return err + } + rel, err := client.GetVolumeReplicationRelationship(ctx, cur.Handle()) + if err != nil { + if errors.Is(err, errs.ErrNotFound) { + return nil // no pairing: an ordinary volume, nothing retired to clean + } + return err + } + if rel.ActiveLvolID == "" || rel.TargetLvolID == "" || rel.TargetLvolID == rel.ActiveLvolID { + return nil + } + target := &lvol.Handle{ClusterID: rel.TargetClusterID, PoolRef: rel.TargetPoolID, VolumeID: rel.TargetLvolID} + targetClient, err := clusters.ReplicationClient(ctx, target.ClusterID) + if err != nil { + return err + } + if err := targetClient.DeleteVolume(ctx, target.Handle()); err != nil { + return err + } + cur = target + } + return nil +} + func (cs *Server) prepareCreateVolumeReq( ctx context.Context, req *csi.CreateVolumeRequest, diff --git a/csi-driver/internal/csi/controller/volume_test.go b/csi-driver/internal/csi/controller/volume_test.go index c35d0f698..67e16f60e 100644 --- a/csi-driver/internal/csi/controller/volume_test.go +++ b/csi-driver/internal/csi/controller/volume_test.go @@ -100,6 +100,87 @@ func TestDeleteVolume_ControlPlaneErrorMapping(t *testing.T) { }) } +// testReplActiveVolumeID stands in for the volume a relocate round trip +// leaves actually serving the workload -- the SECOND hop's clone, which the +// FIRST pairing's records know only as active_lvol_id. +const testReplActiveVolumeID = "88888888-8888-8888-8888-888888888890" + +// Regression: 2026-09-24-delete-foreign-handle-leak — Ramen keeps every PV on +// the ORIGINAL volumeHandle across fail-overs, so the DeleteVolume that +// cleans up a retired side arrives carrying an identity whose own lvol record +// is already reaped, while the actual local copy -- the pairing's superseded +// target clone -- lives on untouched (confirmed live 2026-09-24, relocate +// M-02 round trip: cluster B kept clone 6102a48e, with the replication policy +// still attached to it, after its PV was deleted). DeleteVolume must follow +// the relationship and remove the retired, non-active members. +func TestDeleteVolumeRemovesTheRetiredReplicaBehindAForeignHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-retired-clone", Size: 1 << 30, + } + mock.volumes[testReplActiveVolumeID] = &mockVolume{ + UUID: testReplActiveVolumeID, Name: "repl-vol-active", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + "active": "target", "active_lvol_id": testReplActiveVolumeID, + } + delete(mock.volumes, testReplVolumeID) // the original's record, reaped after fail-over + + _, err := cs.DeleteVolume(context.Background(), &csi.DeleteVolumeRequest{VolumeId: testReplVolID}) + if err != nil { + t.Fatal(err) + } + if _, ok := mock.volumes[testReplTargetVolumeID]; ok { + t.Error("the retired clone still exists: DeleteVolume never followed the relationship to it") + } + if _, ok := mock.volumes[testReplActiveVolumeID]; !ok { + t.Error("the ACTIVE volume was deleted: the workload was running on it") + } +} + +// The safety half of the same contract, pinned so the cleanup above can never +// be "fixed" into deleting the live side: when the pairing's target IS the +// active volume (a fail-over whose destination still serves the workload), +// a DeleteVolume carrying the dead source handle must delete nothing. +func TestDeleteVolumeNeverDeletesTheActiveReplicaThroughADeadHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newReplicationTestServer(t, mock) + mock.volumes[testReplTargetVolumeID] = &mockVolume{ + UUID: testReplTargetVolumeID, Name: "repl-vol-live", Size: 1 << 30, + } + mock.replicationRelationship[testReplVolumeID] = map[string]any{ + "replication_id": "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee", + "direction": "to_target", + "mode": "failover", + "state": "failed_over", + "is_source": true, + "source_cluster_id": sanityClusterID, "source_lvol_id": testReplVolumeID, + "target_cluster_id": sanityClusterID, "target_pool_id": sanityPoolUUID, "target_lvol_id": testReplTargetVolumeID, + "target_nqn": "nqn.test", "target_ns_id": 1, + "active": "target", "active_lvol_id": testReplTargetVolumeID, + } + delete(mock.volumes, testReplVolumeID) + + _, err := cs.DeleteVolume(context.Background(), &csi.DeleteVolumeRequest{VolumeId: testReplVolID}) + if err != nil { + t.Fatal(err) + } + if _, ok := mock.volumes[testReplTargetVolumeID]; !ok { + t.Error("the ACTIVE volume was deleted through the dead source handle") + } +} + // TestControllerExpandVolume_ControlPlaneErrorMapping drives ControllerExpandVolume // through every control-plane response. func TestControllerExpandVolume_ControlPlaneErrorMapping(t *testing.T) { diff --git a/csi-driver/internal/csi/controller/volumegroup.go b/csi-driver/internal/csi/controller/volumegroup.go new file mode 100644 index 000000000..e3ee4b0ed --- /dev/null +++ b/csi-driver/internal/csi/controller/volumegroup.go @@ -0,0 +1,164 @@ +// The csi-addons VolumeGroup (GroupController) service (design +// design-csi-addons-replication.md §14.3): the stock kubernetes-csi-addons +// controller-manager dials it to form a backend consistency group before +// replicating it as one unit. CreateVolumeGroup resolves the group the member +// volumes already belong to (they joined at provisioning by the +// storage.simplyblock.io/consistency-group label) and hands back a group handle +// the Replication verbs route on. Membership and lifecycle are owned by that +// label, not by this service, so ModifyVolumeGroupMembership and +// DeleteVolumeGroup never reshape or delete the backend group. +package controller + +import ( + "context" + + "github.com/csi-addons/spec/lib/go/volumegroup" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" + + "github.com/simplyblock/atlas/lvol" + + "github.com/simplyblock/csi-driver/internal/clusters" +) + +// volumeGroupUnimplemented embeds the csi-addons VolumeGroup unimplemented server +// so Server can satisfy the service's forward-compat guard under a name that does +// not collide with the replication base's own UnimplementedControllerServer. +type volumeGroupUnimplemented struct { + volumegroup.UnimplementedControllerServer +} + +// CreateVolumeGroup resolves the backend consistency group the given member +// volumes belong to and returns its group handle (design §14.3). It creates no +// group of its own: the group is label-formed at provisioning, so this is +// idempotent and returns the existing group's handle. +func (cs *Server) CreateVolumeGroup( + ctx context.Context, + req *volumegroup.CreateVolumeGroupRequest, +) (*volumegroup.CreateVolumeGroupResponse, error) { + clusterID, lvolIDs, err := parseGroupMembers(req.GetVolumeIds()) + if err != nil { + return nil, err + } + client, err := clusters.ReplicationClient(ctx, clusterID) + if err != nil { + return nil, status.Error(codes.Unavailable, err.Error()) + } + groupID, err := client.ConsistencyGroupForLvols(ctx, clusterID, lvolIDs) + if err != nil { + // The PVs may keep the handles of volumes a relocate replaced: their + // data lives at the end of each relationship chain, in the group the + // move formed there. Group that live set instead. + if gh, ok := liveGroupForMembers(ctx, req.GetVolumeIds()); ok { + return &volumegroup.CreateVolumeGroupResponse{ + VolumeGroup: &volumegroup.VolumeGroup{VolumeGroupId: string(gh.Handle())}, + }, nil + } + // No backend group matches the selection exactly: the members are not + // one whole consistency group (design §14.3, the admission webhook's + // invariant), so refuse rather than group a partial set. + return nil, status.Error(codes.FailedPrecondition, err.Error()) + } + gh := lvol.GroupHandle{ClusterID: clusterID, GroupID: groupID} + return &volumegroup.CreateVolumeGroupResponse{ + VolumeGroup: &volumegroup.VolumeGroup{VolumeGroupId: string(gh.Handle())}, + }, nil +} + +// ModifyVolumeGroupMembership is a success no-op: a group's membership is owned +// by the storage.simplyblock.io/consistency-group label at provisioning, and +// dynamic membership through this RPC is deferred (design §14.3, +// design-consistency-groups.md Phase 4). Returning success also lets the +// controller-manager's teardown, which empties a group before deleting it, +// proceed without error. +func (cs *Server) ModifyVolumeGroupMembership( + _ context.Context, + req *volumegroup.ModifyVolumeGroupMembershipRequest, +) (*volumegroup.ModifyVolumeGroupMembershipResponse, error) { + if _, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(req.GetVolumeGroupId())); !ok { + return nil, status.Errorf(codes.InvalidArgument, + "not a consistency-group handle: %q", req.GetVolumeGroupId()) + } + return &volumegroup.ModifyVolumeGroupMembershipResponse{ + VolumeGroup: &volumegroup.VolumeGroup{VolumeGroupId: req.GetVolumeGroupId()}, + }, nil +} + +// DeleteVolumeGroup is a success no-op: the backend consistency group is +// label-formed and lives as long as a labeled member exists, so deleting the +// csi-addons grouping never deletes the backend group or its volumes (design +// §14.3). Idempotent. +func (cs *Server) DeleteVolumeGroup( + _ context.Context, + req *volumegroup.DeleteVolumeGroupRequest, +) (*volumegroup.DeleteVolumeGroupResponse, error) { + if _, ok := lvol.ParseGroupHandle(lvol.VolumeHandle(req.GetVolumeGroupId())); !ok { + return nil, status.Errorf(codes.InvalidArgument, + "not a consistency-group handle: %q", req.GetVolumeGroupId()) + } + return &volumegroup.DeleteVolumeGroupResponse{}, nil +} + +// parseGroupMembers parses the member volume handles of a CreateVolumeGroup +// request into a shared cluster id and the member lvol ids, rejecting a +// malformed handle or members that span clusters. +// liveGroupForMembers resolves each PV handle through its relationship chain to +// the volume serving it now and returns the consistency group that holds +// exactly those volumes, when they all live in one cluster and form one group. +func liveGroupForMembers(ctx context.Context, volumeIDs []string) (lvol.GroupHandle, bool) { + cluster := "" + ids := make([]string, 0, len(volumeIDs)) + for _, vid := range volumeIDs { + h, ok := lvol.ParseHandle(lvol.VolumeHandle(vid)) + if !ok { + return lvol.GroupHandle{}, false + } + client, err := clusters.ReplicationClient(ctx, h.ClusterID) + if err != nil { + return lvol.GroupHandle{}, false + } + hops, _, err := resolveChain(ctx, &h, client) + if err != nil || len(hops) == 0 { + return lvol.GroupHandle{}, false + } + end := hops[len(hops)-1].h + if cluster != "" && end.ClusterID != cluster { + return lvol.GroupHandle{}, false + } + cluster = end.ClusterID + ids = append(ids, end.VolumeID) + } + if cluster == "" { + return lvol.GroupHandle{}, false + } + client, err := clusters.ReplicationClient(ctx, cluster) + if err != nil { + return lvol.GroupHandle{}, false + } + groupID, err := client.ConsistencyGroupForLvols(ctx, cluster, ids) + if err != nil { + return lvol.GroupHandle{}, false + } + return lvol.GroupHandle{ClusterID: cluster, GroupID: groupID}, true +} + +func parseGroupMembers(volumeIDs []string) (clusterID string, lvolIDs []string, err error) { + if len(volumeIDs) == 0 { + return "", nil, status.Error(codes.InvalidArgument, + "CreateVolumeGroup requires at least one volume") + } + lvolIDs = make([]string, 0, len(volumeIDs)) + for _, vid := range volumeIDs { + h, ok := lvol.ParseHandle(lvol.VolumeHandle(vid)) + if !ok { + return "", nil, status.Errorf(codes.InvalidArgument, "malformed volume handle %q", vid) + } + if clusterID != "" && h.ClusterID != clusterID { + return "", nil, status.Error(codes.InvalidArgument, + "a volume group's members must all live in one cluster") + } + clusterID = h.ClusterID + lvolIDs = append(lvolIDs, h.VolumeID) + } + return clusterID, lvolIDs, nil +} diff --git a/csi-driver/internal/csi/controller/volumegroup_test.go b/csi-driver/internal/csi/controller/volumegroup_test.go new file mode 100644 index 000000000..095aa2206 --- /dev/null +++ b/csi-driver/internal/csi/controller/volumegroup_test.go @@ -0,0 +1,130 @@ +package controller + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/volumegroup" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +const ( + vgGroupID = "c9c9c9c9-c9c9-4c9c-8c9c-c9c9c9c9c9c9" + vgMember1 = "a1111111-1111-4111-8111-111111111111" + vgMember2 = "b2222222-2222-4222-8222-222222222222" +) + +func vgHandle(volumeID string) string { + return sanityClusterID + ":" + sanityPoolUUID + ":" + volumeID +} + +// CreateVolumeGroup resolves the backend consistency group its member volumes +// already belong to, and returns that group's handle (design §14.3). +func TestCreateVolumeGroupResolvesTheBackendGroup(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + mock.seedGroup(vgGroupID, vgMember1, vgMember2) + cs := newTestControllerServer(t, mock) + + resp, err := cs.CreateVolumeGroup(context.Background(), &volumegroup.CreateVolumeGroupRequest{ + Name: "vgrcontent-generated-name", + VolumeIds: []string{vgHandle(vgMember1), vgHandle(vgMember2)}, + }) + if err != nil { + t.Fatalf("CreateVolumeGroup: %v", err) + } + want := "cg:" + sanityClusterID + ":" + vgGroupID + if got := resp.GetVolumeGroup().GetVolumeGroupId(); got != want { + t.Fatalf("group handle = %q, want %q", got, want) + } +} + +// A selection that is not exactly one backend group's membership must not be +// silently grouped. +func TestCreateVolumeGroupRefusesAPartialSelection(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + mock.seedGroup(vgGroupID, vgMember1, vgMember2) + cs := newTestControllerServer(t, mock) + + _, err := cs.CreateVolumeGroup(context.Background(), &volumegroup.CreateVolumeGroupRequest{ + VolumeIds: []string{vgHandle(vgMember1)}, // only one of the two members + }) + if status.Code(err) != codes.FailedPrecondition { + t.Fatalf("err = %v, want FailedPrecondition", err) + } +} + +func TestCreateVolumeGroupRejectsAMalformedHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + _, err := cs.CreateVolumeGroup(context.Background(), &volumegroup.CreateVolumeGroupRequest{ + VolumeIds: []string{"not-a-volume-handle"}, + }) + if status.Code(err) != codes.InvalidArgument { + t.Fatalf("err = %v, want InvalidArgument", err) + } +} + +func TestCreateVolumeGroupRequiresAVolume(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + _, err := cs.CreateVolumeGroup(context.Background(), &volumegroup.CreateVolumeGroupRequest{}) + if status.Code(err) != codes.InvalidArgument { + t.Fatalf("err = %v, want InvalidArgument", err) + } +} + +// ModifyVolumeGroupMembership is a success no-op (membership is label-driven, +// design §14.3), so the controller-manager's teardown does not error. +func TestModifyVolumeGroupMembershipIsANoOpSuccess(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + gh := "cg:" + sanityClusterID + ":" + vgGroupID + resp, err := cs.ModifyVolumeGroupMembership(context.Background(), + &volumegroup.ModifyVolumeGroupMembershipRequest{VolumeGroupId: gh}) + if err != nil { + t.Fatalf("ModifyVolumeGroupMembership: %v", err) + } + if got := resp.GetVolumeGroup().GetVolumeGroupId(); got != gh { + t.Fatalf("group handle = %q, want %q", got, gh) + } +} + +// DeleteVolumeGroup is a success no-op: the backend group is label-formed and +// survives the csi-addons grouping (design §14.3). +func TestDeleteVolumeGroupIsANoOpSuccess(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + + gh := "cg:" + sanityClusterID + ":" + vgGroupID + if _, err := cs.DeleteVolumeGroup(context.Background(), + &volumegroup.DeleteVolumeGroupRequest{VolumeGroupId: gh}); err != nil { + t.Fatalf("DeleteVolumeGroup: %v", err) + } +} + +func TestVolumeGroupVerbsRejectANonGroupHandle(t *testing.T) { + mock := newMockSBCLI() + defer mock.Close() + cs := newTestControllerServer(t, mock) + perVolume := vgHandle(vgMember1) // a per-volume handle, not a group handle + + _, err := cs.ModifyVolumeGroupMembership(context.Background(), + &volumegroup.ModifyVolumeGroupMembershipRequest{VolumeGroupId: perVolume}) + if status.Code(err) != codes.InvalidArgument { + t.Errorf("Modify: err = %v, want InvalidArgument", err) + } + if _, err := cs.DeleteVolumeGroup(context.Background(), + &volumegroup.DeleteVolumeGroupRequest{VolumeGroupId: perVolume}); status.Code(err) != codes.InvalidArgument { + t.Errorf("Delete: err = %v, want InvalidArgument", err) + } +} diff --git a/csi-driver/internal/csi/csiaddons/identity/identity.go b/csi-driver/internal/csi/csiaddons/identity/identity.go new file mode 100644 index 000000000..4f4bf3bff --- /dev/null +++ b/csi-driver/internal/csi/csiaddons/identity/identity.go @@ -0,0 +1,99 @@ +// Package identity serves the csi-addons Identity service: GetIdentity, +// GetCapabilities (advertising VOLUME_REPLICATION), and Probe. It is a +// distinct service from the CSI spec's own Identity (served by +// internal/csi/identity), not an extension of it, because the +// kubernetes-csi-addons controller-manager and the CSI sidecars probe two +// separate protocols on the same socket. +package identity + +import ( + "context" + + "github.com/csi-addons/spec/lib/go/identity" +) + +// Server implements the csi-addons Identity service. +type Server struct { + identity.UnimplementedIdentityServer + name string + version string +} + +// New returns an identity.Server reporting name and version as this driver's +// own (the same values the CSI spec's Identity service reports), since both +// protocols identify the one plugin process serving them. +func New(name, version string) *Server { + return &Server{name: name, version: version} +} + +func (s *Server) GetIdentity(context.Context, *identity.GetIdentityRequest) (*identity.GetIdentityResponse, error) { + return &identity.GetIdentityResponse{Name: s.name, VendorVersion: s.version}, nil +} + +func (s *Server) GetCapabilities( + context.Context, *identity.GetCapabilitiesRequest, +) (*identity.GetCapabilitiesResponse, error) { + return &identity.GetCapabilitiesResponse{ + Capabilities: []*identity.Capability{ + { + Type: &identity.Capability_Service_{ + Service: &identity.Capability_Service{ + Type: identity.Capability_Service_CONTROLLER_SERVICE, + }, + }, + }, + { + Type: &identity.Capability_VolumeReplication_{ + VolumeReplication: &identity.Capability_VolumeReplication{ + Type: identity.Capability_VolumeReplication_VOLUME_REPLICATION, + }, + }, + }, + // GetReplicationDestinationInfo (kubernetes-csi-addons >= v0.15.0): + // without it the VolumeGroupReplicationContent's per-volume + // destinations stay empty and Ramen cannot restore a consistency + // group on the target (2026-10-03). + { + Type: &identity.Capability_VolumeReplication_{ + VolumeReplication: &identity.Capability_VolumeReplication{ + Type: identity.Capability_VolumeReplication_GET_REPLICATION_DESTINATION_INFO, + }, + }, + }, + // The VolumeGroup service (design §14.3), which the stock + // kubernetes-csi-addons controller-manager dials to form a backend + // consistency group before replicating it as one unit. + { + Type: &identity.Capability_VolumeGroup_{ + VolumeGroup: &identity.Capability_VolumeGroup{ + Type: identity.Capability_VolumeGroup_VOLUME_GROUP, + }, + }, + }, + { + Type: &identity.Capability_VolumeGroup_{ + VolumeGroup: &identity.Capability_VolumeGroup{ + Type: identity.Capability_VolumeGroup_MODIFY_VOLUME_GROUP, + }, + }, + }, + // DeleteVolumeGroup dissolves the group but keeps its member volumes + // (design §14.3): a group is a label-formed set, never an owner of + // the volumes' lifecycle. + { + Type: &identity.Capability_VolumeGroup_{ + VolumeGroup: &identity.Capability_VolumeGroup{ + Type: identity.Capability_VolumeGroup_DO_NOT_ALLOW_VG_TO_DELETE_VOLUMES, + }, + }, + }, + }, + }, nil +} + +func (s *Server) Probe(context.Context, *identity.ProbeRequest) (*identity.ProbeResponse, error) { + // Ready left nil: per the spec, absent means "assume ready." The + // Replication service's RPCs are stateless and idempotent by contract + // (design §5), so there is no initialization phase to report against. + return &identity.ProbeResponse{}, nil +} diff --git a/csi-driver/internal/csi/csiaddons/identity/identity_test.go b/csi-driver/internal/csi/csiaddons/identity/identity_test.go new file mode 100644 index 000000000..ffbdd0956 --- /dev/null +++ b/csi-driver/internal/csi/csiaddons/identity/identity_test.go @@ -0,0 +1,103 @@ +package identity + +import ( + "context" + "testing" + + "github.com/csi-addons/spec/lib/go/identity" +) + +func TestGetIdentity(t *testing.T) { + s := New("test.csi.simplyblock.io", "v1.2.3") + resp, err := s.GetIdentity(context.Background(), &identity.GetIdentityRequest{}) + if err != nil { + t.Fatal(err) + } + if resp.Name != "test.csi.simplyblock.io" || resp.VendorVersion != "v1.2.3" { + t.Errorf("GetIdentity = %+v", resp) + } +} + +func TestGetCapabilitiesAdvertisesVolumeReplication(t *testing.T) { + s := New("test.csi.simplyblock.io", "v1.2.3") + resp, err := s.GetCapabilities(context.Background(), &identity.GetCapabilitiesRequest{}) + if err != nil { + t.Fatal(err) + } + var sawVolumeReplication, sawControllerService bool + for _, c := range resp.Capabilities { + if vr := c.GetVolumeReplication(); vr != nil && vr.Type == identity.Capability_VolumeReplication_VOLUME_REPLICATION { + sawVolumeReplication = true + } + if svc := c.GetService(); svc != nil && svc.Type == identity.Capability_Service_CONTROLLER_SERVICE { + sawControllerService = true + } + } + if !sawVolumeReplication { + t.Error("capabilities do not advertise VOLUME_REPLICATION") + } + if !sawControllerService { + t.Error("capabilities do not advertise CONTROLLER_SERVICE") + } +} + +func TestGetCapabilitiesAdvertisesVolumeGroup(t *testing.T) { + s := New("test.csi.simplyblock.io", "v1.2.3") + resp, err := s.GetCapabilities(context.Background(), &identity.GetCapabilitiesRequest{}) + if err != nil { + t.Fatal(err) + } + var sawVolumeGroup, sawDoNotDeleteVolumes bool + for _, c := range resp.Capabilities { + vg := c.GetVolumeGroup() + if vg == nil { + continue + } + switch vg.Type { + case identity.Capability_VolumeGroup_VOLUME_GROUP: + sawVolumeGroup = true + case identity.Capability_VolumeGroup_DO_NOT_ALLOW_VG_TO_DELETE_VOLUMES: + sawDoNotDeleteVolumes = true + } + } + if !sawVolumeGroup { + t.Error("capabilities do not advertise VOLUME_GROUP") + } + // DeleteVolumeGroup dissolves the group but keeps its member volumes + // (design §14.3), which is exactly what this capability promises. + if !sawDoNotDeleteVolumes { + t.Error("capabilities do not advertise DO_NOT_ALLOW_VG_TO_DELETE_VOLUMES") + } +} + +func TestProbeReportsReady(t *testing.T) { + s := New("test.csi.simplyblock.io", "v1.2.3") + resp, err := s.Probe(context.Background(), &identity.ProbeRequest{}) + if err != nil { + t.Fatal(err) + } + // Ready left nil means "assume ready" per the spec; asserting nil pins + // that choice rather than a stray true/false creeping in later. + if resp.Ready != nil { + t.Errorf("Ready = %v, want nil (assume ready)", resp.Ready) + } +} + +// Without GET_REPLICATION_DESTINATION_INFO the csi-addons v0.15 controller never +// asks for destinations, a group's VolumeGroupReplicationContent keeps empty +// destination handles, and Ramen cannot restore the group on the target +// (2026-10-03). +func TestGetCapabilitiesAdvertisesReplicationDestinationInfo(t *testing.T) { + s := New("test.csi.simplyblock.io", "v1.2.3") + resp, err := s.GetCapabilities(context.Background(), &identity.GetCapabilitiesRequest{}) + if err != nil { + t.Fatal(err) + } + for _, c := range resp.Capabilities { + if vr := c.GetVolumeReplication(); vr != nil && + vr.Type == identity.Capability_VolumeReplication_GET_REPLICATION_DESTINATION_INFO { + return + } + } + t.Error("capabilities do not advertise GET_REPLICATION_DESTINATION_INFO") +} diff --git a/csi-driver/internal/csi/node/indirection_test.go b/csi-driver/internal/csi/node/indirection_test.go new file mode 100644 index 000000000..9985886a9 --- /dev/null +++ b/csi-driver/internal/csi/node/indirection_test.go @@ -0,0 +1,81 @@ +package node + +import ( + "context" + "strings" + "testing" +) + +// A volume staged behind the dm indirection is torn down through it: the +// record names the layer, and the teardown walks it. +func TestTeardownPlanOfAnIndirectVolumeWalksTheMapping(t *testing.T) { + for _, layers := range [][]string{ + {"fabric", "dmLinear"}, + {"fabric", "dmLinear", "filesystem"}, + } { + ns, _ := newStackedServer(t, newRecordingRunner()) + writeRecord(t, ns.stack, pvcTestHandle, layers) + plan, err := ns.teardownPlan(context.Background(), pvcTestHandle, "/staging", stagedContext()) + if err != nil { + t.Fatalf("teardownPlan(%v): %v", layers, err) + } + if got, want := strings.Join(plan.Names(), " → "), strings.Join(layers, " → "); got != want { + t.Errorf("the teardown walks %s, want %s", got, want) + } + } +} + +// A fresh stage follows SPDKCSI_DM_INDIRECTION; a staged volume follows its +// record, so the layer is never inserted under (or pulled from under) a live +// consumer by a later heal. +// indirectHandle is a second volume, so the record of one never answers for +// another. +const indirectHandle = "0c5b4c4e-6f1d-4b8e-9d2a-3a1f2b7c9e10:pool-1:7d1e0f5a-2b3c-4d5e-8f90-a1b2c3d4e5f6" + +func TestAttachShapeFollowsTheFlagOnlyForAFreshStage(t *testing.T) { + ns, _ := newStackedServer(t, newRecordingRunner()) + + t.Setenv("SPDKCSI_DM_INDIRECTION", "") + if got := ns.attachShape(pvcTestHandle, shapePlain); got != shapePlain { + t.Errorf("flag off, no record: shape %v, want plain", got) + } + t.Setenv("SPDKCSI_DM_INDIRECTION", "true") + if got := ns.attachShape(pvcTestHandle, shapePlain); got != shapeIndirectPlain { + t.Errorf("flag on, no record: shape %v, want indirect plain", got) + } + if got := ns.attachShape(pvcTestHandle, shapeRawBlock); got != shapeIndirectRawBlock { + t.Errorf("flag on, no record: shape %v, want indirect raw block", got) + } + if got := ns.attachShape(pvcTestHandle, shapeLVM); got != shapeLVM { + t.Errorf("an LVM stack has no indirect variant, got %v", got) + } + + writeRecord(t, ns.stack, indirectHandle, []string{"fabric", "filesystem"}) + if got := ns.attachShape(indirectHandle, shapePlain); got != shapePlain { + t.Errorf("flag on, staged without the layer: shape %v, want plain", got) + } + + t.Setenv("SPDKCSI_DM_INDIRECTION", "") + writeRecord(t, ns.stack, indirectHandle, []string{"fabric", "dmLinear", "filesystem"}) + if got := ns.attachShape(indirectHandle, shapePlain); got != shapeIndirectPlain { + t.Errorf("flag off, staged with the layer: shape %v, want indirect plain", got) + } +} + +func TestPlanForBuildsTheIndirectRows(t *testing.T) { + s, _ := newTestStack(t, newRecordingRunner()) + node := s.node("", nil) + vc := stagedContext() + volume := stackVolume("/staging", vc, mountCapability()) + build := func(shape stackShape) string { + return strings.Join(planFor(node, connectionFromContext(vc), volume, vdoOptions(vc), shape).Names(), " → ") + } + got := build(shapeIndirectPlain) + if got != "fabric → dmLinear → filesystem" { + t.Errorf("indirect plain = %s", got) + } + got = build(shapeIndirectRawBlock) + if got != "fabric → dmLinear" { + t.Errorf("indirect raw block = %s", got) + } +} diff --git a/csi-driver/internal/csi/node/plan.go b/csi-driver/internal/csi/node/plan.go index e0dcaa747..fbe2b86cc 100644 --- a/csi-driver/internal/csi/node/plan.go +++ b/csi-driver/internal/csi/node/plan.go @@ -14,6 +14,7 @@ package node import ( "encoding/json" "fmt" + "os" "slices" "strconv" "strings" @@ -43,6 +44,7 @@ const ( layerLVMPhysicalVolume = "lvmPhysicalVolume" layerLVMVolumeGroup = "lvmVolumeGroup" layerLVMLogicalVolume = "lvmLogicalVolume" + layerDMLinear = "dmLinear" ) // vdoPoolName is the pool `lvcreate --type vdo` creates alongside the logical @@ -76,6 +78,15 @@ const ( // asked for client-side compression or deduplication and is opened as a // block device. shapeLVMRawBlock + + // shapeIndirectRawBlock and shapeIndirectPlain are shapeRawBlock and + // shapePlain with the dmLinear indirection above the fabric, so a volume's + // namespace can move to another subsystem under a live consumer + // (consistency-group co-location). Chosen for a fresh stage when the node + // runs with SPDKCSI_DM_INDIRECTION, and afterwards by the stack record: + // a volume never gains or loses the layer under a staged consumer. + shapeIndirectRawBlock + shapeIndirectPlain ) // planFor is the layer list one of the shapes means, built with the seams the @@ -94,6 +105,10 @@ func planFor( switch shape { case shapeRawBlock: return node.RawBlock(connection) + case shapeIndirectRawBlock: + return node.IndirectRawBlock(connection, volume) + case shapeIndirectPlain: + return node.IndirectPlain(connection, volume) case shapeLVM: return node.LVM(connection, volume, options) case shapeLVMRawBlock: @@ -107,6 +122,29 @@ func planFor( // capability decides whether there is a filesystem, and the class parameters // decide whether the LVM layers that provide client-side compression and // deduplication sit between it and the fabric. +// indirect turns a raw block or plain shape into its dmLinear variant. Only +// those two have one: an LVM stack already re-points through its own device +// mapper nodes, and is left as it is. +func indirect(shape stackShape) stackShape { + switch shape { + case shapeRawBlock: + return shapeIndirectRawBlock + case shapePlain: + return shapeIndirectPlain + } + return shape +} + +// dmIndirectionEnabled is SPDKCSI_DM_INDIRECTION: stage new raw block and +// plain volumes behind the dmLinear indirection. Off by default. +func dmIndirectionEnabled() bool { + switch strings.ToLower(strings.TrimSpace(os.Getenv("SPDKCSI_DM_INDIRECTION"))) { + case "1", "true", "yes", "on": + return true + } + return false +} + func shapeFor(vc map[string]string, volCap *csi.VolumeCapability) stackShape { block := volCap.GetBlock() != nil switch { @@ -159,6 +197,10 @@ func stackVolume( vc map[string]string, volCap *csi.VolumeCapability, ) plans.Volume { + // Named by the record or the capability; empty otherwise, so the layer + // mounts what the device carries (a static PV without fsType: a test + // fail-over's clone, a PV Ramen restored without the field) and formats + // a blank device as ext4. fsType := stagedFsType(vc, volCap) return plans.Volume{ UUID: deviceLvolID(vc), @@ -167,8 +209,9 @@ func stackVolume( PVCName: vc[csicommon.CSIStorageNameKey], StagingPath: stagingPath, FsType: fsType, + DefaultFsType: "ext4", MountFlags: volumeMountFlags(volCap), - FormatOptions: mount.FormatOptions(fsType, vc, wantsVDO(vc)), + FormatOptions: mount.FormatOptions(fsTypeOrDefault(volCap), vc, wantsVDO(vc)), ReservedBlocksPercent: vc["tune2fs_reserved_blocks"], Encrypted: boolFromContext(vc[csicommon.ParamEncryption]), } @@ -397,6 +440,8 @@ var recordedShapes = []struct { }{ {[]string{layerFabric}, shapeRawBlock}, {[]string{layerFabric, layerFilesystem}, shapePlain}, + {[]string{layerFabric, layerDMLinear}, shapeIndirectRawBlock}, + {[]string{layerFabric, layerDMLinear, layerFilesystem}, shapeIndirectPlain}, { []string{layerFabric, layerLVMPhysicalVolume, layerLVMVolumeGroup, layerLVMLogicalVolume}, shapeLVMRawBlock, @@ -416,6 +461,7 @@ var knownLayers = map[string]bool{ layerLVMPhysicalVolume: true, layerLVMVolumeGroup: true, layerLVMLogicalVolume: true, + layerDMLinear: true, } // shapeFromRecord is the plan shape a recorded layer list describes. diff --git a/csi-driver/internal/csi/node/stack_test.go b/csi-driver/internal/csi/node/stack_test.go index 94df0cfc8..bf174774f 100644 --- a/csi-driver/internal/csi/node/stack_test.go +++ b/csi-driver/internal/csi/node/stack_test.go @@ -18,6 +18,8 @@ import ( "testing" "github.com/container-storage-interface/spec/lib/go/csi" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" corev1 "k8s.io/api/core/v1" k8smount "k8s.io/mount-utils" @@ -285,6 +287,37 @@ func TestStageBringsTheStackUp(t *testing.T) { } } +// Regression: 2026-09-29-testfailover-nil-volumecontext — a statically +// provisioned PV (the TestFailover bubble PV) carries no csi.volumeAttributes, so +// NodeStageVolume received a nil VolumeContext and panicked with "assignment to +// entry in nil map" at the first vc[...] write. The node plugin crash-looped, its +// socket refused connections, and no pod could mount the clone. Staging must +// tolerate a nil VolumeContext: the volume's identity is re-resolved from its +// handle regardless. +func TestStageToleratesNilVolumeContext(t *testing.T) { + runner := newRecordingRunner() + ns, _ := newStackedServer(t, runner) + + req := &csi.NodeStageVolumeRequest{ + VolumeId: pvcTestHandle, + StagingTargetPath: t.TempDir(), + VolumeCapability: mountCapability(), + VolumeContext: nil, // a static PV with no volumeAttributes + } + + // Before the fix this panicked on the nil map at the first vc[...] write. It + // must return instead. With no control plane in this unit harness to resolve + // the volume's identity from its handle, staging fails cleanly (fail-safe) + // rather than crashing the node plugin or attaching the wrong target. + _, err := ns.NodeStageVolume(context.Background(), req) + if status.Code(err) != codes.Internal { + t.Fatalf("want a clean Internal error for an unresolvable nil-context stage, got %v", err) + } + if runner.called("up") { + t.Fatalf("stage attached a target with no resolved subsystem: %v", runner.calls) + } +} + // An unstage releases and never destroys. It fires whenever no pod on this node // needs the volume mounted, which includes an ordinary pod restart, and // conflating the two verbs is what once ran vgremove over a volume holding diff --git a/csi-driver/internal/csi/node/stage.go b/csi-driver/internal/csi/node/stage.go index bfbd4af4d..c5d59b43d 100644 --- a/csi-driver/internal/csi/node/stage.go +++ b/csi-driver/internal/csi/node/stage.go @@ -18,6 +18,7 @@ import ( "encoding/json" "errors" "fmt" + "slices" "strings" "time" @@ -80,6 +81,13 @@ func (ns *Server) NodeStageVolume( } vc := req.GetVolumeContext() + if vc == nil { + // A statically provisioned PV can carry no csi.volumeAttributes, which + // arrives here as a nil map. The volume's identity is re-resolved from its + // handle by refreshVolumeContext regardless, so an empty context is enough + // to stage; a nil one would panic on the first write below. + vc = map[string]string{} + } vc["stagingParentPath"] = stagingParentPath ns.refreshVolumeContext(ctx, volumeID, vc) @@ -95,7 +103,7 @@ func (ns *Server) NodeStageVolume( return nil, status.Error(codes.Internal, err.Error()) } - ns.rememberStagedVolume(ctx, volumeID, vc, artifact, req.GetVolumeCapability()) + ns.rememberStagedVolume(ctx, volumeID, vc, artifact, plan, req.GetVolumeCapability()) // The CSI spec passes VolumeContext to this RPC and to nothing after it, so // what the later RPCs need is written beside the staging path. @@ -280,7 +288,32 @@ func (ns *Server) attachPlan( } node := ns.stack.node(hostNQN, ns.priorFormat(volumeID, vc)) volume := stackVolume(stagingTargetPath, vc, volCap) - return planFor(node, connection, volume, vdoOptions(vc), shapeFor(vc, volCap)), nil + shape := ns.attachShape(volumeID, shapeFor(vc, volCap)) + return planFor(node, connection, volume, vdoOptions(vc), shape), nil +} + +// attachShape decides whether a stage or heal builds the dmLinear indirection. +// A volume's stack never gains or loses the layer under a staged consumer: +// with a stack record the record decides (the layer is there or it is not), +// and only a fresh stage, which has none, follows SPDKCSI_DM_INDIRECTION. A +// record that cannot be read keeps the shape without the layer, which is what +// every volume staged before the indirection existed is. +func (ns *Server) attachShape(volumeID string, shape stackShape) stackShape { + record, err := ns.stack.store.Load(volumeID) + switch { + case errors.Is(err, volstack.ErrNoRecord): + if dmIndirectionEnabled() { + return indirect(shape) + } + return shape + case err != nil: + klog.Warningf("volume %s: stack record unreadable (%v); staging without the dm indirection", volumeID, err) + return shape + } + if slices.Contains(recordedLayers(record), layerDMLinear) { + return indirect(shape) + } + return shape } // teardownPlan is the plan an unstage walks, which is the shape that was built @@ -590,7 +623,7 @@ func (ns *Server) refreshVolumeContext(ctx context.Context, volumeID string, vc // The source volume was deleted by a migration with --delete-source. // The replication relationship survives it and names the active // volume on the target cluster, which is what this redirects to. - connInfo = ns.redirectToActiveVolume(ctx, sbcClient, spdkVol.VolumeID, volumeID, vc) + connInfo = redirectToActiveVolume(ctx, sbcClient, spdkVol.VolumeID, volumeID, vc) } if connInfo == nil { klog.Warningf("failed to fetch volume connection info for %s: %v", volumeID, infoErr) @@ -612,6 +645,7 @@ func (ns *Server) rememberStagedVolume( volumeID string, vc map[string]string, artifact volstack.Artifact, + plan volstack.Plan, volCap *csi.VolumeCapability, ) { if device, ok := artifact.Device(); ok { @@ -629,10 +663,33 @@ func (ns *Server) rememberStagedVolume( // The device carries this filesystem, because the layer either put it there // or refused to stage a device carrying another. fsType := stagedFsType(vc, volCap) + if fsType == "" { + // Nobody named it: the layer mounted what the device carries, or + // formatted a blank device as its default, and knows which. + fsType = planFsType(plan) + } + if fsType == "" { + return + } vc[stagedFsTypeKey] = fsType ns.recordOnDiskFilesystem(ctx, volumeID, vc, fsType) } +// planFsType is the filesystem the plan's filesystem layer stands for after +// it acted (layers.FilesystemParams), "" when the plan has none. +func planFsType(plan volstack.Plan) string { + for _, l := range plan { + recorded, ok := l.(interface{ Params() any }) + if !ok { + continue + } + if p, ok := recorded.Params().(layers.FilesystemParams); ok { + return p.FsType + } + } + return "" +} + // priorFormat is what the volume is recorded as carrying, for the layer that // has to decide whether a device reading blank is empty or merely unreadable. // @@ -697,13 +754,15 @@ func (ns *Server) volumeIsBeingDeleted(ctx context.Context, volumeID string) boo } // stagedFsType returns the filesystem a volume was staged with: the one -// recorded at stage time when it is there, and otherwise the one the volume -// capability asks for, which is all a volume staged by an older driver has. +// recorded at stage time when it is there, else the one the volume capability +// asks for, else "" -- no opinion, which lets the filesystem layer mount what +// the device carries instead of refusing an XFS volume for not being the ext4 +// nobody asked for (a static PV without fsType, 2026-10-03). func stagedFsType(volumeContext map[string]string, volCap *csi.VolumeCapability) string { if fsType := strings.TrimSpace(volumeContext[stagedFsTypeKey]); fsType != "" { return fsType } - return fsTypeOrDefault(volCap) + return volCap.GetMount().GetFsType() } // fsTypeOrDefault returns the requested filesystem type, defaulting to ext4. diff --git a/csi-driver/internal/csi/node/stats.go b/csi-driver/internal/csi/node/stats.go index 7f7d6c80f..0e24a3085 100644 --- a/csi-driver/internal/csi/node/stats.go +++ b/csi-driver/internal/csi/node/stats.go @@ -99,48 +99,93 @@ func (ns *Server) NodeGetVolumeStats( }, nil } +// clusterClientFor resolves a cluster and pool to a control-plane client. A +// package variable, so redirectToActiveVolume's chain walk is testable with +// fake clients instead of a live secret file. +var clusterClientFor = func(ctx context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + return clusters.Client(ctx, clusterID, poolID) +} + // redirectToActiveVolume is called when VolumeInfo returns ErrVolumeNotFound for -// the source volume, typically after a migration with --delete-source removed it. -// It queries the replication relationship on the source cluster (which survives -// volume deletion) to find the active volume on the target cluster, then fetches -// connection info from the target. Returns nil if redirection is not possible. -func (ns *Server) redirectToActiveVolume( +// the source volume, typically after a migration with --delete-source removed it +// or after a fail-over retired it. It follows the replication relationship +// (which survives volume deletion) to the volume actually serving the data and +// fetches connection info from there. Returns nil if redirection is not possible. +// +// The walk follows CHAINED pairings hop by hop: a relocate round trip leaves +// original -> hop-1 clone -> hop-2 clone, where only the last hop is live +// (each record names it as active_lvol_id, resolved transitively by the +// backend). Each hop uses that record's own target triple -- its cluster, +// pool, and lvol are consistent with EACH OTHER, while active_lvol_id may +// live on an entirely different cluster than the record's target fields +// describe. Pairing the first record's active_lvol_id with its target +// cluster asked cluster B for a volume living on cluster A, fell back to a +// stale stashed context, and timed the mount out (confirmed live 2026-09-24, +// relocate M-02's round trip). +func redirectToActiveVolume( ctx context.Context, srcClient controlplane.ClusterAPI, srcLvolID, volumeID string, vc map[string]string, ) map[string]string { - rel, err := srcClient.GetRelationship(ctx, srcLvolID) - if err != nil || rel == nil { - klog.Warningf("replication relationship lookup failed for deleted volume %s: %v", volumeID, err) - return nil - } - activeLvolID := rel.ActiveLvolID - targetClusterID := rel.TargetClusterID - targetPoolID := rel.TargetPoolID - if activeLvolID == "" || targetClusterID == "" || targetPoolID == "" { - klog.Warningf("relationship for %s has incomplete target info (cluster=%s pool=%s active=%s)", - volumeID, targetClusterID, targetPoolID, activeLvolID) - return nil - } - tgtClient, err := clusters.Client(ctx, targetClusterID, targetPoolID) - if err != nil { - klog.Warningf("target cluster %s not in secret file for deleted volume %s: %v", - targetClusterID, volumeID, err) - return nil - } - connInfo, err := tgtClient.VolumeInfo(ctx, activeLvolID, vc["hostNQN"]) - if err != nil { - klog.Warningf("failed to fetch connection info from target cluster %s for volume %s: %v", - targetClusterID, activeLvolID, err) - return nil + client, lvolID := srcClient, srcLvolID + // One hop per past fail-over, and the chain never shrinks: the PV keeps + // the original handle while every relocate and fail-over appends a clone, + // so a volume moved nine times is nine hops out. A cap of 8 stranded a + // fail-over's clone behind the ninth hop and the node attached the + // partitioned original instead (2026-10-03, WordPress's fifth move of + // the day). The bound is a cycle guard now, not a length estimate. + visited := map[string]bool{} + for range maxChainHops { + if visited[lvolID] { + klog.Warningf("replication chain for deleted volume %s loops at %s", volumeID, lvolID) + return nil + } + visited[lvolID] = true + rel, err := client.GetRelationship(ctx, lvolID) + if err != nil || rel == nil { + klog.Warningf("replication relationship lookup failed for deleted volume %s (at hop %s): %v", + volumeID, lvolID, err) + return nil + } + if rel.TargetLvolID == "" || rel.TargetClusterID == "" || rel.TargetPoolID == "" { + klog.Warningf("relationship for %s has incomplete target info (cluster=%s pool=%s lvol=%s)", + volumeID, rel.TargetClusterID, rel.TargetPoolID, rel.TargetLvolID) + return nil + } + tgtClient, err := clusterClientFor(ctx, rel.TargetClusterID, rel.TargetPoolID) + if err != nil { + klog.Warningf("target cluster %s not in secret file for deleted volume %s: %v", + rel.TargetClusterID, volumeID, err) + return nil + } + if rel.ActiveLvolID != "" && rel.ActiveLvolID != rel.TargetLvolID { + // This pairing's target was itself superseded by a later + // fail-over; keep walking from it toward the active volume. + client, lvolID = tgtClient, rel.TargetLvolID + continue + } + connInfo, err := tgtClient.VolumeInfo(ctx, rel.TargetLvolID, vc["hostNQN"]) + if err != nil { + klog.Warningf("failed to fetch connection info from target cluster %s for volume %s: %v", + rel.TargetClusterID, rel.TargetLvolID, err) + return nil + } + klog.Infof("redirected deleted volume %s → active volume %s on cluster %s", + volumeID, rel.TargetLvolID, rel.TargetClusterID) + // Override cluster_id and poolID so the initiator uses the active + // volume's cluster for any subsequent API calls. Without this the + // initiator inherits the source cluster_id from vc and fails looking + // up the volume there. + connInfo[csicommon.ParamClusterID] = rel.TargetClusterID + connInfo["poolID"] = rel.TargetPoolID + return connInfo } - klog.Infof("redirected deleted volume %s → active volume %s on cluster %s", - volumeID, activeLvolID, targetClusterID) - // Override cluster_id and poolID so the initiator uses the target cluster - // for any subsequent API calls. Without this the initiator inherits the - // source cluster_id from vc and fails looking up the target volume there. - connInfo[csicommon.ParamClusterID] = targetClusterID - connInfo["poolID"] = targetPoolID - return connInfo + klog.Warningf("replication chain for deleted volume %s did not converge within %d hops", volumeID, maxChainHops) + return nil } + +// maxChainHops bounds a replication-chain walk. A chain grows by one member +// per move and is never compacted, so this is a guard against a looping +// record, far above any chain a volume accumulates in its lifetime. +const maxChainHops = 256 diff --git a/csi-driver/internal/csi/node/stats_test.go b/csi-driver/internal/csi/node/stats_test.go new file mode 100644 index 000000000..2ddc16854 --- /dev/null +++ b/csi-driver/internal/csi/node/stats_test.go @@ -0,0 +1,256 @@ +// The redirect-to-active-volume walk: how a NodeStage/NodeGetVolumeStats call +// carrying a volume handle whose own lvol record is gone finds the volume that +// actually serves the data, across one fail-over or a whole relocate round +// trip's chain of them. +package node + +import ( + "context" + "errors" + "fmt" + "testing" + + "github.com/simplyblock/csi-driver/internal/controlplane" + csicommon "github.com/simplyblock/csi-driver/internal/csi/common" +) + +// fakeRelationshipAPI stubs exactly the two ClusterAPI calls the redirect +// walk makes; every other method panics via the embedded nil interface, +// which is the point -- the walk must touch nothing else. +type fakeRelationshipAPI struct { + controlplane.ClusterAPI + rels map[string]*controlplane.ReplicationRelationship + conn map[string]map[string]string +} + +func (f *fakeRelationshipAPI) GetRelationship( + _ context.Context, lvolID string, +) (*controlplane.ReplicationRelationship, error) { + rel, ok := f.rels[lvolID] + if !ok { + return nil, errors.New("no replication relationship") + } + return rel, nil +} + +func (f *fakeRelationshipAPI) VolumeInfo(_ context.Context, lvolID, _ string) (map[string]string, error) { + c, ok := f.conn[lvolID] + if !ok { + return nil, controlplane.ErrVolumeNotFound + } + return c, nil +} + +// Regression: 2026-09-24-chained-relationship-resolves-one-hop-short — after +// a relocate ROUND TRIP the chain is original(A) -> hop-1 clone(B) -> hop-2 +// clone(A, the live volume, named by active_lvol_id on every record). The +// redirect used the FIRST record's active_lvol_id but paired it with that +// same record's target cluster/pool -- fields describing a DIFFERENT hop -- +// and asked cluster B for a volume that lives on cluster A ("volume not +// found," confirmed live 2026-09-24), then fell back to the stale stashed +// context and the mount timed out. The walk must follow each record's own +// consistent target triple, hop by hop, until the hop whose target IS the +// active volume. +func TestRedirectToActiveVolumeWalksAChainedRelationship(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + clusterB = "bbbbbbbb-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + poolB = "bbbbbbbb-0000-0000-0000-00000000000b" + original = "11111111-1111-1111-1111-111111111111" + hop1 = "22222222-2222-2222-2222-222222222222" + active = "33333333-3333-3333-3333-333333333333" + ) + + srcClient := &fakeRelationshipAPI{ + rels: map[string]*controlplane.ReplicationRelationship{ + original: { + SourceLvolID: original, TargetLvolID: hop1, + SourceClusterID: clusterA, TargetClusterID: clusterB, TargetPoolID: poolB, + ActiveLvolID: active, + }, + }, + } + clusterBClient := &fakeRelationshipAPI{ + rels: map[string]*controlplane.ReplicationRelationship{ + hop1: { + SourceLvolID: hop1, TargetLvolID: active, + SourceClusterID: clusterB, TargetClusterID: clusterA, TargetPoolID: poolA, + ActiveLvolID: active, + }, + }, + } + clusterAClient := &fakeRelationshipAPI{ + conn: map[string]map[string]string{ + active: {"nqn": "nqn.test:" + active, "ip": "10.0.0.1", "port": "4420"}, + }, + } + + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(_ context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + switch clusterID + "/" + poolID { + case clusterB + "/" + poolB: + return clusterBClient, nil + case clusterA + "/" + poolA: + return clusterAClient, nil + } + return nil, errors.New("unexpected cluster " + clusterID + "/" + poolID) + } + + connInfo := redirectToActiveVolume(context.Background(), srcClient, original, + clusterA+":"+poolA+":"+original, map[string]string{"hostNQN": "nqn.host"}) + if connInfo == nil { + t.Fatal("redirect returned nil: the walk never reached the active volume") + } + if got := connInfo["nqn"]; got != "nqn.test:"+active { + t.Errorf("connection nqn = %q, want the ACTIVE volume's %q", got, "nqn.test:"+active) + } + if got := connInfo[csicommon.ParamClusterID]; got != clusterA { + t.Errorf("cluster_id = %q, want the active volume's cluster %q, not the first hop's", got, clusterA) + } + if got := connInfo["poolID"]; got != poolA { + t.Errorf("poolID = %q, want the active volume's pool %q", got, poolA) + } +} + +// The single-pairing case the redirect was originally written for (a +// migration with --delete-source, or one fail-over): the first record's +// target IS the active volume, and the walk must behave exactly as the +// one-step redirect always did. +func TestRedirectToActiveVolumeSinglePairingIsUnchanged(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + clusterB = "bbbbbbbb-0000-0000-0000-000000000001" + poolB = "bbbbbbbb-0000-0000-0000-00000000000b" + original = "11111111-1111-1111-1111-111111111111" + active = "22222222-2222-2222-2222-222222222222" + ) + + srcClient := &fakeRelationshipAPI{ + rels: map[string]*controlplane.ReplicationRelationship{ + original: { + SourceLvolID: original, TargetLvolID: active, + SourceClusterID: clusterA, TargetClusterID: clusterB, TargetPoolID: poolB, + ActiveLvolID: active, + }, + }, + } + clusterBClient := &fakeRelationshipAPI{ + conn: map[string]map[string]string{ + active: {"nqn": "nqn.test:" + active}, + }, + } + + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(_ context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + if clusterID == clusterB && poolID == poolB { + return clusterBClient, nil + } + return nil, errors.New("unexpected cluster " + clusterID) + } + + connInfo := redirectToActiveVolume(context.Background(), srcClient, original, + clusterA+":pool:"+original, map[string]string{}) + if connInfo == nil { + t.Fatal("redirect returned nil for the plain single-pairing case") + } + if got := connInfo[csicommon.ParamClusterID]; got != clusterB { + t.Errorf("cluster_id = %q, want %q", got, clusterB) + } +} + +// A volume moved many times: the PV keeps the original handle while every +// relocate and fail-over appends a clone, alternating between the two +// clusters, so the live copy sits one hop further out after each move. The +// walk must reach it however long the chain has grown; a cap of 8 stranded +// the ninth move's clone and the node attached the original on the +// partitioned site instead (2026-10-03). +func TestRedirectToActiveVolumeFollowsALongChain(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + clusterB = "bbbbbbbb-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + poolB = "bbbbbbbb-0000-0000-0000-00000000000b" + moves = 12 + ) + member := func(i int) string { return fmt.Sprintf("%08d-0000-0000-0000-000000000000", i) } + cluster := func(i int) (string, string) { + if i%2 == 0 { + return clusterA, poolA + } + return clusterB, poolB + } + active := member(moves) + clients := map[string]*fakeRelationshipAPI{ + clusterA + "/" + poolA: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, + clusterB + "/" + poolB: { + rels: map[string]*controlplane.ReplicationRelationship{}, + conn: map[string]map[string]string{}, + }, + } + for i := 0; i < moves; i++ { + srcC, srcP := cluster(i) + tgtC, tgtP := cluster(i + 1) + clients[srcC+"/"+srcP].rels[member(i)] = &controlplane.ReplicationRelationship{ + SourceLvolID: member(i), TargetLvolID: member(i + 1), + SourceClusterID: srcC, TargetClusterID: tgtC, TargetPoolID: tgtP, + ActiveLvolID: active, + } + } + activeC, activeP := cluster(moves) + clients[activeC+"/"+activeP].conn[active] = map[string]string{"nqn": "nqn.test:" + active} + + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(_ context.Context, clusterID, poolID string) (controlplane.ClusterAPI, error) { + if c, ok := clients[clusterID+"/"+poolID]; ok { + return c, nil + } + return nil, errors.New("unexpected cluster " + clusterID + "/" + poolID) + } + + connInfo := redirectToActiveVolume(context.Background(), clients[clusterA+"/"+poolA], member(0), + clusterA+":"+poolA+":"+member(0), map[string]string{"hostNQN": "nqn.host"}) + if connInfo == nil { + t.Fatalf("redirect returned nil: the walk gave up before the %d-hop chain's active volume", moves) + } + if got := connInfo["nqn"]; got != "nqn.test:"+active { + t.Errorf("connection nqn = %q, want the active volume's %q", got, "nqn.test:"+active) + } + if got := connInfo[csicommon.ParamClusterID]; got != activeC { + t.Errorf("cluster_id = %q, want the active volume's cluster %q", got, activeC) + } +} + +// A relationship that points back at a member already walked must end the +// walk instead of spinning to the bound. +func TestRedirectToActiveVolumeStopsOnALoop(t *testing.T) { + const ( + clusterA = "aaaaaaaa-0000-0000-0000-000000000001" + poolA = "aaaaaaaa-0000-0000-0000-00000000000a" + x = "11111111-1111-1111-1111-111111111111" + y = "22222222-2222-2222-2222-222222222222" + ) + client := &fakeRelationshipAPI{rels: map[string]*controlplane.ReplicationRelationship{ + x: { + SourceLvolID: x, TargetLvolID: y, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, + y: { + SourceLvolID: y, TargetLvolID: x, SourceClusterID: clusterA, + TargetClusterID: clusterA, TargetPoolID: poolA, ActiveLvolID: "zz", + }, + }} + orig := clusterClientFor + defer func() { clusterClientFor = orig }() + clusterClientFor = func(context.Context, string, string) (controlplane.ClusterAPI, error) { return client, nil } + got := redirectToActiveVolume(context.Background(), client, x, clusterA+":"+poolA+":"+x, map[string]string{}) + if got != nil { + t.Fatalf("a looping chain returned %v, want nil", got) + } +} diff --git a/csi-driver/internal/driver/driver.go b/csi-driver/internal/driver/driver.go index aa30e0311..1589f62e5 100644 --- a/csi-driver/internal/driver/driver.go +++ b/csi-driver/internal/driver/driver.go @@ -28,6 +28,11 @@ import ( "fmt" "github.com/container-storage-interface/spec/lib/go/csi" + csiaddonsidentity "github.com/csi-addons/spec/lib/go/identity" + csiaddonsreplication "github.com/csi-addons/spec/lib/go/replication" + csiaddonsvolumegroup "github.com/csi-addons/spec/lib/go/volumegroup" + "google.golang.org/grpc" + "k8s.io/client-go/dynamic" "k8s.io/client-go/kubernetes" "k8s.io/client-go/rest" "k8s.io/klog" @@ -40,6 +45,7 @@ import ( "github.com/simplyblock/csi-driver/internal/config" csicommon "github.com/simplyblock/csi-driver/internal/csi/common" "github.com/simplyblock/csi-driver/internal/csi/controller" + csiaddonsidentityserver "github.com/simplyblock/csi-driver/internal/csi/csiaddons/identity" "github.com/simplyblock/csi-driver/internal/csi/identity" "github.com/simplyblock/csi-driver/internal/csi/node" "github.com/simplyblock/csi-driver/internal/csilink" @@ -89,12 +95,20 @@ func Run(conf *config.Config) { // its own in-cluster config + clientset. A missing in-cluster config is // non-fatal, and the features that need it degrade to no-ops. var kubeClient kubernetes.Interface + var dynClient dynamic.Interface if k8sConfig, err := rest.InClusterConfig(); err != nil { klog.Warningf("no in-cluster config; Kubernetes API features disabled: %v", err) } else if clientset, err := kubernetes.NewForConfig(k8sConfig); err != nil { klog.Warningf("failed to create kubernetes client; Kubernetes API features disabled: %v", err) } else { kubeClient = clientset + // The consistency-group watcher requests VolumeMigrations (the + // pre-join live migration) without importing the operator's types. + if d, err := dynamic.NewForConfig(k8sConfig); err != nil { + klog.Warningf("failed to create dynamic client; pre-join migrations disabled: %v", err) + } else { + dynClient = d + } } if conf.IsNodeServer { @@ -117,7 +131,7 @@ func Run(conf *config.Config) { // without a Kubernetes client, like the other kube-backed features. watcherCtx, watcherCancel := context.WithCancel(context.Background()) defer watcherCancel() - controller.StartConsistencyGroupLabelWatcher(watcherCtx, kubeClient, conf.DriverName) + controller.StartConsistencyGroupLabelWatcher(watcherCtx, kubeClient, dynClient, conf.DriverName) } // The link to the operator, when enabled. It is independent of the CSI @@ -131,8 +145,29 @@ func Run(conf *config.Config) { } } + // The csi-addons Identity and Replication services register alongside the + // CSI services on the same socket. Identity is always registered (it just + // answers capability probes); Replication only when this process serves + // the controller (cs is nil on a node-only process). + register := []func(*grpc.Server){ + func(gs *grpc.Server) { + csiaddonsidentity.RegisterIdentityServer(gs, csiaddonsidentityserver.New(conf.DriverName, conf.DriverVersion)) + }, + } + if cs != nil { + register = append(register, func(gs *grpc.Server) { + csiaddonsreplication.RegisterControllerServer(gs, cs) + }) + // The csi-addons VolumeGroup service (design §14.3): the stock + // controller-manager dials it to form a backend consistency group before + // replicating the group as one unit. + register = append(register, func(gs *grpc.Server) { + csiaddonsvolumegroup.RegisterControllerServer(gs, cs) + }) + } + s := csicommon.NewNonBlockingGRPCServer() - s.Start(conf.Endpoint, ids, cs, ns) + s.Start(conf.Endpoint, ids, cs, ns, register...) s.Wait() } diff --git a/fleet/config/crd/bases/fleet.simplyblock.io_driverdeployments.yaml b/fleet/config/crd/bases/fleet.simplyblock.io_driverdeployments.yaml index a2331a6a9..41c615170 100644 --- a/fleet/config/crd/bases/fleet.simplyblock.io_driverdeployments.yaml +++ b/fleet/config/crd/bases/fleet.simplyblock.io_driverdeployments.yaml @@ -298,13 +298,29 @@ spec: type: object sidecarImages: description: |- - SidecarImages overrides the six CSI sidecars, one field each. Unset takes - the version this operator release ships. + SidecarImages overrides the seven CSI sidecars, one field each. Unset + takes the version this operator release ships. properties: attacher: description: Attacher is csi-attacher, on the controller plugin. pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ type: string + csiAddons: + description: |- + CSIAddons is the kubernetes-csi-addons sidecar, on the controller + plugin. It connects to the plugin's socket, probes the csi-addons + Identity service for capabilities, and publishes a CSIAddonsNode so the + kubernetes-csi-addons controller-manager (design + design-csi-addons-replication.md §4.1) can reach the Replication + service this driver serves. + + Unlike the other sidecars above, this one's upstream home is the + csi-addons project's own registry, not simplyblock's: the allowlist + carries quay.io/csiaddons alongside the simplyblock registries so a + deployment can run the stock kubernetes-csi-addons sidecar image + directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string healthMonitor: description: |- HealthMonitor is csi-external-health-monitor-controller, on the diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_controlplanes.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_controlplanes.yaml index 94b0cad35..c426685ee 100644 --- a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_controlplanes.yaml +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_controlplanes.yaml @@ -163,6 +163,32 @@ spec: local: description: Local is a control plane the operator installs. properties: + adminTokenSecretRef: + description: |- + AdminTokenSecretRef names a Secret in this namespace holding a static + admin bearer token this control plane accepts, under the `token` key, in + addition to this deployment's own Kubernetes identity + (SB_K8S_ADMIN_SERVICE_ACCOUNTS). It is what lets a cluster this control + plane manages remotely (spec.source.managed there, + ManagedControlPlane.CredentialsSecretRef naming the same value) + authenticate a CreateCluster call, since a Kubernetes TokenReview can + never cross a cluster boundary. + + The Secret is projected into the management API container's environment + with secretKeyRef, so this operator never itself reads the plaintext. + Absent grants no credential beyond the operator's own service account. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic foundationDB: description: FoundationDB sizes the FoundationDB the management API stores its state in. properties: @@ -513,6 +539,20 @@ spec: so a loopback or link-local address is rejected. pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ type: string + storageNodeImage: + description: |- + StorageNodeImage is the storage-node image a StorageCluster on this + Kubernetes cluster defaults to when its own spec.storageNodes.image is + unset. A local control plane's own spec.source.local.image doubles as + this default (StorageNodeWorkloadReconciler.image), because a + self-hosted deployment's control plane and its storage nodes are one + release. A managed one is a different Kubernetes cluster's install and + says nothing about what this cluster's storage nodes should run, so + there is no equivalent to fall back to without this field -- every + StorageCluster on a managed deployment must get an image from here or + from its own spec. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string required: - endpoint type: object diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_simplyblockdrivers.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_simplyblockdrivers.yaml index 94d87edeb..e4a7d6c6b 100644 --- a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_simplyblockdrivers.yaml +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_simplyblockdrivers.yaml @@ -299,13 +299,29 @@ spec: type: object sidecarImages: description: |- - SidecarImages overrides the six CSI sidecars, one field each. Unset takes - the version this operator release ships. + SidecarImages overrides the seven CSI sidecars, one field each. Unset + takes the version this operator release ships. properties: attacher: description: Attacher is csi-attacher, on the controller plugin. pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ type: string + csiAddons: + description: |- + CSIAddons is the kubernetes-csi-addons sidecar, on the controller + plugin. It connects to the plugin's socket, probes the csi-addons + Identity service for capabilities, and publishes a CSIAddonsNode so the + kubernetes-csi-addons controller-manager (design + design-csi-addons-replication.md §4.1) can reach the Replication + service this driver serves. + + Unlike the other sidecars above, this one's upstream home is the + csi-addons project's own registry, not simplyblock's: the allowlist + carries quay.io/csiaddons alongside the simplyblock registries so a + deployment can run the stock kubernetes-csi-addons sidecar image + directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string healthMonitor: description: |- HealthMonitor is csi-external-health-monitor-controller, on the diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..393fd46a5 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1053 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: |- + Count is the number of journal managers to configure. The control plane + requires at least 3. + format: int32 + minimum: 3 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml new file mode 100644 index 000000000..d9117bb4c --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/crds/storage.simplyblock.io_testfailovers.yaml @@ -0,0 +1,336 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: testfailovers.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: TestFailover + listKind: TestFailoverList + plural: testfailovers + shortNames: + - tfo + singular: testfailover + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.scope + name: Scope + type: string + - jsonPath: .spec.sourceRef + name: Source + type: string + - jsonPath: .spec.sourceCluster + name: "On" + type: string + - jsonPath: .spec.bubbleCluster + name: Bubble + type: string + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.step.state + name: Step + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + TestFailover is a one-way, non-disruptive test-failover drill. It recovers a + source volume, or a consistency group, from a snapshot into an isolated + namespace on a chosen cluster as bound PVCs, without touching the source. The + hub reads the source on its cluster and places the bubble on the recovery + cluster through OCM. It runs to a terminal phase, or holds Ready until it is + deleted, and deletion reclaims the clones and any snapshots the drill took. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + TestFailoverSpec is the request for one non-disruptive test-failover drill. + + The source is named by where it runs and what it is, so the hub can find it + without anyone extracting a backend handle by hand. SourceNamespace is + required for a Volume drill, where the source is a PVC, and unused for a Group + drill, where SourceRef names a consistency group. + properties: + bubbleCluster: + description: |- + BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + the replicated point, or another cluster. It must differ from SourceCluster; + test-failover recovers onto a different cluster, never in place. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + bubbleNamespace: + default: bubble + description: |- + BubbleNamespace is the namespace on the bubble cluster where the recovered + PVCs are created. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + scope: + description: Scope selects what the drill recovers. Immutable. + enum: + - Volume + - Group + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceCluster: + description: |- + SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + the source there through a ManagedClusterView. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceNamespace: + description: |- + SourceNamespace is the namespace of the source PVC on SourceCluster. + Required for scope=Volume. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceRef: + description: |- + SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + SourceNamespace (scope=Volume), or a consistency group (scope=Group). + Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + ttlSeconds: + description: |- + TTLSeconds is an optional maximum lifetime: the drill is torn down after it + even without a delete, so a forgotten drill cannot hold a clone forever. + format: int64 + minimum: 0 + type: integer + required: + - bubbleCluster + - scope + - sourceCluster + - sourceRef + type: object + x-kubernetes-validations: + - message: field sourceNamespace is immutable once set + rule: '!has(oldSelf.sourceNamespace) || has(self.sourceNamespace)' + - message: field bubbleNamespace is immutable once set + rule: '!has(oldSelf.bubbleNamespace) || has(self.bubbleNamespace)' + - message: sourceNamespace is required for scope=Volume + rule: self.scope != 'Volume' || has(self.sourceNamespace) + status: + description: TestFailoverStatus is the observed state of one drill. + properties: + clones: + description: Clones is one entry per recovered volume. + items: + description: |- + TestFailoverClone is one recovered volume: the source it came from, the + snapshot and clone the drill built, and the PVC placed on the bubble cluster. + properties: + cloneID: + description: CloneID is the backend id of the writable clone. + type: string + pvcName: + description: PVCName is the bound PVC in the bubble namespace + on the bubble cluster. + type: string + sizeBytes: + description: SizeBytes is the recovered volume's size. + format: int64 + type: integer + snapshotID: + description: |- + SnapshotID is the recovery-point snapshot: the replicated snapshot already on + the bubble cluster's backend that the clone is built from. + type: string + sourceFSType: + description: |- + SourceFSType is the source PV's CSI fsType, carried onto the bubble PV so the + node plugin stages the clone with the filesystem it actually carries. The + clone is a block copy of the source, so its filesystem is the source's; an + empty fsType makes the node plugin default to ext4 and refuse to mount an XFS + volume. + type: string + sourceHandle: + description: SourceHandle is the source volume's backend handle, + read from its PV. + type: string + sourceRef: + description: |- + SourceRef is the source volume, or group member, the recovered volume maps + to. + type: string + sourceVolumeContext: + additionalProperties: + type: string + description: |- + SourceVolumeContext is the source PV's CSI volumeAttributes, minus the + identity and provisioner keys, carried onto the bubble PV so the node plugin + receives a non-nil VolumeContext when it stages the clone. The clone's own + identity (NQN, connections, nsId, and so on) is re-resolved from the clone + handle at stage time, so only the class-level parameters are carried; the + identity keys are dropped so a failed clone lookup can never point the mount + back at the source. + type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string + required: + - sourceRef + type: object + type: array + x-kubernetes-list-map-keys: + - sourceRef + x-kubernetes-list-type: map + completedAt: + description: CompletedAt is when the drill reached a terminal phase. + format: date-time + type: string + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as the + drill moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from, so a stale status can be told from a current one. + format: int64 + type: integer + phase: + description: Phase is the drill's own progress. + enum: + - Pending + - Provisioning + - Ready + - Failed + - TearingDown + type: string + readyAt: + description: ReadyAt is when every recovered PVC became bound. + format: date-time + type: string + report: + description: Report is the drill's evidence, populated as it reaches + Ready. + properties: + bubbleCluster: + description: BubbleCluster is the cluster the drill recovered + onto. + type: string + invariantsHeld: + description: |- + InvariantsHeld is true only when the source fingerprint taken before the + drill matches the one taken at Ready. A Ready drill with this false is a + defect. + type: boolean + recoveryPoint: + description: RecoveryPoint is the snapshot or group generation + the drill recovered. + type: string + recoveryPointAgeSeconds: + description: RecoveryPointAgeSeconds is the drill time minus the + recovery-point time. + format: int64 + type: integer + recoveryPointTime: + description: RecoveryPointTime is when that point was taken. + format: date-time + type: string + type: object + startedAt: + description: StartedAt is when the drill started. + format: date-time + type: string + step: + description: Step is the position of the running drill's state machine. + properties: + claim: + description: |- + Claim records that this state's side effect was started. Absent means + no pass has started it since the state was entered. + properties: + attempt: + description: Attempt counts the claims taken on this state, + starting at 1. + format: int32 + type: integer + leaseUntil: + description: |- + LeaseUntil is when the claim expires and the side effect may be fired + again. + format: date-time + type: string + state: + description: State is the state the claim was taken in. + type: string + required: + - attempt + - leaseUntil + - state + type: object + deadline: + description: |- + Deadline is when that state expires, absent when it has none. It is an + absolute instant, so a state whose deadline passed while the controller + was down restores as already expired. + format: date-time + type: string + state: + description: |- + State is the state the machine was in. Empty means the resource has not + been reconciled yet, and restores to the graph's initial state. + type: string + type: object + x-kubernetes-validations: + - message: unknown step + rule: '!has(self.state) || self.state in [''ResolvingSource'',''ResolvingPoint'',''Shipping'',''Cloning'',''Placing'',''Releasing'']' + triggered: + description: |- + Triggered records that the current step's side effect was issued, so a + restart does not repeat it. + type: boolean + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl index c2486ddbf..d2ec36174 100644 --- a/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl +++ b/helm-charts/charts/simplyblock-operator/templates/_control_center_helpers.tpl @@ -47,8 +47,18 @@ helm.sh/chart: {{ printf "%s-%s" .Name .Version | replace "+" "_" | trunc 63 | t {{/* Default upstream URLs. This chart names its objects statically (simplyblock-operator, simplyblock-prometheus, …) rather than deriving them from the release, so the fallbacks here are static too. */}} +{{/* The operator's HTTP API, when this release serves one. Unset, it is the + simplyblock-operator Service if that exists in the release namespace, and + empty otherwise: nginx refuses to start on an upstream host it cannot + resolve ("host not found in upstream"), and an empty upstream makes the + console answer 503 "not part of this deployment" on the operator-backed + screens instead (2026-10-03, a release without the operator API). */}} {{- define "sbcc.operatorUrl" -}} -{{- .Values.controlCenter.operatorUrl | default "http://simplyblock-operator:8080" -}} +{{- if .Values.controlCenter.operatorUrl -}} +{{- .Values.controlCenter.operatorUrl -}} +{{- else if (lookup "v1" "Service" .Release.Namespace "simplyblock-operator") -}} +http://simplyblock-operator:8080 +{{- end -}} {{- end -}} {{/* The control plane API the console reads storage from. Fully qualified: diff --git a/helm-charts/charts/simplyblock-operator/templates/control-center.yaml b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml index e5989e7ed..f32c90ef0 100644 --- a/helm-charts/charts/simplyblock-operator/templates/control-center.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/control-center.yaml @@ -227,6 +227,9 @@ spec: - name: http port: {{ $cc.service.port }} targetPort: http + {{- with $cc.service.nodePort }} + nodePort: {{ . }} + {{- end }} selector: {{- include "sbcc.selectorLabels" . | nindent 4 }} {{- if $cc.ingress.enabled }} diff --git a/helm-charts/charts/simplyblock-operator/templates/controlplane_cr.yaml b/helm-charts/charts/simplyblock-operator/templates/controlplane_cr.yaml index 43606c988..e61ef15f6 100644 --- a/helm-charts/charts/simplyblock-operator/templates/controlplane_cr.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/controlplane_cr.yaml @@ -95,6 +95,10 @@ spec: {{- end }} {{- end }} {{- end }} + {{- if .Values.controlplane.local.adminTokenSecretRef }} + adminTokenSecretRef: + name: {{ .Values.controlplane.local.adminTokenSecretRef | quote }} + {{- end }} {{- else }} # A control plane elsewhere manages this cluster's storage. The operator # installs nothing here: it resolves the endpoint, probes it, and reports. @@ -108,5 +112,8 @@ spec: caBundleSecretRef: name: {{ .Values.controlplane.managed.caBundleSecretRef | quote }} {{- end }} + {{- if .Values.controlplane.managed.storageNodeImage }} + storageNodeImage: {{ .Values.controlplane.managed.storageNodeImage | quote }} + {{- end }} {{- end }} {{- end }} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_csiaddonsnodes.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_csiaddonsnodes.yaml new file mode 100644 index 000000000..34670e2ea --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_csiaddonsnodes.yaml @@ -0,0 +1,168 @@ +# CSIAddonsNode CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/CSIAddonsNode")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: csiaddonsnodes.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: CSIAddonsNode + listKind: CSIAddonsNodeList + plural: csiaddonsnodes + singular: csiaddonsnode + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .metadata.namespace + name: namespace + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + - jsonPath: .spec.driver.name + name: DriverName + type: string + - jsonPath: .spec.driver.endpoint + name: Endpoint + type: string + - jsonPath: .spec.driver.nodeID + name: NodeID + type: string + name: v1alpha1 + schema: + openAPIV3Schema: + description: CSIAddonsNode is the Schema for the csiaddonsnode API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: CSIAddonsNodeSpec defines the desired state of CSIAddonsNode + properties: + driver: + description: |- + Driver is the information of the CSI Driver existing on a node. + If the driver is uninstalled, this can become empty. + properties: + endpoint: + description: |- + EndPoint is url that contains the ip-address to which the CSI-Addons + side-car listens to. + type: string + name: + description: |- + Name is the name of the CSI driver that this object refers to. + This must be the same name returned by the CSI-Addons GetIdentity() + call for that driver. The name of the driver is in the format: + `example.csi.ceph.com` + type: string + x-kubernetes-validations: + - message: name is immutable + rule: self == oldSelf + nodeID: + description: |- + NodeID is the ID of the node to identify on which node the side-car + is running. + type: string + x-kubernetes-validations: + - message: nodeID is immutable + rule: self == oldSelf + required: + - endpoint + - name + - nodeID + type: object + required: + - driver + type: object + status: + description: CSIAddonsNodeStatus defines the observed state of CSIAddonsNode + properties: + capabilities: + description: A list of capabilities advertised by the sidecar + items: + type: string + type: array + message: + description: |- + Message is a human-readable message indicating details about why the CSIAddonsNode + is in this state. + type: string + networkFenceClientStatus: + description: NetworkFenceClientStatus contains the status of the clients + required for fencing. + items: + description: NetworkFenceClientStatus contains the status of the + clients required for fencing. + properties: + ClientDetails: + items: + description: ClientDetail contains the details of the client + required for fencing. + properties: + cidrs: + description: Cidrs is the list of CIDR blocks that are + fenced. + items: + type: string + type: array + id: + description: Id is the unique identifier of the client + where it belongs to. + type: string + required: + - cidrs + - id + type: object + type: array + networkFenceClassName: + type: string + required: + - ClientDetails + - networkFenceClassName + type: object + type: array + reason: + description: |- + Reason is a brief CamelCase string that describes any failure and is meant + for machine parsing and tidy display in the CLI. + type: string + state: + description: |- + State represents the state of the CSIAddonsNode object. + It informs whether or not the CSIAddonsNode is Connected + to the CSI Driver. + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationcronjobs.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationcronjobs.yaml new file mode 100644 index 000000000..e0e2ea4a4 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationcronjobs.yaml @@ -0,0 +1,253 @@ +# EncryptionKeyRotationCronJob CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/EncryptionKeyRotationCronJob")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: encryptionkeyrotationcronjobs.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: EncryptionKeyRotationCronJob + listKind: EncryptionKeyRotationCronJobList + plural: encryptionkeyrotationcronjobs + singular: encryptionkeyrotationcronjob + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.schedule + name: Schedule + type: string + - jsonPath: .spec.suspend + name: Suspend + type: boolean + - jsonPath: .status.active.name + name: Active + type: string + - jsonPath: .status.lastScheduleTime + name: Lastschedule + type: date + - jsonPath: .status.lastSuccessfulTime + name: Lastsuccessfultime + priority: 1 + type: date + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: EncryptionKeyRotationCronJob is the Schema for the encryptionkeyrotationcronjobs + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: EncryptionKeyRotationCronJobSpec defines the desired state + of EncryptionKeyRotationCronJob + properties: + concurrencyPolicy: + default: Forbid + description: |- + Specifies how to treat concurrent executions of a Job. + Valid values are: + - "Forbid" (default): forbids concurrent runs, skipping next run if + previous run hasn't finished yet; + - "Replace": cancels currently running job and replaces it + with a new one + enum: + - Forbid + - Replace + type: string + failedJobsHistoryLimit: + default: 1 + description: |- + The number of failed finished jobs to retain. Value must be non-negative integer. + Defaults to 1. + format: int32 + maximum: 60 + minimum: 0 + type: integer + jobTemplate: + description: Specifies the job that will be created when executing + a CronJob. + properties: + metadata: + description: |- + Standard object's metadata of the jobs created from this template. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#metadata + type: object + spec: + description: |- + Specification of the desired behavior of the job. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#spec-and-status + properties: + backOffLimit: + default: 6 + description: |- + BackOffLimit specifies the number of retries allowed before marking reclaim + space operation as failed. If not specified, defaults to 6. Maximum allowed + value is 60 and minimum allowed value is 0. + format: int32 + maximum: 60 + minimum: 0 + type: integer + retryDeadlineSeconds: + default: 600 + description: |- + RetryDeadlineSeconds specifies the duration in seconds relative to the + start time that the operation may be retried; value MUST be positive integer. + If not specified, defaults to 600 seconds. Maximum allowed + value is 1800. + format: int64 + maximum: 1800 + minimum: 0 + type: integer + target: + description: |- + Target represents tvolume target on which operation will be + performed. + properties: + persistentVolumeClaim: + description: PersistentVolumeClaim specifies the target + PersistentVolumeClaim name. + type: string + x-kubernetes-validations: + - message: persistentVolumeClaim is immutable + rule: self == oldSelf + type: object + timeout: + description: |- + Timeout specifies the timeout in seconds for the grpc request sent to the + CSI driver. + Minimum allowed value is 60. + format: int64 + minimum: 60 + type: integer + required: + - target + type: object + required: + - spec + type: object + schedule: + description: |- + The schedule in Cron format, see https://en.wikipedia.org/wiki/Cron. + A deterministic, UID-based stagger offset is applied to spread + execution across the "cronjob-stagger-window" (default: 2 hours, + set to 0 to disable) configured in the csi-addons-config ConfigMap. + pattern: .+ + type: string + startingDeadlineSeconds: + description: |- + Optional deadline in seconds for starting the job if it misses scheduled + time for any reason. Missed jobs executions will be counted as failed ones. + format: int64 + type: integer + successfulJobsHistoryLimit: + default: 3 + description: |- + The number of successful finished jobs to retain. Value must be non-negative integer. + Defaults to 3. + format: int32 + maximum: 60 + minimum: 0 + type: integer + suspend: + description: |- + This flag tells the controller to suspend subsequent executions, it does + not apply to already started executions. Defaults to false. + type: boolean + required: + - jobTemplate + - schedule + type: object + status: + description: EncryptionKeyRotationCronJobStatus defines the observed state + of EncryptionKeyRotationCronJob + properties: + active: + description: A pointer to currently running job. + properties: + apiVersion: + description: API version of the referent. + type: string + fieldPath: + description: |- + If referring to a piece of an object instead of an entire object, this string + should contain a valid JSON/Go field access statement, such as desiredState.manifest.containers[2]. + For example, if the object reference is to a container within a pod, this would take on a value like: + "spec.containers{name}" (where "name" refers to the name of the container that triggered + the event) or if no container name is specified "spec.containers[2]" (container with + index 2 in this pod). This syntax is chosen only to have some well-defined way of + referencing a part of an object. + type: string + kind: + description: |- + Kind of the referent. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + name: + description: |- + Name of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + namespace: + description: |- + Namespace of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/namespaces/ + type: string + resourceVersion: + description: |- + Specific resourceVersion to which this reference is made, if any. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#concurrency-control-and-consistency + type: string + uid: + description: |- + UID of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#uids + type: string + type: object + x-kubernetes-map-type: atomic + lastScheduleTime: + description: Information when was the last time the job was successfully + scheduled. + format: date-time + type: string + lastSuccessfulTime: + description: Information when was the last time the job successfully + completed. + format: date-time + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationjobs.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationjobs.yaml new file mode 100644 index 000000000..8346e40b3 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_encryptionkeyrotationjobs.yaml @@ -0,0 +1,196 @@ +# EncryptionKeyRotationJob CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/EncryptionKeyRotationJob")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: encryptionkeyrotationjobs.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: EncryptionKeyRotationJob + listKind: EncryptionKeyRotationJobList + plural: encryptionkeyrotationjobs + singular: encryptionkeyrotationjob + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .metadata.namespace + name: Namespace + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + - jsonPath: .status.retries + name: Retries + type: integer + - jsonPath: .status.result + name: Result + type: string + name: v1alpha1 + schema: + openAPIV3Schema: + description: EncryptionKeyRotationJob is the Schema for the encryptionkeyrotationjobs + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: EncryptionKeyRotationJobSpec defines the desired state of + EncryptionKeyRotationJob + properties: + backOffLimit: + default: 6 + description: |- + BackOffLimit specifies the number of retries allowed before marking reclaim + space operation as failed. If not specified, defaults to 6. Maximum allowed + value is 60 and minimum allowed value is 0. + format: int32 + maximum: 60 + minimum: 0 + type: integer + retryDeadlineSeconds: + default: 600 + description: |- + RetryDeadlineSeconds specifies the duration in seconds relative to the + start time that the operation may be retried; value MUST be positive integer. + If not specified, defaults to 600 seconds. Maximum allowed + value is 1800. + format: int64 + maximum: 1800 + minimum: 0 + type: integer + target: + description: |- + Target represents tvolume target on which operation will be + performed. + properties: + persistentVolumeClaim: + description: PersistentVolumeClaim specifies the target PersistentVolumeClaim + name. + type: string + x-kubernetes-validations: + - message: persistentVolumeClaim is immutable + rule: self == oldSelf + type: object + timeout: + description: |- + Timeout specifies the timeout in seconds for the grpc request sent to the + CSI driver. + Minimum allowed value is 60. + format: int64 + minimum: 60 + type: integer + required: + - target + type: object + status: + description: EncryptionKeyRotationJobStatus defines the observed state + of EncryptionKeyRotationJob + properties: + completionTime: + format: date-time + type: string + conditions: + description: Conditions are the list of conditions and their status. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + message: + description: Message contains any message from the EncryptionKeyRotationJob. + type: string + result: + description: Result indicates the result of EncryptionKeyRotationJob. + type: string + retries: + description: Retries indicates the number of times the operation is + retried. + format: int32 + type: integer + startTime: + format: date-time + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfenceclasses.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfenceclasses.yaml new file mode 100644 index 000000000..541ad3ba1 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfenceclasses.yaml @@ -0,0 +1,83 @@ +# NetworkFenceClass CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/NetworkFenceClass")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: networkfenceclasses.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: NetworkFenceClass + listKind: NetworkFenceClassList + plural: networkfenceclasses + singular: networkfenceclass + scope: Cluster + versions: + - name: v1alpha1 + schema: + openAPIV3Schema: + description: NetworkFenceClass is the Schema for the networkfenceclasses API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + NetworkFenceClassSpec specifies parameters that an underlying storage system uses + to get client for network fencing. Upon creating a NetworkFenceClass object, a RPC will be set + to the storage system that matches the provisioner to get the client for network fencing. + properties: + parameters: + additionalProperties: + type: string + description: |- + Parameters is a key-value map with storage provisioner specific configurations for + creating volume replicas + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: self == oldSelf + provisioner: + description: Provisioner is the name of storage provisioner + type: string + x-kubernetes-validations: + - message: provisioner is immutable + rule: self == oldSelf + required: + - provisioner + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: has(self.parameters) == has(oldSelf.parameters) + status: + description: NetworkFenceClassStatus defines the observed state of NetworkFenceClass + type: object + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfences.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfences.yaml new file mode 100644 index 000000000..31fc28027 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_networkfences.yaml @@ -0,0 +1,209 @@ +# NetworkFence CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/NetworkFence")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: networkfences.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: NetworkFence + listKind: NetworkFenceList + plural: networkfences + singular: networkfence + scope: Cluster + versions: + - additionalPrinterColumns: + - jsonPath: .spec.driver + name: Driver + type: string + - jsonPath: .spec.cidrs + name: Cidrs + type: string + - jsonPath: .spec.fenceState + name: FenceState + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + - jsonPath: .status.result + name: Result + type: string + name: v1alpha1 + schema: + openAPIV3Schema: + description: NetworkFence is the Schema for the networkfences API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: NetworkFenceSpec defines the desired state of NetworkFence + properties: + cidrs: + description: Cidrs contains a list of CIDR blocks, which are required + to be fenced. + items: + type: string + type: array + driver: + description: Driver contains the name of CSI driver, required if NetworkFenceClassName + is absent + type: string + x-kubernetes-validations: + - message: driver is immutable + rule: self == oldSelf + fenceState: + default: Fenced + description: |- + FenceState contains the desired state for the CIDRs + mentioned in the Spec. i.e. Fenced or Unfenced + enum: + - Fenced + - Unfenced + type: string + networkFenceClassName: + description: NetworkFenceClassName contains the name of the NetworkFenceClass + type: string + x-kubernetes-validations: + - message: networkFenceClassName is immutable + rule: self == oldSelf + parameters: + additionalProperties: + type: string + description: Parameters is used to pass additional parameters to the + CSI driver. + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: self == oldSelf + secret: + description: Secret is a kubernetes secret, which is required to perform + the fence/unfence operation. + properties: + name: + description: Name specifies the name of the secret. + type: string + x-kubernetes-validations: + - message: name is immutable + rule: self == oldSelf + namespace: + description: |- + Namespace specifies the namespace in which the secret + is located. + type: string + x-kubernetes-validations: + - message: namespace is immutable + rule: self == oldSelf + type: object + x-kubernetes-validations: + - message: secret is immutable + rule: self == oldSelf + required: + - cidrs + - fenceState + type: object + x-kubernetes-validations: + - message: one of driver or networkFenceClassName must be present + rule: has(self.driver) || has(self.networkFenceClassName) + - message: secret must be present when networkFenceClassName is not specified + rule: has(self.networkFenceClassName) || has(self.secret) + status: + description: NetworkFenceStatus defines the observed state of NetworkFence + properties: + conditions: + description: Conditions are the list of conditions and their status. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + message: + description: Message contains any message from the NetworkFence operation. + type: string + result: + description: Result indicates the result of Network Fence/Unfence + operation. + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacecronjobs.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacecronjobs.yaml new file mode 100644 index 000000000..ef2c90840 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacecronjobs.yaml @@ -0,0 +1,251 @@ +# ReclaimSpaceCronJob CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/ReclaimSpaceCronJob")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: reclaimspacecronjobs.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: ReclaimSpaceCronJob + listKind: ReclaimSpaceCronJobList + plural: reclaimspacecronjobs + singular: reclaimspacecronjob + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.schedule + name: Schedule + type: string + - jsonPath: .spec.suspend + name: Suspend + type: boolean + - jsonPath: .status.active.name + name: Active + type: string + - jsonPath: .status.lastScheduleTime + name: Lastschedule + type: date + - jsonPath: .status.lastSuccessfulTime + name: Lastsuccessfultime + priority: 1 + type: date + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: ReclaimSpaceCronJob is the Schema for the reclaimspacecronjobs + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: ReclaimSpaceCronJobSpec defines the desired state of ReclaimSpaceJob + properties: + concurrencyPolicy: + default: Forbid + description: |- + Specifies how to treat concurrent executions of a Job. + Valid values are: + - "Forbid" (default): forbids concurrent runs, skipping next run if + previous run hasn't finished yet; + - "Replace": cancels currently running job and replaces it + with a new one + enum: + - Forbid + - Replace + type: string + failedJobsHistoryLimit: + default: 1 + description: |- + The number of failed finished jobs to retain. Value must be non-negative integer. + Defaults to 1. + format: int32 + maximum: 60 + minimum: 0 + type: integer + jobTemplate: + description: Specifies the job that will be created when executing + a CronJob. + properties: + metadata: + description: |- + Standard object's metadata of the jobs created from this template. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#metadata + type: object + spec: + description: |- + Specification of the desired behavior of the job. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#spec-and-status + properties: + backOffLimit: + default: 6 + description: |- + BackOffLimit specifies the number of retries allowed before marking reclaim + space operation as failed. If not specified, defaults to 6. Maximum allowed + value is 60 and minimum allowed value is 0. + format: int32 + maximum: 60 + minimum: 0 + type: integer + retryDeadlineSeconds: + default: 600 + description: |- + RetryDeadlineSeconds specifies the duration in seconds relative to the + start time that the operation may be retried; value MUST be positive integer. + If not specified, defaults to 600 seconds. Maximum allowed + value is 1800. + format: int64 + maximum: 1800 + minimum: 0 + type: integer + target: + description: |- + Target represents volume target on which the operation will be + performed. + properties: + persistentVolumeClaim: + description: PersistentVolumeClaim specifies the target + PersistentVolumeClaim name. + type: string + x-kubernetes-validations: + - message: persistentVolumeClaim is immutable + rule: self == oldSelf + type: object + timeout: + description: |- + Timeout specifies the timeout in seconds for the grpc request sent to the + CSI driver. If not specified, defaults to global reclaimspace timeout. + Minimum allowed value is 60. + format: int64 + minimum: 60 + type: integer + required: + - target + type: object + required: + - spec + type: object + schedule: + description: |- + The schedule in Cron format, see https://en.wikipedia.org/wiki/Cron. + A deterministic, UID-based stagger offset is applied to spread + execution across the "cronjob-stagger-window" (default: 2 hours, + set to 0 to disable) configured in the csi-addons-config ConfigMap. + pattern: .+ + type: string + startingDeadlineSeconds: + description: |- + Optional deadline in seconds for starting the job if it misses scheduled + time for any reason. Missed jobs executions will be counted as failed ones. + format: int64 + type: integer + successfulJobsHistoryLimit: + default: 3 + description: |- + The number of successful finished jobs to retain. Value must be non-negative integer. + Defaults to 3. + format: int32 + maximum: 60 + minimum: 0 + type: integer + suspend: + description: |- + This flag tells the controller to suspend subsequent executions, it does + not apply to already started executions. Defaults to false. + type: boolean + required: + - jobTemplate + - schedule + type: object + status: + description: ReclaimSpaceCronJobStatus defines the observed state of ReclaimSpaceJob + properties: + active: + description: A pointer to currently running job. + properties: + apiVersion: + description: API version of the referent. + type: string + fieldPath: + description: |- + If referring to a piece of an object instead of an entire object, this string + should contain a valid JSON/Go field access statement, such as desiredState.manifest.containers[2]. + For example, if the object reference is to a container within a pod, this would take on a value like: + "spec.containers{name}" (where "name" refers to the name of the container that triggered + the event) or if no container name is specified "spec.containers[2]" (container with + index 2 in this pod). This syntax is chosen only to have some well-defined way of + referencing a part of an object. + type: string + kind: + description: |- + Kind of the referent. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + name: + description: |- + Name of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + namespace: + description: |- + Namespace of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/namespaces/ + type: string + resourceVersion: + description: |- + Specific resourceVersion to which this reference is made, if any. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#concurrency-control-and-consistency + type: string + uid: + description: |- + UID of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#uids + type: string + type: object + x-kubernetes-map-type: atomic + lastScheduleTime: + description: Information when was the last time the job was successfully + scheduled. + format: date-time + type: string + lastSuccessfulTime: + description: Information when was the last time the job successfully + completed. + format: date-time + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacejobs.yaml b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacejobs.yaml new file mode 100644 index 000000000..894848b3d --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/csiaddons.openshift.io_reclaimspacejobs.yaml @@ -0,0 +1,204 @@ +# ReclaimSpaceJob CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "csiaddons.openshift.io/v1alpha1/ReclaimSpaceJob")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: reclaimspacejobs.csiaddons.openshift.io +spec: + group: csiaddons.openshift.io + names: + kind: ReclaimSpaceJob + listKind: ReclaimSpaceJobList + plural: reclaimspacejobs + singular: reclaimspacejob + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .metadata.namespace + name: Namespace + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + - jsonPath: .status.retries + name: Retries + type: integer + - jsonPath: .status.result + name: Result + type: string + - jsonPath: .status.reclaimedSpace + name: ReclaimedSpace + priority: 1 + type: string + name: v1alpha1 + schema: + openAPIV3Schema: + description: ReclaimSpaceJob is the Schema for the reclaimspacejobs API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: ReclaimSpaceJobSpec defines the desired state of ReclaimSpaceJob + properties: + backOffLimit: + default: 6 + description: |- + BackOffLimit specifies the number of retries allowed before marking reclaim + space operation as failed. If not specified, defaults to 6. Maximum allowed + value is 60 and minimum allowed value is 0. + format: int32 + maximum: 60 + minimum: 0 + type: integer + retryDeadlineSeconds: + default: 600 + description: |- + RetryDeadlineSeconds specifies the duration in seconds relative to the + start time that the operation may be retried; value MUST be positive integer. + If not specified, defaults to 600 seconds. Maximum allowed + value is 1800. + format: int64 + maximum: 1800 + minimum: 0 + type: integer + target: + description: |- + Target represents volume target on which the operation will be + performed. + properties: + persistentVolumeClaim: + description: PersistentVolumeClaim specifies the target PersistentVolumeClaim + name. + type: string + x-kubernetes-validations: + - message: persistentVolumeClaim is immutable + rule: self == oldSelf + type: object + timeout: + description: |- + Timeout specifies the timeout in seconds for the grpc request sent to the + CSI driver. If not specified, defaults to global reclaimspace timeout. + Minimum allowed value is 60. + format: int64 + minimum: 60 + type: integer + required: + - target + type: object + status: + description: ReclaimSpaceJobStatus defines the observed state of ReclaimSpaceJob + properties: + completionTime: + format: date-time + type: string + conditions: + description: Conditions are the list of conditions and their status. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + message: + description: Message contains any message from the ReclaimSpaceJob. + type: string + reclaimedSpace: + anyOf: + - type: integer + - type: string + description: ReclaimedSpace indicates the amount of space reclaimed. + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + result: + description: Result indicates the result of ReclaimSpaceJob. + type: string + retries: + description: Retries indicates the number of times the operation is + retried. + format: int32 + type: integer + startTime: + format: date-time + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/rbac-csi-addons-controller.yaml b/helm-charts/charts/simplyblock-operator/templates/rbac-csi-addons-controller.yaml new file mode 100644 index 000000000..1a4879189 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/rbac-csi-addons-controller.yaml @@ -0,0 +1,299 @@ +# Identity and permissions of the csi-addons controller-manager, vendored from +# csi-addons/kubernetes-csi-addons v0.15.0 (deploy/controller/rbac.yaml) with +# names prefixed simplyblock- and namespaces bound to the release. The upstream +# file's four unbound editor and viewer convenience roles are deliberately not +# vendored: nothing binds them, so shipping them would grant nothing and audit +# as dead surface. What remains is exactly what the manager's ServiceAccount +# holds. +{{- if .Values.csiaddons.create -}} +--- +apiVersion: v1 +kind: ServiceAccount +metadata: + name: simplyblock-csi-addons-controller-manager + namespace: {{ .Release.Namespace }} +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: simplyblock-csi-addons-leader-election-role + namespace: {{ .Release.Namespace }} +rules: + # rbac-justified: controller-runtime leader election holds a Lease (and the + # legacy ConfigMap lock) in the manager's own namespace, and records election + # events there. + - apiGroups: + - "" + resources: + - configmaps + verbs: + - get + - list + - watch + - create + - update + - patch + - delete + - apiGroups: + - coordination.k8s.io + resources: + - leases + verbs: + - get + - list + - watch + - create + - update + - patch + - delete + - apiGroups: + - "" + resources: + - events + verbs: + - create + - patch +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: simplyblock-csi-addons-manager-role +rules: + # rbac-justified: the reconcilers emit events on the objects they drive + # (VolumeReplication, ReclaimSpaceJob, NetworkFence) in every namespace. + - apiGroups: + - "" + resources: + - events + verbs: + - create + - patch + # rbac-justified: the manager resolves the sidecar pod behind each + # CSIAddonsNode to build its gRPC connection, and lists namespaces to fan + # per-namespace ReclaimSpaceCronJob scheduling out cluster-wide. + - apiGroups: + - "" + resources: + - namespaces + - pods + verbs: + - get + - list + - watch + # rbac-justified: the VolumeReplication reconciler resolves the protected + # PVC, and the reclaim-space and volume-health paths annotate it. The + # finalizer update protects a replicated PVC from deletion mid-operation. + - apiGroups: + - "" + resources: + - persistentvolumeclaims + verbs: + - get + - list + - patch + - update + - watch + # rbac-justified: an addition beyond the stock upstream manifest, which + # omits this too (it's an opt-in feature, not something every deployment + # uses). The manager's sidecar-facing ReplicationServer proxy + # (internal/sidecar/service/volumereplication.go) hard-requires resolving + # a Secret named by VolumeReplicationClass's + # replication.storage.openshift.io/replication-secret-name(space) + # parameters before it will proxy ANY Replication RPC, even to a driver + # (like this one) that never reads the secret's own contents -- confirmed + # against a live cluster, where every EnableVolumeReplication attempt + # failed with "Failed to get secret ... resource name may not be empty" + # before ever reaching the sidecar. get only, no list/watch: the manager + # reads one named Secret per VolumeReplicationClass, never enumerates. + - apiGroups: + - "" + resources: + - secrets + verbs: + - get + - apiGroups: + - "" + resources: + - persistentvolumeclaims/finalizers + verbs: + - update + # rbac-justified: the PVC's volume handle (the id every csi-addons RPC + # addresses) lives on the PV, and the reclaim-space controller records + # progress annotations there. + - apiGroups: + - "" + resources: + - persistentvolumes + verbs: + - get + - list + - update + - watch + # rbac-justified: the manager reads the sidecars' leader-election Leases to + # pick the active sidecar per driver. Its own election lock is the + # namespaced Role above. + - apiGroups: + - coordination.k8s.io + resources: + - leases + verbs: + - get + - list + - watch + # rbac-justified: the manager owns these kinds: sidecars publish + # CSIAddonsNode, cron controllers materialize their child jobs, and every + # reconciler maintains status and finalizers on its own objects. + - apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes + - encryptionkeyrotationcronjobs + - encryptionkeyrotationjobs + - networkfenceclasses + - networkfences + - reclaimspacecronjobs + - reclaimspacejobs + verbs: + - create + - delete + - get + - list + - patch + - update + - watch + - apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes/finalizers + - encryptionkeyrotationcronjobs/finalizers + - encryptionkeyrotationjobs/finalizers + - networkfenceclasses/finalizers + - networkfences/finalizers + - reclaimspacecronjobs/finalizers + - reclaimspacejobs/finalizers + verbs: + - update + - apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes/status + - encryptionkeyrotationcronjobs/status + - encryptionkeyrotationjobs/status + - networkfenceclasses/status + - networkfences/status + - reclaimspacecronjobs/status + - reclaimspacejobs/status + verbs: + - get + - patch + - update + # rbac-justified: classes are read-only inputs (parameters, provisioner + # matching). The replication objects themselves are reconciled, so they and + # their status and finalizers are written, and VolumeGroupReplicationContent + # is created by the group flow the manager owns. + - apiGroups: + - replication.storage.openshift.io + resources: + - volumegroupreplicationclasses + - volumereplicationclasses + verbs: + - get + - list + - watch + - apiGroups: + - replication.storage.openshift.io + resources: + - volumegroupreplicationcontents + verbs: + - create + - delete + - get + - list + - patch + - update + - watch + - apiGroups: + - replication.storage.openshift.io + resources: + - volumegroupreplicationcontents/finalizers + - volumegroupreplications/finalizers + - volumereplications/finalizers + verbs: + - update + - apiGroups: + - replication.storage.openshift.io + resources: + - volumegroupreplicationcontents/status + - volumegroupreplications/status + verbs: + - get + - patch + - update + - apiGroups: + - replication.storage.openshift.io + resources: + - volumegroupreplications + verbs: + - get + - list + - patch + - update + - watch + - apiGroups: + - replication.storage.openshift.io + resources: + - volumereplications + verbs: + - create + - delete + - get + - list + - update + - watch + - apiGroups: + - replication.storage.openshift.io + resources: + - volumereplications/status + verbs: + - get + - list + - update + # rbac-justified: the scheduling and peer-matching logic resolves a PVC's + # StorageClass and its attachment state, both read-only. + - apiGroups: + - storage.k8s.io + resources: + - storageclasses + - volumeattachments + verbs: + - get + - list + - watch +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: RoleBinding +metadata: + name: simplyblock-csi-addons-leader-election-rolebinding + namespace: {{ .Release.Namespace }} +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: Role + name: simplyblock-csi-addons-leader-election-role +subjects: + - kind: ServiceAccount + name: simplyblock-csi-addons-controller-manager + namespace: {{ .Release.Namespace }} +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: simplyblock-csi-addons-manager-rolebinding +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: simplyblock-csi-addons-manager-role +subjects: + - kind: ServiceAccount + name: simplyblock-csi-addons-controller-manager + namespace: {{ .Release.Namespace }} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationclasses.yaml b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationclasses.yaml new file mode 100644 index 000000000..2ea2a447c --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationclasses.yaml @@ -0,0 +1,94 @@ +# VolumeGroupReplicationClass CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "replication.storage.openshift.io/v1alpha1/VolumeGroupReplicationClass")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: volumegroupreplicationclasses.replication.storage.openshift.io +spec: + group: replication.storage.openshift.io + names: + kind: VolumeGroupReplicationClass + listKind: VolumeGroupReplicationClassList + plural: volumegroupreplicationclasses + shortNames: + - vgrc + singular: volumegroupreplicationclass + scope: Cluster + versions: + - additionalPrinterColumns: + - jsonPath: .spec.provisioner + name: provisioner + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: VolumeGroupReplicationClass is the Schema for the volumegroupreplicationclasses + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + VolumeGroupReplicationClassSpec specifies parameters that an underlying storage system uses + when creating a volumegroup replica. A specific VolumeGroupReplicationClass is used by specifying + its name in a VolumeGroupReplication object. + properties: + parameters: + additionalProperties: + type: string + description: |- + Parameters is a key-value map with storage provisioner specific configurations for + creating volume group replicas + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: self == oldSelf + provisioner: + description: Provisioner is the name of storage provisioner + type: string + x-kubernetes-validations: + - message: provisioner is immutable + rule: self == oldSelf + required: + - provisioner + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: has(self.parameters) == has(oldSelf.parameters) + status: + description: VolumeGroupReplicationClassStatus defines the observed state + of VolumeGroupReplicationClass + type: object + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationcontents.yaml b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationcontents.yaml new file mode 100644 index 000000000..1a6374eb4 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplicationcontents.yaml @@ -0,0 +1,241 @@ +# VolumeGroupReplicationContent CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "replication.storage.openshift.io/v1alpha1/VolumeGroupReplicationContent")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: volumegroupreplicationcontents.replication.storage.openshift.io +spec: + group: replication.storage.openshift.io + names: + kind: VolumeGroupReplicationContent + listKind: VolumeGroupReplicationContentList + plural: volumegroupreplicationcontents + shortNames: + - vgrcontent + singular: volumegroupreplicationcontent + scope: Cluster + versions: + - additionalPrinterColumns: + - jsonPath: .spec.volumeGroupReplicationClassName + name: VolumeGroupReplicationClass + type: string + - jsonPath: .spec.volumeGroupReplicationRef.name + name: VolumeGroupReplication + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: VolumeGroupReplicationContent is the Schema for the volumegroupreplicationcontents + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: VolumeGroupReplicationContentSpec defines the desired state + of VolumeGroupReplicationContent + properties: + provisioner: + description: |- + provisioner is the name of the CSI driver used to create the physical + volume group on + the underlying storage system. + This MUST be the same as the name returned by the CSI GetPluginName() call for + that driver. + Required. + type: string + source: + description: |- + Source specifies whether the volume group is (or should be) dynamically provisioned + or already exists using the volumes listed here, and just requires a + Kubernetes object representation. + Required. + properties: + volumeHandles: + description: |- + VolumeHandles is a list of volume handles on the backend to be grouped + and replicated. + items: + type: string + type: array + required: + - volumeHandles + type: object + volumeGroupAttributes: + additionalProperties: + type: string + description: volumeGroupAttributes holds the contextual information + of the volume group. + type: object + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + volumeGroupReplicationClassName: + description: |- + VolumeGroupReplicationClassName is the name of the VolumeGroupReplicationClass from + which this group replication was (or will be) created. + Required. + type: string + volumeGroupReplicationHandle: + description: |- + VolumeGroupReplicationHandle is a unique id returned by the CSI driver + to identify the VolumeGroupReplication on the storage system. + type: string + volumeGroupReplicationRef: + description: |- + VolumeGroupreplicationRef specifies the VolumeGroupReplication object to which this + VolumeGroupReplicationContent object is bound. + VolumeGroupReplication.Spec.VolumeGroupReplicationContentName field must reference to + this VolumeGroupReplicationContent's name for the bidirectional binding to be valid. + For a pre-existing VolumeGroupReplication object, MUST provide an empty/nil value for + VolumeGroupReplicationRef for the auto-binding to happen. + properties: + apiVersion: + description: API version of the referent. + type: string + fieldPath: + description: |- + If referring to a piece of an object instead of an entire object, this string + should contain a valid JSON/Go field access statement, such as desiredState.manifest.containers[2]. + For example, if the object reference is to a container within a pod, this would take on a value like: + "spec.containers{name}" (where "name" refers to the name of the container that triggered + the event) or if no container name is specified "spec.containers[2]" (container with + index 2 in this pod). This syntax is chosen only to have some well-defined way of + referencing a part of an object. + type: string + kind: + description: |- + Kind of the referent. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + name: + description: |- + Name of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + namespace: + description: |- + Namespace of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/namespaces/ + type: string + resourceVersion: + description: |- + Specific resourceVersion to which this reference is made, if any. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#concurrency-control-and-consistency + type: string + uid: + description: |- + UID of the referent. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#uids + type: string + type: object + x-kubernetes-map-type: atomic + x-kubernetes-validations: + - message: volumeGroupReplicationRef.name, volumeGroupReplicationRef.namespace + and volumeGroupReplicationRef.uid must be set if volumeGroupReplicationRef + is defined + rule: 'self != null ? has(self.name) && has(self.__namespace__) + && has(self.uid) : true' + required: + - provisioner + - source + - volumeGroupReplicationClassName + type: object + status: + description: VolumeGroupReplicationContentStatus defines the status of + VolumeGroupReplicationContent + properties: + destinationVolumeGroupID: + description: |- + DestinationVolumeGroupID is the volume group ID on the + destination/target side, as reported by the SP. + type: string + persistentVolumeMappingList: + description: |- + PersistentVolumeMappingList is the list of PVs for the group + replication, enriched with source and destination volume handles. + This replaces PersistentVolumeRefList. When both fields are + present, consumers SHOULD prefer this field. + The maximum number of allowed PVs in the group is 100. + items: + description: |- + PersistentVolumeMapping contains the PV reference along with its + source and destination volume handles. The destination handle is + populated only when the SP supports GET_REPLICATION_DESTINATION_INFO + and destination info is available. + properties: + destinationVolumeHandle: + description: |- + DestinationVolumeHandle is the CSI volume handle on the + destination/target cluster, as reported by the SP. + This field is empty when destination info is not available. + type: string + name: + description: Name is the name of the PersistentVolume. + type: string + volumeHandle: + description: |- + VolumeHandle is the CSI volume handle of this PV on the + source cluster (i.e. PV.Spec.CSI.VolumeHandle). + type: string + required: + - name + - volumeHandle + type: object + type: array + persistentVolumeRefList: + description: |- + PersistentVolumeRefList is the list of PV for the group replication + The maximum number of allowed PV in the group is 100. + Deprecated: Use PersistentVolumeMappingList instead, which includes + source and destination volume handle information per PV. + items: + description: |- + LocalObjectReference contains enough information to let you locate the + referenced object inside the same namespace. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + type: array + type: object + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplications.yaml b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplications.yaml new file mode 100644 index 000000000..5d212a201 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumegroupreplications.yaml @@ -0,0 +1,309 @@ +# VolumeGroupReplication CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "replication.storage.openshift.io/v1alpha1/VolumeGroupReplication")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: volumegroupreplications.replication.storage.openshift.io +spec: + group: replication.storage.openshift.io + names: + kind: VolumeGroupReplication + listKind: VolumeGroupReplicationList + plural: volumegroupreplications + shortNames: + - vgr + singular: volumegroupreplication + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.volumeGroupReplicationClassName + name: VolumeGroupReplicationClass + type: string + - jsonPath: .spec.volumeGroupReplicationContentName + name: VolumeGroupReplicationContent + type: string + - jsonPath: .spec.replicationState + name: desiredState + type: string + - jsonPath: .status.state + name: currentState + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: VolumeGroupReplication is the Schema for the volumegroupreplications + API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: VolumeGroupReplicationSpec defines the desired state of VolumeGroupReplication + properties: + autoResync: + default: false + description: |- + AutoResync represents the group to be auto resynced when + ReplicationState is "secondary" + type: boolean + external: + default: false + description: |- + External represents if VolumeGroupReplication should be reconciled by the csi-addons controller + or an external controller managed by the storage vendor. + type: boolean + x-kubernetes-validations: + - message: source is immutable + rule: self == oldSelf + replicationState: + description: |- + ReplicationState represents the replication operation to be performed on the group. + Supported operations are "primary", "secondary" and "resync" + enum: + - primary + - secondary + - resync + type: string + source: + description: |- + Source specifies where a group replications will be created from. + This field is immutable after creation. + Required. + properties: + selector: + description: |- + Selector is a label query over persistent volume claims that are to be + grouped together for replication. + properties: + matchExpressions: + description: matchExpressions is a list of label selector + requirements. The requirements are ANDed. + items: + description: |- + A label selector requirement is a selector that contains values, a key, and an operator that + relates the key and values. + properties: + key: + description: key is the label key that the selector + applies to. + type: string + operator: + description: |- + operator represents a key's relationship to a set of values. + Valid operators are In, NotIn, Exists and DoesNotExist. + type: string + values: + description: |- + values is an array of string values. If the operator is In or NotIn, + the values array must be non-empty. If the operator is Exists or DoesNotExist, + the values array must be empty. This array is replaced during a strategic + merge patch. + items: + type: string + type: array + x-kubernetes-list-type: atomic + required: + - key + - operator + type: object + type: array + x-kubernetes-list-type: atomic + matchLabels: + additionalProperties: + type: string + description: |- + matchLabels is a map of {key,value} pairs. A single {key,value} in the matchLabels + map is equivalent to an element of matchExpressions, whose key field is "key", the + operator is "In", and the values array contains only "value". The requirements are ANDed. + type: object + type: object + x-kubernetes-map-type: atomic + x-kubernetes-validations: + - message: selector is immutable + rule: self == oldSelf + required: + - selector + type: object + x-kubernetes-validations: + - message: source is immutable + rule: self == oldSelf + volumeGroupReplicationClassName: + description: volumeGroupReplicationClassName is the volumeGroupReplicationClass + name for this VolumeGroupReplication resource + type: string + x-kubernetes-validations: + - message: volumeGroupReplicationClassName is immutable + rule: self == oldSelf + volumeGroupReplicationContentName: + description: Name of the VolumeGroupReplicationContent object created + for this volumeGroupReplication + type: string + x-kubernetes-validations: + - message: volumeGroupReplicationContentName is immutable + rule: self == oldSelf + volumeReplicationClassName: + description: |- + volumeReplicationClassName is the volumeReplicationClass name for the VolumeReplication object + created for this volumeGroupReplication + type: string + x-kubernetes-validations: + - message: volumeReplicationClassName is immutable + rule: self == oldSelf + volumeReplicationName: + description: Name of the VolumeReplication object created for this + volumeGroupReplication + type: string + x-kubernetes-validations: + - message: volumeReplicationName is immutable + rule: self == oldSelf + required: + - autoResync + - replicationState + - source + - volumeGroupReplicationClassName + type: object + status: + description: VolumeGroupReplicationStatus defines the observed state of + VolumeGroupReplication + properties: + conditions: + description: Conditions are the list of conditions and their status. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + destinationVolumeID: + description: |- + DestinationVolumeID is the volume ID on the destination/target side. + This field is set when the SP reports different source and destination + volume IDs. + type: string + lastCompletionTime: + format: date-time + type: string + lastStartTime: + format: date-time + type: string + lastSyncBytes: + format: int64 + type: integer + lastSyncDuration: + type: string + lastSyncTime: + format: date-time + type: string + message: + type: string + observedGeneration: + description: observedGeneration is the last generation change the + operator has dealt with + format: int64 + type: integer + persistentVolumeClaimsRefList: + description: |- + PersistentVolumeClaimsRefList is the list of PVCs for the volume group replication. + The maximum number of allowed PVCs in the group is 100. + items: + description: |- + LocalObjectReference contains enough information to let you locate the + referenced object inside the same namespace. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + type: array + state: + description: State captures the latest state of the replication operation. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplicationclasses.yaml b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplicationclasses.yaml new file mode 100644 index 000000000..b434380df --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplicationclasses.yaml @@ -0,0 +1,96 @@ +# VolumeReplicationClass CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "replication.storage.openshift.io/v1alpha1/VolumeReplicationClass")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: volumereplicationclasses.replication.storage.openshift.io +spec: + group: replication.storage.openshift.io + names: + kind: VolumeReplicationClass + listKind: VolumeReplicationClassList + plural: volumereplicationclasses + shortNames: + - vrc + singular: volumereplicationclass + scope: Cluster + versions: + - additionalPrinterColumns: + - jsonPath: .spec.provisioner + name: provisioner + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha1 + schema: + openAPIV3Schema: + description: VolumeReplicationClass is the Schema for the volumereplicationclasses + API. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + VolumeReplicationClassSpec specifies parameters that an underlying storage system uses + when creating a volume replica. A specific VolumeReplicationClass is used by specifying + its name in a VolumeReplication object. + properties: + parameters: + additionalProperties: + type: string + description: |- + Parameters is a key-value map with storage provisioner specific configurations for + creating volume replicas + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: self == oldSelf + provisioner: + description: Provisioner is the name of storage provisioner + type: string + x-kubernetes-validations: + - message: provisioner is immutable + rule: self == oldSelf + required: + - provisioner + type: object + x-kubernetes-validations: + - message: parameters are immutable + rule: has(self.parameters) == has(oldSelf.parameters) + status: + description: VolumeReplicationClassStatus defines the observed state of + VolumeReplicationClass. + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplications.yaml b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplications.yaml new file mode 100644 index 000000000..3deaca0e5 --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/replication.storage.openshift.io_volumereplications.yaml @@ -0,0 +1,225 @@ +# VolumeReplication CRD, vendored verbatim from csi-addons/kubernetes-csi-addons +# v0.15.0 (deploy/controller/crds.yaml), plus the chart's resource-policy +# annotation. The stock csi-addons controller-manager starts a controller +# per kind unconditionally, so every CRD it watches must exist for the +# manager to come up. This file installs one of them unless the cluster +# already serves the API. +{{- if and .Values.csiaddons.create (not (.Capabilities.APIVersions.Has "replication.storage.openshift.io/v1alpha1/VolumeReplication")) -}} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + helm.sh/resource-policy: keep + controller-gen.kubebuilder.io/version: v0.20.1 + name: volumereplications.replication.storage.openshift.io +spec: + group: replication.storage.openshift.io + names: + kind: VolumeReplication + listKind: VolumeReplicationList + plural: volumereplications + shortNames: + - vr + singular: volumereplication + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + - jsonPath: .spec.volumeReplicationClass + name: volumeReplicationClass + type: string + - jsonPath: .spec.dataSource.kind + name: SourceKind + type: string + - jsonPath: .spec.dataSource.name + name: SourceName + type: string + - jsonPath: .spec.replicationState + name: desiredState + type: string + - jsonPath: .status.state + name: currentState + type: string + name: v1alpha1 + schema: + openAPIV3Schema: + description: VolumeReplication is the Schema for the volumereplications API. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: VolumeReplicationSpec defines the desired state of VolumeReplication. + properties: + autoResync: + default: false + description: |- + AutoResync represents the volume to be auto resynced when + ReplicationState is "secondary" + type: boolean + dataSource: + description: DataSource represents the object associated with the + volume + properties: + apiGroup: + description: |- + APIGroup is the group for the resource being referenced. + If APIGroup is not specified, the specified Kind must be in the core API group. + For any other third-party types, APIGroup is required. + type: string + kind: + description: Kind is the type of resource being referenced + type: string + name: + description: Name is the name of resource being referenced + type: string + required: + - kind + - name + type: object + x-kubernetes-map-type: atomic + x-kubernetes-validations: + - message: dataSource is immutable + rule: self == oldSelf + replicationHandle: + description: replicationHandle represents an existing (but new) replication + id + type: string + replicationState: + description: |- + ReplicationState represents the replication operation to be performed on the volume. + Supported operations are "primary", "secondary" and "resync" + enum: + - primary + - secondary + - resync + type: string + volumeReplicationClass: + description: VolumeReplicationClass is the VolumeReplicationClass + name for this VolumeReplication resource + type: string + x-kubernetes-validations: + - message: volumeReplicationClass is immutable + rule: self == oldSelf + required: + - autoResync + - dataSource + - replicationState + - volumeReplicationClass + type: object + status: + description: VolumeReplicationStatus defines the observed state of VolumeReplication. + properties: + conditions: + description: Conditions are the list of conditions and their status. + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + destinationVolumeID: + description: |- + DestinationVolumeID is the volume ID on the destination/target side. + This field is set when the SP reports different source and destination + volume IDs. + type: string + lastCompletionTime: + format: date-time + type: string + lastStartTime: + format: date-time + type: string + lastSyncBytes: + format: int64 + type: integer + lastSyncDuration: + type: string + lastSyncTime: + format: date-time + type: string + message: + type: string + observedGeneration: + description: observedGeneration is the last generation change the + operator has dealt with + format: int64 + type: integer + state: + description: State captures the latest state of the replication operation. + type: string + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml index b0d444214..3f8c28526 100644 --- a/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml +++ b/helm-charts/charts/simplyblock-operator/templates/roles/manager_role.yaml @@ -176,6 +176,44 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch +- apiGroups: + - coordination.k8s.io + resources: + - leases + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes/status + verbs: + - get + - patch + - update - apiGroups: - discovery.k8s.io resources: @@ -295,7 +333,9 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks + - testfailovers - volumemigrations verbs: - create @@ -330,7 +370,9 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers + - testfailovers/finalizers - volumemigrations/finalizers verbs: - update @@ -362,7 +404,9 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status + - testfailovers/status - volumegroupsnapshotops/status - volumemigrations/status verbs: @@ -389,3 +433,27 @@ rules: - get - list - watch +- apiGroups: + - view.open-cluster-management.io + resources: + - managedclusterviews + verbs: + - create + - delete + - get + - list + - patch + - update + - watch +- apiGroups: + - work.open-cluster-management.io + resources: + - manifestworks + verbs: + - create + - delete + - get + - list + - patch + - update + - watch diff --git a/helm-charts/charts/simplyblock-operator/templates/setup-csi-addons-controller.yaml b/helm-charts/charts/simplyblock-operator/templates/setup-csi-addons-controller.yaml new file mode 100644 index 000000000..6ca2b2e7b --- /dev/null +++ b/helm-charts/charts/simplyblock-operator/templates/setup-csi-addons-controller.yaml @@ -0,0 +1,78 @@ +# The stock kubernetes-csi-addons controller-manager (design +# design-csi-addons-replication.md §4.1, prerequisite P0-5): it discovers +# driver endpoints through CSIAddonsNode objects and reconciles +# VolumeReplication by calling the driver's Replication gRPC through the +# csi-addons sidecar. Vendored from csi-addons/kubernetes-csi-addons v0.15.0 +# (deploy/controller/setup-controller.yaml) with the image pinned instead of +# :latest, the namespace bound to the release, and the legacy +# ControllerManagerConfig ConfigMap dropped (the manager neither mounts nor +# reads it). Deployed into the release namespace rather than kube-system: this +# is a chart component, not base cluster functionality. +{{- if .Values.csiaddons.create -}} +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + labels: + app.kubernetes.io/name: csi-addons + name: simplyblock-csi-addons-controller-manager + namespace: {{ .Release.Namespace }} +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: csi-addons + template: + metadata: + annotations: + kubectl.kubernetes.io/default-container: manager + labels: + app.kubernetes.io/name: csi-addons + control-plane: controller-manager + spec: + containers: + - name: manager + image: "{{ .Values.image.csiAddonsController.repository }}:{{ .Values.image.csiAddonsController.tag }}" + imagePullPolicy: {{ .Values.image.csiAddonsController.pullPolicy }} + command: + - /csi-addons-manager + args: + - --namespace=$(POD_NAMESPACE) + - --leader-elect + - --automaxprocs + env: + - name: POD_NAMESPACE + valueFrom: + fieldRef: + fieldPath: metadata.namespace + livenessProbe: + httpGet: + path: /healthz + port: 8081 + initialDelaySeconds: 15 + periodSeconds: 20 + readinessProbe: + httpGet: + path: /readyz + port: 8081 + initialDelaySeconds: 5 + periodSeconds: 10 + ports: + - containerPort: 8443 + name: metrics + protocol: TCP + resources: + limits: + cpu: 1000m + memory: 512Mi + requests: + cpu: 10m + memory: 64Mi + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + securityContext: + runAsNonRoot: true + serviceAccountName: simplyblock-csi-addons-controller-manager + terminationGracePeriodSeconds: 10 +{{- end -}} diff --git a/helm-charts/charts/simplyblock-operator/values.schema.json b/helm-charts/charts/simplyblock-operator/values.schema.json index e6307361f..d9184342f 100644 --- a/helm-charts/charts/simplyblock-operator/values.schema.json +++ b/helm-charts/charts/simplyblock-operator/values.schema.json @@ -433,6 +433,10 @@ "nodeDriverRegistrar": { "type": "string", "pattern": "^($|(quay\\.io/simplyblock-io|docker\\.io/simplyblock|public\\.ecr\\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$" + }, + "csiAddons": { + "type": "string", + "pattern": "^($|(quay\\.io/simplyblock-io|docker\\.io/simplyblock|public\\.ecr\\.aws/simply-block|quay\\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$" } } } diff --git a/helm-charts/charts/simplyblock-operator/values.yaml b/helm-charts/charts/simplyblock-operator/values.yaml index 8ef2b6202..9b27b9ea0 100644 --- a/helm-charts/charts/simplyblock-operator/values.yaml +++ b/helm-charts/charts/simplyblock-operator/values.yaml @@ -7,6 +7,14 @@ image: repository: quay.io/simplyblock-io/snapshot-controller tag: v8.2.0 pullPolicy: Always + csiAddonsController: + # Upstream registry until the image is mirrored to quay.io/simplyblock-io, + # which is the convention every other component follows. Keep the tag in + # lockstep with the vendored CRDs and RBAC (templates/*csi-addons*, + # templates/csiaddons.openshift.io_*, templates/replication.storage.*). + repository: quay.io/csiaddons/k8s-controller + tag: v0.15.0 + pullPolicy: Always simplyblock: repository: quay.io/simplyblock-io/simplyblock tag: "main" @@ -32,10 +40,21 @@ rbac: snapshotcontroller: create: true +csiaddons: + # Deploy the kubernetes-csi-addons machinery: the CRDs (VolumeReplication + # and friends) and the stock controller-manager. Off until the driver's + # csi-addons Replication service ships (design-csi-addons-replication.md + # Phase 1). The vendored manifests are the design's prerequisite P0-5. + create: false + externallyManagedSecret: # Specifies whether a externallyManagedSecret should be created create: true +spdkdev: + # Specifies whether a spdkdev should be created + create: false + # Configuration for the CSI to connect to the cluster csiConfig: simplybk: @@ -47,13 +66,19 @@ csiSecret: simplybk: secret: + +podAnnotations: {} + +# SimplyBlock Daemonset, Deployment, Statefulset annotations +simplyBlockAnnotations: {} + multiCluster: enable: false clusters: - cluster_id: secret: workers: - + deployment: # One of: standalone, managed, empty. # @@ -404,9 +429,12 @@ driver: # accounts nothing presents. enableServiceAccountAuth: false - # Overrides for the six CSI sidecars, as full `repository:tag` references. + # Overrides for the seven CSI sidecars, as full `repository:tag` references. # Empty takes the version this operator release ships, which is the - # combination it was tested against. + # combination it was tested against. csiAddons's own default is already the + # real upstream image (quay.io/csiaddons/k8s-sidecar) rather than a + # quay.io/simplyblock-io mirror, which does not exist yet -- see + # SimplyblockDriverSpec.SidecarImages.CSIAddons. sidecarImages: provisioner: "" attacher: "" @@ -414,8 +442,20 @@ driver: snapshotter: "" healthMonitor: "" nodeDriverRegistrar: "" + csiAddons: "" controlplane: + # deployment.profile: standalone options -- this cluster's own control plane. + local: + # Secret in this namespace holding a static admin bearer token this + # control plane accepts, under `token`, in addition to this operator's + # own Kubernetes identity. It is what lets a cluster this control plane + # manages remotely (deployment.profile: managed there, + # controlplane.managed.credentialsSecretRef naming the same value) create + # its backend cluster identity, since a Kubernetes TokenReview can never + # cross a cluster boundary. Empty grants no additional admin credential. + adminTokenSecretRef: "" + # Where the control plane is, for deployment.profile: managed. managed: # Base URL of the management API. Required for that profile. A loopback or @@ -427,6 +467,13 @@ controlplane: # Secret holding the CA the endpoint is verified against, under `ca.crt` or # `tls.crt`. Empty uses the system trust store. Named and unusable fails. caBundleSecretRef: "" + # Default storage-node image for a StorageCluster on this Kubernetes + # cluster whose own spec.storageNodes.image is unset. The managed control + # plane is a different Kubernetes cluster's install and says nothing about + # what this cluster's storage nodes should run, so there is no equivalent + # to the local profile's spec.image to fall back to without this -- + # required in practice for any StorageCluster that does not set its own. + storageNodeImage: "" observability: enabled: false @@ -447,12 +494,15 @@ controlplane: graylog: repository: quay.io/simplyblock-io/graylog tag: "5.0" + pullPolicy: Always rootPasswordSha2: "b87c15a8ae4736d771ca60a7cc2014baaeab19b11c31f5fedef9421958a403c9" passwordSecret: "is6SP2EdWg0NdmVGv6CEp5hRHNL7BKVMFem4t9pouMqDQnHwXMSomas1qcbKSt5yISr8eBHv4Y7Dbswhyz84Ut0TW6kqsiPs" maxNumberIndex: "3" + retentionPeriod: "7d" grafana: repository: quay.io/simplyblock-io/grafana tag: 10.0.12 + pullPolicy: IfNotPresent endpoint: "" # Receivers Grafana provisions into the "grafana-alerts" contact point. # Every enabled channel is notified for every simplyblock alert. With none @@ -511,6 +561,18 @@ controlplane: thanos: repository: quay.io/simplyblock-io/thanos tag: v0.31.0 + pullPolicy: IfNotPresent + minio: + repository: quay.io/simplyblock-io/minio + tag: RELEASE.2024-01-16T16-07-38Z + pullPolicy: IfNotPresent + mcRepository: quay.io/simplyblock-io/minio-client + mcTag: RELEASE.2024-01-16T16-06-34Z + bucket: thanos + accessKey: minioadmin + secretKey: minioadmin + storageSize: 50Gi + storageClass: fluentbit: repository: quay.io/simplyblock-io/fluent-bit tag: "1.8.11" @@ -1191,3 +1253,193 @@ tls: # and FoundationDB's peers, so there is nothing left for a deployment to # provision before it can be required. mutual_enabled: true + +# The Control Center — the simplyblock web console (control-center/ in the +# monorepo). Off by default: it proxies the Kubernetes API with the pod's +# ServiceAccount token, so enabling it is a deliberate decision about who may +# reach the Service. See control-center/README.md, "Who can do what". +controlCenter: + enabled: false + + # Naming. The console templates are self-contained and do not borrow the + # chart's helpers, so these control its object names. Default object name: + # -control-center. + nameOverride: "" + fullnameOverride: "" + + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center + # inherits .Chart.AppVersion when unset. This chart's appVersion is a + # floating tag, so pin a released console tag here for production. + tag: "" + pullPolicy: IfNotPresent + imagePullSecrets: [] + + # Stateless — location lives in the browser — so two replicas cost little and + # keep the console up through a node loss, which is when it matters most. + replicas: 2 + + # full | dr + # + # full: the storage console — clusters, Kubernetes, control plane — with a + # Disaster recovery section that reads the DR hub's dr.simplyblock.io CRDs + # when dr-simplyblock is installed on the same cluster. + # dr: the DR-only console. Nothing but the DR section; no storage CRDs, no + # operator API, no Prometheus. This is what the dr-simplyblock-hub chart + # deploys (console.enabled) on a hub without a simplyblock control plane; + # set it here only to run the same stripped-down console from this chart. + mode: full + # Ramen's ops namespace on the DR hub: where discovered ProtectedApplications + # live and where the console asks the API server what it may do. + drNamespace: ramen-ops + + # serviceaccount | passthrough + # + # serviceaccount: this pod attaches its own token to proxied requests. The + # browser holds no credential. The console's authority is the ClusterRole, + # shared by everyone who can reach it, so put authentication in front. + # + # passthrough: the browser supplies the bearer token and Kubernetes enforces + # that user's own RBAC. Per-user authority and a real audit trail, at the + # cost of needing an auth proxy that injects the token. If you use this, + # set rbac.create=false — the pod needs no permissions of its own. + authMode: serviceaccount + tokenRefreshSeconds: 600 + + kubernetesApi: https://kubernetes.default.svc + # SNI and Host presented to the API server. Must be a name on its + # certificate — change this only together with kubernetesApi. + kubernetesApiHost: kubernetes.default.svc + kubernetesApiCaFile: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt + # Empty defaults resolve to this chart's own services: the operator API on + # http://simplyblock-operator:8080 and Prometheus per + # prometheus.simplyblock.prometheusURL/prometheusPORT. + operatorUrl: "" + helmUrl: "" + prometheusUrl: "" + + # The control plane API, read-only, behind the console's own proxy. On a hub + # whose control plane manages storage clusters on other sites the clusters + # are not CRDs here, and this is where the Clusters and Control plane screens + # read them from. Every response is scrubbed of credentials before it leaves + # the console's pod, and the proxy refuses anything but GET. + controlPlane: + enabled: true + # Empty: this release's management API when deployment.profile is + # standalone (https when tls.enabled), and off otherwise. + url: "" + # Have the operator add the console's service account to the management + # API's admin accounts (serviceaccount mode). The console proxies only GETs. + trustServiceAccount: true + # Or a Secret (key: token) holding one of the control plane's static admin + # tokens (controlplane.local.adminTokenSecretRef); it wins when set. + tokenSecret: "" + + # The log store the Logs view searches: Graylog's search API (the shipped + # logs of every pod fluent-bit collects), read-only through the console's + # proxy, credentials attached server-side, responses scrubbed of credentials. + # Without it the Logs view falls back to a pod's live tail. + logStore: + enabled: true + # Empty: this release's Graylog when controlplane.observability.enabled. + url: "" + user: admin + # The Secret and key holding the user's password. Empty: the observability + # stack's own (simplyblock-grafana-secrets / MONITORING_SECRET). + passwordSecret: "" + passwordKey: "" + # Or a Secret (key: token) holding a Graylog access token; it wins when set. + tokenSecret: "" + + rbac: + # Required in serviceaccount mode: the proxied token needs these rules or + # every request returns 403 and the console renders empty. Set false only + # with authMode: passthrough, where each user's own RBAC applies instead. + create: true + # Let "Label for DR" patch the hub's own StorageClasses, nodes, PVCs and + # workloads. RBAC cannot narrow a patch to label keys, so this grants + # patch on those kinds; managed sites are labelled through dr-agent + # (LabelRequest) and need no grant here. + hubLabelling: false + + service: + type: ClusterIP + port: 80 + # With type NodePort: the node port to pin, else Kubernetes picks one. + nodePort: null + + ingress: + enabled: false + className: nginx + host: "" + annotations: + # Authentication is not optional in serviceaccount mode. Replace this with + # your OIDC/OAuth2 proxy annotations, or keep basic auth as a stop-gap. + nginx.ingress.kubernetes.io/auth-type: basic + nginx.ingress.kubernetes.io/auth-secret: simplyblock-control-center-auth + nginx.ingress.kubernetes.io/auth-realm: simplyblock Control Center + nginx.ingress.kubernetes.io/proxy-body-size: 2m + nginx.ingress.kubernetes.io/proxy-read-timeout: "3600" + tls: [] + + networkPolicy: + # Caps the console's egress to its three upstreams plus DNS. Off by default + # because the API-server CIDRs below are cluster-specific: narrow them to + # your API server endpoint (`kubectl get endpoints kubernetes -n default`) + # or your service CIDR before enabling. + enabled: false + apiServerCidrs: + - 10.0.0.0/8 + - 172.16.0.0/12 + - 192.168.0.0/16 + # Namespaces allowed to reach the console, in addition to the release + # namespace itself — add your ingress controller's namespace when the + # Ingress is enabled, e.g. [ingress-nginx]. + ingressFromNamespaces: [] + + # Deploys sb-mock (control-center/mock) next to the console and points the + # console's proxy at it instead of the real cluster: the Kubernetes API, + # operator API and Prometheus are all impersonated, reads come from a + # generated dataset, and writes persist without performing any change. For + # testing the UI and demo installs only — never enable in production. + mock: + enabled: false + image: + # defaults to quay.io when unset + registry: "" + repository: simplyblock-io/control-center-mock + # inherits .Chart.AppVersion when unset + tag: "" + pullPolicy: IfNotPresent + # small-healthy | medium-degraded | large-scale | dr-failover | chaos, + # or auto for a seeded random pick + dataset: auto + # 0 picks a random seed; a fixed value makes the world reproducible + seed: 0 + # simulator tick (Ops phase progression); "0" disables the simulator so + # e2e suites can drive it deterministically via POST /mockctl/advance + simInterval: 4s + # probability a simulated operation ends Failed + failRate: 0.1 + resources: + requests: + cpu: 10m + memory: 32Mi + limits: + cpu: 200m + memory: 256Mi + + resources: + requests: + cpu: 20m + memory: 48Mi + limits: + cpu: 500m + memory: 192Mi + + nodeSelector: {} + tolerations: [] + podAnnotations: {} + diff --git a/operator/Dockerfile b/operator/Dockerfile index 332bdd507..c51792d67 100644 --- a/operator/Dockerfile +++ b/operator/Dockerfile @@ -36,7 +36,7 @@ COPY operator/ . RUN CC=$([ "${TARGETARCH}" = "arm64" ] && echo "aarch64-linux-gnu-gcc" || echo "gcc") \ CGO_ENABLED=1 GOEXPERIMENT=boringcrypto \ GOOS=${TARGETOS:-linux} GOARCH=${TARGETARCH} \ - go build -a -o manager cmd/main.go + go build -a -o manager ./cmd # The worker probe ships in this image rather than in one of its own, so that # the Job a discovery run creates cannot be a version out of step with the diff --git a/operator/Makefile b/operator/Makefile index b03a001a7..5acd38bf0 100644 --- a/operator/Makefile +++ b/operator/Makefile @@ -177,11 +177,11 @@ lint-config: golangci-lint ## Verify golangci-lint linter configuration .PHONY: build build: manifests generate fmt vet ## Build manager binary. mkdir -p build - go build -o build/manager cmd/main.go + go build -o build/manager ./cmd .PHONY: run run: manifests generate fmt vet ## Run a controller from your host. - go run ./cmd/main.go + go run ./cmd # The API upgrade tool, built on its own and not in the operator image: it is a # prerequisite for running the new operator, so shipping it inside that diff --git a/operator/api/v1alpha2/controlplane_types.go b/operator/api/v1alpha2/controlplane_types.go index 53b0cca77..cef68a8f1 100644 --- a/operator/api/v1alpha2/controlplane_types.go +++ b/operator/api/v1alpha2/controlplane_types.go @@ -250,6 +250,20 @@ type LocalControlPlane struct { // +optional TLS ControlPlaneTLS `json:"tls,omitempty"` + // AdminTokenSecretRef names a Secret in this namespace holding a static + // admin bearer token this control plane accepts, under the `token` key, in + // addition to this deployment's own Kubernetes identity + // (SB_K8S_ADMIN_SERVICE_ACCOUNTS). It is what lets a cluster this control + // plane manages remotely (spec.source.managed there, + // ManagedControlPlane.CredentialsSecretRef naming the same value) + // authenticate a CreateCluster call, since a Kubernetes TokenReview can + // never cross a cluster boundary. + // + // The Secret is projected into the management API container's environment + // with secretKeyRef, so this operator never itself reads the plaintext. + // Absent grants no credential beyond the operator's own service account. + // +optional + AdminTokenSecretRef *corev1.LocalObjectReference `json:"adminTokenSecretRef,omitempty"` // Observability connects this control plane to the monitoring stack // installed beside it: Graylog, Grafana, and OpenSearch. Absent leaves // monitoring off. @@ -328,6 +342,20 @@ type ManagedControlPlane struct { // is verified against. Absent means the system trust store. // +optional CABundleSecretRef *corev1.LocalObjectReference `json:"caBundleSecretRef,omitempty"` + + // StorageNodeImage is the storage-node image a StorageCluster on this + // Kubernetes cluster defaults to when its own spec.storageNodes.image is + // unset. A local control plane's own spec.source.local.image doubles as + // this default (StorageNodeWorkloadReconciler.image), because a + // self-hosted deployment's control plane and its storage nodes are one + // release. A managed one is a different Kubernetes cluster's install and + // says nothing about what this cluster's storage nodes should run, so + // there is no equivalent to fall back to without this field -- every + // StorageCluster on a managed deployment must get an image from here or + // from its own spec. + // +kubebuilder:validation:Pattern=`^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$` + // +optional + StorageNodeImage string `json:"storageNodeImage,omitempty"` } // ControlPlaneSource selects whether this cluster hosts its control plane or is diff --git a/operator/api/v1alpha2/simplyblockdriver_types.go b/operator/api/v1alpha2/simplyblockdriver_types.go index 6ac16aeb2..2fde2d5c8 100644 --- a/operator/api/v1alpha2/simplyblockdriver_types.go +++ b/operator/api/v1alpha2/simplyblockdriver_types.go @@ -84,6 +84,22 @@ type SidecarImages struct { // +kubebuilder:validation:Pattern=`^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$` // +optional NodeDriverRegistrar string `json:"nodeDriverRegistrar,omitempty"` + + // CSIAddons is the kubernetes-csi-addons sidecar, on the controller + // plugin. It connects to the plugin's socket, probes the csi-addons + // Identity service for capabilities, and publishes a CSIAddonsNode so the + // kubernetes-csi-addons controller-manager (design + // design-csi-addons-replication.md §4.1) can reach the Replication + // service this driver serves. + // + // Unlike the other sidecars above, this one's upstream home is the + // csi-addons project's own registry, not simplyblock's: the allowlist + // carries quay.io/csiaddons alongside the simplyblock registries so a + // deployment can run the stock kubernetes-csi-addons sidecar image + // directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + // +kubebuilder:validation:Pattern=`^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$` + // +optional + CSIAddons string `json:"csiAddons,omitempty"` } // DriverTLSProvider is where the TLS certificate on this connection comes @@ -203,8 +219,8 @@ type SimplyblockDriverSpec struct { // +optional NodeResources corev1.ResourceRequirements `json:"nodeResources,omitempty"` - // SidecarImages overrides the six CSI sidecars, one field each. Unset takes - // the version this operator release ships. + // SidecarImages overrides the seven CSI sidecars, one field each. Unset + // takes the version this operator release ships. // +optional SidecarImages SidecarImages `json:"sidecarImages,omitempty"` diff --git a/operator/api/v1alpha2/storagesitedeployment_types.go b/operator/api/v1alpha2/storagesitedeployment_types.go new file mode 100644 index 000000000..93a18006c --- /dev/null +++ b/operator/api/v1alpha2/storagesitedeployment_types.go @@ -0,0 +1,298 @@ +// The storage deployment of a managed site, requested from the hub. +// +// A site's storage cluster is built from objects that live on the site's API +// server: an OperatorOps discovery, the ClusterDeploymentConfig draft it +// writes, and the StorageCluster the approved draft expands into. A hub that +// manages the site through Open Cluster Management does not reach that API +// server, so this kind is the hub-side request: it names the site and the +// sizing, and a controller carries the request to the site through a +// ManifestWork and projects the site's answer back through ManagedClusterViews. +// Approval is the same one-way gate the draft has on the site; it is flipped +// here and delivered there. +// +// Deleting the object withdraws nothing on the site: the storage cluster it +// requested stays, as a storage cluster is never torn down by deleting a +// request. Specified by docs/design/control-center-managed-discovery.md of the +// simplyblock-dr repository. + +package v1alpha2 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// StorageSiteDeploymentPhase is the request's own progress. +// +kubebuilder:validation:Enum=Pending;Discovering;Drafted;Deploying;Online;Failed +type StorageSiteDeploymentPhase string + +const ( + // StorageSiteDeploymentPhasePending is the request before the hub delivered + // anything to the site. + StorageSiteDeploymentPhasePending StorageSiteDeploymentPhase = "Pending" + + // StorageSiteDeploymentPhaseDiscovering is the discovery running on the site: + // the draft is not written yet, or names no node yet. + StorageSiteDeploymentPhaseDiscovering StorageSiteDeploymentPhase = "Discovering" + + // StorageSiteDeploymentPhaseDrafted is a draft with nodes on the site, sized + // as the request says, awaiting approval. + StorageSiteDeploymentPhaseDrafted StorageSiteDeploymentPhase = "Drafted" + + // StorageSiteDeploymentPhaseDeploying is an approved draft expanding into a + // StorageCluster that is not Online yet. + StorageSiteDeploymentPhaseDeploying StorageSiteDeploymentPhase = "Deploying" + + // StorageSiteDeploymentPhaseOnline is the StorageCluster Online on the site. + StorageSiteDeploymentPhaseOnline StorageSiteDeploymentPhase = "Online" + + // StorageSiteDeploymentPhaseFailed is the site's own failure: the draft or the + // StorageCluster failed, or the hub could not deliver the request. + StorageSiteDeploymentPhaseFailed StorageSiteDeploymentPhase = "Failed" +) + +// StorageSiteDiscovery is the discovery the site runs: which nodes are +// inspected. It is the hub-side form of OperatorOps.spec.discover. +type StorageSiteDiscovery struct { + // EnableControlPlaneNodes lets the discovery consider the nodes that run the + // API server. Every server of a small distribution is one, so a three-node + // site has no storage without it. + // +optional + EnableControlPlaneNodes *bool `json:"enableControlPlaneNodes,omitempty"` + + // Workers limits the discovery to these nodes. Empty is every worker. + // +optional + // +listType=set + Workers []string `json:"workers,omitempty"` + + // NodeSelector limits the discovery to the nodes carrying these labels. + // +optional + NodeSelector map[string]string `json:"nodeSelector,omitempty"` +} + +// StorageSiteSizing is the cluster template written onto the draft before it +// is approved: the fields of ClusterDeploymentConfig.spec.cluster a reviewer +// decides. Absent fields keep what the discovery wrote. +type StorageSiteSizing struct { + // Name is the StorageCluster's name on the site. + // +kubebuilder:validation:MaxLength=63 + // +optional + Name string `json:"name,omitempty"` + + // VCPUCount is the number of vCPUs each storage node takes. + // +kubebuilder:validation:Minimum=1 + // +optional + VCPUCount *int32 `json:"vcpuCount,omitempty"` + + // MinHugePagesSize is the hugepage memory each storage node takes, as a + // quantity ("8G"). + // +optional + MinHugePagesSize string `json:"minHugePagesSize,omitempty"` + + // MaxSubsystemCount is the number of NVMe-oF subsystems each node serves. + // +kubebuilder:validation:Minimum=1 + // +optional + MaxSubsystemCount *int32 `json:"maxSubsystemCount,omitempty"` + + // EnableDriveFormat lets the deployment format the devices it takes. + // +optional + EnableDriveFormat *bool `json:"enableDriveFormat,omitempty"` + + // EnableJournalDevice dedicates one device per node to the journal. + // +optional + EnableJournalDevice *bool `json:"enableJournalDevice,omitempty"` + + // Stripe is the erasure-coding layout. + // +optional + Stripe *StripeSpec `json:"stripe,omitempty"` +} + +// StorageSiteDeploymentSpec is the request for one site's storage cluster. +// +kubebuilder:validation:XValidation:rule="!has(oldSelf.approved) || !oldSelf.approved || self.approved",message="approval is one-way: an approved deployment cannot be un-approved" +type StorageSiteDeploymentSpec struct { + // Cluster is the OCM ManagedCluster the storage is deployed on. The request's + // ManifestWork and views live in its namespace on the hub. Immutable. + // +kubebuilder:validation:Required + // +kubebuilder:validation:MinLength=1 + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="cluster is immutable" + Cluster string `json:"cluster"` + + // SiteNamespace is the simplyblock operator's namespace on the site, where + // the discovery and the draft live. + // +kubebuilder:default=simplyblock + // +kubebuilder:validation:MaxLength=63 + // +optional + SiteNamespace string `json:"siteNamespace,omitempty"` + + // DraftName is the ClusterDeploymentConfig the discovery writes on the site + // and the request sizes and approves. Immutable. + // +kubebuilder:default=site-draft + // +kubebuilder:validation:MaxLength=63 + // +kubebuilder:validation:XValidation:rule="self == oldSelf",message="draftName is immutable" + // +optional + DraftName string `json:"draftName,omitempty"` + + // Discover is the discovery the site runs first. Changing it runs another + // discovery, which rewrites the draft. + // +optional + Discover StorageSiteDiscovery `json:"discover,omitempty"` + + // Sizing is written onto the draft's cluster template once the draft exists, + // so the reviewer sees the sized draft before approving it. + // +optional + Sizing *StorageSiteSizing `json:"sizing,omitempty"` + + // Approved is the review gate, delivered to the draft on the site. One-way, + // as the draft's own gate is. + // +kubebuilder:default=false + // +optional + Approved bool `json:"approved"` +} + +// StorageSiteDraft is the draft as the site reports it. +type StorageSiteDraft struct { + // Name is the ClusterDeploymentConfig on the site. + Name string `json:"name"` + + // Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + // Failed). + // +optional + Phase string `json:"phase,omitempty"` + + // Message is what the site says about the draft: validation findings while + // it is a draft, the expansion's step afterwards. + // +optional + Message string `json:"message,omitempty"` + + // Approved is whether the draft is approved on the site. + // +optional + Approved bool `json:"approved,omitempty"` + + // Cluster is the draft's cluster template, with the sizing applied. + // +optional + Cluster *ClusterTemplate `json:"cluster,omitempty"` + + // NodeSets are the nodes and devices the discovery found, for review. + // +optional + NodeSets []NodeSet `json:"nodeSets,omitempty"` + + // NodeRefs are the StorageNode objects the expansion created. + // +optional + // +listType=set + NodeRefs []string `json:"nodeRefs,omitempty"` +} + +// StorageSiteNode is one storage node of the deployed cluster, as the site +// reports it. +type StorageSiteNode struct { + // Name is the StorageNode object on the site. + Name string `json:"name"` + + // Phase is the node's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Hostname is the Kubernetes node it runs on. + // +optional + Hostname string `json:"hostname,omitempty"` +} + +// StorageSiteCluster is the StorageCluster the approved draft produced. +type StorageSiteCluster struct { + // Name is the StorageCluster object on the site. + Name string `json:"name"` + + // UUID is the storage cluster's id in the control plane, which a + // StorageClass names in cluster_id. + // +optional + UUID string `json:"uuid,omitempty"` + + // Phase is the StorageCluster's phase on the site. + // +optional + Phase string `json:"phase,omitempty"` + + // Pool is the pool the cluster was created with, which a StorageClass names + // in pool_name. + // +optional + Pool string `json:"pool,omitempty"` + + // Nodes are the cluster's storage nodes. + // +optional + // +listType=map + // +listMapKey=name + Nodes []StorageSiteNode `json:"nodes,omitempty"` +} + +// StorageSiteDeploymentStatus is what the site reports back, projected. +type StorageSiteDeploymentStatus struct { + // Phase is the request's own progress. + // +optional + Phase StorageSiteDeploymentPhase `json:"phase,omitempty"` + + // Message is the reason the phase is what it is: one sentence, replaced as + // the request moves, and never a log. + // +optional + Message string `json:"message,omitempty"` + + // ObservedGeneration is the generation the rest of this status was computed + // from. + // +optional + ObservedGeneration int64 `json:"observedGeneration,omitempty"` + + // WorkName is the ManifestWork carrying the request to the site. + // +optional + WorkName string `json:"workName,omitempty"` + + // Draft is the draft as the site reports it. + // +optional + Draft *StorageSiteDraft `json:"draft,omitempty"` + + // StorageCluster is the cluster the approved draft produced. + // +optional + StorageCluster *StorageSiteCluster `json:"storageCluster,omitempty"` + + // Conditions: Delivered (the work is applied on the site), Discovered (the + // draft names nodes), Approved (the site's draft is approved), Ready (the + // StorageCluster is Online). + // +optional + // +listType=map + // +listMapKey=type + Conditions []metav1.Condition `json:"conditions,omitempty"` +} + +// +kubebuilder:object:root=true +// +kubebuilder:subresource:status +// +kubebuilder:resource:scope=Namespaced,shortName=sbsd +// +kubebuilder:printcolumn:name="Cluster",type=string,JSONPath=".spec.cluster" +// +kubebuilder:printcolumn:name="Approved",type=boolean,JSONPath=".spec.approved" +// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=".status.phase" +// +kubebuilder:printcolumn:name="Draft",type=string,JSONPath=".status.draft.phase" +// +kubebuilder:printcolumn:name="Storage",type=string,JSONPath=".status.storageCluster.phase" +// +kubebuilder:printcolumn:name="Message",type=string,JSONPath=".status.message",priority=1 +// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=".metadata.creationTimestamp" + +// StorageSiteDeployment requests a managed site's storage cluster from the hub: +// a discovery on the site, the sizing of the draft it writes, and the approval +// that expands the draft into a StorageCluster. The hub carries the request +// through OCM and projects the site's draft and cluster into the status. +// Deleting the request leaves the storage cluster alone. +type StorageSiteDeployment struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + Spec StorageSiteDeploymentSpec `json:"spec,omitempty"` + Status StorageSiteDeploymentStatus `json:"status,omitempty"` +} + +// +kubebuilder:object:root=true + +// StorageSiteDeploymentList contains a list of StorageSiteDeployment. +type StorageSiteDeploymentList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + Items []StorageSiteDeployment `json:"items"` +} + +func init() { + SchemeBuilder.Register(&StorageSiteDeployment{}, &StorageSiteDeploymentList{}) +} diff --git a/operator/api/v1alpha2/testfailover_types.go b/operator/api/v1alpha2/testfailover_types.go new file mode 100644 index 000000000..cf2e1e852 --- /dev/null +++ b/operator/api/v1alpha2/testfailover_types.go @@ -0,0 +1,297 @@ +// One non-disruptive test-failover drill. +// +// A test failover proves an application can be recovered from a point-in-time +// copy, in isolation, without disturbing the running production. The recovery +// point is always a snapshot and the result is always a clone, so the source is +// never touched. The hub coordinates the drill: it reads the source on its +// cluster, resolves a recovery point on the recovery cluster's backend, clones +// it there, and places the clone as a bound PVC in an isolated namespace on the +// recovery cluster. Deleting the object reclaims the clone and any snapshot the +// drill took. +// +// Specified by operator/docs/designs/design-test-failover.md, whose Appendix A +// is this file. + +package v1alpha2 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "github.com/simplyblock/atlas/statemachine" +) + +// TestFailoverScope selects what a drill recovers. +// +kubebuilder:validation:Enum=Volume;Group +type TestFailoverScope string + +const ( + // TestFailoverScopeVolume recovers a single source volume, named by a PVC. + TestFailoverScopeVolume TestFailoverScope = "Volume" + + // TestFailoverScopeGroup recovers a consistency group from one + // group-consistent point. + TestFailoverScopeGroup TestFailoverScope = "Group" +) + +// TestFailoverPhase is the drill's own progress. +// +kubebuilder:validation:Enum=Pending;Provisioning;Ready;Failed;TearingDown +type TestFailoverPhase string + +const ( + TestFailoverPhasePending TestFailoverPhase = "Pending" + TestFailoverPhaseProvisioning TestFailoverPhase = "Provisioning" + TestFailoverPhaseReady TestFailoverPhase = "Ready" + TestFailoverPhaseFailed TestFailoverPhase = "Failed" + TestFailoverPhaseTearingDown TestFailoverPhase = "TearingDown" +) + +// TestFailoverStep is one step of a running drill. Which steps belong to which +// phase of the flow is declared by the drill's graph rather than by this type, +// which is why the enum stays flat. +// +kubebuilder:validation:Enum=ResolvingSource;ResolvingPoint;Shipping;Cloning;Placing;Releasing +type TestFailoverStep string + +const ( + // TestFailoverStepResolvingSource reads the source PVC and PV on the source + // cluster through a ManagedClusterView to learn the source volume's handle. + TestFailoverStepResolvingSource TestFailoverStep = "ResolvingSource" + + // TestFailoverStepResolvingPoint resolves the recovery point on the recovery + // cluster's backend: a fresh source snapshot, or the latest replicated one. + TestFailoverStepResolvingPoint TestFailoverStep = "ResolvingPoint" + + // TestFailoverStepShipping ships the recovery point to a backend that holds no + // copy of it (the cross-cluster, separate-backend case). + TestFailoverStepShipping TestFailoverStep = "Shipping" + + // TestFailoverStepCloning clones the recovery point into a writable volume on + // the recovery cluster's backend. + TestFailoverStepCloning TestFailoverStep = "Cloning" + + // TestFailoverStepPlacing delivers the bubble PV and PVC to the recovery + // cluster and waits for the PVC to bind. + TestFailoverStepPlacing TestFailoverStep = "Placing" + + // TestFailoverStepReleasing tears the drill down: removes the placed objects, + // reclaims the clone, and deletes any snapshot the drill took. + TestFailoverStepReleasing TestFailoverStep = "Releasing" +) + +// TestFailoverSpec is the request for one non-disruptive test-failover drill. +// +// The source is named by where it runs and what it is, so the hub can find it +// without anyone extracting a backend handle by hand. SourceNamespace is +// required for a Volume drill, where the source is a PVC, and unused for a Group +// drill, where SourceRef names a consistency group. +// +kubebuilder:validation:XValidation:rule="self.scope != 'Volume' || has(self.sourceNamespace)",message="sourceNamespace is required for scope=Volume" +type TestFailoverSpec struct { + // Scope selects what the drill recovers. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + Scope TestFailoverScope `json:"scope"` + + // SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + // the source there through a ManagedClusterView. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + SourceCluster string `json:"sourceCluster"` + + // SourceNamespace is the namespace of the source PVC on SourceCluster. + // Required for scope=Volume. Immutable. + // +optional + // +k8s:immutable + SourceNamespace string `json:"sourceNamespace,omitempty"` + + // SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + // SourceNamespace (scope=Volume), or a consistency group (scope=Group). + // Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + SourceRef string `json:"sourceRef"` + + // BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + // the replicated point, or another cluster. It must differ from SourceCluster; + // test-failover recovers onto a different cluster, never in place. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + BubbleCluster string `json:"bubbleCluster"` + + // BubbleNamespace is the namespace on the bubble cluster where the recovered + // PVCs are created. Immutable. + // +kubebuilder:default=bubble + // +optional + // +k8s:immutable + BubbleNamespace string `json:"bubbleNamespace,omitempty"` + + // TTLSeconds is an optional maximum lifetime: the drill is torn down after it + // even without a delete, so a forgotten drill cannot hold a clone forever. + // +kubebuilder:validation:Minimum=0 + // +optional + TTLSeconds *int64 `json:"ttlSeconds,omitempty"` +} + +// TestFailoverClone is one recovered volume: the source it came from, the +// snapshot and clone the drill built, and the PVC placed on the bubble cluster. +type TestFailoverClone struct { + // SourceRef is the source volume, or group member, the recovered volume maps + // to. + SourceRef string `json:"sourceRef"` + + // SourceHandle is the source volume's backend handle, read from its PV. + // +optional + SourceHandle string `json:"sourceHandle,omitempty"` + + // SnapshotID is the recovery-point snapshot: the replicated snapshot already on + // the bubble cluster's backend that the clone is built from. + // +optional + SnapshotID string `json:"snapshotID,omitempty"` + + // CloneID is the backend id of the writable clone. + // +optional + CloneID string `json:"cloneID,omitempty"` + + // PVCName is the bound PVC in the bubble namespace on the bubble cluster. + // +optional + PVCName string `json:"pvcName,omitempty"` + + // SizeBytes is the recovered volume's size. + // +optional + SizeBytes int64 `json:"sizeBytes,omitempty"` + + // SourceVolumeContext is the source PV's CSI volumeAttributes, minus the + // identity and provisioner keys, carried onto the bubble PV so the node plugin + // receives a non-nil VolumeContext when it stages the clone. The clone's own + // identity (NQN, connections, nsId, and so on) is re-resolved from the clone + // handle at stage time, so only the class-level parameters are carried; the + // identity keys are dropped so a failed clone lookup can never point the mount + // back at the source. + // +optional + SourceVolumeContext map[string]string `json:"sourceVolumeContext,omitempty"` + + // SourceFSType is the source PV's CSI fsType, carried onto the bubble PV so the + // node plugin stages the clone with the filesystem it actually carries. The + // clone is a block copy of the source, so its filesystem is the source's; an + // empty fsType makes the node plugin default to ext4 and refuse to mount an XFS + // volume. + // +optional + SourceFSType string `json:"sourceFSType,omitempty"` + // SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + // carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + // claim that omitted the mode defaulted to Filesystem and the kubelet asked + // the node plugin to mount a raw guest disk (2026-10-03). + // +optional + SourceVolumeMode string `json:"sourceVolumeMode,omitempty"` +} + +// TestFailoverReport is the evidence a drill produces. +type TestFailoverReport struct { + // BubbleCluster is the cluster the drill recovered onto. + // +optional + BubbleCluster string `json:"bubbleCluster,omitempty"` + + // RecoveryPoint is the snapshot or group generation the drill recovered. + // +optional + RecoveryPoint string `json:"recoveryPoint,omitempty"` + + // RecoveryPointTime is when that point was taken. + // +optional + RecoveryPointTime *metav1.Time `json:"recoveryPointTime,omitempty"` + + // RecoveryPointAgeSeconds is the drill time minus the recovery-point time. + // +optional + RecoveryPointAgeSeconds int64 `json:"recoveryPointAgeSeconds,omitempty"` + + // InvariantsHeld is true only when the source fingerprint taken before the + // drill matches the one taken at Ready. A Ready drill with this false is a + // defect. + // +optional + InvariantsHeld bool `json:"invariantsHeld,omitempty"` +} + +// TestFailoverStatus is the observed state of one drill. +type TestFailoverStatus struct { + // Phase is the drill's own progress. + // +optional + Phase TestFailoverPhase `json:"phase,omitempty"` + + // Step is the position of the running drill's state machine. + // +kubebuilder:validation:XValidation:rule="!has(self.state) || self.state in ['ResolvingSource','ResolvingPoint','Shipping','Cloning','Placing','Releasing']",message="unknown step" + // +optional + Step statemachine.KubeSnapshot `json:"step,omitempty"` + + // Message is the reason the phase is what it is: one sentence, replaced as the + // drill moves, and never a log. + // +optional + Message string `json:"message,omitempty"` + + // Triggered records that the current step's side effect was issued, so a + // restart does not repeat it. + // +optional + Triggered bool `json:"triggered,omitempty"` + + // ObservedGeneration is the generation the rest of this status was computed + // from, so a stale status can be told from a current one. + // +optional + ObservedGeneration int64 `json:"observedGeneration,omitempty"` + + // Clones is one entry per recovered volume. + // +optional + // +listType=map + // +listMapKey=sourceRef + Clones []TestFailoverClone `json:"clones,omitempty"` + + // Report is the drill's evidence, populated as it reaches Ready. + // +optional + Report *TestFailoverReport `json:"report,omitempty"` + + // StartedAt is when the drill started. + // +optional + StartedAt *metav1.Time `json:"startedAt,omitempty"` + + // ReadyAt is when every recovered PVC became bound. + // +optional + ReadyAt *metav1.Time `json:"readyAt,omitempty"` + + // CompletedAt is when the drill reached a terminal phase. + // +optional + CompletedAt *metav1.Time `json:"completedAt,omitempty"` +} + +// +kubebuilder:object:root=true +// +kubebuilder:subresource:status +// +kubebuilder:resource:scope=Namespaced,shortName=tfo +// +kubebuilder:printcolumn:name="Scope",type=string,JSONPath=".spec.scope" +// +kubebuilder:printcolumn:name="Source",type=string,JSONPath=".spec.sourceRef" +// +kubebuilder:printcolumn:name="On",type=string,JSONPath=".spec.sourceCluster" +// +kubebuilder:printcolumn:name="Bubble",type=string,JSONPath=".spec.bubbleCluster" +// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=".status.phase" +// +kubebuilder:printcolumn:name="Step",type=string,JSONPath=".status.step.state" +// +kubebuilder:printcolumn:name="Message",type=string,JSONPath=".status.message",priority=1 +// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=".metadata.creationTimestamp" + +// TestFailover is a one-way, non-disruptive test-failover drill. It recovers a +// source volume, or a consistency group, from a snapshot into an isolated +// namespace on a chosen cluster as bound PVCs, without touching the source. The +// hub reads the source on its cluster and places the bubble on the recovery +// cluster through OCM. It runs to a terminal phase, or holds Ready until it is +// deleted, and deletion reclaims the clones and any snapshots the drill took. +type TestFailover struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + Spec TestFailoverSpec `json:"spec,omitempty"` + Status TestFailoverStatus `json:"status,omitempty"` +} + +// +kubebuilder:object:root=true + +// TestFailoverList contains a list of TestFailover. +type TestFailoverList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + Items []TestFailover `json:"items"` +} + +func init() { + SchemeBuilder.Register(&TestFailover{}, &TestFailoverList{}) +} diff --git a/operator/api/v1alpha2/zz_generated.deepcopy.go b/operator/api/v1alpha2/zz_generated.deepcopy.go index d6c84863d..2f890f8f0 100644 --- a/operator/api/v1alpha2/zz_generated.deepcopy.go +++ b/operator/api/v1alpha2/zz_generated.deepcopy.go @@ -1151,6 +1151,11 @@ func (in *LocalControlPlane) DeepCopyInto(out *LocalControlPlane) { } } in.TLS.DeepCopyInto(&out.TLS) + if in.AdminTokenSecretRef != nil { + in, out := &in.AdminTokenSecretRef, &out.AdminTokenSecretRef + *out = new(v1.LocalObjectReference) + **out = **in + } if in.Observability != nil { in, out := &in.Observability, &out.Observability *out = new(ControlPlaneObservability) @@ -3646,6 +3651,257 @@ func (in *StoragePoolStatus) DeepCopy() *StoragePoolStatus { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteCluster) DeepCopyInto(out *StorageSiteCluster) { + *out = *in + if in.Nodes != nil { + in, out := &in.Nodes, &out.Nodes + *out = make([]StorageSiteNode, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteCluster. +func (in *StorageSiteCluster) DeepCopy() *StorageSiteCluster { + if in == nil { + return nil + } + out := new(StorageSiteCluster) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeployment) DeepCopyInto(out *StorageSiteDeployment) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + in.Spec.DeepCopyInto(&out.Spec) + in.Status.DeepCopyInto(&out.Status) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeployment. +func (in *StorageSiteDeployment) DeepCopy() *StorageSiteDeployment { + if in == nil { + return nil + } + out := new(StorageSiteDeployment) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeployment) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentList) DeepCopyInto(out *StorageSiteDeploymentList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]StorageSiteDeployment, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentList. +func (in *StorageSiteDeploymentList) DeepCopy() *StorageSiteDeploymentList { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *StorageSiteDeploymentList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentSpec) DeepCopyInto(out *StorageSiteDeploymentSpec) { + *out = *in + in.Discover.DeepCopyInto(&out.Discover) + if in.Sizing != nil { + in, out := &in.Sizing, &out.Sizing + *out = new(StorageSiteSizing) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentSpec. +func (in *StorageSiteDeploymentSpec) DeepCopy() *StorageSiteDeploymentSpec { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDeploymentStatus) DeepCopyInto(out *StorageSiteDeploymentStatus) { + *out = *in + if in.Draft != nil { + in, out := &in.Draft, &out.Draft + *out = new(StorageSiteDraft) + (*in).DeepCopyInto(*out) + } + if in.StorageCluster != nil { + in, out := &in.StorageCluster, &out.StorageCluster + *out = new(StorageSiteCluster) + (*in).DeepCopyInto(*out) + } + if in.Conditions != nil { + in, out := &in.Conditions, &out.Conditions + *out = make([]metav1.Condition, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDeploymentStatus. +func (in *StorageSiteDeploymentStatus) DeepCopy() *StorageSiteDeploymentStatus { + if in == nil { + return nil + } + out := new(StorageSiteDeploymentStatus) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDiscovery) DeepCopyInto(out *StorageSiteDiscovery) { + *out = *in + if in.EnableControlPlaneNodes != nil { + in, out := &in.EnableControlPlaneNodes, &out.EnableControlPlaneNodes + *out = new(bool) + **out = **in + } + if in.Workers != nil { + in, out := &in.Workers, &out.Workers + *out = make([]string, len(*in)) + copy(*out, *in) + } + if in.NodeSelector != nil { + in, out := &in.NodeSelector, &out.NodeSelector + *out = make(map[string]string, len(*in)) + for key, val := range *in { + (*out)[key] = val + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDiscovery. +func (in *StorageSiteDiscovery) DeepCopy() *StorageSiteDiscovery { + if in == nil { + return nil + } + out := new(StorageSiteDiscovery) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteDraft) DeepCopyInto(out *StorageSiteDraft) { + *out = *in + if in.Cluster != nil { + in, out := &in.Cluster, &out.Cluster + *out = new(ClusterTemplate) + (*in).DeepCopyInto(*out) + } + if in.NodeSets != nil { + in, out := &in.NodeSets, &out.NodeSets + *out = make([]NodeSet, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } + if in.NodeRefs != nil { + in, out := &in.NodeRefs, &out.NodeRefs + *out = make([]string, len(*in)) + copy(*out, *in) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteDraft. +func (in *StorageSiteDraft) DeepCopy() *StorageSiteDraft { + if in == nil { + return nil + } + out := new(StorageSiteDraft) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteNode) DeepCopyInto(out *StorageSiteNode) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteNode. +func (in *StorageSiteNode) DeepCopy() *StorageSiteNode { + if in == nil { + return nil + } + out := new(StorageSiteNode) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *StorageSiteSizing) DeepCopyInto(out *StorageSiteSizing) { + *out = *in + if in.VCPUCount != nil { + in, out := &in.VCPUCount, &out.VCPUCount + *out = new(int32) + **out = **in + } + if in.MaxSubsystemCount != nil { + in, out := &in.MaxSubsystemCount, &out.MaxSubsystemCount + *out = new(int32) + **out = **in + } + if in.EnableDriveFormat != nil { + in, out := &in.EnableDriveFormat, &out.EnableDriveFormat + *out = new(bool) + **out = **in + } + if in.EnableJournalDevice != nil { + in, out := &in.EnableJournalDevice, &out.EnableJournalDevice + *out = new(bool) + **out = **in + } + if in.Stripe != nil { + in, out := &in.Stripe, &out.Stripe + *out = new(StripeSpec) + (*in).DeepCopyInto(*out) + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new StorageSiteSizing. +func (in *StorageSiteSizing) DeepCopy() *StorageSiteSizing { + if in == nil { + return nil + } + out := new(StorageSiteSizing) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *StripeSpec) DeepCopyInto(out *StripeSpec) { *out = *in @@ -3671,6 +3927,166 @@ func (in *StripeSpec) DeepCopy() *StripeSpec { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailover) DeepCopyInto(out *TestFailover) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ObjectMeta.DeepCopyInto(&out.ObjectMeta) + in.Spec.DeepCopyInto(&out.Spec) + in.Status.DeepCopyInto(&out.Status) +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailover. +func (in *TestFailover) DeepCopy() *TestFailover { + if in == nil { + return nil + } + out := new(TestFailover) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *TestFailover) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailoverClone) DeepCopyInto(out *TestFailoverClone) { + *out = *in + if in.SourceVolumeContext != nil { + in, out := &in.SourceVolumeContext, &out.SourceVolumeContext + *out = make(map[string]string, len(*in)) + for key, val := range *in { + (*out)[key] = val + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailoverClone. +func (in *TestFailoverClone) DeepCopy() *TestFailoverClone { + if in == nil { + return nil + } + out := new(TestFailoverClone) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailoverList) DeepCopyInto(out *TestFailoverList) { + *out = *in + out.TypeMeta = in.TypeMeta + in.ListMeta.DeepCopyInto(&out.ListMeta) + if in.Items != nil { + in, out := &in.Items, &out.Items + *out = make([]TestFailover, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailoverList. +func (in *TestFailoverList) DeepCopy() *TestFailoverList { + if in == nil { + return nil + } + out := new(TestFailoverList) + in.DeepCopyInto(out) + return out +} + +// DeepCopyObject is an autogenerated deepcopy function, copying the receiver, creating a new runtime.Object. +func (in *TestFailoverList) DeepCopyObject() runtime.Object { + if c := in.DeepCopy(); c != nil { + return c + } + return nil +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailoverReport) DeepCopyInto(out *TestFailoverReport) { + *out = *in + if in.RecoveryPointTime != nil { + in, out := &in.RecoveryPointTime, &out.RecoveryPointTime + *out = (*in).DeepCopy() + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailoverReport. +func (in *TestFailoverReport) DeepCopy() *TestFailoverReport { + if in == nil { + return nil + } + out := new(TestFailoverReport) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailoverSpec) DeepCopyInto(out *TestFailoverSpec) { + *out = *in + if in.TTLSeconds != nil { + in, out := &in.TTLSeconds, &out.TTLSeconds + *out = new(int64) + **out = **in + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailoverSpec. +func (in *TestFailoverSpec) DeepCopy() *TestFailoverSpec { + if in == nil { + return nil + } + out := new(TestFailoverSpec) + in.DeepCopyInto(out) + return out +} + +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *TestFailoverStatus) DeepCopyInto(out *TestFailoverStatus) { + *out = *in + in.Step.DeepCopyInto(&out.Step) + if in.Clones != nil { + in, out := &in.Clones, &out.Clones + *out = make([]TestFailoverClone, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } + if in.Report != nil { + in, out := &in.Report, &out.Report + *out = new(TestFailoverReport) + (*in).DeepCopyInto(*out) + } + if in.StartedAt != nil { + in, out := &in.StartedAt, &out.StartedAt + *out = (*in).DeepCopy() + } + if in.ReadyAt != nil { + in, out := &in.ReadyAt, &out.ReadyAt + *out = (*in).DeepCopy() + } + if in.CompletedAt != nil { + in, out := &in.CompletedAt, &out.CompletedAt + *out = (*in).DeepCopy() + } +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new TestFailoverStatus. +func (in *TestFailoverStatus) DeepCopy() *TestFailoverStatus { + if in == nil { + return nil + } + out := new(TestFailoverStatus) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *ThroughputLimits) DeepCopyInto(out *ThroughputLimits) { *out = *in diff --git a/operator/cmd/main.go b/operator/cmd/main.go index f8cb1dff3..faa203344 100644 --- a/operator/cmd/main.go +++ b/operator/cmd/main.go @@ -38,6 +38,7 @@ import ( _ "k8s.io/client-go/plugin/pkg/client/auth" apiextensionsv1 "k8s.io/apiextensions-apiserver/pkg/apis/apiextensions/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" utilruntime "k8s.io/apimachinery/pkg/util/runtime" @@ -53,6 +54,7 @@ import ( volumegroupsnapshotv1beta1 "github.com/kubernetes-csi/external-snapshotter/client/v8/apis/volumegroupsnapshot/v1beta1" snapshotv1 "github.com/kubernetes-csi/external-snapshotter/client/v8/apis/volumesnapshot/v1" + workv1 "open-cluster-management.io/api/work/v1" simplyblockv1alpha1 "github.com/simplyblock/simplyblock-operator/api/v1alpha1" simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" @@ -82,12 +84,46 @@ var ( const ( openShiftConfigAPIGroup = "config.openshift.io" certManagerAPIGroup = "cert-manager.io" + // ocmWorkGroupVersion is OCM's work API group/version, and ocmManifestWorkResource + // the ManifestWork resource within it. ManifestWork is served only on the hub; + // a managed cluster serves the SAME group/version for AppliedManifestWork (the + // work-agent's local record) but NOT ManifestWork, so the presence check must + // be resource-level, not group-level — a group-level check sees the group as + // served everywhere AppliedManifestWork exists. The TestFailover controller + // watches ManifestWork, so it is registered only where ManifestWork is served. + ocmWorkGroupVersion = "work.open-cluster-management.io/v1" + ocmManifestWorkResource = "manifestworks" ) type serverGroupsGetter interface { ServerGroups() (*metav1.APIGroupList, error) } +type serverResourcesGetter interface { + ServerResourcesForGroupVersion(groupVersion string) (*metav1.APIResourceList, error) +} + +// serverHasResource reports whether the API server serves the named resource in +// the given group/version. Used to skip a controller that watches a kind the +// cluster does not serve, since a watch whose cache can never sync takes the +// whole manager down at startup. A group/version the server does not serve at +// all is reported as absent rather than an error. +func serverHasResource(discoveryClient serverResourcesGetter, groupVersion, resource string) (bool, error) { + list, err := discoveryClient.ServerResourcesForGroupVersion(groupVersion) + if err != nil { + if apierrors.IsNotFound(err) { + return false, nil + } + return false, fmt.Errorf("discover resources for %s: %w", groupVersion, err) + } + for _, r := range list.APIResources { + if r.Name == resource { + return true, nil + } + } + return false, nil +} + func init() { utilruntime.Must(clientgoscheme.AddToScheme(scheme)) // The cert-manager webhook provisioner injects its CA bundle into the @@ -106,6 +142,9 @@ func init() { // external-snapshotter VolumeSnapshot: the VolumeGroupSnapshotOps restore // enumerates a group snapshot's member snapshots (design §7.4). utilruntime.Must(snapshotv1.AddToScheme(scheme)) + // OCM ManifestWork: the TestFailover controller places the bubble PV/PVC on a + // recovery cluster through it (design-test-failover.md §7.6). + utilruntime.Must(workv1.Install(scheme)) // +kubebuilder:scaffold:scheme } @@ -169,7 +208,9 @@ func main() { "turn it back on while migrations raised against it drain. A rename and a scope "+ "change make a new CRD rather than a new version, so an in-flight migration cannot "+ "be carried across.") + var csiLinkEnabled bool var csiLinkAddr, csiLinkCertPath, csiLinkCertName, csiLinkCertKey, csiLinkAudience string + flag.BoolVar(&csiLinkEnabled, "csi-link", false, "Serve the CSI link.") flag.StringVar(&csiLinkAddr, "csi-link-bind-address", ":9500", "The address the CSI link endpoint binds to.") flag.StringVar(&csiLinkCertPath, "csi-link-cert-path", "", @@ -332,27 +373,29 @@ func main() { // plugins; a reconciler reaching a node goes through it, and treats // link.ErrNoSession as a requeue rather than a failure. // - // Always served, because both plugins always dial it. TLS when a - // certificate is configured, plaintext when none is. - var certFile, keyFile string - if csiLinkCertPath != "" { - certFile = filepath.Join(csiLinkCertPath, csiLinkCertName) - keyFile = filepath.Join(csiLinkCertPath, csiLinkCertKey) - } - csiPeers, err := csilink.Setup(mgr, csilink.Config{ - BindAddress: csiLinkAddr, - CertFile: certFile, - KeyFile: keyFile, - Namespace: operatorNamespace, - Audiences: []string{csiLinkAudience}, - NodeServiceAccount: "simplyblock-csi-node-sa", - ControllerServiceAccount: "simplyblock-csi-controller-sa", - }) - if err != nil { - setupLog.Error(err, "unable to set up the CSI link") - os.Exit(1) + // Off by default, on with --csi-link. TLS when a certificate is + // configured, plaintext when none is. + if csiLinkEnabled { + var certFile, keyFile string + if csiLinkCertPath != "" { + certFile = filepath.Join(csiLinkCertPath, csiLinkCertName) + keyFile = filepath.Join(csiLinkCertPath, csiLinkCertKey) + } + csiPeers, err := csilink.Setup(mgr, csilink.Config{ + BindAddress: csiLinkAddr, + CertFile: certFile, + KeyFile: keyFile, + Namespace: operatorNamespace, + Audiences: []string{csiLinkAudience}, + NodeServiceAccount: "simplyblock-csi-node-sa", + ControllerServiceAccount: "simplyblock-csi-controller-sa", + }) + if err != nil { + setupLog.Error(err, "unable to set up the CSI link") + os.Exit(1) + } + _ = csiPeers // handed to reconcilers as they start using it } - _ = csiPeers // handed to reconcilers as they start using it // Control-plane SSE push subscriptions: one leader-only manager, streams // driven by scopes that reconcilers register (the StorageNode controller adds @@ -518,6 +561,14 @@ func main() { controlPlaneEndpoint := controlplanecontroller.NewEndpointResolver( mgr.GetClient(), operatorNamespace) + // What a managed control plane's CreateCluster and ClusterByName calls + // authenticate with, read from the same ControlPlane object + // (spec.source.managed.credentialsSecretRef) for the same reason: a + // StorageCluster CR applied against a remote control plane has no other + // credential to create its backend identity with. + controlPlaneCredential := controlplanecontroller.NewCredentialResolver( + mgr.GetClient(), operatorNamespace) + if err := (&controlplanecontroller.ControlPlaneReconciler{ Client: mgr.GetClient(), Scheme: mgr.GetScheme(), @@ -538,7 +589,7 @@ func main() { Client: mgr.GetClient(), Scheme: mgr.GetScheme(), Recorder: mgr.GetEventRecorder("storagecluster-controller"), - API: clustercontroller.NewControlPlane(controlPlaneEndpoint), + API: clustercontroller.NewControlPlane(controlPlaneEndpoint, controlPlaneCredential), Namespace: operatorNamespace, Clusters: clusterSubscription, Tasks: taskSubscription, @@ -572,10 +623,11 @@ func main() { os.Exit(1) } if err := (&pool.StoragePoolReconciler{ - Client: mgr.GetClient(), - Scheme: mgr.GetScheme(), - Recorder: mgr.GetEventRecorder("storagepool-controller"), - VolumeScopes: volumeScopes, + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorder("storagepool-controller"), + VolumeScopes: volumeScopes, + EndpointResolver: controlPlaneEndpoint, }).SetupWithManager(mgr); err != nil { setupLog.Error(err, "unable to create controller", "controller", "StoragePool") os.Exit(1) @@ -806,7 +858,7 @@ func main() { Client: mgr.GetClient(), Scheme: mgr.GetScheme(), Recorder: mgr.GetEventRecorder("storageclusterops-controller"), - API: clustercontroller.NewControlPlane(controlPlaneEndpoint), + API: clustercontroller.NewControlPlane(controlPlaneEndpoint, controlPlaneCredential), Clusters: clusterSubscription, Nodes: nodeSubscription, Tasks: taskSubscription, @@ -842,8 +894,9 @@ func main() { os.Exit(1) } if err := (&controller.ReplicationPolicyReconciler{ - Client: mgr.GetClient(), - Scheme: mgr.GetScheme(), + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + EndpointResolver: controlPlaneEndpoint, }).SetupWithManager(mgr); err != nil { setupLog.Error(err, "unable to create controller", "controller", "ReplicationPolicy") os.Exit(1) @@ -856,8 +909,9 @@ func main() { os.Exit(1) } if err := (&controller.ReplicationPairReconciler{ - Client: mgr.GetClient(), - Scheme: mgr.GetScheme(), + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + EndpointResolver: controlPlaneEndpoint, }).SetupWithManager(mgr); err != nil { setupLog.Error(err, "unable to create controller", "controller", "ReplicationPair") os.Exit(1) @@ -886,6 +940,55 @@ func main() { setupLog.Error(err, "unable to create controller", "controller", "VolumeGroupSnapshotOps") os.Exit(1) } + // The TestFailover controller watches OCM ManifestWork, which only the hub + // serves; registering it where the work API is absent leaves a watch whose + // cache never syncs and the manager exits at startup, taking every other + // controller with it. It is a hub-only drill, so skip it off the hub. + workDiscovery, err := discovery.NewDiscoveryClientForConfig(cfg) + if err != nil { + setupLog.Error(err, "unable to build a discovery client to check for the OCM work API") + os.Exit(1) + } + hasManifestWork, err := serverHasResource(workDiscovery, ocmWorkGroupVersion, ocmManifestWorkResource) + if err != nil { + setupLog.Error(err, "unable to determine whether the OCM ManifestWork resource is served") + os.Exit(1) + } + registerOCMControllers := func() error { + if err := (&controller.TestFailoverReconciler{ + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorder("testfailover-controller"), + }).SetupWithManager(mgr); err != nil { + return fmt.Errorf("controller TestFailover: %w", err) + } + // A managed site's storage deployment is requested from the hub through + // the same work API, so the controller is hub-only too. + if err := (&controller.StorageSiteDeploymentReconciler{ + Client: mgr.GetClient(), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorder("storagesitedeployment-controller"), + }).SetupWithManager(mgr); err != nil { + return fmt.Errorf("controller StorageSiteDeployment: %w", err) + } + return nil + } + if hasManifestWork { + if err := registerOCMControllers(); err != nil { + setupLog.Error(err, "unable to create controller") + os.Exit(1) + } + } else { + // OCM may be installed after the operator (the DR stack brings it): keep + // looking, and start the controllers once ManifestWork is served. + setupLog.Info("OCM ManifestWork resource not served yet; the TestFailover and "+ + "StorageSiteDeployment controllers start once it is (hub-only)", + "groupVersion", ocmWorkGroupVersion, "resource", ocmManifestWorkResource) + if err := mgr.Add(&ocmLateStart{log: setupLog, disc: workDiscovery, register: registerOCMControllers}); err != nil { + setupLog.Error(err, "unable to add the OCM ManifestWork watcher") + os.Exit(1) + } + } // +kubebuilder:scaffold:builder // Provision the admission webhooks' serving certificate at runtime (self-signed diff --git a/operator/cmd/main_test.go b/operator/cmd/main_test.go index 938893810..d6c615cb1 100644 --- a/operator/cmd/main_test.go +++ b/operator/cmd/main_test.go @@ -4,7 +4,9 @@ import ( "strings" "testing" + apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime/schema" "github.com/simplyblock/simplyblock-operator/internal/utils" ) @@ -21,6 +23,71 @@ func (f fakeServerGroupsGetter) ServerGroups() (*metav1.APIGroupList, error) { return list, nil } +type fakeServerResourcesGetter struct { + // groupVersion -> resource names served. A missing key means the server does + // not serve that group/version at all (discovery returns NotFound). + resources map[string][]string +} + +func (f fakeServerResourcesGetter) ServerResourcesForGroupVersion(gv string) (*metav1.APIResourceList, error) { + names, ok := f.resources[gv] + if !ok { + return nil, apierrors.NewNotFound(schema.GroupResource{Resource: gv}, "") + } + list := &metav1.APIResourceList{GroupVersion: gv} + for _, n := range names { + list.APIResources = append(list.APIResources, metav1.APIResource{Name: n}) + } + return list, nil +} + +// Regression: 2026-10-01 — the operator crashed at startup on every cluster +// outside the hub. The TestFailover controller unconditionally watched OCM's +// ManifestWork (.Owns(&workv1.ManifestWork{})), but ManifestWork is served only +// on the hub — managed clusters serve the same group/version for +// AppliedManifestWork (the work-agent's local record) yet never serve +// ManifestWork. The controller's cache never syncs and the manager exits +// ("failed to wait for testfailover caches to sync ... *v1.ManifestWork"), +// taking every other controller (replication, storagecluster, …) down with it. +// Registration must be gated on the ManifestWork RESOURCE being served, not the +// work API group — a group-level check sees the group as served wherever +// AppliedManifestWork exists and so does not skip the controller (live 2026-10-01). +func TestServerHasManifestWork(t *testing.T) { + tests := []struct { + name string + resources map[string][]string + want bool + }{ + { + name: "hub serves ManifestWork", + resources: map[string][]string{ocmWorkGroupVersion: {"manifestworks", "appliedmanifestworks"}}, + want: true, + }, + { + name: "managed cluster serves only AppliedManifestWork", + resources: map[string][]string{ocmWorkGroupVersion: {"appliedmanifestworks"}}, + want: false, + }, + { + name: "cluster without OCM at all", + resources: map[string][]string{}, + want: false, + }, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + got, err := serverHasResource( + fakeServerResourcesGetter{resources: tc.resources}, ocmWorkGroupVersion, ocmManifestWorkResource) + if err != nil { + t.Fatalf("serverHasResource returned error: %v", err) + } + if got != tc.want { + t.Fatalf("serverHasResource = %v, want %v", got, tc.want) + } + }) + } +} + func TestValidateTLSConfiguration(t *testing.T) { tests := []struct { name string diff --git a/operator/cmd/ocm_late.go b/operator/cmd/ocm_late.go new file mode 100644 index 000000000..1a38c14af --- /dev/null +++ b/operator/cmd/ocm_late.go @@ -0,0 +1,70 @@ +package main + +import ( + "context" + "time" + + "github.com/go-logr/logr" +) + +const ( + // ocmPollInterval is how often the operator looks again for the OCM + // ManifestWork resource when it was not served at start; it doubles after + // each miss up to ocmPollMaxInterval. + ocmPollInterval = 60 * time.Second + ocmPollMaxInterval = 10 * time.Minute +) + +// waitForResource polls discovery until the API server serves resource in +// groupVersion, and reports true; false when ctx ends first. Discovery errors +// are logged and polled through: they are as likely to be a passing API server +// hiccup as a real absence. after is time.After outside of tests. +func waitForResource(ctx context.Context, log logr.Logger, disc serverResourcesGetter, groupVersion, resource string, + interval, maxInterval time.Duration, after func(time.Duration) <-chan time.Time) bool { + for { + select { + case <-ctx.Done(): + return false + case <-after(interval): + } + served, err := serverHasResource(disc, groupVersion, resource) + if err != nil { + log.Info("could not check for the OCM ManifestWork resource; trying again", + "groupVersion", groupVersion, "resource", resource, "error", err.Error()) + } else if served { + return true + } + if interval *= 2; interval > maxInterval { + interval = maxInterval + } + } +} + +// ocmLateStart registers the hub-only controllers once OCM's ManifestWork is +// served. The operator is commonly installed before OCM (the DR stack brings +// it); checking only at process start left every TestFailover without a +// reconciler until the operator happened to restart (2026-10-04). Controllers +// added to a running manager are started by controller-runtime right away +// (manager runnableGroup.Add), and neither controller registers field indexes, +// so no restart of the process is needed. +type ocmLateStart struct { + log logr.Logger + disc serverResourcesGetter + register func() error + after func(time.Duration) <-chan time.Time +} + +// Start implements manager.Runnable. +func (o *ocmLateStart) Start(ctx context.Context) error { + after := o.after + if after == nil { + after = time.After + } + if !waitForResource(ctx, o.log, o.disc, ocmWorkGroupVersion, ocmManifestWorkResource, + ocmPollInterval, ocmPollMaxInterval, after) { + return nil + } + o.log.Info("OCM ManifestWork resource is now served; starting the TestFailover and "+ + "StorageSiteDeployment controllers", "groupVersion", ocmWorkGroupVersion, "resource", ocmManifestWorkResource) + return o.register() +} diff --git a/operator/cmd/ocm_late_test.go b/operator/cmd/ocm_late_test.go new file mode 100644 index 000000000..9f2494189 --- /dev/null +++ b/operator/cmd/ocm_late_test.go @@ -0,0 +1,117 @@ +package main + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/go-logr/logr" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime/schema" +) + +// scriptedDiscovery answers ServerResourcesForGroupVersion from a script, one +// answer per call; the last answer repeats. +type scriptedDiscovery struct { + answers []func() (*metav1.APIResourceList, error) + calls int +} + +func (d *scriptedDiscovery) ServerResourcesForGroupVersion(string) (*metav1.APIResourceList, error) { + i := d.calls + if i >= len(d.answers) { + i = len(d.answers) - 1 + } + d.calls++ + return d.answers[i]() +} + +func notServed() (*metav1.APIResourceList, error) { + return nil, apierrors.NewNotFound(schema.GroupResource{Group: "work.open-cluster-management.io"}, "v1") +} + +func servedWithout() (*metav1.APIResourceList, error) { + // a managed cluster: the group serves AppliedManifestWork, not ManifestWork + return &metav1.APIResourceList{APIResources: []metav1.APIResource{{Name: "appliedmanifestworks"}}}, nil +} + +func served() (*metav1.APIResourceList, error) { + return &metav1.APIResourceList{APIResources: []metav1.APIResource{{Name: ocmManifestWorkResource}}}, nil +} + +func discoveryError() (*metav1.APIResourceList, error) { + return nil, errors.New("the server is currently unable to handle the request") +} + +// instant fires every wait at once and records the intervals asked for. +func instant(waits *[]time.Duration) func(time.Duration) <-chan time.Time { + return func(d time.Duration) <-chan time.Time { + *waits = append(*waits, d) + ch := make(chan time.Time, 1) + ch <- time.Time{} + return ch + } +} + +func TestWaitForResourceFindsManifestWorkInstalledLater(t *testing.T) { + disc := &scriptedDiscovery{answers: []func() (*metav1.APIResourceList, error){ + notServed, servedWithout, discoveryError, served}} + var waits []time.Duration + ok := waitForResource(context.Background(), logr.Discard(), disc, ocmWorkGroupVersion, ocmManifestWorkResource, + time.Minute, 3*time.Minute, instant(&waits)) + if !ok { + t.Fatal("waitForResource gave up although ManifestWork became served") + } + if disc.calls != 4 { + t.Errorf("discovery called %d times, want 4 (absent, other kind only, error, served)", disc.calls) + } + want := []time.Duration{time.Minute, 2 * time.Minute, 3 * time.Minute, 3 * time.Minute} + if len(waits) != len(want) { + t.Fatalf("waits %v, want %v", waits, want) + } + for i := range want { + if waits[i] != want[i] { + t.Errorf("wait %d = %v, want %v (doubling, capped)", i, waits[i], want[i]) + } + } +} + +func TestWaitForResourceStopsWithItsContext(t *testing.T) { + disc := &scriptedDiscovery{answers: []func() (*metav1.APIResourceList, error){notServed}} + ctx, cancel := context.WithCancel(context.Background()) + cancel() + never := func(time.Duration) <-chan time.Time { return make(chan time.Time) } + if waitForResource(ctx, logr.Discard(), disc, ocmWorkGroupVersion, ocmManifestWorkResource, + time.Minute, time.Minute, never) { + t.Fatal("waitForResource reported the resource served after its context ended") + } + if disc.calls != 0 { + t.Errorf("discovery called %d times after cancellation, want 0", disc.calls) + } +} + +func TestOCMLateStartRegistersTheControllersOnce(t *testing.T) { + disc := &scriptedDiscovery{answers: []func() (*metav1.APIResourceList, error){notServed, served}} + var waits []time.Duration + registered := 0 + o := &ocmLateStart{log: logr.Discard(), disc: disc, after: instant(&waits), + register: func() error { registered++; return nil }} + if err := o.Start(context.Background()); err != nil { + t.Fatalf("Start: %v", err) + } + if registered != 1 { + t.Errorf("controllers registered %d times, want 1", registered) + } +} + +func TestOCMLateStartReportsARegistrationFailure(t *testing.T) { + disc := &scriptedDiscovery{answers: []func() (*metav1.APIResourceList, error){served}} + var waits []time.Duration + o := &ocmLateStart{log: logr.Discard(), disc: disc, after: instant(&waits), + register: func() error { return errors.New("boom") }} + if err := o.Start(context.Background()); err == nil { + t.Fatal("Start hid the registration failure") + } +} diff --git a/operator/config/crd/bases/storage.simplyblock.io_controlplanes.yaml b/operator/config/crd/bases/storage.simplyblock.io_controlplanes.yaml index 94b0cad35..c426685ee 100644 --- a/operator/config/crd/bases/storage.simplyblock.io_controlplanes.yaml +++ b/operator/config/crd/bases/storage.simplyblock.io_controlplanes.yaml @@ -163,6 +163,32 @@ spec: local: description: Local is a control plane the operator installs. properties: + adminTokenSecretRef: + description: |- + AdminTokenSecretRef names a Secret in this namespace holding a static + admin bearer token this control plane accepts, under the `token` key, in + addition to this deployment's own Kubernetes identity + (SB_K8S_ADMIN_SERVICE_ACCOUNTS). It is what lets a cluster this control + plane manages remotely (spec.source.managed there, + ManagedControlPlane.CredentialsSecretRef naming the same value) + authenticate a CreateCluster call, since a Kubernetes TokenReview can + never cross a cluster boundary. + + The Secret is projected into the management API container's environment + with secretKeyRef, so this operator never itself reads the plaintext. + Absent grants no credential beyond the operator's own service account. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic foundationDB: description: FoundationDB sizes the FoundationDB the management API stores its state in. properties: @@ -513,6 +539,20 @@ spec: so a loopback or link-local address is rejected. pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ type: string + storageNodeImage: + description: |- + StorageNodeImage is the storage-node image a StorageCluster on this + Kubernetes cluster defaults to when its own spec.storageNodes.image is + unset. A local control plane's own spec.source.local.image doubles as + this default (StorageNodeWorkloadReconciler.image), because a + self-hosted deployment's control plane and its storage nodes are one + release. A managed one is a different Kubernetes cluster's install and + says nothing about what this cluster's storage nodes should run, so + there is no equivalent to fall back to without this field -- every + StorageCluster on a managed deployment must get an image from here or + from its own spec. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string required: - endpoint type: object diff --git a/operator/config/crd/bases/storage.simplyblock.io_simplyblockdrivers.yaml b/operator/config/crd/bases/storage.simplyblock.io_simplyblockdrivers.yaml index 94d87edeb..e4a7d6c6b 100644 --- a/operator/config/crd/bases/storage.simplyblock.io_simplyblockdrivers.yaml +++ b/operator/config/crd/bases/storage.simplyblock.io_simplyblockdrivers.yaml @@ -299,13 +299,29 @@ spec: type: object sidecarImages: description: |- - SidecarImages overrides the six CSI sidecars, one field each. Unset takes - the version this operator release ships. + SidecarImages overrides the seven CSI sidecars, one field each. Unset + takes the version this operator release ships. properties: attacher: description: Attacher is csi-attacher, on the controller plugin. pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ type: string + csiAddons: + description: |- + CSIAddons is the kubernetes-csi-addons sidecar, on the controller + plugin. It connects to the plugin's socket, probes the csi-addons + Identity service for capabilities, and publishes a CSIAddonsNode so the + kubernetes-csi-addons controller-manager (design + design-csi-addons-replication.md §4.1) can reach the Replication + service this driver serves. + + Unlike the other sidecars above, this one's upstream home is the + csi-addons project's own registry, not simplyblock's: the allowlist + carries quay.io/csiaddons alongside the simplyblock registries so a + deployment can run the stock kubernetes-csi-addons sidecar image + directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string healthMonitor: description: |- HealthMonitor is csi-external-health-monitor-controller, on the diff --git a/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..393fd46a5 --- /dev/null +++ b/operator/config/crd/bases/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1053 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: |- + Count is the number of journal managers to configure. The control plane + requires at least 3. + format: int32 + minimum: 3 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml new file mode 100644 index 000000000..d9117bb4c --- /dev/null +++ b/operator/config/crd/bases/storage.simplyblock.io_testfailovers.yaml @@ -0,0 +1,336 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: testfailovers.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: TestFailover + listKind: TestFailoverList + plural: testfailovers + shortNames: + - tfo + singular: testfailover + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.scope + name: Scope + type: string + - jsonPath: .spec.sourceRef + name: Source + type: string + - jsonPath: .spec.sourceCluster + name: "On" + type: string + - jsonPath: .spec.bubbleCluster + name: Bubble + type: string + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.step.state + name: Step + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + TestFailover is a one-way, non-disruptive test-failover drill. It recovers a + source volume, or a consistency group, from a snapshot into an isolated + namespace on a chosen cluster as bound PVCs, without touching the source. The + hub reads the source on its cluster and places the bubble on the recovery + cluster through OCM. It runs to a terminal phase, or holds Ready until it is + deleted, and deletion reclaims the clones and any snapshots the drill took. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + TestFailoverSpec is the request for one non-disruptive test-failover drill. + + The source is named by where it runs and what it is, so the hub can find it + without anyone extracting a backend handle by hand. SourceNamespace is + required for a Volume drill, where the source is a PVC, and unused for a Group + drill, where SourceRef names a consistency group. + properties: + bubbleCluster: + description: |- + BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + the replicated point, or another cluster. It must differ from SourceCluster; + test-failover recovers onto a different cluster, never in place. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + bubbleNamespace: + default: bubble + description: |- + BubbleNamespace is the namespace on the bubble cluster where the recovered + PVCs are created. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + scope: + description: Scope selects what the drill recovers. Immutable. + enum: + - Volume + - Group + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceCluster: + description: |- + SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + the source there through a ManagedClusterView. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceNamespace: + description: |- + SourceNamespace is the namespace of the source PVC on SourceCluster. + Required for scope=Volume. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceRef: + description: |- + SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + SourceNamespace (scope=Volume), or a consistency group (scope=Group). + Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + ttlSeconds: + description: |- + TTLSeconds is an optional maximum lifetime: the drill is torn down after it + even without a delete, so a forgotten drill cannot hold a clone forever. + format: int64 + minimum: 0 + type: integer + required: + - bubbleCluster + - scope + - sourceCluster + - sourceRef + type: object + x-kubernetes-validations: + - message: field sourceNamespace is immutable once set + rule: '!has(oldSelf.sourceNamespace) || has(self.sourceNamespace)' + - message: field bubbleNamespace is immutable once set + rule: '!has(oldSelf.bubbleNamespace) || has(self.bubbleNamespace)' + - message: sourceNamespace is required for scope=Volume + rule: self.scope != 'Volume' || has(self.sourceNamespace) + status: + description: TestFailoverStatus is the observed state of one drill. + properties: + clones: + description: Clones is one entry per recovered volume. + items: + description: |- + TestFailoverClone is one recovered volume: the source it came from, the + snapshot and clone the drill built, and the PVC placed on the bubble cluster. + properties: + cloneID: + description: CloneID is the backend id of the writable clone. + type: string + pvcName: + description: PVCName is the bound PVC in the bubble namespace + on the bubble cluster. + type: string + sizeBytes: + description: SizeBytes is the recovered volume's size. + format: int64 + type: integer + snapshotID: + description: |- + SnapshotID is the recovery-point snapshot: the replicated snapshot already on + the bubble cluster's backend that the clone is built from. + type: string + sourceFSType: + description: |- + SourceFSType is the source PV's CSI fsType, carried onto the bubble PV so the + node plugin stages the clone with the filesystem it actually carries. The + clone is a block copy of the source, so its filesystem is the source's; an + empty fsType makes the node plugin default to ext4 and refuse to mount an XFS + volume. + type: string + sourceHandle: + description: SourceHandle is the source volume's backend handle, + read from its PV. + type: string + sourceRef: + description: |- + SourceRef is the source volume, or group member, the recovered volume maps + to. + type: string + sourceVolumeContext: + additionalProperties: + type: string + description: |- + SourceVolumeContext is the source PV's CSI volumeAttributes, minus the + identity and provisioner keys, carried onto the bubble PV so the node plugin + receives a non-nil VolumeContext when it stages the clone. The clone's own + identity (NQN, connections, nsId, and so on) is re-resolved from the clone + handle at stage time, so only the class-level parameters are carried; the + identity keys are dropped so a failed clone lookup can never point the mount + back at the source. + type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string + required: + - sourceRef + type: object + type: array + x-kubernetes-list-map-keys: + - sourceRef + x-kubernetes-list-type: map + completedAt: + description: CompletedAt is when the drill reached a terminal phase. + format: date-time + type: string + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as the + drill moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from, so a stale status can be told from a current one. + format: int64 + type: integer + phase: + description: Phase is the drill's own progress. + enum: + - Pending + - Provisioning + - Ready + - Failed + - TearingDown + type: string + readyAt: + description: ReadyAt is when every recovered PVC became bound. + format: date-time + type: string + report: + description: Report is the drill's evidence, populated as it reaches + Ready. + properties: + bubbleCluster: + description: BubbleCluster is the cluster the drill recovered + onto. + type: string + invariantsHeld: + description: |- + InvariantsHeld is true only when the source fingerprint taken before the + drill matches the one taken at Ready. A Ready drill with this false is a + defect. + type: boolean + recoveryPoint: + description: RecoveryPoint is the snapshot or group generation + the drill recovered. + type: string + recoveryPointAgeSeconds: + description: RecoveryPointAgeSeconds is the drill time minus the + recovery-point time. + format: int64 + type: integer + recoveryPointTime: + description: RecoveryPointTime is when that point was taken. + format: date-time + type: string + type: object + startedAt: + description: StartedAt is when the drill started. + format: date-time + type: string + step: + description: Step is the position of the running drill's state machine. + properties: + claim: + description: |- + Claim records that this state's side effect was started. Absent means + no pass has started it since the state was entered. + properties: + attempt: + description: Attempt counts the claims taken on this state, + starting at 1. + format: int32 + type: integer + leaseUntil: + description: |- + LeaseUntil is when the claim expires and the side effect may be fired + again. + format: date-time + type: string + state: + description: State is the state the claim was taken in. + type: string + required: + - attempt + - leaseUntil + - state + type: object + deadline: + description: |- + Deadline is when that state expires, absent when it has none. It is an + absolute instant, so a state whose deadline passed while the controller + was down restores as already expired. + format: date-time + type: string + state: + description: |- + State is the state the machine was in. Empty means the resource has not + been reconciled yet, and restores to the graph's initial state. + type: string + type: object + x-kubernetes-validations: + - message: unknown step + rule: '!has(self.state) || self.state in [''ResolvingSource'',''ResolvingPoint'',''Shipping'',''Cloning'',''Placing'',''Releasing'']' + triggered: + description: |- + Triggered records that the current step's side effect was issued, so a + restart does not repeat it. + type: boolean + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/config/crd/kustomization.yaml b/operator/config/crd/kustomization.yaml index 66e4a63c0..520d8f4be 100644 --- a/operator/config/crd/kustomization.yaml +++ b/operator/config/crd/kustomization.yaml @@ -30,6 +30,8 @@ resources: - bases/storage.simplyblock.io_persistentvolumeops.yaml - bases/storage.simplyblock.io_controlplaneops.yaml - bases/storage.simplyblock.io_storagedeviceops.yaml +- bases/storage.simplyblock.io_testfailovers.yaml +- bases/storage.simplyblock.io_storagesitedeployments.yaml # +kubebuilder:scaffold:crdkustomizeresource patches: [] diff --git a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml index 802b65438..ced5c9b1d 100644 --- a/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml +++ b/operator/config/manifests/bases/simplyblock-operator.clusterserviceversion.yaml @@ -593,6 +593,22 @@ spec: displayName: Tasks path: tasks version: v1alpha1 + - description: TestFailover is a one-way, non-disruptive test-failover drill. + It recovers a source volume, or a consistency group, from a snapshot into + an isolated namespace on a chosen cluster as bound PVCs, without touching + the source. + displayName: Test Failover + kind: TestFailover + name: testfailovers.storage.simplyblock.io + version: v1alpha2 + - description: Deploys a managed site's storage from the hub, through OCM. + Discovers the site's nodes, applies the requested sizing to the draft and, + once approved, deploys the storage cluster; status projects the draft and + the resulting storage cluster. + displayName: Storage Site Deployment + kind: StorageSiteDeployment + name: storagesitedeployments.storage.simplyblock.io + version: v1alpha2 description: The Simplyblock Operator helps with installation, operation, and management of Simplyblock Control Planes, Storage Planes, and the CSI Driver. displayName: Simplyblock Operator diff --git a/operator/config/rbac/role.yaml b/operator/config/rbac/role.yaml index 7a08f3c24..1335dea16 100644 --- a/operator/config/rbac/role.yaml +++ b/operator/config/rbac/role.yaml @@ -176,6 +176,44 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch +- apiGroups: + - coordination.k8s.io + resources: + - leases + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes/status + verbs: + - get + - patch + - update - apiGroups: - discovery.k8s.io resources: @@ -295,7 +333,9 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks + - testfailovers - volumemigrations verbs: - create @@ -330,7 +370,9 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers + - testfailovers/finalizers - volumemigrations/finalizers verbs: - update @@ -362,7 +404,9 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status + - testfailovers/status - volumegroupsnapshotops/status - volumemigrations/status verbs: @@ -389,3 +433,27 @@ rules: - get - list - watch +- apiGroups: + - view.open-cluster-management.io + resources: + - managedclusterviews + verbs: + - create + - delete + - get + - list + - patch + - update + - watch +- apiGroups: + - work.open-cluster-management.io + resources: + - manifestworks + verbs: + - create + - delete + - get + - list + - patch + - update + - watch diff --git a/operator/dist/install.yaml b/operator/dist/install.yaml index 047ebaf1f..83fb44aa5 100644 --- a/operator/dist/install.yaml +++ b/operator/dist/install.yaml @@ -2175,6 +2175,32 @@ spec: local: description: Local is a control plane the operator installs. properties: + adminTokenSecretRef: + description: |- + AdminTokenSecretRef names a Secret in this namespace holding a static + admin bearer token this control plane accepts, under the `token` key, in + addition to this deployment's own Kubernetes identity + (SB_K8S_ADMIN_SERVICE_ACCOUNTS). It is what lets a cluster this control + plane manages remotely (spec.source.managed there, + ManagedControlPlane.CredentialsSecretRef naming the same value) + authenticate a CreateCluster call, since a Kubernetes TokenReview can + never cross a cluster boundary. + + The Secret is projected into the management API container's environment + with secretKeyRef, so this operator never itself reads the plaintext. + Absent grants no credential beyond the operator's own service account. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic foundationDB: description: FoundationDB sizes the FoundationDB the management API stores its state in. @@ -2535,6 +2561,20 @@ spec: so a loopback or link-local address is rejected. pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ type: string + storageNodeImage: + description: |- + StorageNodeImage is the storage-node image a StorageCluster on this + Kubernetes cluster defaults to when its own spec.storageNodes.image is + unset. A local control plane's own spec.source.local.image doubles as + this default (StorageNodeWorkloadReconciler.image), because a + self-hosted deployment's control plane and its storage nodes are one + release. A managed one is a different Kubernetes cluster's install and + says nothing about what this cluster's storage nodes should run, so + there is no equivalent to fall back to without this field -- every + StorageCluster on a managed deployment must get an image from here or + from its own spec. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string required: - endpoint type: object @@ -4555,13 +4595,29 @@ spec: type: object sidecarImages: description: |- - SidecarImages overrides the six CSI sidecars, one field each. Unset takes - the version this operator release ships. + SidecarImages overrides the seven CSI sidecars, one field each. Unset + takes the version this operator release ships. properties: attacher: description: Attacher is csi-attacher, on the controller plugin. pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ type: string + csiAddons: + description: |- + CSIAddons is the kubernetes-csi-addons sidecar, on the controller + plugin. It connects to the plugin's socket, probes the csi-addons + Identity service for capabilities, and publishes a CSIAddonsNode so the + kubernetes-csi-addons controller-manager (design + design-csi-addons-replication.md §4.1) can reach the Replication + service this driver serves. + + Unlike the other sidecars above, this one's upstream home is the + csi-addons project's own registry, not simplyblock's: the allowlist + carries quay.io/csiaddons alongside the simplyblock registries so a + deployment can run the stock kubernetes-csi-addons sidecar image + directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string healthMonitor: description: |- HealthMonitor is csi-external-health-monitor-controller, on the @@ -11530,142 +11586,37 @@ kind: CustomResourceDefinition metadata: annotations: controller-gen.kubebuilder.io/version: v0.21.0 - name: tasks.storage.simplyblock.io -spec: - group: storage.simplyblock.io - names: - kind: Task - listKind: TaskList - plural: tasks - singular: task - scope: Namespaced - versions: - - name: v1alpha1 - schema: - openAPIV3Schema: - description: Task is the Schema for the tasks API - properties: - apiVersion: - description: |- - APIVersion defines the versioned schema of this representation of an object. - Servers should convert recognized schemas to the latest internal value, and - may reject unrecognized values. - More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources - type: string - kind: - description: |- - Kind is a string value representing the REST resource this object represents. - Servers may infer this from the endpoint the client submits requests to. - Cannot be updated. - In CamelCase. - More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds - type: string - metadata: - type: object - spec: - description: spec defines the desired state of Task - properties: - clusterName: - description: ClusterName is the target storage cluster name. - type: string - subtasks: - description: |- - Subtasks includes related child subtasks when supported by the backend. - FIXME: Unused for now - type: boolean - taskID: - description: TaskID filters results to a specific backend task when - set. - type: string - required: - - clusterName - type: object - status: - description: status defines the observed state of Task - properties: - tasks: - description: Tasks is the currently reported task list for the query - scope. - items: - properties: - canceled: - description: Canceled indicates whether the task was canceled. - type: boolean - parentTask: - description: |- - ParentTask is the parent task UUID when this task is a subtask. - FIXME: Unused for now - type: string - retried: - description: Retried is the number of retry attempts made for - the task. - format: int32 - type: integer - startedAt: - description: |- - StartedAt is the backend-reported task start timestamp. - FIXME: Unused for now - format: date-time - type: string - taskResult: - description: TaskResult is the backend result payload/message. - type: string - taskStatus: - description: TaskStatus is the backend lifecycle status for - the task. - type: string - taskType: - description: TaskType is the backend task function/type name. - type: string - uuid: - description: UUID is the backend task UUID. - type: string - type: object - type: array - type: object - required: - - spec - type: object - served: true - storage: true - subresources: - status: {} ---- -apiVersion: apiextensions.k8s.io/v1 -kind: CustomResourceDefinition -metadata: - annotations: - controller-gen.kubebuilder.io/version: v0.21.0 - name: volumegroupsnapshotops.storage.simplyblock.io + name: storagesitedeployments.storage.simplyblock.io spec: group: storage.simplyblock.io names: - kind: VolumeGroupSnapshotOps - listKind: VolumeGroupSnapshotOpsList - plural: volumegroupsnapshotops + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments shortNames: - - vgsops - singular: volumegroupsnapshotops + - sbsd + singular: storagesitedeployment scope: Namespaced versions: - additionalPrinterColumns: - - jsonPath: .spec.volumeGroupSnapshotRef - name: GroupSnapshot - type: string - - jsonPath: .spec.action - name: Action + - jsonPath: .spec.cluster + name: Cluster type: string + - jsonPath: .spec.approved + name: Approved + type: boolean - jsonPath: .status.phase name: Phase type: string - - jsonPath: .status.step.state - name: Step + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage type: string - - jsonPath: .status.membersBound - name: Bound - type: integer - jsonPath: .status.message name: Message + priority: 1 type: string - jsonPath: .metadata.creationTimestamp name: Age @@ -11674,11 +11625,11 @@ spec: schema: openAPIV3Schema: description: |- - VolumeGroupSnapshotOps is a one-shot operation on a VolumeGroupSnapshot, - analogous to a Kubernetes Job: it drives its action to completion, records - the result, and is then inert. The Restore action creates one - PersistentVolumeClaim per member snapshot of the target's generation and - waits for every claim to bind. + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. properties: apiVersion: description: |- @@ -11698,101 +11649,1595 @@ spec: metadata: type: object spec: - description: |- - VolumeGroupSnapshotOpsSpec defines the requested operation. The whole spec is - immutable: the object is a request. + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. properties: - action: - description: Action is the operation to perform. - enum: - - Restore + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 type: string x-kubernetes-validations: - - message: field is immutable + - message: cluster is immutable rule: self == oldSelf - restore: + discover: description: |- - Restore carries the parameters of the Restore action. Ignored for any - other action. + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. properties: - consistencyGroup: - description: |- - ConsistencyGroup labels every restored claim with - storage.simplyblock.io/consistency-group: , so the clones form a - new group at provisioning under the mandatory placement rule. Empty - leaves the clones as independent, mutually consistent volumes. - type: string - x-kubernetes-validations: - - message: field is immutable - rule: self == oldSelf - enablePartialRestore: + enableControlPlaneNodes: description: |- - EnablePartialRestore restores the members an incomplete generation still - has instead of failing the operation. Off by default: an incomplete - generation fails, naming the missing members. + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. type: boolean - x-kubernetes-validations: - - message: field is immutable - rule: self == oldSelf - namePrefix: - description: |- - NamePrefix prefixes every restored claim's name: - -, the source name read from the member - snapshot's spec.source.persistentVolumeClaimName. Defaults to the - operation's own name. - type: string - x-kubernetes-validations: - - message: field is immutable - rule: self == oldSelf - storageClassName: - description: |- - StorageClassName is the class every restored claim requests. When empty, - each claim inherits the class of its member snapshot's source claim, and - the restore fails for a member whose source claim no longer exists. - type: string - x-kubernetes-validations: - - message: field is immutable - rule: self == oldSelf + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set type: object - x-kubernetes-validations: - - message: field is immutable - rule: self == oldSelf - - message: field namePrefix is immutable once set - rule: '!has(oldSelf.namePrefix) || has(self.namePrefix)' - - message: field storageClassName is immutable once set - rule: '!has(oldSelf.storageClassName) || has(self.storageClassName)' - - message: field consistencyGroup is immutable once set - rule: '!has(oldSelf.consistencyGroup) || has(self.consistencyGroup)' - - message: field enablePartialRestore is immutable once set - rule: '!has(oldSelf.enablePartialRestore) || has(self.enablePartialRestore)' - volumeGroupSnapshotRef: + draftName: + default: site-draft description: |- - VolumeGroupSnapshotRef names the VolumeGroupSnapshot, in this namespace, - the operation acts on. Resolved at admission: a create naming a - VolumeGroupSnapshot that does not exist is rejected. + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 type: string x-kubernetes-validations: - - message: field is immutable + - message: draftName is immutable rule: self == oldSelf - required: - - action - - volumeGroupSnapshotRef - type: object - x-kubernetes-validations: - - message: field restore is immutable once set - rule: '!has(oldSelf.restore) || has(self.restore)' - status: - description: VolumeGroupSnapshotOpsStatus holds the observed state of - the operation. - properties: - completedAt: - description: CompletedAt is when the operation finished, successfully - or not. - format: date-time + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 type: string - members: - description: Members records each member snapshot and its restored - claim. + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: |- + Count is the number of journal managers to configure. The control plane + requires at least 3. + format: int32 + minimum: 3 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: tasks.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: Task + listKind: TaskList + plural: tasks + singular: task + scope: Namespaced + versions: + - name: v1alpha1 + schema: + openAPIV3Schema: + description: Task is the Schema for the tasks API + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: spec defines the desired state of Task + properties: + clusterName: + description: ClusterName is the target storage cluster name. + type: string + subtasks: + description: |- + Subtasks includes related child subtasks when supported by the backend. + FIXME: Unused for now + type: boolean + taskID: + description: TaskID filters results to a specific backend task when + set. + type: string + required: + - clusterName + type: object + status: + description: status defines the observed state of Task + properties: + tasks: + description: Tasks is the currently reported task list for the query + scope. + items: + properties: + canceled: + description: Canceled indicates whether the task was canceled. + type: boolean + parentTask: + description: |- + ParentTask is the parent task UUID when this task is a subtask. + FIXME: Unused for now + type: string + retried: + description: Retried is the number of retry attempts made for + the task. + format: int32 + type: integer + startedAt: + description: |- + StartedAt is the backend-reported task start timestamp. + FIXME: Unused for now + format: date-time + type: string + taskResult: + description: TaskResult is the backend result payload/message. + type: string + taskStatus: + description: TaskStatus is the backend lifecycle status for + the task. + type: string + taskType: + description: TaskType is the backend task function/type name. + type: string + uuid: + description: UUID is the backend task UUID. + type: string + type: object + type: array + type: object + required: + - spec + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: testfailovers.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: TestFailover + listKind: TestFailoverList + plural: testfailovers + shortNames: + - tfo + singular: testfailover + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.scope + name: Scope + type: string + - jsonPath: .spec.sourceRef + name: Source + type: string + - jsonPath: .spec.sourceCluster + name: "On" + type: string + - jsonPath: .spec.bubbleCluster + name: Bubble + type: string + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.step.state + name: Step + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + TestFailover is a one-way, non-disruptive test-failover drill. It recovers a + source volume, or a consistency group, from a snapshot into an isolated + namespace on a chosen cluster as bound PVCs, without touching the source. The + hub reads the source on its cluster and places the bubble on the recovery + cluster through OCM. It runs to a terminal phase, or holds Ready until it is + deleted, and deletion reclaims the clones and any snapshots the drill took. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + TestFailoverSpec is the request for one non-disruptive test-failover drill. + + The source is named by where it runs and what it is, so the hub can find it + without anyone extracting a backend handle by hand. SourceNamespace is + required for a Volume drill, where the source is a PVC, and unused for a Group + drill, where SourceRef names a consistency group. + properties: + bubbleCluster: + description: |- + BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + the replicated point, or another cluster. It must differ from SourceCluster; + test-failover recovers onto a different cluster, never in place. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + bubbleNamespace: + default: bubble + description: |- + BubbleNamespace is the namespace on the bubble cluster where the recovered + PVCs are created. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + scope: + description: Scope selects what the drill recovers. Immutable. + enum: + - Volume + - Group + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceCluster: + description: |- + SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + the source there through a ManagedClusterView. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceNamespace: + description: |- + SourceNamespace is the namespace of the source PVC on SourceCluster. + Required for scope=Volume. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceRef: + description: |- + SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + SourceNamespace (scope=Volume), or a consistency group (scope=Group). + Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + ttlSeconds: + description: |- + TTLSeconds is an optional maximum lifetime: the drill is torn down after it + even without a delete, so a forgotten drill cannot hold a clone forever. + format: int64 + minimum: 0 + type: integer + required: + - bubbleCluster + - scope + - sourceCluster + - sourceRef + type: object + x-kubernetes-validations: + - message: field sourceNamespace is immutable once set + rule: '!has(oldSelf.sourceNamespace) || has(self.sourceNamespace)' + - message: field bubbleNamespace is immutable once set + rule: '!has(oldSelf.bubbleNamespace) || has(self.bubbleNamespace)' + - message: sourceNamespace is required for scope=Volume + rule: self.scope != 'Volume' || has(self.sourceNamespace) + status: + description: TestFailoverStatus is the observed state of one drill. + properties: + clones: + description: Clones is one entry per recovered volume. + items: + description: |- + TestFailoverClone is one recovered volume: the source it came from, the + snapshot and clone the drill built, and the PVC placed on the bubble cluster. + properties: + cloneID: + description: CloneID is the backend id of the writable clone. + type: string + pvcName: + description: PVCName is the bound PVC in the bubble namespace + on the bubble cluster. + type: string + sizeBytes: + description: SizeBytes is the recovered volume's size. + format: int64 + type: integer + snapshotID: + description: |- + SnapshotID is the recovery-point snapshot: the replicated snapshot already on + the bubble cluster's backend that the clone is built from. + type: string + sourceFSType: + description: |- + SourceFSType is the source PV's CSI fsType, carried onto the bubble PV so the + node plugin stages the clone with the filesystem it actually carries. The + clone is a block copy of the source, so its filesystem is the source's; an + empty fsType makes the node plugin default to ext4 and refuse to mount an XFS + volume. + type: string + sourceHandle: + description: SourceHandle is the source volume's backend handle, + read from its PV. + type: string + sourceRef: + description: |- + SourceRef is the source volume, or group member, the recovered volume maps + to. + type: string + sourceVolumeContext: + additionalProperties: + type: string + description: |- + SourceVolumeContext is the source PV's CSI volumeAttributes, minus the + identity and provisioner keys, carried onto the bubble PV so the node plugin + receives a non-nil VolumeContext when it stages the clone. The clone's own + identity (NQN, connections, nsId, and so on) is re-resolved from the clone + handle at stage time, so only the class-level parameters are carried; the + identity keys are dropped so a failed clone lookup can never point the mount + back at the source. + type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string + required: + - sourceRef + type: object + type: array + x-kubernetes-list-map-keys: + - sourceRef + x-kubernetes-list-type: map + completedAt: + description: CompletedAt is when the drill reached a terminal phase. + format: date-time + type: string + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as the + drill moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from, so a stale status can be told from a current one. + format: int64 + type: integer + phase: + description: Phase is the drill's own progress. + enum: + - Pending + - Provisioning + - Ready + - Failed + - TearingDown + type: string + readyAt: + description: ReadyAt is when every recovered PVC became bound. + format: date-time + type: string + report: + description: Report is the drill's evidence, populated as it reaches + Ready. + properties: + bubbleCluster: + description: BubbleCluster is the cluster the drill recovered + onto. + type: string + invariantsHeld: + description: |- + InvariantsHeld is true only when the source fingerprint taken before the + drill matches the one taken at Ready. A Ready drill with this false is a + defect. + type: boolean + recoveryPoint: + description: RecoveryPoint is the snapshot or group generation + the drill recovered. + type: string + recoveryPointAgeSeconds: + description: RecoveryPointAgeSeconds is the drill time minus the + recovery-point time. + format: int64 + type: integer + recoveryPointTime: + description: RecoveryPointTime is when that point was taken. + format: date-time + type: string + type: object + startedAt: + description: StartedAt is when the drill started. + format: date-time + type: string + step: + description: Step is the position of the running drill's state machine. + properties: + claim: + description: |- + Claim records that this state's side effect was started. Absent means + no pass has started it since the state was entered. + properties: + attempt: + description: Attempt counts the claims taken on this state, + starting at 1. + format: int32 + type: integer + leaseUntil: + description: |- + LeaseUntil is when the claim expires and the side effect may be fired + again. + format: date-time + type: string + state: + description: State is the state the claim was taken in. + type: string + required: + - attempt + - leaseUntil + - state + type: object + deadline: + description: |- + Deadline is when that state expires, absent when it has none. It is an + absolute instant, so a state whose deadline passed while the controller + was down restores as already expired. + format: date-time + type: string + state: + description: |- + State is the state the machine was in. Empty means the resource has not + been reconciled yet, and restores to the graph's initial state. + type: string + type: object + x-kubernetes-validations: + - message: unknown step + rule: '!has(self.state) || self.state in [''ResolvingSource'',''ResolvingPoint'',''Shipping'',''Cloning'',''Placing'',''Releasing'']' + triggered: + description: |- + Triggered records that the current step's side effect was issued, so a + restart does not repeat it. + type: boolean + type: object + type: object + served: true + storage: true + subresources: + status: {} +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: volumegroupsnapshotops.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: VolumeGroupSnapshotOps + listKind: VolumeGroupSnapshotOpsList + plural: volumegroupsnapshotops + shortNames: + - vgsops + singular: volumegroupsnapshotops + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.volumeGroupSnapshotRef + name: GroupSnapshot + type: string + - jsonPath: .spec.action + name: Action + type: string + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.step.state + name: Step + type: string + - jsonPath: .status.membersBound + name: Bound + type: integer + - jsonPath: .status.message + name: Message + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + VolumeGroupSnapshotOps is a one-shot operation on a VolumeGroupSnapshot, + analogous to a Kubernetes Job: it drives its action to completion, records + the result, and is then inert. The Restore action creates one + PersistentVolumeClaim per member snapshot of the target's generation and + waits for every claim to bind. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + VolumeGroupSnapshotOpsSpec defines the requested operation. The whole spec is + immutable: the object is a request. + properties: + action: + description: Action is the operation to perform. + enum: + - Restore + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + restore: + description: |- + Restore carries the parameters of the Restore action. Ignored for any + other action. + properties: + consistencyGroup: + description: |- + ConsistencyGroup labels every restored claim with + storage.simplyblock.io/consistency-group: , so the clones form a + new group at provisioning under the mandatory placement rule. Empty + leaves the clones as independent, mutually consistent volumes. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + enablePartialRestore: + description: |- + EnablePartialRestore restores the members an incomplete generation still + has instead of failing the operation. Off by default: an incomplete + generation fails, naming the missing members. + type: boolean + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + namePrefix: + description: |- + NamePrefix prefixes every restored claim's name: + -, the source name read from the member + snapshot's spec.source.persistentVolumeClaimName. Defaults to the + operation's own name. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + storageClassName: + description: |- + StorageClassName is the class every restored claim requests. When empty, + each claim inherits the class of its member snapshot's source claim, and + the restore fails for a member whose source claim no longer exists. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + type: object + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + - message: field namePrefix is immutable once set + rule: '!has(oldSelf.namePrefix) || has(self.namePrefix)' + - message: field storageClassName is immutable once set + rule: '!has(oldSelf.storageClassName) || has(self.storageClassName)' + - message: field consistencyGroup is immutable once set + rule: '!has(oldSelf.consistencyGroup) || has(self.consistencyGroup)' + - message: field enablePartialRestore is immutable once set + rule: '!has(oldSelf.enablePartialRestore) || has(self.enablePartialRestore)' + volumeGroupSnapshotRef: + description: |- + VolumeGroupSnapshotRef names the VolumeGroupSnapshot, in this namespace, + the operation acts on. Resolved at admission: a create naming a + VolumeGroupSnapshot that does not exist is rejected. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + required: + - action + - volumeGroupSnapshotRef + type: object + x-kubernetes-validations: + - message: field restore is immutable once set + rule: '!has(oldSelf.restore) || has(self.restore)' + status: + description: VolumeGroupSnapshotOpsStatus holds the observed state of + the operation. + properties: + completedAt: + description: CompletedAt is when the operation finished, successfully + or not. + format: date-time + type: string + members: + description: Members records each member snapshot and its restored + claim. items: description: |- RestoredMemberStatus records one member snapshot and the claim restored @@ -12333,6 +13778,44 @@ rules: - patch - update - watch +- apiGroups: + - cluster.open-cluster-management.io + resources: + - managedclusters + verbs: + - get + - list + - watch +- apiGroups: + - coordination.k8s.io + resources: + - leases + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes + verbs: + - create + - delete + - get + - list + - update + - watch +- apiGroups: + - csiaddons.openshift.io + resources: + - csiaddonsnodes/status + verbs: + - get + - patch + - update - apiGroups: - discovery.k8s.io resources: @@ -12452,7 +13935,9 @@ rules: - storagenodes - storagepoolops - storagepools + - storagesitedeployments - tasks + - testfailovers - volumemigrations verbs: - create @@ -12487,7 +13972,9 @@ rules: - storagenodes/finalizers - storagepoolops/finalizers - storagepools/finalizers + - storagesitedeployments/finalizers - tasks/finalizers + - testfailovers/finalizers - volumemigrations/finalizers verbs: - update @@ -12519,7 +14006,9 @@ rules: - storagenodesets/status - storagepoolops/status - storagepools/status + - storagesitedeployments/status - tasks/status + - testfailovers/status - volumegroupsnapshotops/status - volumemigrations/status verbs: @@ -12546,6 +14035,30 @@ rules: - get - list - watch +- apiGroups: + - view.open-cluster-management.io + resources: + - managedclusterviews + verbs: + - create + - delete + - get + - list + - patch + - update + - watch +- apiGroups: + - work.open-cluster-management.io + resources: + - manifestworks + verbs: + - create + - delete + - get + - list + - patch + - update + - watch --- apiVersion: rbac.authorization.k8s.io/v1 kind: ClusterRole diff --git a/operator/docs/designs/design-consistency-groups.md b/operator/docs/designs/design-consistency-groups.md index aa17b552c..d8c605819 100644 --- a/operator/docs/designs/design-consistency-groups.md +++ b/operator/docs/designs/design-consistency-groups.md @@ -208,6 +208,17 @@ A group lives exactly as long as its members. Removing the last member deletes t ### 4.5 Membership after provisioning (Phase 4 — Planned) +> **Update 2026-10-03 (co-location).** Group-wide migration, the pre-join +> live migration of a late joiner, create-time subsystem forcing and namespace +> moves between subsystems are specified in sbcli +> `docs/consistency-group-colocation.md`. In short: the control plane refuses a +> migration that would split a group and migrates a group's whole scope (every +> subsystem holding a member) through `.../consistency-groups/{g}/migration`; +> the CSI watcher turns a join refused for placement into a VolumeMigration to +> the pin and joins afterwards; the rebalancer skips group members; namespace +> moves and the node's dm-linear indirection are behind flags. + + Phase 4 makes the membership label live for the volume's whole life rather than read once at creation. Adding `storage.simplyblock.io/consistency-group` to an existing PVC joins its volume to the named group, and removing the label detaches the volume in §8.2's sense: the epoch closes and every generation that contains the member stays restorable. **What made this safe.** The group's subsystem-scoped operations reach every volume that shares a member's NVMe subsystem: a migration moves the whole subsystem (§9.5), and the frozen cut stalls a shared subsystem's I/O path as one unit. While a subsystem could hold volumes of more than one storage pool, only the create path could guarantee that a member's subsystem held nothing those operations must not touch, which is why membership was fixed at creation. The control plane now enforces subsystem and pool alignment as an invariant (P0-5): a shared subsystem belongs to exactly one pool, checked at lvol placement and again inside the transactional namespace-slot claim, so any volume in the group's pool sits in a subsystem wholly owned by that pool. A late join can therefore no longer entangle another pool's volumes in the group's freeze or migration scope, and the join reduces to the epoch bookkeeping the backend already has. @@ -358,6 +369,17 @@ If a member's snapshot in some generation is gone (pruned, or its volume hard-de ### 8.4 Members are excluded from migration +> **Update 2026-10-03: superseded for whole groups.** Group-wide migration, the pre-join +> live migration of a late joiner, create-time subsystem forcing and namespace +> moves between subsystems are specified in sbcli +> `docs/consistency-group-colocation.md`. In short: the control plane refuses a +> migration that would split a group and migrates a group's whole scope (every +> subsystem holding a member) through `.../consistency-groups/{g}/migration`; +> the CSI watcher turns a join refused for placement into a VolumeMigration to +> the pin and joins afterwards; the rebalancer skips group members; namespace +> moves and the node's dm-linear indirection are behind flags. + + A group's members are pinned to one logical volume store (§4.2), so a member that moved off the store would break the frozen group snapshot. For now, the design excludes consistency-group members from volume migration entirely, and it does so at two layers. The operator's validating webhook on `VolumeMigration` (§9.5) declines the request at `kubectl apply` when the target PV's backing volume is a group member, so the operator never even starts the migration. The backend refusing to migrate a group member is the last line of defense behind it, catching any migration reached by a path the webhook does not cover. Between the two, a group's placement stays fixed for its life and the frozen-snapshot invariant holds by construction rather than being checked after the fact. This is the conservative first cut, and it keeps the feature simple while the group model settles. Migrating a whole group as a unit, moving the shared placement pin together so every member stays colocated, is a later option and is Open Question 2. The §12 failure path, a member found off the pinned store at snapshot time, stays as defense in depth against a member moved by some path other than migration (a node-failure recovery, say), because a group snapshot must fail loudly rather than freeze an inconsistent subset. diff --git a/operator/docs/designs/design-csi-addons-replication.md b/operator/docs/designs/design-csi-addons-replication.md new file mode 100644 index 000000000..8b773548b --- /dev/null +++ b/operator/docs/designs/design-csi-addons-replication.md @@ -0,0 +1,503 @@ +# Design Document: csi-addons Volume Replication + +**Status:** Phase 3 Implemented (Phase 4, group replication: Planned) +**Author:** Israel Geoffrey (geoffrey1330) +**Date:** 2026-09-16 (last updated 2026-09-25) +**Test Plan:** [`tests/test-plan-csi-addons-replication.md`](../tests/test-plan-csi-addons-replication.md) + +--- + +## Phasing Overview + +| Phase | Status | Scope | Sections | +|-------------|-------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------| +| **Phase 1** | Implemented | The csi-addons machinery and the steady-state contract: CRDs, controller-manager, sidecar, the Replication and csi-addons Identity gRPC services with `EnableVolumeReplication`, `DisableVolumeReplication`, and `GetVolumeReplicationInfo`, backed by a typed backend status endpoint | §4, §5.1, §6 | +| **Phase 2** | Implemented | The lifecycle verbs: `PromoteVolume` (planned and forced), `DemoteVolume`, and `ResyncVolume`. Validation end to end against a Ramen `VolumeReplicationGroup` in async mode is still outstanding (§12, E-06/E-07) | §5.2, §9 | +| **Phase 3** | Implemented | §11 (the Prometheus metrics). §7.2's peerClasses preflight is out of this design's scope entirely (it's Ramen's own `DRPolicy` mechanism) and is deferred to a future Ramen-integration design | §7.1, §11 | +| **Phase 4** | Planned | §14 (`VolumeGroupReplication`): the group-level csi-addons surface, driven by the stock controller-manager through a new driver VolumeGroup service and a group handle the existing Replication verbs act on, backed by new group-replication endpoints (P0-6/P0-7). This is the gap analysis's own Phase 2 group-replication item | §14 | + +Phase 1 is independently useful: a `VolumeReplication` object per PVC whose status truthfully reports the relationship, which no surface provides today. Phase 2 makes the object drivable, which is what Ramen actually needs. Phase 3 makes the whole thing operable at fleet scale. Phase 4 lifts the same surface from one volume to a consistency group, replicated as one unit. + +The phase numbers above are this document's own, not the DR storage foundation gap analysis's (§1): its Phase 0 (shipping the csi-addons contract itself) is this design's Phase 1, and its Phase 1 (promote, demote, and resync end to end through Ramen) is this design's Phase 2. + +--- + +## Phase 0 — External Prerequisites + +| # | Prerequisite | Kind | Blocks | Status | +|------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------|---------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| P0-1 | A typed, steady-state per-volume replication status read: `GET .../volumes/{id}/replication/status` serving what `lvol_controller.get_replication_info` computes today (state, lag, outstanding bytes, failure counters), available for the volume's whole replicated life | Control plane (`sbcli`) | Phase 1 | Shipped: `GET .../volumes/{v}/replication/status` → `ReplicationStatusDTO` (`simplyblock_web/api/v2/cluster/storage_pool/volume/replication.py:58-69`) | +| P0-2 | Idempotent attach and detach: attaching a volume to the policy it already follows returns success, and detaching a non-attached volume returns success | Control plane (`sbcli`) | Phase 1 | Shipped: `replication_policy_controller.attach_policy`/`detach_policy` (`simplyblock_core/controllers/replication_policy_controller.py:208-262`) | +| P0-3 | A standalone demote verb: `POST .../volumes/{id}/replication/demote` that converges the peer while still serving (repeated snapshot-and-ship until the remaining delta is small), then quiesces, ships the final delta, confirms it landed on the peer, and fences the data path | Control plane (`sbcli`) | Phase 2 | Shipped: `POST .../volumes/{v}/replication/demote` → `lvol_controller.demote_lvol` | +| P0-4 | An `rpo_target_seconds` field on `ReplicationPolicy`, so RPO compliance is computable against a declared target rather than the derived lag budget | Control plane (`sbcli`) | Phase 3 | Shipped: `ReplicationPolicy.rpo_target_seconds` (`simplyblock_core/models/replication.py:89`), wired through the API (`PolicyParams.rpo_target_seconds`) and CLI (`--rpo-target-sec`) | +| P0-5 | csi-addons upstream: the `VolumeReplication` and `VolumeReplicationClass` CRDs (`replication.storage.openshift.io/v1alpha1`), the kubernetes-csi-addons controller-manager image, and the csi-addons sidecar image | Ecosystem | Phase 1 | Vendored in the chart at v0.15.0 behind `csiaddons.create` (all twelve upstream CRDs, since the stock manager starts a controller per kind); sidecar wiring is Phase 1 | +| P0-6 | Group-level replication in the control plane: a consistency group replicated as one unit, with attach/detach to a group replication policy, group failover, group demote, group failback, and a typed group status, using group snapshots (`bdev_lvol_snapshot_group`) as the recovery generations (§14.4, §14.5) | Control plane (`sbcli`) | Phase 4 | Not shipped. New group-replication engine and the `/consistency-groups/{id}/replication/*` endpoints | +| P0-7 | A group replication policy: cadence and target for a whole consistency group, the group twin of `ReplicationPolicy` (Open Question 7) | Control plane (`sbcli`) | Phase 4 | Not shipped | + +Everything the per-volume adapter (Phases 1-3) needs already exists: the attach and detach calls, failover, the failback and commit pair, the relationship read, and the backlog arithmetic inside `get_replication_info`. That adapter is thin precisely because the engine is complete. Group replication (Phase 4) is the exception: P0-6/P0-7 are genuinely new backend work, because the engine replicates per volume today and a consistency group has no group-level replication of its own (§14.4). + +--- + +## Table of Contents + +1. [Background](#1-background) +2. [Goals and Non-Goals](#2-goals-and-non-goals) +3. [Architecture Overview](#3-architecture-overview) +4. [The csi-addons Machinery](#4-the-csi-addons-machinery) +5. [The Replication Service](#5-the-replication-service) +6. [Steady-State Status and Conditions](#6-steady-state-status-and-conditions) +7. [VolumeReplicationClass and peerClasses](#7-volumereplicationclass-and-peerclasses) +8. [Coexistence with the Legacy Replication Kinds](#8-coexistence-with-the-legacy-replication-kinds) +9. [Backend API Requirements](#9-backend-api-requirements) +10. [Failure Modes and Fallback](#10-failure-modes-and-fallback) +11. [Observability](#11-observability) +12. [Testing Strategy](#12-testing-strategy) +13. [Migration Strategy](#13-migration-strategy) +14. [VolumeGroupReplication](#14-volumegroupreplication) +15. [Open Questions](#15-open-questions) + +--- + +## Overview + +simplyblock's replication engine is complete and works: interval snapshots ship per volume under a `ReplicationPolicy`, failover clones the last replicated generation on the target, and the failback-plus-commit pair performs a fenced, lossless cutover back. What it is not is drivable by anything outside simplyblock. A Ramen `VolumeReplicationGroup` speaks exactly one per-volume replication dialect, the csi-addons `VolumeReplication` object, and simplyblock implements none of it. + +This design exposes the existing engine behind that dialect. The CSI controller plugin gains the csi-addons Replication and Identity gRPC services, a csi-addons sidecar and controller-manager reconcile `VolumeReplication` objects into those RPCs, and each RPC is a thin adapter onto a backend call that already exists (or is named in Phase 0). No replication mechanism is rebuilt, and the engine keeps shipping snapshots exactly as it does today. + +| Concern | Mechanism | Decided when | +|--------------------------------|-------------------------------------------------------------------------|-------------------------------------| +| Per-volume replication intent | `VolumeReplication` (`spec.replicationState: primary\|secondary`) | Reconciled continuously | +| Cadence, retention, and target | `VolumeReplicationClass` parameters naming a `ReplicationPolicy` | At class authoring | +| Promote, demote, and resync | The Replication gRPC verbs, adapted onto failover, demote, and failback | On each reconcile until `Completed` | +| Relationship health | `VolumeReplication.status.conditions` from the typed status read | On every reconcile | +| RPO figures | Prometheus metrics from the control plane (§11) | Continuously | + +A reader who stops here has the model: the engine is unchanged, the csi-addons surface is the adapter, and the volume handle (`{clusterID}:{poolID}:{volumeID}`) is the one identity that lets either cluster's driver address the same backend relationship. + +--- + +## 1. Background + +**The engine.** A volume replicates when `do_replicate` is set and a `ReplicationPolicy` names its cadence and target. The snapshot monitor takes interval snapshots, the shipping runner transfers each one to a landing volume on the target cluster and chains it there, and retention prunes behind the shipped frontier. Failover (`POST .../replication/failover`) clones the last fully replicated snapshot into a writable volume on the target with the same NVMe identity. Failback is a reversed replication seeded by `data_uuid` matching, and commit (`POST .../replication/commit`) runs the fenced final-step cutover through a task runner. All of this is driven today by the operator's `ReplicationPair`, `ReplicationPolicy`, `ReplicationSlot`, and `ReplicationOps` kinds, which build their HTTP calls inline against the same endpoints. + +**The contract this design targets.** Ramen's dr-cluster operator reconciles one `VolumeReplication` per protected PVC. It flips `spec.replicationState` between `primary` and `secondary` and waits for the driver's conditions (`Completed`, `Degraded`, `Resyncing`) to report the operation done and the relationship healthy. It reads `status.lastSyncTime` for RPO. It never calls a vendor API. The interfaces are the csi-addons specification's Replication gRPC, served by the driver, and the kubernetes-csi-addons controller-manager, which turns `VolumeReplication` objects into those RPCs through a per-driver sidecar. + +**The direction is already committed.** The CRD redesign excludes the four replication kinds from its model because "that subsystem is being redesigned against the CSI Addons specification, whose `VolumeReplication` and `VolumeGroupReplication` kinds already carry the per-volume and per-group replication contract that a backup tool or a DR orchestrator understands" (`crd-redesign/design-crd-model.md`, Non-Goals). This document is that redesign's replication chapter: the per-volume half in §5, and the per-group half (`VolumeGroupReplication`) in §14. The DR storage foundation gap analysis (Phase 0 and Appendix A of that document) is its requirements source. + +**Three facts about today's surface shape the design.** + +1. **The steady-state status has no home.** The typed relationship read (`GET .../replication/relationships/{lvol}` and its per-volume twin) serves an `LVolReplication` record that is only created at cutover or failover, so it returns 404 for a volume's entire healthy replicated life. The real steady-state verdict lives in `lvol_controller.get_replication_info`, an untyped dict reachable only as `rep_info` on the volume DTO. The operator's slot controller documents the consequence in its own comments: `status.lastReplicatedAt` is stamped once at attach and never refreshed while replication is healthy. +2. **There is no demote and no resync verb.** The backend surface is attach, detach, failover, failback, commit, and cutover-proceed. Demote exists only as an internal composition (ANA suspend plus fencing) inside the failover and cutover paths. Resync exists only as the failback direction reversal. +3. **The secondary has no volume object.** During steady-state replication the target cluster holds replicated snapshots and transient landing volumes, never a secondary lvol. A writable volume materializes on the target only at promotion. The `VolumeReplication` on the DR cluster therefore describes a relationship addressed by handle, not a local volume, and the driver's multi-cluster `secret.json` resolution is what makes that address work from either side. + +--- + +## 2. Goals and Non-Goals + +### Goals + +- A `VolumeReplication` object per PVC, reconciled by the stock kubernetes-csi-addons controller-manager against this driver, drives simplyblock replication: enable, disable, promote (planned and forced), demote, and resync. +- The driver's conditions follow the csi-addons contract: `Completed` reports the last requested state change finished, `Degraded` reports relationship health, and `Resyncing` reports a divergence catch-up in flight. Steady-state healthy async is `Completed=True, Degraded=False, Resyncing=False`, and lag alone trips none of them. +- `status.lastSyncTime` is truthful for the volume's whole replicated life, sourced from a typed backend status read rather than the cutover-time relationship record. +- Every verb is idempotent, because Ramen re-drives every reconcile. +- The adapter reuses one shared control-plane client (`atlas-lib/controlplane`), ending the pattern where each consumer hand-rolls the same replication HTTP calls. +- A `StorageClass` and `VolumeSnapshotClass` naming convention across paired clusters that Ramen's peerClasses can express (§7.2 documents what Ramen's own contract requires; verifying it is a future Ramen-integration design's concern, not this one's). +- The RPO and backlog figures Ramen cannot carry (`bytesBehind`, throughput, RPO compliance) are exported as Prometheus metrics from the control plane. + +### Non-Goals + +- **Global `VolumeGroupReplication`.** The base group surface (one VRG, one vendor's PVCs, one `VolumeGroupReplication` on top of a consistency group) is specified in §14. RamenDR's newer multi-VRG "Global VGR" consensus, spanning a replication group across several applications, is out of scope (§14.8, Open Question 4). The group snapshot surface (`design-consistency-groups.md`) is untouched. +- **VolSync and the S3 backup path.** The recurrent-immutable-snapshots DR type is a separate phase of the gap analysis and does not pass through this adapter. +- **Synchronous replication.** The engine is asynchronous snapshot shipping, and nothing here changes that. +- **Ramen hub components.** DRPolicy, DRPC, and hub orchestration are consumers of this contract, not part of it. +- **Retiring the legacy replication kinds.** `ReplicationPair`, `ReplicationPolicy`, `ReplicationSlot`, and `ReplicationOps` keep working unchanged. §8 defines coexistence and §13 the consolidation direction. Removal is its own change once the adapter is proven. +- **A simplyblock DR-status CRD.** The observability surface here is metrics. A rollup CRD is future work in the gap analysis's Appendix B. + +--- + +## 3. Architecture Overview + +``` +┌──────────────────────────────────────────────────────────────────────────────┐ +│ Kubernetes (each cluster) │ +│ │ +│ VolumeReplicationClass VolumeReplication (one per protected PVC) │ +│ (names a ReplicationPolicy) spec.replicationState: primary|secondary │ +│ │ │ │ +│ ▼ ▼ │ +│ ┌──────────────────────────────────────────────────────────────────┐ │ +│ │ kubernetes-csi-addons controller-manager (chart-deployed): │ │ +│ │ reconciles VolumeReplication → Replication gRPC on the driver │ │ +│ └──────────────────────────────┬───────────────────────────────────┘ │ +│ │ via CSIAddonsNode registration │ +│ ┌──────────────────────────────▼───────────────────────────────────┐ │ +│ │ CSI controller StatefulSet (SimplyblockDriver reconciler): │ │ +│ │ … csi-provisioner, csi-snapshotter, …, csi-addons sidecar, │ │ +│ │ and the controller plugin, all on one socket. The plugin │ │ +│ │ serves CSI Identity/Controller/GroupController, plus │ │ +│ │ csi-addons Identity and Replication (this design) │ │ +│ └──────────────────────────────┬───────────────────────────────────┘ │ +│ │ +│ operator: PVCAnnotationWatcher skips csi-addons-managed volumes (§8); │ +│ ReplicationPair/Policy author the backend target and policy │ +└─────────────────────────────────┬────────────────────────────────────────────┘ + │ HTTP, resolved per volume handle +┌─────────────────────────────────▼────────────────────────────────────────────┐ +│ simplyblock control plane │ +│ PUT .../volumes/{v} {replication_policy_id} (attach, │ +│ detach) │ +│ GET .../volumes/{v}/replication/status (P0-1, steady state) │ +│ POST .../volumes/{v}/replication/failover (promote: planned gate │ +│ or forced) │ +│ POST .../volumes/{v}/replication/demote (P0-3: converge, │ +│ quiesce, flush, fence) │ +│ POST .../volumes/{v}/replication/failback (resync) │ +│ GET .../replication/relationships/{lvol} (cutover records) │ +│ exports simplyblock_replication_* metrics: lag, backlog, RPO (§11) │ +└──────────────────────────────────────────────────────────────────────────────┘ +``` + +**One relationship, addressed from either cluster.** The `VolumeReplication.spec.dataSource` resolves to a PV whose handle is `{clusterID}:{poolID}:{volumeID}`. The driver resolves the backend from the handle through `clusters.Client`, exactly as every other RPC does, so the DR cluster's driver can drive the same backend relationship as the source cluster's without either side holding special state. This is the same stateless addressing the node plugin already uses to redirect a staged volume after failover. + +**The adapter holds no state.** csi-addons RPCs are stateless and idempotent by contract. Every answer the driver gives is derived on the spot from the backend status read and the relationship record. There is no driver-side cache, no persisted step, and no state machine. `PromoteVolumeResponse` and `DemoteVolumeResponse` carry no fields at all in `csi-addons/spec` v0.2.0 -- there is no response field to report partial progress in. A verb whose backend work outlives the call (a demote converging a busy peer) instead returns a retryable `ABORTED` error, and the vendored controller-manager's own reconcile requeue is the retry loop; only once the backend reports the target state reached does the call return success, which the controller then reflects as the CR's `Completed` condition. + +**The operator's part is small and off the data path.** The kinds that author the backend state (`ReplicationPair` for the target, `ReplicationPolicy` for cadence and retention) keep working unchanged, and the classes name what they author. On top of them the operator teaches the `PVCAnnotationWatcher` the one-owner rule (§8) so the legacy annotation path and a `VolumeReplication` never fight over one volume. Everything imperative it used to own (`ReplicationOps`, the commit cutover, the cutover-proceed handshake) is off this contract and confined to the legacy path. + +--- + +## 4. The csi-addons Machinery + +### 4.1 What gets deployed + +- **CRDs** (chart): `volumereplications.replication.storage.openshift.io`, `volumereplicationclasses.replication.storage.openshift.io`, and the kubernetes-csi-addons operational CRDs (`csiaddonsnodes.csiaddons.openshift.io`). Vendored at a pinned upstream version, the same way the `VolumeGroupSnapshot` CRDs are. +- **Controller-manager** (chart): the stock kubernetes-csi-addons manager. It discovers driver endpoints through `CSIAddonsNode` objects and reconciles `VolumeReplication` by calling the driver's Replication gRPC. +- **Sidecar** (operator, `SimplyblockDriver` reconciler): the csi-addons sidecar in the controller StatefulSet. It connects to the plugin's socket, probes the csi-addons Identity service for capabilities, publishes a `CSIAddonsNode`, and proxies the manager's gRPC to the plugin. Wiring follows the existing sidecar pattern: a seventh field on `spec.sidecarImages`, a pinned default in `sidecars.go`, an appended `controllerSidecar` entry in `workloads.go` (appended, because the per-container tweaks there are positional), and a new RBAC component whose group alias is `replication.storage.openshift.io` plus `csiaddons.openshift.io`. + +### 4.2 What the plugin serves + +The plugin registers two additional gRPC services on the existing socket, beside the CSI services, following the GroupController precedent in `csicommon.NonBlockingGRPCServer`: + +- **csi-addons Identity:** `GetIdentity`, `GetCapabilities` (advertising `VOLUME_REPLICATION`), and `Probe`. This is a distinct service from CSI Identity, so it is a new small server type, not an extension of the existing one. +- **Replication:** the six verbs of §5, implemented on the controller `Server` through an embedded `replication.UnimplementedControllerServer` from `github.com/csi-addons/spec` (pinned at v0.2.0), mirroring how the GroupController embeds its unimplemented base. The same verbs accept a group handle for group replication (§14.4). +- **csi-addons VolumeGroup (GroupController)** (Phase 4, Planned): `CreateVolumeGroup`, `ModifyVolumeGroupMembership`, and `DeleteVolumeGroup`, mapping a set of volume handles to the backend consistency group and back (§14.3). csi-addons Identity advertises the capability so the stock controller-manager dials it. + +`NonBlockingGRPCServer.Start` today takes exactly the three CSI servers. It gains a registration hook so the driver package can register additional services without `csicommon` importing csi-addons. + +### 4.3 The shared client + +The generated control-plane client already declares every replication endpoint, but it is `internal/` to atlas-lib. This design adds `atlas-lib/controlplane/replication.go`, a typed wrapper over those operations (attach, detach, status, failover, demote, failback, relationship, and commit for the legacy `ReplicationOps` path), following the shape `migrations.go` set for long-running backend operations. The driver adapter consumes it, and the operator's replication reconcilers move onto it opportunistically (§13), ending the three-copies-of-the-same-HTTP-call pattern the gap analysis lists as an inconsistency. + +--- + +## 5. The Replication Service + +`volume_id` on every request is the CSI volume handle. The driver parses it with the existing helper and resolves the cluster client from `secret.json`. Every verb is idempotent: repeating a completed operation returns success, and repeating an in-flight one returns the in-flight answer. Backend errors map to gRPC codes through the existing per-RPC classifier, extended with replication constructors. + +### 5.1 Phase 1 verbs + +| Verb | Backend mapping | Semantics | +|----------------------------|------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `EnableVolumeReplication` | `PUT .../volumes/{v}` body `{"replication_policy_id": }` | The policy comes from the `VolumeReplicationClass` parameters (§7). Attach is synchronous on the backend; already-attached to the same policy is success (P0-2). Attaching to a *different* policy is **not** refused: sbcli's `attach_policy` silently re-attaches onto the new policy rather than returning `FAILED_PRECONDITION`, and no endpoint yet exposes the policy id a volume is currently attached to for the driver to compare against before attaching (§15, Open Question 3). | +| `DisableVolumeReplication` | `PUT .../volumes/{v}` body `{"replication_policy_id": null}` | Detach. A 409 (cutover in flight) maps to `ABORTED`, retryable. Not-attached is success (P0-2). | +| `GetVolumeReplicationInfo` | `GET .../volumes/{v}/replication/status` (P0-1) | Returns `lastSyncTime` (newest fully replicated snapshot's creation time). `csi-addons/spec` v0.2.0's `GetVolumeReplicationInfoResponse` carries no `lastSyncDuration` or `lastSyncBytes` field, so the status read's cycle-duration and shipped-size data has no response field to land in until a newer spec version adds one. | + +### 5.2 Phase 2 verbs + +| Verb | Backend mapping | Semantics | +|------------------------------------|--------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `PromoteVolume` with `force=true` | `POST .../volumes/{v}/replication/failover` (already shipped, unchanged) | Unplanned promote: clone the last fully replicated generation on the target, retire the (assumed dead) source. Idempotent by the backend's own NQN probe. Ignores demote state entirely -- its whole premise is that the peer may never have been reachable to demote. | +| `PromoteVolume` with `force=false` | `POST .../volumes/{v}/replication/failover?planned=true` (P0-3 adds the `planned` parameter to the existing route) | The same promote as the forced form, refused unless the peer already holds every acknowledged write, which is exactly the state a completed demote leaves behind: the source fenced and the final flush confirmed. Demote still converging surfaces as `ABORTED` (retryable); no demote ever requested surfaces as `FAILED_PRECONDITION` -- the split matters because the vendored csi-addons controller auto-escalates ANY `FAILED_PRECONDITION` from a force=false promote to force=true inline, with no wait-and-retry grace period of its own (§15, resolves the risk this design's original `FAILED_PRECONDITION`-only wording would have created). | +| `DemoteVolume` | `POST .../volumes/{v}/replication/demote` (P0-3, new route) | Fence the source (ANA inaccessible) FIRST, then trigger and confirm one final internal snapshot. This is the lossless half of a planned swap: after demote, the peer's planned promote loses nothing. Synchronous and re-drivable, not queued: returns `ABORTED` while the snapshot is still converging, success only once confirmed. Deliberately does not reuse `replication/commit`'s live-cutover engine (shrink rounds, the synchronous hub transfer) -- that machinery exists to bound a freeze window against a *live* writer, and by the time Kubernetes calls `DemoteVolume` the workload has already unmounted, so there is no moving target to protect against, only the ordinary fence-before-snapshot discipline. It also never touches a target volume; that is `PromoteVolume`'s job, on a separate, later call, possibly on a different cluster. | +| `ResyncVolume` | `POST .../volumes/{v}/replication/failback` | Reverse the shipping direction, seeded by `data_uuid` matching so a recovered source resyncs by delta rather than full copy. Response's `ready` field is false while lag exceeds the budget and true once caught up. Resync reconciles the diverged old primary from the current primary; it never merges, and never calls `commit` -- that would be a cutover, not a resync. | + +**Promote is one operation; planned and unplanned differ only in what precedes it.** Both forms clone the last fully replicated generation on the target and serve it under the preserved NVMe identity, exactly as failover does today. The planned form is lossless not because it runs different machinery but because a completed demote guarantees the last replicated generation contains every acknowledged write, and the planned gate refuses the promote until that holds. The forced form skips the gate and accepts the RPO loss, because its premise is that the source is gone. The engine's commit cutover (`replication/commit`, the `FN_REPLICATION_FINAL` runner with its shrink rounds and cutover-proceed handshake) is deliberately NOT part of this contract and remains behind only the legacy `ReplicationOps` migration path (§8, §13) -- its convergence job does NOT move into the demote verb as this design originally assumed; see the `DemoteVolume` row above for why that engine is the wrong shape for a post-unmount demote. + +**The planned form's no-demote branch splits on the source's own health, not on `role`.** The vendored controller-manager has no "already primary" awareness of its own: `markVolumeAsPrimary` calls Promote unconditionally, every time a `VolumeReplication` first declares primary intent (`internal/controller/replication.storage/volumereplication_controller.go`), including day-one protection of a volume that has always lived here and has never failed over. `get_replication_info`'s `role` field cannot distinguish that case from a volume genuinely awaiting a planned re-promotion after a completed demote -- `demote_lvol` writes only `LVol.replication_demote_state`, a field `role`'s computation never reads, so a fenced, demoted volume still reports `role: source`, identically to one that was never touched at all. A role-based short-circuit before the backend call was tried and reverted for exactly this reason: it silently skipped the real `failover?planned=true` call a completed demote is waiting on, leaving the volume fenced while csi-addons reported `Completed=True`. The fix instead reads the source's *own storage node* status (`lvol_controller.replication_source_online`, mirroring the target-node health check `replicate_lvol_on_target_cluster` already makes for the destination side): when `planned=true` and no demote was ever requested, a genuinely online source has nothing to fail over and the call succeeds as the no-op it is, with no clone; a source that is not online falls through to `FAILED_PRECONDITION` exactly as before, letting the vendored controller's force-escalation run for a real disaster. This narrows, but does not close, the race a staleness-tolerant health field always carries: a source that died within the last health-check interval still reads online and is treated as a no-op for one reconcile, correcting itself once the node's status catches up and the controller retries. Test 6 of `regression_test/21/test_csi_addons_replication.sh` needs updating to match -- its `$FRESH_PVC_NAME` scenario now reaches Primary via this no-op path, not via force-escalation, since nothing in that scenario's setup makes the source anything but healthy. + +--- + +## 6. Steady-State Status and Conditions + +### 6.1 The typed status read (P0-1) + +`GET .../volumes/{v}/replication/status` serves, as a typed DTO, what `get_replication_info` computes today: + +``` +role: source | secondary | failed_over | none +state: in_sync | replicating | lagging | degraded | error | not_replicating +last_replicated_at: timestamp of the newest fully replicated snapshot +lag_seconds: now - last_replicated_at +lag_budget_seconds: derived budget (or the policy's rpo_target_seconds, P0-4) +outstanding_count: snapshots queued but not yet shipped +outstanding_bytes: sum of used_size over the outstanding snapshots +failing_count: shipping tasks currently suspended on errors +max_retry_reached: whether any task exhausted its retries +last_cycle_bytes: used_size of the last shipped snapshot +last_cycle_seconds: duration of the last shipping cycle +resyncing: whether a failback catch-up is in flight +``` + +This endpoint exists for the volume's whole replicated life. It does not replace the relationship read: `GET .../replication/relationships/{lvol}` keeps its cutover-record semantics (including surviving source deletion, which the node redirect depends on), and the status read is the steady-state complement. The operator's slot controller can also poll it to keep `status.lastReplicatedAt` honest, independent of this design's adapter. + +### 6.2 Condition mapping + +Following the csi-addons contract: conditions describe relationship and operation health, never instantaneous RPO. Steady-state healthy async is `Completed=True, Degraded=False, Resyncing=False`, and lag alone trips none of them. + +| Condition | True when | simplyblock source | +|-------------|------------------------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------| +| `Completed` | The last requested state change (enable, promote, demote) has finished | Enable: attach returned. Promote (either form): the relationship reports the target active. Demote: the P0-3 verb confirmed the final flush landed. | +| `Degraded` | The relationship is unhealthy or not progressing | `state` is `degraded` (suspended shipping tasks) or `error` (retries exhausted), or `lag_seconds` exceeds `lag_budget_seconds`. | +| `Resyncing` | A divergence catch-up is reconciling the secondary | The status read's `resyncing` flag: a failback direction reversal or a `FN_REPLICATION_FINAL` task in flight. | + +`status.state` mirrors `spec.replicationState` once `Completed=True`, from the status read's `role`. + +--- + +## 7. VolumeReplicationClass and peerClasses + +### 7.1 The class + +A `VolumeReplicationClass` binds a `VolumeReplication` to a backend policy: + +```yaml +apiVersion: replication.storage.openshift.io/v1alpha1 +kind: VolumeReplicationClass +metadata: + name: simplyblock-async-5m +spec: + provisioner: csi.simplyblock.io + parameters: + replicationPolicy: dr-policy-5m + schedulingInterval: 5m +``` + +`replicationPolicy` names the backend `ReplicationPolicy` (resolved per cluster by name), which owns cadence, retention, mode, and the replication target. `schedulingInterval` restates the policy's interval for Ramen's `DRPolicy` matching. No secrets parameter is needed: the driver's credentials come from `secret.json`, as for every other RPC. The chart ships no default class. Classes are the user's to author, matching the `VolumeGroupSnapshotClass` decision. + +One class per (policy, cadence) is the authoring model: a `VolumeReplicationClass` names exactly one policy, and a volume needing a different cadence follows a different policy under a different class. Per-volume interval overrides are not provided. + +### 7.2 peerClasses: Ramen's own mechanism, out of this design's scope + +`peerClasses` is Ramen's `DRPolicy` computation, not this design's: Ramen's hub-side controller pairs each managed cluster's `StorageClass`/`VolumeReplicationClass` objects itself, through the OCM hub-spoke visibility it already has, and a `DRPolicy` that cannot find a valid pairing already reports that failure on its own. Building any verification of that pairing into this operator -- a preflight, an admission check, or otherwise -- belongs to the future design that actually wires Ramen/OCM into this operator, not here: this design's own scope stops at §7.1, authoring one `VolumeReplicationClass` per policy, which is sufficient for the Phase 1/2 adapter to work whether or not Ramen, OCM, or peerClasses are ever in the picture. + +What Ramen's contract requires of a pairing, for whoever writes that future design: + +**Contract (Ramen requires this):** + +- **The `StorageClass` name exists on both clusters.** A failover restores the protected PVC with its original `spec.storageClassName`, a by-name reference, so an equivalent class of that exact name must exist on the peer. The peerClasses computation also joins classes across the two clusters by `StorageClass` name. +- **The Ramen identity labels.** Each cluster's `StorageClass` carries `ramendr.openshift.io/storageid` (differing per cluster, since the backends differ), and the two `VolumeReplicationClass` objects representing one relationship carry an equal `ramendr.openshift.io/replicationid`. The `VolumeReplicationClass` is selected per cluster by `spec.replicationClassSelector` labels plus `provisioner` and a `schedulingInterval` equal to the `DRPolicy`'s, never by name. + +**Convention (this design's own authoring choice, independent of Ramen):** + +- **Same `StorageClass` parameters on both clusters** apart from `cluster_id` (necessarily) and pool when pools differ. The replication target's pool mapping already handles the pool difference at shipping time. +- **Same `VolumeSnapshotClass` and `VolumeReplicationClass` names on both clusters**, each side's `replicationPolicy` naming that cluster's policy toward its peer. Ramen does not require the names to match, but one name per relationship is what keeps a fleet legible. + +--- + +## 8. Coexistence with the Legacy Replication Kinds + +Two control paths now reach the same backend relationship: the annotation-driven slot family, and `VolumeReplication`. They must not fight. + +- **One owner per volume.** A volume is managed either by the annotation path (a `ReplicationSlot` exists for its PVC) or by a `VolumeReplication`, never both. The adapter's `EnableVolumeReplication` refuses (`FAILED_PRECONDITION`) whenever a slot manages the volume, even when the slot's policy and the class's agree: ownership is the conflict, not the policy value, because two owners turn every ownership flap into a detach and re-attach, and a re-attach is a full re-sync on the backend. The `PVCAnnotationWatcher` is taught to skip PVCs that have a `VolumeReplication` (one informer lookup), so the annotation cannot re-attach behind the adapter's back. Moving a volume between the two paths is deliberate: remove the annotation (the slot detaches), then create the `VolumeReplication`. +- **`ReplicationOps` stays imperative and internal.** Bulk failover of a whole policy or target remains a `ReplicationOps` concern until `VolumeGroupReplication` lands. Nothing in this design calls it, and it does not touch csi-addons-managed volumes because the mutual-exclusion rule above keeps the sets disjoint. +- **The slot's status problem is fixed as a side effect.** The typed status read (P0-1) gives the slot controller a truthful `lastReplicatedAt` source, whether or not the adapter is in use. + +The consolidation direction (§13) is that the annotation path becomes a compatibility layer and the four kinds retire once Ramen-driven replication is proven, exactly as the CRD redesign anticipates. + +--- + +## 9. Backend API Requirements + +Every endpoint is scoped as today: volume-scoped under `/api/v2/clusters/{c}/storage-pools/{p}/volumes/{v}`, cluster-scoped under `/api/v2/clusters/{c}/replication`. + +| Method | Endpoint | Notes | +|--------|---------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `PUT` | `.../volumes/{v}` (`replication_policy_id`) | Existing attach and detach. P0-2 makes both idempotent: same-policy attach and non-attached detach return success. | +| `GET` | `.../volumes/{v}/replication/status` | **New (P0-1).** The typed steady-state status of §6.1. Never 404s for a volume that exists; `state: not_replicating, role: none` is a valid answer. | +| `POST` | `.../volumes/{v}/replication/failover` (+ planned gate) | Existing; the one promote, both forms. Gains a planned form that is refused unless the peer holds every acknowledged write (a completed demote). Idempotent by NQN probe. | +| `POST` | `.../volumes/{v}/replication/demote` | **New (P0-3).** Quiesce, final ship, confirm on peer, fence. Idempotent: demoting a demoted volume returns success. | +| `POST` | `.../volumes/{v}/replication/failback` | Existing. Resync (direction reversal, delta-seeded). | +| `POST` | `.../volumes/{v}/replication/commit` | Existing, unchanged, and NOT part of this contract: it stays behind the legacy `ReplicationOps` migration path only (§13). | +| `GET` | `.../replication/relationships/{lvol}` | Existing, unchanged. Cutover records only; the node redirect depends on its survive-deletion semantics. | +| `POST` | `.../volumes/{v}/replication/cutover-proceed` | Existing, unchanged, legacy path only: the adapter never reaches it, because the commit cutover is off this contract. | + +The unused backend verbs the operator never calls (`start`, `stop`, `trigger`, `tasks`) are unaffected, and `start` and `stop` remain the policy-less legacy path. + +--- + +## 10. Failure Modes and Fallback + +| Failure | Detection | Behavior | +|--------------------------------------------------------------------------|---------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| Enable against a volume attached to a different policy | Adapter compares the status read's policy | `FAILED_PRECONDITION`; the message names both policies. Never silently re-attaches, because that is a full re-sync. | +| Disable during a cutover | Backend 409 | `ABORTED`, retryable. Ramen re-drives; the detach succeeds after the cutover settles. | +| Planned promote without a completed demote (source live or peer lagging) | The planned gate on the failover endpoint refuses | `FAILED_PRECONDITION` naming the un-flushed tail. The caller demotes (or resyncs) first, and `Resyncing` reports progress. | +| Forced promote when the source is alive | Backend fences the source data path as part of failover | Split-brain is prevented structurally: the source is retired (ANA inaccessible, subsystem removed) before the target serves. | +| Backend unreachable | HTTP error from the shared client | The RPC returns `UNAVAILABLE`; the controller-manager retries. Conditions keep their last-observed values, so a blip does not flap `Degraded`. | +| Demote cannot complete its flush (peer slow or unreachable) | The demote's confirm step times out | `Completed` stays `False`, and the volume is not left fenced without a decision: whether the backend rolls back to serving primary or holds quiesced is Open Question 1. | +| Status read missing (pre-P0-1 backend) | 404 from the status endpoint | The csi-addons capability is not advertised, so the controller-manager never drives this driver. The adapter ships dark until the backend is current. | +| Both an annotation slot and a VolumeReplication claim a volume | Adapter check plus watcher skip (§8) | The first owner wins; the second surfaces `FAILED_PRECONDITION` (adapter) or a skip event (watcher). Never two writers to one relationship. | + +--- + +## 11. Observability + +**Baseline.** The replication engine's telemetry today is log lines (the `XFER-TIMING` phases) and the `rep_info` dict. Nothing is exported as metrics, and the `VolumeReplication` surface adds only `lastSyncTime`. The figures an operator actually watches live here. + +### Kubernetes Events + +The kubernetes-csi-addons controller-manager owns events on `VolumeReplication` (promote, demote, and resync outcomes), and this design adds none there. This design defines no events of its own on `ReplicationPair` either: the peerClasses preflight that would have emitted them is Ramen's own concern, out of scope here (§7.2). + +### Prometheus Metrics (Implemented) + +Exported by the control plane's existing v2 `Collector`-pattern exporter (`simplyblock_web/api/v2/metrics.py`), rebuilt from FDB on every scrape like every other series in that file. Labeled `lvol`/`lvol_name`/`pvc_name`/`pool`/`pool_name` (not the bare `volume` this section originally specified: the exporter's existing lvol-scoped metrics already use `lvol`/`lvol_name`, and joining the new series against them needs a shared label name) plus `policy`/`policy_name`/`peer_cluster`: + +| Metric | Description | +|---------------------------------------------|---------------------------------------------------------------------------------| +| `simplyblock_replication_lag_seconds` | Now minus the newest fully replicated snapshot's creation time. | +| `simplyblock_replication_backlog_bytes` | `outstanding_bytes`: the queued-but-unshipped snapshot sizes. | +| `simplyblock_replication_last_sync_seconds` | Duration of the last shipping cycle. | +| `simplyblock_replication_last_sync_bytes` | Size of the last shipped snapshot. | +| `simplyblock_replication_rpo_violation` | 1 while `lag_seconds` exceeds the policy's `rpo_target_seconds` (P0-4), else 0. | +| `simplyblock_replication_degraded` | 1 while the status read's state is `degraded` or `error`. | + +`simplyblock_replication_rpo_violation` is the alert: it is the declared objective against the measured lag, which no timestamp alone can express. `simplyblock_replication_backlog_bytes` is the second load-bearing figure, because it is the input to any honest RTO estimate and the number that distinguishes "slow cycle" from "falling behind." One caveat is recorded rather than hidden: `outstanding_bytes` measures queued snapshot sizes, not dirty bytes written since the last snapshot, so intra-interval writes are invisible to it. A true dirty-delta figure needs storage-plane support and is future work. + +Values are computed by `lvol_controller.get_replication_info_bulk`, a bulk-friendly sibling of `get_replication_info` (P0-1) that reads a whole cluster's job tasks, policies, and targets once rather than once per replicating volume -- `get_replication_info` itself makes several effectively global FDB scans internally (repeated per call), which is fine for a single-volume status read but not for a per-scrape loop across a fleet. `simplyblock_replication_last_sync_seconds`/`_bytes` are omitted per volume when no cycle has completed yet, and `rpo_violation` is omitted entirely for a volume whose policy declares no `rpo_target_seconds`, matching the exporter's existing "omit rather than fabricate" convention (`_health_family`) -- a 0 there would misread as "in compliance" absent a declared target. + +--- + +## 12. Testing Strategy + +Full scenario matrix and coverage status: [`tests/test-plan-csi-addons-replication.md`](../tests/test-plan-csi-addons-replication.md) + +- **Unit (driver):** each verb against a mock control plane: the idempotency table (repeat enable, repeat disable, repeat promote), the refusal paths (different-policy enable, lagging planned promote, disable during cutover), the condition derivation from every status-read state, and handle parsing failures. +- **Unit (operator):** the `PVCAnnotationWatcher` skip when a `VolumeReplication` exists. +- **Unit (driver, §14):** the VolumeGroup service verbs against a mock control plane (`CreateVolumeGroup` resolves the label-formed group idempotently; `ModifyVolumeGroupMembership` refused off-placement; `DeleteVolumeGroup` leaves members), and the group-handle branch of each Replication verb routing to the group endpoints. +- **Unit (backend, §14.5):** the group-replication engine in `sbcli` (group failover clones the last group generation for every member atomically; group demote quiesces all then ships one final group snapshot; group failback reverses direction), tests-first. +- **Integration:** the csi-addons sidecar and controller-manager against the driver with a mock backend under envtest or kind: a `VolumeReplication` flipped `primary` to `secondary` and back walks the verbs in order and lands the conditions. +- **E2E (two live clusters):** the Ramen-shaped lifecycle without Ramen: enable on the source, write data, and verify `lastSyncTime` advances; forced promote on the DR side, verifying the clone serves with the source fenced; and resync back with a planned swap (demote then promote), verifying zero loss with a hashed writer. Then the same driven by an actual Ramen VRG in async mode, which is Phase 2's acceptance gate. + +The risk concentrates in the demote verb's flush-confirmation (the lossless-swap guarantee), in idempotency under Ramen's aggressive re-drive, and in the coexistence rules of §8. Those must not be cut. + +--- + +## 13. Migration Strategy + +Three replication control surfaces exist today: the operator's kinds, the stale CSI-shipped CRDs (`replications.simplyblock.com`, invalid as written, plus the orphaned `SnapshotReplication`), and the backend-driven node redirect. The target is one: csi-addons, with the node redirect unchanged beneath it. + +1. **Phase 1 and 2 (this design):** the adapter ships alongside the legacy kinds. The mutual-exclusion rule (§8) keeps the two paths disjoint per volume. The shared `atlas-lib/controlplane/replication.go` client lands, and the driver uses it from day one. +2. **Opportunistic:** the operator's replication reconcilers move their inline HTTP calls onto the shared client, and the slot controller sources `lastReplicatedAt` from the typed status read. No behavior change, one client. +3. **After Ramen validation:** the annotation path is declared a compatibility layer, and new volumes are protected through `VolumeReplication`. The stale CSI-shipped CRDs are deleted (they were never installable), and `SnapshotReplication` is retired from the charts. +4. **The redesign's replication chapter:** whether `ReplicationPair` and `ReplicationPolicy` survive as the backend-policy authoring surface (classes need policies to name) or are re-cut is decided there, not here. This design only requires that a policy exists per cluster pair, however it is authored. +5. **Removing the offloaded group reconciler.** An earlier revision implemented `VolumeGroupReplication` as an operator-owned, `external: true` path: `VolumeGroupReplicationReconciler` (`operator/internal/controller/volumegroupreplication_controller.go`) and `VolumeGroupReplicationValidator` (`operator/internal/webhook/volumegroupreplication_validator.go`), which fanned a group out to per-member `VolumeReplication` objects. §14 supersedes that with the stock, `external: false` driver path, so both files, their unit tests, the `main.go` wiring, and the webhook registration are removed. Group protection then rides the driver's VolumeGroup service and the group-replication endpoints (P0-6/P0-7), with no operator code on the path. + +--- + +## 14. VolumeGroupReplication + +`VolumeGroupReplication` protects a consistency group as one replicated unit. It is the group-level sibling of the `VolumeReplication` surface (§5), in the same csi-addons `replication.storage.openshift.io` API group, and it is driven entirely by the stock kubernetes-csi-addons machinery: the generic controller-manager forms a backend volume group through the driver's csi-addons **VolumeGroup** service, then replicates that group as a single unit through the Replication service (§5), addressing it by one group handle. No operator reconciler and no vendor-specific controller take part. This is the gap analysis's Phase 2 group-replication item, and it is **Planned** (§13 records the earlier attempt this supersedes). + +Ramen's `VolumeReplicationGroup` creates one `VolumeGroupReplication` when a protected application's PVCs share a `StorageClass` carrying `ramendr.openshift.io/groupreplicationid` (the group-level sibling of the per-volume `replicationid` of §7.1). The end-to-end Ramen validation is `design-ramen-integration.md` §6, whose test plan carries its E2E scenario as M-05. + +### 14.1 Why the driver, not the operator + +csi-addons's `VolumeGroupReplication` controller (v0.15.0) branches on `spec.external`: + +- **`external: false` (this design's path).** The stock controller creates a `VolumeGroupReplicationContent`, calls the driver's VolumeGroup service to group the member volumes and obtain a group handle, then creates **one** `VolumeReplication` addressed by that group handle, which the ordinary `VolumeReplication` controller drives through the driver's Replication service (§5). The whole group is one replicated object, exactly as a single volume is. +- **`external: true`.** The stock controller skips the object entirely, handing it to a vendor controller. + +This design takes the `external: false` path: the group is a first-class backend object the driver creates, and the standard machinery replicates it. The alternative, an operator `VolumeGroupReplicationReconciler` owning `external: true` objects and fanning them out to per-member `VolumeReplication` objects, is not used. It re-implements in the operator what the csi-addons controller already does, and it drives the group through per-member calls rather than one group operation, losing the crash-consistent, all-at-one-point failover a consistency group exists to give. An earlier revision built that reconciler and its admission webhook; §13 removes them. + +### 14.2 Ramen configuration + +Ramen sets `spec.external` from whether the member PVCs' `StorageClass` carries `ramendr.openshift.io/offloaded`. For this driver-driven design the `StorageClass` carries `groupreplicationid` but **not** `offloaded`, so Ramen creates the object with `external: false` and a real `volumeReplicationClassName`, and the stock controller reconciles it. + +```yaml +apiVersion: storage.k8s.io/v1 +kind: StorageClass +metadata: + name: simplyblock-group-sc + labels: + ramendr.openshift.io/storageid: # differs per cluster + ramendr.openshift.io/replicationid: # names the per-volume relationship + ramendr.openshift.io/groupreplicationid: # triggers group replication; no `offloaded` label +provisioner: csi.simplyblock.io +parameters: { cluster_id: , pool_name: } # plus the usual StorageClass parameters +--- +apiVersion: replication.storage.openshift.io/v1alpha1 +kind: VolumeGroupReplicationClass +metadata: + name: simplyblock-group-async-5m + labels: + ramendr.openshift.io/groupreplicationid: # must equal the StorageClass's + ramendr.openshift.io/storageid: # must equal the StorageClass's +spec: + provisioner: csi.simplyblock.io + parameters: + schedulingInterval: "5m" # must equal the DRPolicy's +``` + +Member PVCs carry `storage.simplyblock.io/consistency-group` (backend grouping and group snapshots, `design-consistency-groups.md`) and the `pvcSelector` label the VRG matches. Ramen additionally reads its own `ramendr.openshift.io/consistency-group` label to build the group's selector, so a Ramen-protected member carries both consistency-group labels with the same value. + +### 14.3 The VolumeGroup service (driver) + +The plugin serves the csi-addons **VolumeGroup** (GroupController) service beside csi-addons Identity and Replication (§4.2), advertising the capability through csi-addons Identity so the stock controller-manager dials it. Its verbs map onto the backend consistency group that already exists as a first-class object (`design-consistency-groups.md`): + +| Verb | Backend mapping | Semantics | +|-------------------------------|------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `CreateVolumeGroup` | Resolve (or ensure) the consistency group over the given volume handles; return its id as the group handle | Idempotent over the group the `storage.simplyblock.io/consistency-group` label already formed at provisioning: the members are co-placed, so the call returns the existing group's id. | +| `ModifyVolumeGroupMembership` | Add or remove members | Refused when placement forbids it (a volume not co-placed on the group's node/LVS); dynamic membership is `design-consistency-groups.md`'s own Phase 4. | +| `DeleteVolumeGroup` | Dissolve the group | The member volumes survive; only the grouping is removed. Idempotent. | + +The label remains the single source of truth for membership (`design-consistency-groups.md`): `CreateVolumeGroup` does not introduce a competing grouping, it addresses the same backend group the label formed. One backend consistency group is therefore reachable three ways: by label at provisioning, by label for `VolumeGroupSnapshot`, and by handle for group replication. + +### 14.4 Group-wide replication (driver and backend) + +Once grouped, csi-addons addresses the whole group by its handle through the **same** Replication verbs of §5. Each verb operates on the group as one unit and maps to a new backend group-replication endpoint (§9, and P0-6/P0-7): + +| Verb (on the group handle) | Backend group operation | +|---------------------------------|---------------------------------------------------------------------------------------------------------------------------| +| `EnableVolumeReplication` | Attach the consistency group to a group replication policy (cadence and target for the whole group). | +| `PromoteVolume` (force/planned) | Group failover: clone the last fully replicated group-snapshot generation for **every** member on the target, atomically. | +| `DemoteVolume` | Group demote: quiesce every member, take and ship one final group snapshot, confirm it landed, then fence every member. | +| `ResyncVolume` | Group failback: reverse the shipping direction for the whole group. | +| `GetVolumeReplicationInfo` | Group `lastSyncTime`: the creation time of the newest fully replicated group-snapshot generation. | +| `DisableVolumeReplication` | Detach the group from its group replication policy. | + +The recovery generation is a **group snapshot** (`bdev_lvol_snapshot_group`, `design-consistency-groups.md` P0-1): one frozen snapshot of every member at a single point, shipped atomically, so a group failover always lands every member at one crash-consistent point. Group promote, demote, and failback are all-or-nothing across members (Open Question 6). The driver detects a group handle (versus a per-volume handle) and routes to these endpoints; a per-volume handle still takes the §5 path unchanged. + +### 14.5 Backend API + +The group-replication endpoints, cluster-scoped like the per-volume ones under `/api/v2/clusters/{c}`, are new backend work (P0-6, P0-7): + +| Method | Endpoint | Notes | +|--------|---------------------------------------------------------|-------------------------------------------------------------------------------------------------------| +| `PUT` | `.../consistency-groups/{id}` (`replication_policy_id`) | Attach and detach the group to a group replication policy. Idempotent (P0-2's group twin). | +| `POST` | `.../consistency-groups/{id}/replication/failover` | Group promote, planned and forced. Clones the last group generation for every member atomically. | +| `POST` | `.../consistency-groups/{id}/replication/demote` | Group demote: quiesce all, ship one final group snapshot, confirm, fence all. | +| `POST` | `.../consistency-groups/{id}/replication/failback` | Group resync (direction reversal). | +| `GET` | `.../consistency-groups/{id}/replication/status` | Typed group status: role, group `lastSyncTime`, lag, per-member health rollup. | +| `GET` | `.../consistency-groups/` (by name), `/{id}/members` | Existing reads (`design-consistency-groups.md`); the VolumeGroup service resolves the handle by them. | + +### 14.6 Observability + +The stock kubernetes-csi-addons controller-manager owns Kubernetes events on `VolumeGroupReplication` (the group promote, demote, and resync outcomes), the same way it owns them for `VolumeReplication`; this design adds none there. The per-member replication metrics of §11 cover each group member; a group rollup (`simplyblock_replication_group_lag_seconds`, keyed by `consistency_group`) is derivable from the group status read of §14.5 and is exported by the same v2 exporter, its worst-member lag being the figure a group RPO alert watches. + +### 14.7 Scope + +The base case is one VRG, one storage vendor's PVCs, one `VolumeGroupReplication`. RamenDR's newer multi-VRG "Global VGR" consensus, spanning a replication group across several applications' VRGs, is out of scope (Open Question 4). `VolumeGroupReplicationContent` is created and owned by the stock controller-manager on the `external: false` path (it carries the group handle `CreateVolumeGroup` returns); this design writes nothing to it directly. + +--- + +## 15. Open Questions + +| # | Question | Owner | +|-----|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------| +| 1 | **Demote semantics for the application.** The P0-3 demote fences the volume (ANA inaccessible) after the final flush, and with convergence folded into the verb it is now the only place a planned swap can stall. This is also the one verb the planned promote's lossless guarantee entirely depends on (§5.2): a planned promote is refused unless a completed demote already fenced the source and confirmed the final delta landed, so an unresolved failure mode here is an unresolved gap in the whole "zero loss" claim. Ramen relocation unmounts the workload first, so the fence is ordinarily unopposed, but that is Ramen's choreography, not a guarantee the driver can rely on: a stuck termination, a stale mount that never released, or a demote invoked outside Ramen's normal flow can all leave writes still arriving when quiesce fires. Confirm the verb's behavior when writes are still in flight at quiesce (block versus fail), whether the converge phase has its own budget separate from the quiesced flush, and whether a timeout in either phase must abort back to serving primary or leave the volume fenced with no automatic recovery. | Backend team | +| 2 | **Per-volume policy granularity.** A `VolumeReplicationClass` names one policy, and today one policy implies one target and cadence for all its volumes. Confirm one class per (policy, cadence) is an acceptable authoring model for Ramen's `replicationClassSelector`, or whether per-volume interval overrides are needed. | Operator / Backend team | +| 3 | ~~**Avoiding the clone on day-one protection.**~~ **Resolved:** `POST .../replication/failover?planned=true`'s no-demote branch now checks `lvol_controller.replication_source_online` (the source's own storage-node status) before falling through to `FAILED_PRECONDITION` -- an online source is a no-op (§5.2), so a healthy volume's first-ever `PromoteVolume` no longer materializes a clone. The remaining residual: a source that dies within the last health-check interval still briefly reads online, so one reconcile can treat a genuine disaster as a no-op before the node's status catches up and the controller retries -- bounded by the health-check detection window, not open-ended. | Backend team | +| 4 | **Is Global VGR needed (§14.7).** §14 covers the base case: one VRG, one vendor's PVCs, one group. Confirm whether any planned simplyblock deployment spans a replication group across more than one application's VRG before treating RamenDR's multi-VRG "Global VGR" consensus as work this design should also specify. | Operator team | +| 5 | **`VolumeGroupReplicationContent` on the `external: false` path (§14.7).** The stock controller-manager creates and owns the `Content`, populating it with the group handle `CreateVolumeGroup` returns. Confirm the driver need only return a stable handle and never reads or writes the `Content` itself, across the csi-addons versions in scope. | Operator team | +| 6 | **Atomicity of group failover, demote, and failback (§14.4).** The design states all-or-nothing across members. Confirm the backend can guarantee it, and define the behavior when one member cannot complete: abort the whole group, or serve a partial group and report it. | Backend team | +| 7 | **The group replication policy (§14.4, P0-7).** `design-consistency-groups.md` removed the replication policy from the consistency group; group replication needs a cadence and target back. Confirm a group replication policy attached to the CG (the group twin of `ReplicationPolicy`) is the model, versus per-member policies coordinated at the group level. | Backend team | +| 8 | **Grouping already-provisioned, non-co-placed volumes (§14.3).** `CreateVolumeGroup` is idempotent over the label-formed, co-placed group. Confirm whether `ModifyVolumeGroupMembership` must support adding a volume that is not already co-placed (a data move), or whether that stays refused as `design-consistency-groups.md`'s Phase 4 dynamic-membership work. | Backend team | diff --git a/operator/docs/designs/design-ramen-integration.md b/operator/docs/designs/design-ramen-integration.md new file mode 100644 index 000000000..851815218 --- /dev/null +++ b/operator/docs/designs/design-ramen-integration.md @@ -0,0 +1,182 @@ +# Design Document: Ramen Integration + +**Status:** Draft (§3 confirms peerClasses needs no operator code; §4 defers `VolumeGroupReplication` to `design-csi-addons-replication.md` §14, driver-and-backend, Planned; E2E validation pending) +**Author:** Israel Geoffrey (geoffrey1330) +**Date:** 2026-09-21 +**Test Plan:** [`tests/test-plan-ramen-integration.md`](../tests/test-plan-ramen-integration.md) + +--- + +## Phase 0: External Prerequisites + +| # | Prerequisite | Kind | Blocks | Status | +|------|--------------------------------------------------------------------------------------------------------------------|-----------|----------------------------------------------|-----------------------------------------------------------------------------------------------------------| +| P0-1 | A live OCM hub with at least two registered managed clusters | Ecosystem | The E2E validation (§6) | Unknown | +| P0-2 | Ramen installed on the hub and on each managed cluster, with a `DRPolicy` naming both clusters | Ecosystem | The E2E validation (§6) | Unknown | +| P0-3 | SiteMap, or a hand-authored `DRPlacementControl` standing in for it, driving the `DRPolicy` | Ecosystem | The E2E validation (§6) | Not shipped. SiteMap is an external document and system, and storage is explicitly outside its own scope. | +| P0-4 | `csi-addons/spec` at a version whose `GetVolumeReplicationInfoResponse` carries `lastSyncBytes`/`lastSyncDuration` | Ecosystem | Full Appendix A.3 `GetVolumeReplicationInfo` | Not shipped: pinned at v0.2.0 today, which has neither field. | + +Without P0-1 through P0-3 nothing in §6 can run, because the validation is E2E-only, live-cluster work that no mock or `envtest` substitutes for. §4's `VolumeGroupReplication` work is a driver-and-backend feature specified in `design-csi-addons-replication.md` §14 and does not depend on the OCM hub for its own unit and backend tests, only for the E2E group scenario (M-05). P0-4's absence is narrower: it leaves `GetVolumeReplicationInfo` reporting only `lastSyncTime`, never cycle size or duration, but Ramen's own `PeerReady` gate (§5.1) does not read either field, so P0-4 does not block the validation itself. + +--- + +## Table of Contents + +1. [Background](#1-background) +2. [Goals and Non-Goals](#2-goals-and-non-goals) +3. [peerClasses: Ramen's Own Mechanism](#3-peerclasses-ramens-own-mechanism) +4. [VolumeGroupReplication](#4-volumegroupreplication) +5. [The Contract, Confirmed](#5-the-contract-confirmed) +6. [E2E Validation Plan](#6-e2e-validation-plan) +7. [Testing Strategy](#7-testing-strategy) +8. [Open Questions](#8-open-questions) + +--- + +## Overview + +`design-csi-addons-replication.md` builds the storage-level adapter Ramen's per-volume DR contract requires, and validates it end to end "without Ramen" (its own §12): real backend, real csi-addons machinery, but a hand-driven `VolumeReplication` object rather than a real Ramen reconcile loop. That document's own §7.2 named `peerClasses` verification Ramen's own hub-side mechanism, out of its scope, and deferred `VolumeGroupReplication` as "the next design," strictly per volume itself. `design-consistency-groups.md` deferred `VolumeGroupReplication` too, as future work independent of any replication policy. This document records Ramen's part in the two pieces that needed new design: peerClasses (§3) and `VolumeGroupReplication` (§4), the latter now specified in full in `design-csi-addons-replication.md` §14. §3 confirms, rather than reopens, `design-csi-addons-replication.md` §7.2's original position on peerClasses, after this document's own history of first building a same-cluster preflight for it and then removing that preflight once it became clear a same-cluster check cannot verify a cross-cluster pairing. §5 through §6 confirm the rest of the per-volume contract against what already shipped and specify the E2E validation that closes `design-csi-addons-replication.md` §12's outstanding acceptance gate (E-06, E-07). + +--- + +## 1. Background + +The gap analysis's own headline finding (§2) was that simplyblock's DR machinery was "a complete, parallel, out-of-band system that implements none of the interfaces Ramen drives." Appendix A answered that with a concise spec: the six csi-addons `Replication` gRPC verbs, the one info query, and the `VolumeReplication.status` conditions Ramen actually reads (`Completed`, `Degraded`, `Resyncing`), aggregated by Ramen's own VRG into `DataProtected` and by the DRPC into `PeerReady`, the boolean that gates `Relocate` and failback. Appendix B specified the RPO/RTO figures Ramen cannot carry natively, as metrics. + +`design-csi-addons-replication.md`'s three phases implement that spec. This document does not repeat what it built. §5 below cites, section by section, where each Appendix A and Appendix B item now lives in the shipped code. + +The other half of Appendix A's own premise is that Ramen's hub, not this operator, computes `peerClasses` and drives the VRG, through OCM's hub-spoke visibility into every managed cluster. That hub, and the OCM/SiteMap layer above it, is external to this repository (confirmed this session: no `ManagedCluster`, `DRPolicy`, or `VolumeReplicationGroup` reference exists anywhere in this codebase outside design-doc prose, and no mechanism for this operator to reach a peer cluster's Kubernetes API exists or is needed. `design-management-hub.md`, this repo's own hub design, specifies a real hub component of its own (the `fleet-manager`, reading member clusters through OCM's `ManagedClusterView`), but it is a separate fleet-config-distribution concern, and nothing in it builds or is intended to build Ramen's own pairing). Ramen's hub already reports when it cannot find a valid `StorageClass`/`VolumeReplicationClass` pairing across two clusters, once it looks, and only the hub's own cross-cluster visibility can make that comparison at all: a same-cluster read sees whether this cluster's own label is present, never whether its value agrees with the peer's, which is the only question a pairing check actually needs answered. §3 explains why this operator's own history of trying to close that gap locally settled on not closing it. + +The gap analysis's own §6 named a second piece of genuinely new work, in its Phase 2: `VolumeGroupReplication`, "on top of the CG primitive," so that the VRG async group path can protect and fail over a multi-volume app at one point rather than only snapshot it. `design-consistency-groups.md` built the CG primitive that gap analysis cites (the `storage.simplyblock.io/consistency-group` label, `VolumeGroupSnapshot`, the `GroupController`) but explicitly left group replication for later, independent of any policy, exactly so this document could attach it without reshaping the group. §4 records Ramen's part in that attachment; the group surface itself is specified as driver-and-backend work in `design-csi-addons-replication.md` §14 (Planned). Everything else Appendix A and Appendix B specify is confirmed, in §5, against code `design-csi-addons-replication.md` already shipped. + +--- + +## 2. Goals and Non-Goals + +### Goals + +- Confirm that Ramen's `peerClasses` pairing needs no operator-side code, settling the question this document's own earlier attempt at a same-cluster preflight left open (§3). +- Confirm Ramen drives `VolumeGroupReplication` through the stock csi-addons machinery (the `external: false` path) against the driver-and-backend group surface `design-csi-addons-replication.md` §14 specifies, satisfying Appendix A.4's group-readiness query and the gap analysis's own Phase 2 ask (§4). +- Confirm, against the actual shipped code, that every condition, verb, and status query Appendix A specifies is satisfied, or state precisely which is not and why (§5). +- Confirm, against the actual shipped code, which Appendix B metrics are delivered, which are derivable from what already exists, and which remain future work (§5.4). +- Specify a live-cluster validation that closes `design-csi-addons-replication.md` §12's outstanding acceptance gate: a real Ramen VRG, on a real OCM-registered pair of clusters, driving the adapter through protect, planned relocate, and unplanned failover (§6). + +### Non-Goals + +- **Building any hub, OCM, or cross-cluster Kubernetes access mechanism.** That is Ramen's and OCM's job, external to this operator, confirmed in §1. Nothing in §6's validation plan asks this operator to reach a peer cluster's API server: every step drives objects on the cluster where the workload currently runs, exactly as `design-csi-addons-replication.md`'s own architecture already assumes. +- **A same-cluster `peerClasses` preflight, in any form.** Tried once, in this document's own history, and removed (§3): a same-cluster read can confirm a label is present, never that its value agrees with the peer's, and a pairing check that cannot verify agreement is not a pairing check. That comparison needs the hub's own visibility into both clusters and is Ramen's job alone, once a `DRPolicy` exists. +- **Global VGR.** RamenDR's newer multi-VRG consensus feature for a replication group spanning several applications is out of scope. §4 covers the base case: one VRG, one storage vendor's PVCs, one `VolumeGroupReplication` (§8 Open Question 4). +- **A new group gRPC verb on the driver's Replication service.** The same `PromoteVolume`/`DemoteVolume`/`ResyncVolume` verbs act on a group handle rather than a single volume (`design-csi-addons-replication.md` §14.4). The new driver surface is the csi-addons VolumeGroup service (`CreateVolumeGroup` and siblings), not a fourth Replication verb. +- **SiteMap.** A separate external document and system. Where §6's topology needs a `DRPlacementControl` and SiteMap is not available to author one, a hand-authored stand-in is explicitly permitted (P0-3). +- **`bytesBehind`'s remaining Appendix B siblings** (throughput, RTO estimation, backup RPO/RTO). §5.4 accounts for each, and none blocks the validation this document specifies. +- **Widening `csi-addons/spec` past v0.2.0.** P0-4 is recorded as a prerequisite, not solved here: it is an upstream dependency version, not something this repository's own code can add a field to. + +--- + +## 3. peerClasses: Ramen's Own Mechanism + +Ramen's hub pairs a `StorageClass` and a `VolumeReplicationClass` across two managed clusters into a `peerClasses` entry on its `DRPolicy`, through OCM's hub-spoke visibility into both clusters at once. That visibility is what makes the pairing check meaningful, and it is exactly what a reconciler running inside one managed cluster does not have and cannot substitute for: the only question worth asking about a pairing is whether the two clusters' objects *agree*, and agreement is a cross-cluster comparison, not a same-cluster one. + +This document tried the same-cluster version anyway, once. A `ReplicationPairReconciler` preflight was specified and implemented, confirming this cluster's own `VolumeReplicationClass` carried the `ramendr.openshift.io/replicationid` label and a `schedulingInterval`, and emitting `PeerClassesVerified`/`PeerClassesMismatch` accordingly. That check has a real, narrow use (it catches an operator who forgot the label entirely), but it is not peerClasses verification: a cluster whose label carries a value that does not match its peer's passes identically to one whose value is correct, because presence, not agreement, is everything a same-cluster read can check. Reporting `PeerClassesVerified` under that name risked being read as a stronger guarantee than it delivered, so it has been removed. Nothing in this repository performs this check today, which is the position `design-csi-addons-replication.md` §7.2 already took before this document first tried to revisit it. + +The exact contract Ramen's pairing requires of the two clusters' objects (the identity labels, the `StorageClass` name-matching, and the authoring convention this repository follows beyond what Ramen strictly requires) is recorded in `design-csi-addons-replication.md` §7.2, unchanged by this document. `VolumeGroupReplication` (§4) is unaffected by any of this: it needs no cross-cluster comparison of its own, only this cluster's own consistency-group membership. + +--- + +## 4. VolumeGroupReplication + +The gap analysis's Phase 2 named this the piece consistency groups still owed: "Add `VolumeGroupReplication` (csi-addons) on top of the CG primitive so the VRG async group path can protect and fail over multi-volume apps at one point, not just snapshot them." It is now specified in full in `design-csi-addons-replication.md` §14, as a **driver-and-backend** feature driven by the stock kubernetes-csi-addons machinery (Phase 4, Planned there). This section records only what Ramen contributes to that path and where this document validates it. + +**Ramen's part.** Ramen's VRG (`RamenDR/ramen`'s `vrg_volgrouprep.go`, confirmed present at `v0.1.0-rc1`) creates one `VolumeGroupReplication` when the member PVCs' `StorageClass` carries `ramendr.openshift.io/groupreplicationid` (the group-level sibling of the per-volume `replicationid` of `design-csi-addons-replication.md` §7.1), and sets `spec.external` from whether that same `StorageClass` carries `ramendr.openshift.io/offloaded`. simplyblock takes the `external: false` path (`design-csi-addons-replication.md` §14.1): the `StorageClass` carries `groupreplicationid` but **not** `offloaded`, so the stock controller-manager owns the object, groups the members through the driver's csi-addons VolumeGroup service, and replicates the whole group as one unit through the driver's Replication service, addressed by a group handle. No operator code is on the path. + +**No new group verb.** Group promote, demote, and resync are the same three verbs `design-csi-addons-replication.md` §5 already ships, applied to the group handle rather than a single volume (§14.4). The recovery generation is a group snapshot, so a group failover lands every member at one crash-consistent point. + +**Global VGR out of scope.** RamenDR's newer multi-VRG consensus, spanning a replication group across several applications' VRGs, is not covered here; §8 Open Question 4. + +**An earlier operator-owned revision is removed.** A previous version of this section specified a `VolumeGroupReplicationReconciler` and a `VolumeGroupReplicationValidator` in this operator, owning `external: true` objects and fanning them out to per-member `VolumeReplication` objects. `design-csi-addons-replication.md` §13 removes them in favor of the driver-driven path above: it re-implemented in the operator what the stock controller already does, and drove the group through per-member calls rather than one group operation. Nothing in this operator handles `VolumeGroupReplication` now. + +--- + +## 5. The Contract, Confirmed + +### 5.1 Conditions (Appendix A.2) + +Appendix A.2 specified `Completed`, `Degraded`, `Resyncing`, with steady-state healthy async as all three settled (`Completed=True, Degraded=False, Resyncing=False`) and lag alone tripping none of them. `design-csi-addons-replication.md` §6.2 implements exactly this mapping: `Completed` from the last requested state change finishing, `Degraded` from the status read's `degraded`/`error` state or `lag_seconds` exceeding `lag_budget_seconds`, `Resyncing` from the status read's own `resyncing` flag. This exceeds Appendix A.2's own "interim" fallback (a staleness heuristic on `lastReplicatedAt`): `Degraded` is sourced from the backend's real failing-task signal, not a timestamp guess. + +Appendix A.3's `PeerReady` (`Completed=True && Degraded=False && Resyncing=False` on the peer) needs no code of this repository's own. It is Ramen's own DRPC-level aggregation of the three conditions above, computed client-side from what §6.2 already reports. + +### 5.2 gRPC verbs (Appendix A.3) + +All six verbs Appendix A.3 specifies are implemented and idempotent, per `design-csi-addons-replication.md` §5: + +| csi-addons verb | Appendix A.3 requirement | Status | +|----------------------------|--------------------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| `EnableVolumeReplication` | Explicit, idempotent attach. `Completed` once the relationship is established. | Implemented (§5.1, P0-2) | +| `DisableVolumeReplication` | Explicit stop/teardown verb (implicit before this design) | Implemented (§5.1, P0-2) | +| `PromoteVolume` (`force`) | Planned (peer alive) vs. unplanned (peer gone), wired through `force` | Implemented (§5.2), including the source-health no-op that keeps a healthy day-one promote from cloning unnecessarily | +| `DemoteVolume` | A new standalone verb: quiesce, final flush, confirm on peer | Implemented (§5.2, P0-3), exactly the "new standalone demote" Appendix A called for | +| `ResyncVolume` | A direction-reversing resync with `Resyncing`→`Completed` progress | Implemented (§5.2), maps onto `failback` | +| `GetVolumeReplicationInfo` | `lastSyncTime`, plus `lastSyncBytes`/`lastSyncDuration` | `lastSyncTime` implemented (§5.1, P0-1). `lastSyncBytes`/`lastSyncDuration` blocked on P0-4: the response type in the pinned `csi-addons/spec` v0.2.0 has no field for either, so the backend's own cycle-size and cycle-duration reads (`get_replication_info`'s `last_cycle_bytes`/`last_cycle_seconds`) have nowhere to land in the RPC response. | + +### 5.3 Status queries (Appendix A.4) + +1. **Per-slot health/role, never 404ing:** implemented (P0-1, `design-csi-addons-replication.md` §6.1). `state: not_replicating, role: none` is the valid answer for an unreplicated volume, closing exactly the 404 gap Appendix A.4.1 named. +2. **The `PeerReady` boolean:** covered by §5.1 above, Ramen's own aggregation, needing no new code. +3. **Group readiness:** out of scope here (§2 Non-Goals), `design-consistency-groups.md`'s future work. + +### 5.4 Observability (Appendix B) + +| Appendix B concept | Metric | Status | +|--------------------------|------------------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| RPO, time gap | `simplyblock_replication_lag_seconds` | Implemented (`design-csi-addons-replication.md` §11) | +| RPO, data gap | `simplyblock_replication_backlog_bytes` | Implemented. This is Appendix B's "one genuinely new backend measurement," `bytesBehind`. `get_replication_info`'s existing `outstanding_bytes` (queued-but-unshipped snapshot sizes) already reports it. | +| Last cycle size/duration | `simplyblock_replication_last_sync_bytes`/`_seconds` | Implemented | +| RPO compliance | `simplyblock_replication_rpo_violation` | Implemented, keyed on `ReplicationPolicy.rpo_target_seconds` | +| Degraded state | `simplyblock_replication_degraded` | Implemented | +| Throughput (smoothed) | (none) | Not implemented as its own series. Derivable via PromQL (`last_sync_bytes / last_sync_seconds`) from what already exists. Non-blocking. | +| RTO (estimated) | (none) | Not implemented. Appendix B itself frames this as "always an estimate," lowest priority of the set. Non-blocking. | +| Backup RPO/RTO (S3) | (none) | Out of scope. `design-csi-addons-replication.md` §2's own Non-Goals exclude the VolSync/S3 backup path entirely, a separate phase of the gap analysis. | + +Nothing in this row set blocks §6's validation: every metric `PeerReady` or a planned/unplanned promote depends on is already implemented. + +--- + +## 6. E2E Validation Plan + +### 6.1 Topology + +Two Kubernetes clusters, each running a `SimplyblockDriver` against its own simplyblock storage cluster, paired exactly as `regression_test/21/test_csi_addons_replication.sh` already sets up (one shared simplyblock control plane, `ReplicationPair`/`ReplicationPolicy` authoring the backend relationship). On top of that pair: an OCM hub (a third cluster, or the hub role colocated on one of the two), both storage clusters registered as `ManagedCluster`s, Ramen's dr-cluster operator installed on each, Ramen's hub operator installed on the hub, and one `DRPolicy` naming both clusters via their `StorageClass`/`VolumeReplicationClass` labels (`design-csi-addons-replication.md` §7.1's `ramendr.openshift.io/storageid`/`replicationid` labels, already implemented). A `DRPlacementControl` drives the workload, authored by SiteMap where available and hand-authored otherwise (P0-3). + +### 6.2 Test flow + +1. **Protect.** Deploy a workload with a PVC on cluster A, under a `StorageClass`/`VolumeReplicationClass` pair carrying the Ramen labels. Create the `DRPlacementControl`. Confirm Ramen's VRG creates one `VolumeReplication` for the PVC, `EnableVolumeReplication` fires, and `status.lastSyncTime` advances on the policy's ordinary cadence. +2. **Planned relocate.** Trigger Ramen's `Relocate` action. Confirm the sequence design-csi-addons-replication.md §5.2 documents drives correctly through Ramen rather than by hand: the source-side VRG demotes (fence, final flush, `Completed` settles once confirmed), then the target-side VRG promotes with `force=false`, gated on `PeerReady` from step 1's demote. Confirm the workload comes up on cluster B with the data a hashed writer wrote before relocation, and that Ramen's own `PeerReady`/`DataProtected` aggregation reports correctly throughout, not just this design's own conditions in isolation. +3. **Unplanned failover.** Simulate cluster B's loss (or reachability loss, matching this session's `replication_source_online` distinction) and trigger Ramen's `Failover` action against cluster A. Confirm the force-escalation path Ramen's own controller drives (§5.2's "no wait-and-retry grace period" behavior) still lands correctly when Ramen, not a test script, is the one issuing the calls. +4. **Resync.** Recover the failed cluster and confirm Ramen drives `ResyncVolume` to reconcile the diverged copy, without merging or re-triggering a full cutover. + +### 6.3 Pass/fail criteria + +Every step's pass criterion is an observable Ramen already reports on its own objects (`VolumeReplication.status.conditions`, the VRG's `DataProtected`, the DRPC's `PeerReady`), not a simplyblock-specific read: the point of this validation is that Ramen's own surface reflects reality, not that a side-channel confirms it. Data correctness is checked with a hashed writer across every promote, matching `design-csi-addons-replication.md` §12's existing E2E discipline. + +--- + +## 7. Testing Strategy + +Two different classes of coverage, for the two different things this document specifies. §3 adds nothing to test: it confirms that no operator-side code exists for peerClasses, and there is no code left to exercise once the same-cluster preflight was removed. + +- **§4's group surface is tested where it lives**, in `design-csi-addons-replication.md` §14: the driver's VolumeGroup service and group-handle Replication routing, and the backend group-replication engine, as unit and backend scenarios in that document's test plan (U-62 … U-69). This document adds no unit test of its own for it, since no operator code implements it any longer. +- **§5 confirms existing code** and adds nothing to test on its own. **§6 is exclusively E2E, live-cluster validation** with no smaller harness to substitute, including a group-protect scenario (M-05) confirming a real VRG's `VolumeGroupReplication` driving the driver's group surface end to end. + +Full scenario detail: [`tests/test-plan-ramen-integration.md`](../tests/test-plan-ramen-integration.md). Its M-01 and M-02 close `design-csi-addons-replication.md` test plan's E-06 and E-07, which have carried no implementing test since they were written. + +--- + +## 8. Open Questions + +| # | Question | Owner | +|-----|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------| +| 1 | **Is a real OCM hub with Ramen already available for this validation**, or does P0-1/P0-2 need to be stood up from scratch? The answer decides whether §6 is schedulable now or needs its own infrastructure work first. | Operator team / Infra | +| 2 | **Does SiteMap exist in a runnable form yet**, or does every run of §6 use a hand-authored `DRPlacementControl` stand-in (P0-3)? If SiteMap is not yet runnable, note that explicitly rather than blocking on it indefinitely. | Operator team | +| 3 | **`csi-addons/spec` version floor (P0-4).** Confirm whether a newer pinned version already carries `lastSyncBytes`/`lastSyncDuration` before treating this as a real upstream gap to track. | Operator team | +| 4 | **Is Global VGR needed.** §4 scopes `VolumeGroupReplication` to the base, single-VRG case. Confirm whether any planned simplyblock deployment spans a replication group across more than one application's VRG before treating RamenDR's multi-VRG consensus feature as work this document should also specify. Tracked as `design-csi-addons-replication.md` Open Question 4. | Operator team | +| 5 | ~~**`VolumeGroupReplicationContent`'s exact contract.**~~ **Moved.** With group replication now the driver-and-backend, `external: false` path (§4), the `Content` object is created and owned by the stock controller-manager; its contract is tracked in `design-csi-addons-replication.md` Open Question 5, not here. | Operator team | diff --git a/operator/docs/designs/design-test-failover.md b/operator/docs/designs/design-test-failover.md new file mode 100644 index 000000000..930da44a2 --- /dev/null +++ b/operator/docs/designs/design-test-failover.md @@ -0,0 +1,612 @@ +# Design Document: Non-Disruptive Test Failover + +**Status:** Draft +**Author:** Israel Geoffrey (geoffrey1330) +**Date:** 2026-09-29 +**Test Plan:** [`tests/test-plan-test-failover.md`](../tests/test-plan-test-failover.md) + +--- + +## Phasing Overview + +| Phase | Status | Where the bubble runs | Recovery point | New capability | Sections | +|-------------|---------|---------------------------------|-----------------------------------------------|------------------------------------------------------------------|-----------------------| +| **Phase 1** | Planned | The DR target cluster | The replicated snapshot already on the target | Cross-cluster read and placement from the hub (OCM) | §4, §5.1–§5.5, §6, §7 | +| **Phase 2** | Planned | A cluster that holds no replica | A point shipped there on demand | On-demand shipping of a recovery point to a named backend (P0-6) | §5.6 | + +Test-failover recovers onto a cluster other than the source's, never in place: the point is to rehearse the site a real failover would move to, and recovering into the source's own backend would clone into the subsystem serving the live volume. The two phases differ on one axis: whether the bubble cluster already holds a replicated copy of the source. In Phase 1 the bubble runs on the DR target, where replication has already landed a snapshot, so no data moves. In Phase 2 the bubble runs on a cluster with no copy, the only case that needs data shipped on demand and the design's long pole. + +The hub coordinates both phases. It never has to run the source or the bubble itself, and both may be any managed cluster, but the bubble must differ from the source. What Phase 1 needs, and Phase 2 inherits, is the ability to read the source object on its cluster and place the bubble object on the recovery cluster, both from the hub, which OCM provides. + +--- + +## Phase 0 — External Prerequisites + +| # | Prerequisite | Kind | Blocks | Status | +|------|----------------------------------------------------------------------------------------------------------------------------------------------|-------------------------|---------|----------------------------------------------------------------------------------------------------------------------------------------------------------| +| P0-2 | Clone a snapshot into a writable volume in a chosen pool, and return the clone's volume handle | Control plane (`sbcli`) | All | Shipped: `snapshot_controller.clone`, and the CSI clone-from-snapshot path | +| P0-3 | Delete a volume, idempotent | Control plane (`sbcli`) | All | Shipped | +| P0-4 | Resolve the latest replicated snapshot on a DR-target backend for a source relationship | Control plane (`sbcli`) | Phase 1 | Shipped: `lvol_controller.latest_replicated_snapshot` and `replication_policy_controller.latest_replicated_generation`, with v2 endpoints | +| P0-5 | From the hub, read an object on a managed cluster (`ManagedClusterView`) and place objects on it (`ManifestWork`), each with status feedback | Ecosystem (OCM) | Phase 1 | Available: both are OCM primitives, once the cluster is a registered `ManagedCluster` with a working view controller (04-bootstrap-ocm.sh) | +| P0-6 | On-demand shipping of a specific snapshot or group generation to a named target backend that holds no copy, followed by a clone there | Control plane (`sbcli`) | Phase 2 | Not shipped. The long pole of Phase 2. Today's cross-cluster reach is the continuous replication engine or the S3 backup path, neither an on-demand push | + +The recovery point numbering keeps its original P0- ids so cross-references hold; P0-1 (taking a fresh source snapshot) is gone with the in-place case. Phase 1 has no unmet storage prerequisite: the recovery point is the replicated snapshot already on the target backend (P0-4), and it clones and deletes with calls that ship. Its one non-storage need is OCM (P0-5), which the DR setup already establishes, and it is needed because the hub, as coordinator, reaches the source and the bubble on their own clusters through it. Phase 2 is the exception: P0-6 is genuinely new, because moving one recovery point to a backend that holds no copy of it, on demand, is a capability the engine does not have. + +--- + +## Table of Contents + +1. [Background](#1-background) +2. [Goals and Non-Goals](#2-goals-and-non-goals) +3. [Architecture Overview](#3-architecture-overview) +4. [API Design — New CRD](#4-api-design--new-crd) +5. [Core Mechanism](#5-core-mechanism) +6. [State Machine](#6-state-machine) +7. [Controller Design](#7-controller-design) +8. [Backend API Requirements](#8-backend-api-requirements) +9. [Configuration](#9-configuration) +10. [Failure Modes and Fallback](#10-failure-modes-and-fallback) +11. [Observability](#11-observability) +12. [Testing Strategy](#12-testing-strategy) +13. [Open Questions](#13-open-questions) +14. [Appendix A: `testfailover_types.go`](#appendix-a-testfailover_typesgo) + +--- + +## Overview + +A test failover proves an application can be recovered from a point-in-time copy, in isolation, without disturbing the running production. The recovery point is always a snapshot and the result is always a clone, so the source is never touched. What varies is where the bubble runs, which is what a real DR test cares about: recovering onto the site a real failover would move to. + +The feature is a new CRD, `TestFailover`, and its controller, both on the hub, which coordinates the drill. The object names the source, by the cluster it runs on and its PVC, and a place to recover it, the `bubbleCluster`, which must be a different cluster. The controller reads the source PVC on its cluster to learn its volume, resolves the recovery point on the bubble's backend, clones it there, and places the clone on the bubble cluster as a bound PVC in an isolated namespace, `bubble` by default. The operator boots the application there, confirms the data, and deletes the drill, which reclaims the clone. The source serves throughout. + +`bubbleCluster` selects the topology, and always names a cluster other than the source's. Naming the DR target recovers from the replicated snapshot already sitting on its backend (Phase 1), a genuine "fail over to the target site" test with no data moved. Naming a cluster that holds no copy is the case that needs the point shipped there first (Phase 2). Both are one object, one controller, and one state machine, differing only in whether the point is already on the bubble's backend or must be shipped there. + +--- + +## 1. Background + +simplyblock has a real failover, driven either by Ramen (`DRPC.spec.action: Failover`) or imperatively by the simplyblock-native `ReplicationOps` CR. Both promote a replicated copy on the DR target and land the recovered workload there. That copy is, mechanically, a clone of the last replicated snapshot on the target, so a real failover is a clone-and-promote of a recovery point that already lives on the target's backend. + +Three facts shape a test failover. First, replication is between two clusters: `add_target` refuses `target_cluster_id == cluster_id` with "A cluster cannot replicate to itself" (`simplyblock_core/controllers/replication_policy_controller.py`). The DR target holds the replicated copy, which `lvol_controller.latest_replicated_snapshot` resolves without triggering anything, so the drill recovers there and never in the source's own cluster. Second, the primitives to recover from that point already exist: `snapshot_controller.clone` clones the replicated snapshot into a writable volume, and a volume can be deleted. A clone is a first-class volume with its own handle, which CSI static provisioning adopts as a `PersistentVolume`. Third, the hub already reaches its managed clusters both ways in this deployment: it reads an object on one through an OCM `ManagedClusterView`, the way Ramen reads a spoke's status, and writes one through a `ManifestWork`, the way Ramen places a workload. + +What is missing is the orchestration: an object that finds the source on its cluster, resolves the right recovery point, clones it on the right backend, places the bubble PVC on the right cluster, proves the copy is recoverable, and tears it down, all without touching the source. Ramen orchestrates none of it, because Ramen only fails over and relocates for real. This design is that object. + +--- + +## 2. Goals and Non-Goals + +### Goals + +- A `TestFailover` CRD and controller that, from one object on the hub, produce a bound PVC per source volume in an isolated namespace, on the cluster the drill recovers onto. +- Locate the source from the hub. The object names the source by its cluster and PVC, and the controller reads that PVC through OCM to learn its volume, so nobody has to hand-extract a backend handle. +- Recover onto the DR target site, from the replicated snapshot already there, with the bubble PVC placed on that cluster. This is the case that makes a test failover a real rehearsal of the DR target. +- Non-disruptive by construction. The recovery point is a snapshot and the result is a clone, so the source's data and I/O are never touched, and the controller records a before-and-after fingerprint of the source and its replication relationship so a regression is caught rather than assumed. +- Volume-scoped and consistency-group-scoped drills. A group drill recovers one PVC per member from one group-consistent point. +- DR target now, an arbitrary cluster later (§5), behind one object and one state machine. +- A finalizer-driven teardown that reclaims the clone, removes the placed PVC from the bubble cluster, and proves nothing test-labeled remains. +- Restart safety. The controller records which side effect each step issued, so a restart mid-drill resumes rather than repeats. + +### Non-Goals + +- **Bringing up the application.** The object produces bound PVCs and stops. The workload that consumes them is the operator's to deploy. An application lifecycle and its workload spec are a separate concern, out of scope here. +- **A test failback.** The drill is one-way. Tearing it down reclaims the clone. There is no promote-back. +- **Recovering in place, on the source's own cluster.** `bubbleCluster` must name a different cluster, and a same-cluster drill is rejected up front. Recovering into the source's backend would clone into the subsystem serving the live volume, and it rehearses no failover site. +- **Recovering onto a cluster with no copy in Phase 1.** A bubble cluster whose backend holds no replica needs the point shipped there, which is Phase 2 (§5.6), gated on a backend primitive that does not exist (P0-6). +- **Snapshot scheduling and evidence export.** A recurring schedule and a signed test report are a layer above this object and are out of scope here. +- **Replacing Ramen's own test paths.** Ramen has no non-disruptive test. This design does not add one to Ramen. It is a simplyblock-native object that reuses OCM only as the transport for reading the source and placing the bubble. + +--- + +## 3. Architecture Overview + +``` + TestFailover CR ──▶┌──────────────────────────────────────────────────────┐ + (on the hub) │ hub operator: TestFailoverReconciler │ + spec: scope, │ 1. read source PVC on sourceCluster (ManagedClusterView) → handle │ + sourceCluster, │ 2. resolve recovery point on the bubble's backend │ + sourceNamespace, │ replicated snapshot (P0-4) │ + sourceRef, │ 3. [Phase 2] ship point to that backend (P0-6) │ + bubbleCluster, │ 4. clone the point on that backend (P0-2) │ + bubbleNamespace │ 5. place PV + PVC on bubbleCluster (ManifestWork) │ + │ 6. Ready; hold until deleted │ + │ 7. finalizer: remove PVC, reclaim clone │ + └──┬──────────────┬───────────────────────┬────────────┘ + ManagedClusterView│ │ REST (webapi.Client) │ ManifestWork + (read source) ▼ ▼ ▼ (place bubble) + ┌────────────────────────┐ ┌──────────────────┐ ┌────────────────────────┐ + │ source cluster │ │ control plane │ │ bubble cluster │ + │ PVC + PV (volumeHandle)│ │ clone P0-2 │ │ work-agent applies: │ + │ projected to the hub │ │ delete P0-3 │ │ PersistentVolume │ + └────────────────────────┘ │ latest-repl P0-4│ │ PersistentVolumeClaim│ + │ ship P0-6 (P2) │ │ (bound to the clone │ + └──────────────────┘ │ on its backend) │ + └────────────────────────┘ +``` + +The controller runs on the hub, which coordinates the drill and runs neither the source nor the bubble. It learns the source's volume by reading the source PVC and its PV on `sourceCluster` through an OCM `ManagedClusterView`, which projects their current state back to the hub. It reads the source only to fingerprint it, never to change it. The clone is built on the bubble cluster's own backend, and the bubble cluster's CSI driver adopts it through a static `PersistentVolume` naming the clone's handle, with a `PersistentVolumeClaim` bound to it in the bubble namespace. + +Placement is uniform: the controller delivers the bubble PV and PVC to `bubbleCluster` as an OCM `ManifestWork`, whose work-agent applies them and reports the bind result back through the `ManifestWork` status. The hub addresses both the source and the bubble this way, including its own cluster when it is self-managed. The trust boundary for storage is the control-plane REST API, reached with the admin bearer token the operator already uses. The trust boundary for cross-cluster read and write is OCM, which the DR setup already establishes (04-bootstrap-ocm.sh). + +--- + +## 4. API Design — New CRD + +`TestFailover` is a namespaced object in the `storage.simplyblock.io` group, created on the hub. One object drives one drill. Its spec is immutable, because the object is a request and a drill whose target moved under the controller mid-flight has no coherent meaning. The full type is [Appendix A](#appendix-a-testfailover_typesgo), and the body shows only the fields an argument turns on. + +### 4.1 `TestFailover` Spec + +The spec names the source by where it runs and what it is, and names where to recover it. `sourceCluster` and `sourceRef` are what let the hub find the source without anyone extracting a backend handle by hand. + +```go +// SourceCluster is the OCM ManagedCluster the source runs on. The hub reads the +// source there through a ManagedClusterView. Immutable. +// +kubebuilder:validation:Required +// +k8s:immutable +SourceCluster string `json:"sourceCluster"` + +// SourceRef names the source on SourceCluster: a PersistentVolumeClaim in +// SourceNamespace (scope=Volume), or a consistency group (scope=Group). +// Immutable. +// +kubebuilder:validation:Required +// +k8s:immutable +SourceRef string `json:"sourceRef"` + +// BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding +// the replicated point, or another cluster. It must differ from SourceCluster; +// test-failover recovers onto a different cluster, never in place. Immutable. +// +kubebuilder:validation:Required +// +k8s:immutable +BubbleCluster string `json:"bubbleCluster"` +``` + +`sourceCluster` answers "where is the PVC to test," and the controller reads it there rather than requiring a handle. `bubbleCluster` selects the topology (§5) and carries the "recover onto the target site" intent; it must name a cluster other than the source's, and a same-cluster drill is rejected up front. `bubbleNamespace` defaults to `bubble` and is the namespace on the bubble cluster where every recovered PVC lands, isolated so it cannot collide with the source workload's PVCs, which carry the same names. + +### 4.2 `TestFailover` Status + +The status carries the state-machine position, the resolved source and recovery point, one entry per recovered volume, and a report. `phase` is the coarse lifecycle and `step` is the durable machine position with its deadline (§6). `clones` is the list the finalizer reclaims from, and `report` is the evidence a reader takes away. + +```go +// Clones is one entry per recovered volume: the source it came from, the +// snapshot and clone the drill built, and the PVC placed on the bubble cluster. +// +optional +// +listType=map +// +listMapKey=sourceRef +Clones []TestFailoverClone `json:"clones,omitempty"` + +// Report is the drill's evidence: the source and point recovered, its age, the +// cluster it ran on, and whether the source was untouched. Populated as the +// drill reaches Ready. +// +optional +Report *TestFailoverReport `json:"report,omitempty"` +``` + +An invariant the controller enforces and the status records: the drill is non-disruptive. `status.report.invariantsHeld` is set only when the fingerprint of the source and its replication relationship taken before the drill matches the one taken after (§7.4). A drill that reached `Ready` with `invariantsHeld: false` is a defect, not a passing test. + +The object owns a finalizer, `storage.simplyblock.io/testfailover-teardown`. Deletion runs the teardown state (§6) before the finalizer is removed, so a clone or a placed PVC is never orphaned by a delete that races the controller. + +--- + +## 5. Core Mechanism + +### 5.1 Locating the source + +The hub does not run the source, so it reads it. The controller creates an OCM `ManagedClusterView` on `sourceCluster` for the source PVC named by `sourceRef` in `sourceNamespace`, and for the PV it is bound to, which projects their current state back to the hub. From the PV's `spec.csi.volumeHandle` it learns the source volume's backend handle, the identity every later step keys on. For a group drill, `sourceRef` names a consistency group, whose member volumes the control plane resolves from the group id on `sourceCluster`'s backend, so the drill recovers the whole set. + +Reading the source is also where the non-disruptiveness fingerprint begins: the handle, the PVC's binding, and the replication relationship's state are captured here and compared again at the end (§7.4). + +### 5.2 Resolving the recovery point + +The recovery point is a snapshot already on the bubble cluster's backend. For a volume drill it is the latest replicated snapshot for the source relationship, which `latest_replicated_snapshot` resolves from the source handle (P0-4). For a group drill it is one group-consistent snapshot, resolved as the latest replicated generation on the target. Either way the resolution triggers nothing, because replication already produced the point. + +Resolving the replicated point touches nothing: replication already produced it, and the drill neither promotes nor commits anything, which is what keeps the source and the live replication relationship untouched. The drill takes no snapshot of its own, so it has none to delete on teardown. + +### 5.3 Cloning the point on the bubble's backend + +The controller clones the recovery-point snapshot into a writable volume on the bubble cluster's own backend (P0-2), and the backend returns the clone's volume handle. The replicated snapshot already sits on that backend, so the clone is local to the point and no data crosses a cluster boundary. The clone is a first-class volume, tagged with the drill's `test-id` for reclaim. + +### 5.4 Placing the bubble PVC + +The clone is adopted as a static `PersistentVolume` whose `spec.csi.volumeHandle` is the clone's handle, with a `PersistentVolumeClaim` bound to it in the bubble namespace, sized from the source's request. The controller delivers the PV and PVC to `bubbleCluster` as an OCM `ManifestWork` (P0-5), and that cluster's work-agent applies them and reports the PVC bound through the `ManifestWork` status. The PV carries `persistentVolumeReclaimPolicy: Retain`, so deleting the PVC does not delete the clone, which the controller reclaims at the backend on teardown. + +The result is one bound PVC per source volume in the bubble namespace on the bubble cluster. For a group drill, one PVC per member from the one point, which is what makes the recovered set crash-consistent. + +### 5.5 Recovering onto the DR target (Phase 1) + +The DR-target case is why the source is named by cluster and PVC rather than assumed local. The source runs on one cluster and the bubble on the DR target, so the controller, from the hub, reads the source PVC on its cluster (§5.1), resolves its handle to the replicated snapshot on the target's backend (P0-4), clones it there (§5.3), and places the bubble PVC on the target through `ManifestWork` (§5.4). Nothing is shipped, because replication already put the point on the target. This is a true rehearsal of the site that would take over in a real failover, and it leaves the running replication relationship exactly as it was. + +### 5.6 Shipping to a cluster with no copy (Phase 2) + +A bubble cluster whose backend holds no replica of the source needs the point moved there before it can be cloned. The controller calls the shipping verb (P0-6), which replicates the specific recovery point to that backend as a cloneable object, and then the clone and placement steps run there as in §5.3 and §5.4. + +This is the design's long pole, because the shipping primitive does not exist. The continuous replication engine is pair-scoped and aimed at the DR target, and the S3 backup path is not a cluster-to-cluster push. Neither is an on-demand "ship this one point to cluster X now." P0-6 is that new capability, and Phase 2 does not ship until it does. + +### 5.7 Teardown + +Deleting the `TestFailover` runs the teardown state before the finalizer clears. The controller removes the PVC and its static PV from the bubble cluster by deleting their `ManifestWork`, removes the `ManagedClusterView` it created on the source, reclaims each clone at the backend, then enumerates by the drill's `test-id` label to prove nothing remains. The recovery point is a replicated snapshot the drill only resolved, never created, so it is left alone. Only then is the finalizer removed. A teardown that cannot confirm a reclaim holds the object in `TearingDown` with the reason on `status.message`, rather than removing the finalizer and orphaning backend storage. + +--- + +## 6. State Machine + +``` +Pending + │ spec admitted, finalizer added + ▼ +Provisioning ──(step: ResolvingSource)──▶ read source PVC/PV on sourceCluster + │ via ManagedClusterView → handle + │ (step: ResolvingPoint) latest replicated snapshot (P0-4) + │ ← status.report.recoveryPoint set + │ (step: Shipping) [Phase 2 only] ship point to the bubble's backend (P0-6) + │ (step: Cloning) clone on the bubble's backend (P0-2) + │ ← status.clones[].cloneID set + │ (step: Placing) deliver PV + PVC to bubbleCluster via ManifestWork; + │ wait Bound + ▼ +Ready ────────────────────────────────── every PVC Bound; report populated + │ (holds here until the object is deleted) + │ .metadata.deletionTimestamp set + ▼ +TearingDown ──(step: Releasing)──▶ delete ManifestWork + ManagedClusterView, + │ reclaim clones (the replicated point is left alone) + ▼ +(finalizer removed, object gone) + +Failed ◀── any step's deadline expires, or a backend, read, or placement call + fails terminally (object stays; teardown still runs on delete) +``` + +The machine position lives in `status.step` as a snapshot carrying the state and the deadline that state expires at, so a restored controller times a stalled step out rather than waiting forever. `status.phase` is the coarse view for `kubectl get`. Each step records `status.step.triggered` once its side effect is issued, so a restart between cloning and recording the handle does not clone twice, and a restart between placing and confirming does not place twice. + +| Condition | Step | Result | +|---------------------------------|-----------------|----------------------------------------------------------------------------------------------------| +| User deletes mid-drill | any | `TearingDown`: reclaim whatever `status.clones` records, then clear the finalizer | +| Operator restart | any | resume from `status.step`, and `triggered` prevents re-issuing the current step's side effect | +| Source PVC not found on cluster | ResolvingSource | `Failed`: the view returns nothing for `sourceRef` on `sourceCluster` | +| Source cluster not managed | ResolvingSource | `Failed`: `sourceCluster` is not a registered `ManagedCluster` | +| No replicated point on target | ResolvingPoint | `Failed` for a Phase 1 drill: replication has landed nothing on the bubble's backend yet | +| Bubble equals source cluster | ResolvingSource | `Failed`: `bubbleCluster` must differ from `sourceCluster`; a same-cluster drill is rejected | +| Backend clone error | Cloning | `Failed`. The replicated point is left alone; the clone is reclaimed on delete | +| Bubble cluster not managed | Placing | `Failed`: `bubbleCluster` is not a registered `ManagedCluster` | +| PVC never binds | Placing | `Failed` at the deadline. The clone is recorded and reclaimed on delete | +| Clone reclaim fails | Releasing | hold in `TearingDown` with the reason. The finalizer is not removed until the reclaim is confirmed | + +--- + +## 7. Controller Design + +### 7.1 Location + +`internal/controller/testfailover_controller.go`, `TestFailoverReconciler`, on the hub. It reuses the `webapi.Client` the replication controllers use for control-plane calls, an OCM `ManagedClusterView` to read the source, and an OCM `ManifestWork` to place the bubble. + +### 7.2 Reconciliation Trigger + +Watches `TestFailover` and owns the `ManagedClusterView` and `ManifestWork` it creates for each drill, reconciling on their status feedback, which carries the source's projected state and the placed PVC's bind state back to the hub. It requeues on its own step deadline so a stalled step is detected without an external event. + +### 7.3 Concurrency and Mutual Exclusion + +A drill does not mutate the source, so two drills against one source cannot corrupt it. What they can do is duplicate snapshots and clones, so the controller keys one active drill per resolved `(scope, sourceCluster, sourceRef, bubbleCluster)` and refuses a second with an admission rule where CEL can express it and a `Failed` phase otherwise. This is a lighter lock than the `ReplicationOps` entity lock (`ActiveOpsRef`), because there is no production object whose single-writer invariant has to be defended. + +### 7.4 Interaction with Existing Controllers + +The drill is invisible to production by construction. It only reads the source, resolves or takes a snapshot, and clones the snapshot, none of which changes the source volume or the live replication. To make "invisible" checkable rather than asserted, the controller fingerprints the source at `ResolvingSource` and again at `Ready`: the source PVC is still bound to the same volume, and the relationship's `ReplicationSlot`, and any `VolumeReplication` or `VolumeGroupReplication` object, are unchanged in state and lag. A mismatch sets `status.report.invariantsHeld: false` and moves the object to `Failed`, because a drill that changed production has failed at its one core promise. + +### 7.5 RBAC + +New rules: `testfailovers` and `testfailovers/status` and `testfailovers/finalizers` (full), and `create`/`delete`/`get`/`list`/`watch` on `managedclusterviews` and `manifestworks` in a cluster's namespace on the hub. The source's projection and the bubble's PV and PVC are handled by the managed clusters' own agents, so the hub operator needs no direct `persistentvolume`, `persistentvolumeclaim`, or snapshot-API permission. No new permission on any production replication CR either: the controller reads them, and read is a permission the operator already holds. + +### 7.6 Cross-Cluster Read and Placement + +The hub reaches its managed clusters through OCM, the transport already established for DR. To read the source, the controller creates a `ManagedClusterView` in `sourceCluster`'s namespace naming the PVC and PV, and the view controller on that cluster projects them back into the view's status. To place the bubble, the controller creates a `ManifestWork` in `bubbleCluster`'s namespace carrying the PV and PVC, and the work-agent on that cluster applies them and reports their status back, which is how the hub learns the bubble is bound without a direct connection to either cluster's API server. Both clusters must be registered `ManagedCluster`s (04-bootstrap-ocm.sh). Teardown deletes both objects, and OCM garbage-collects the applied PV and PVC on the bubble cluster. + +--- + +## 8. Backend API Requirements + +| Method | Endpoint | Notes | +|--------|---------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| GET | `.../clusters/{c}/replication/relationships/{lvol}/latest-snapshot` | **Exists** (P0-4). Resolves the latest replicated snapshot on the DR-target backend. Pure read. Group form: `latest-generation` | +| POST | `.../clusters/{c}/snapshots/{id}/clone` | **Exists** (P0-2). Clones the snapshot into a chosen pool and returns the clone's volume handle. Backed by `snapshot_controller.clone` | +| DELETE | `.../clusters/{c}/storage-pools/{p}/volumes/{id}` | **Exists** (P0-3). Reclaims a clone. Idempotent | +| POST | `.../clusters/{c}/replication/ship-snapshot` | **New** (P0-6, Phase 2). Ships a named recovery point to a backend that holds no copy. Idempotent per `(recovery-point, target)`. Long-running: returns a handle to poll | + +Phase 1 uses only endpoints that exist. The one mutating call the controller may retry after a restart, the clone, is made idempotent by keying on the drill's `test-id`, so a retry that finds a matching clone reuses it. The exact v2 route spellings mirror the existing clone route and are confirmed against the API at implementation time. P0-6's ship is a long-running call and returns a handle the controller polls, with a deadline that moves the step to `Failed` on expiry. The source read and bubble placement are OCM, not backend calls (§7.6). + +--- + +## 9. Configuration + +| Field | Type | Default | Description | +|------------------------|--------|-----------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------| +| `spec.sourceCluster` | string | (required) | The OCM `ManagedCluster` the source runs on. Immutable | +| `spec.sourceNamespace` | string | (required for Volume) | The namespace of the source PVC. Immutable | +| `spec.sourceRef` | string | (required) | The source PVC (Volume) or consistency group (Group) on `sourceCluster`. Immutable | +| `spec.bubbleCluster` | string | (required) | The OCM `ManagedCluster` to recover onto. Must differ from `sourceCluster`. Immutable | +| `spec.bubbleNamespace` | string | `bubble` | Namespace on the bubble cluster the recovered PVCs are created in. Immutable | +| `spec.ttlSeconds` | int | unset | Optional maximum lifetime. When set, the drill is torn down after the deadline even without a delete, so a forgotten drill cannot hold a clone forever | + +`ttlSeconds` is the only runtime-relevant knob and it is advisory: the controller reads it once at `Ready` and schedules a teardown. Changing the rest of the spec after creation is refused by immutability, because a drill whose target moved mid-flight has no coherent meaning. + +--- + +## 10. Failure Modes and Fallback + +| Failure | Detection | Behavior | +|------------------------------------|------------------------------|----------------------------------------------------------------------------------------------------------------------| +| Control plane unreachable | REST call error | Requeue with backoff, and the step deadline eventually moves the object to `Failed`. No partial state is committed | +| Source cluster not managed | ManagedClusterView error | `Failed` at `ResolvingSource`: `sourceCluster` must be registered on the hub | +| Source PVC not found | view projects nothing | `Failed` at `ResolvingSource` with the `sourceRef` in `status.message` | +| No replicated point on the target | latest-snapshot 404 | `Failed` at `ResolvingPoint` for a Phase 1 drill. Replication has landed nothing on the bubble's backend yet | +| Bubble cluster not managed | ManifestWork placement error | `Failed` at `Placing`: the target must be registered on the hub | +| PVC never binds on the bubble | ManifestWork status timeout | `Failed` at `Placing`. The clone is recorded and reclaimed on delete | +| Fingerprint drift (source changed) | Compare at `Ready` | `Failed`, `invariantsHeld: false`. This is the guard, not an expected path | +| Reclaim cannot be confirmed | delete call non-success | Hold in `TearingDown`, finalizer retained, reason on `status.message`. Never orphan a clone, snapshot, or placed PVC | + +Every path degrades to a named state. The one path that must never degrade silently is the fingerprint guard: a drill that cannot prove it left the source untouched fails, rather than passing on the assumption that it did. + +--- + +## 11. Observability + +The operator has no metrics or events for a non-disruptive test today, because the capability does not exist. Everything below is new, on the new kind. + +### Kubernetes Events + +Events land on the `TestFailover` object, which lives on the hub and outlives each step it reports. + +| Event | Type | Reason | +|---------------------------------------------------------------|---------|-----------------------| +| The source resolved to volume X on cluster Y | Normal | SourceResolved | +| The recovery point resolved to snapshot X at time T | Normal | RecoveryPointResolved | +| The clone was built on the bubble's backend, source untouched | Normal | CloneBuilt | +| The bubble PVC is bound on cluster X and the drill is ready | Normal | BubbleReady | +| The drill failed because the source changed during the drill | Warning | InvariantViolated | +| A clone, snapshot, or placed PVC could not be reclaimed | Warning | ReclaimPending | + +`InvariantViolated` and `ReclaimPending` are the two that matter most. The first says the drill stopped being non-disruptive, and the second is a teardown correctly refusing to orphan storage or a peer object, which is a hold that would otherwise look like a hang. + +### Prometheus Metrics + +| Metric | Labels | Description | +|-------------------------------------------------------|----------------------------|----------------------------------------------------------------------------------| +| `simplyblock_testfailover_drills_total` | `scope`, `phase`, `result` | Counter of completed drills by outcome (`ready`, `failed`, `invariant_violated`) | +| `simplyblock_testfailover_duration_seconds` | `scope`, `phase`, `step` | Histogram of time spent per step, for the recover-time estimate | +| `simplyblock_testfailover_active` | `phase` | Gauge of drills currently holding a clone, for capacity watch | +| `simplyblock_testfailover_recovery_point_age_seconds` | `scope` | Gauge of the recovery point's age at drill time (now minus the snapshot time) | + +The two load-bearing metrics are `simplyblock_testfailover_drills_total` with `result="invariant_violated"`, which is the alert that a test failover stopped being non-disruptive, and `simplyblock_testfailover_active`, which is the alert that clones are accumulating because teardowns are not completing. + +--- + +## 12. Testing Strategy + +Full scenario matrix, coverage status, and hand-off test concepts: +[`tests/test-plan-test-failover.md`](../tests/test-plan-test-failover.md) + +- **Unit:** the state-machine transitions and their deadlines, the fingerprint comparison (drift detected and no-drift accepted), the source resolution from a `ManagedClusterView` projection, the DR-target recovery-point resolution, the same-cluster rejection, the idempotency keying, and the scope-to-source resolution. +- **Integration:** the reconcile loop against `envtest`, a mock control plane, and a mock OCM (`ManagedClusterView` and `ManifestWork` with status feedback). The DR-target drill, and the teardown that reclaims and proves no leftovers. The restart case per step. The non-disruptiveness guard, asserting a source change fails the drill. +- **E2E:** a live two-cluster DR setup where the replicated point on the target is cloned there, the bubble PVC is placed on the target, a pod boots on it, and the recovered marker matches, with the source and its replication lag asserted untouched. +- **Load / long-running:** none in Phase 1 or 2. + +The Phase 2 scenarios (§5.6) become testable only when P0-6 exists. The risk concentrates in the non-disruptiveness guard (§7.4), the cross-cluster read and placement and their status feedback (§7.6), and the teardown reclaim (§5.7): the first is the feature's core promise, and the last is where a bug leaks backend storage or a peer object. + +--- + +## 13. Open Questions + +| # | Question | Owner | +|-----|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|---------------------| +| 1 | **Group point on the replica.** `latest_replicated_generation` resolves a group generation on the DR target, but a group-consistent point on a replica was flagged as unconfirmed. Is the generation crash-consistent across members on the target, or only on the source? | SPDK / Backend team | +| 2 | **Clone accounting.** A drill's clone consumes the bubble backend's lvstore object budget for the drill's life. Does the backend expose a per-clone reservation the controller can pre-check, or does a drill risk failing at `Cloning` on a full lvstore with no admission-time warning? | SPDK / Backend team | +| 3 | **Leftover proof under a disabled `LIST_VOLUMES`.** The CSI driver does not advertise `LIST_VOLUMES`, so teardown cannot cross-check backend volumes against Kubernetes objects through CSI. Is the label enumeration sufficient, or is a backend enumeration needed to guarantee no leaked clone or snapshot? | SPDK / Backend team | +| 4 | **P0-6 shape (Phase 2).** Is on-demand shipping a new engine mode (a one-shot pair-and-transfer) or a distinct primitive? Its API shape (§8) is provisional until this is decided. | SPDK / Backend team | + +--- + +## Appendix A: `testfailover_types.go` + +```go +package v1alpha2 + +import ( + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "github.com/simplyblock/atlas/statemachine" +) + +// TestFailoverScope selects what a drill recovers. +// +kubebuilder:validation:Enum=Volume;Group +type TestFailoverScope string + +const ( + // TestFailoverScopeVolume recovers a single source volume. + TestFailoverScopeVolume TestFailoverScope = "Volume" + // TestFailoverScopeGroup recovers a consistency group from one point. + TestFailoverScopeGroup TestFailoverScope = "Group" +) + +// TestFailoverPhase is the coarse lifecycle phase of a drill. +// +kubebuilder:validation:Enum=Pending;Provisioning;Ready;Failed;TearingDown +type TestFailoverPhase string + +const ( + TestFailoverPhasePending TestFailoverPhase = "Pending" + TestFailoverPhaseProvisioning TestFailoverPhase = "Provisioning" + TestFailoverPhaseReady TestFailoverPhase = "Ready" + TestFailoverPhaseFailed TestFailoverPhase = "Failed" + TestFailoverPhaseTearingDown TestFailoverPhase = "TearingDown" +) + +// TestFailoverStep is one step of a running drill. The enum is the union of every +// phase's steps; which steps belong to which phase is declared by the graph rather +// than by this type. +// +kubebuilder:validation:Enum=ResolvingSource;ResolvingPoint;Shipping;Cloning;Placing;Releasing +type TestFailoverStep string + +const ( + TestFailoverStepResolvingSource TestFailoverStep = "ResolvingSource" + TestFailoverStepResolvingPoint TestFailoverStep = "ResolvingPoint" + TestFailoverStepShipping TestFailoverStep = "Shipping" + TestFailoverStepCloning TestFailoverStep = "Cloning" + TestFailoverStepPlacing TestFailoverStep = "Placing" + TestFailoverStepReleasing TestFailoverStep = "Releasing" +) + +// TestFailoverSpec is the request for one non-disruptive test-failover drill. +type TestFailoverSpec struct { + // Scope selects what the drill recovers. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + Scope TestFailoverScope `json:"scope"` + + // SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + // the source there through a ManagedClusterView. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + SourceCluster string `json:"sourceCluster"` + + // SourceNamespace is the namespace of the source PVC on SourceCluster. + // Required for scope=Volume. Immutable. + // +optional + // +k8s:immutable + SourceNamespace string `json:"sourceNamespace,omitempty"` + + // SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + // SourceNamespace (scope=Volume), or a consistency group (scope=Group). + // Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + SourceRef string `json:"sourceRef"` + + // BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + // the replicated point, or another cluster. It must differ from SourceCluster; + // test-failover recovers onto a different cluster, never in place. Immutable. + // +kubebuilder:validation:Required + // +k8s:immutable + BubbleCluster string `json:"bubbleCluster"` + + // BubbleNamespace is the namespace on the bubble cluster where the recovered + // PVCs are created. Immutable. + // +kubebuilder:default=bubble + // +optional + // +k8s:immutable + BubbleNamespace string `json:"bubbleNamespace,omitempty"` + + // TTLSeconds is an optional maximum lifetime: the drill is torn down after it + // even without a delete, so a forgotten drill cannot hold a clone forever. + // +optional + // +kubebuilder:validation:Minimum=0 + TTLSeconds *int64 `json:"ttlSeconds,omitempty"` +} + +// TestFailoverClone is one recovered volume: the source it came from, the +// snapshot and clone the drill built, and the PVC placed on the bubble cluster. +type TestFailoverClone struct { + // SourceRef is the source volume (or group member) the recovered volume maps to. + SourceRef string `json:"sourceRef"` + // SourceHandle is the source volume's backend handle, read from its PV. + // +optional + SourceHandle string `json:"sourceHandle,omitempty"` + // SnapshotID is the recovery-point snapshot: the replicated snapshot already on + // the bubble cluster's backend that the clone is built from. + // +optional + SnapshotID string `json:"snapshotID,omitempty"` + // CloneID is the backend id of the writable clone. + // +optional + CloneID string `json:"cloneID,omitempty"` + // PVCName is the bound PVC in the bubble namespace on the bubble cluster. + // +optional + PVCName string `json:"pvcName,omitempty"` + // SizeBytes is the recovered volume's size. + // +optional + SizeBytes int64 `json:"sizeBytes,omitempty"` +} + +// TestFailoverReport is the evidence a drill produces. +type TestFailoverReport struct { + // BubbleCluster is the cluster the drill recovered onto. + // +optional + BubbleCluster string `json:"bubbleCluster,omitempty"` + // RecoveryPoint is the snapshot or group generation the drill recovered. + // +optional + RecoveryPoint string `json:"recoveryPoint,omitempty"` + // RecoveryPointTime is when that point was taken. + // +optional + RecoveryPointTime *metav1.Time `json:"recoveryPointTime,omitempty"` + // RecoveryPointAgeSeconds is the drill time minus the recovery-point time. + // +optional + RecoveryPointAgeSeconds int64 `json:"recoveryPointAgeSeconds,omitempty"` + // InvariantsHeld is true only when the source fingerprint taken before the + // drill matches the one taken at Ready. A Ready drill with this false is a + // defect. + // +optional + InvariantsHeld bool `json:"invariantsHeld,omitempty"` +} + +// TestFailoverStatus is the observed state of a drill. +type TestFailoverStatus struct { + // +optional + Phase TestFailoverPhase `json:"phase,omitempty"` + // +kubebuilder:validation:XValidation:rule="!has(self.state) || self.state in ['ResolvingSource','ResolvingPoint','Shipping','Cloning','Placing','Releasing']",message="unknown step" + // +optional + Step statemachine.KubeSnapshot `json:"step,omitempty"` + // +optional + Message string `json:"message,omitempty"` + // Triggered records that the current step's side effect was issued, so a + // restart does not repeat it. + // +optional + Triggered bool `json:"triggered,omitempty"` + // +optional + ObservedGeneration int64 `json:"observedGeneration,omitempty"` + // +optional + // +listType=map + // +listMapKey=sourceRef + Clones []TestFailoverClone `json:"clones,omitempty"` + // +optional + Report *TestFailoverReport `json:"report,omitempty"` + // +optional + StartedAt *metav1.Time `json:"startedAt,omitempty"` + // +optional + ReadyAt *metav1.Time `json:"readyAt,omitempty"` + // +optional + CompletedAt *metav1.Time `json:"completedAt,omitempty"` +} + +// +kubebuilder:object:root=true +// +kubebuilder:subresource:status +// +kubebuilder:resource:scope=Namespaced,shortName=tfo +// +kubebuilder:printcolumn:name="Scope",type=string,JSONPath=`.spec.scope` +// +kubebuilder:printcolumn:name="Source",type=string,JSONPath=`.spec.sourceRef` +// +kubebuilder:printcolumn:name="On",type=string,JSONPath=`.spec.sourceCluster` +// +kubebuilder:printcolumn:name="Bubble",type=string,JSONPath=`.spec.bubbleCluster` +// +kubebuilder:printcolumn:name="Phase",type=string,JSONPath=`.status.phase` +// +kubebuilder:printcolumn:name="Step",type=string,JSONPath=`.status.step.state` +// +kubebuilder:printcolumn:name="Age",type=date,JSONPath=`.metadata.creationTimestamp` + +// TestFailover is a one-way, non-disruptive test-failover drill. It recovers a +// source volume, or a consistency group, from a snapshot into an isolated +// namespace on a chosen cluster as bound PVCs, without touching the source: the +// recovery point is a snapshot and the result is a clone. The hub reads the +// source on its cluster and places the bubble on the recovery cluster through +// OCM, which must differ from the source cluster. Deleting the object reclaims +// the clones; the replicated recovery point is left alone. +type TestFailover struct { + metav1.TypeMeta `json:",inline"` + metav1.ObjectMeta `json:"metadata,omitempty"` + + Spec TestFailoverSpec `json:"spec,omitempty"` + Status TestFailoverStatus `json:"status,omitempty"` +} + +// +kubebuilder:object:root=true + +// TestFailoverList contains a list of TestFailover. +type TestFailoverList struct { + metav1.TypeMeta `json:",inline"` + metav1.ListMeta `json:"metadata,omitempty"` + Items []TestFailover `json:"items"` +} + +func init() { + SchemeBuilder.Register(&TestFailover{}, &TestFailoverList{}) +} +``` diff --git a/operator/docs/tests/test-plan-csi-addons-replication.md b/operator/docs/tests/test-plan-csi-addons-replication.md new file mode 100644 index 000000000..9280fb443 --- /dev/null +++ b/operator/docs/tests/test-plan-csi-addons-replication.md @@ -0,0 +1,247 @@ +# Test Plan: csi-addons Volume Replication + +Related design: [`designs/design-csi-addons-replication.md`](../designs/design-csi-addons-replication.md) + +Scope is the CSI driver's Replication service, the operator's preflight and coexistence rules, and the deployment of the csi-addons machinery. The replication engine itself (snapshot shipping, failover cloning, the cutover task runner) is the control plane's to prove and is exercised here only through the adapter's boundary. The kubernetes-csi-addons controller-manager is stock upstream and is not re-tested; what is tested is this driver's conformance to the contract it drives. + +Scenario IDs are permanent and are never reused or renumbered. `U-` is unit (no cluster: mock control plane, fake `client.Client`), `I-` is integration (the sidecar and controller-manager against the driver with a mock backend), `E-` is end-to-end (two live simplyblock clusters), and `M-` is manual. Types are `Positive`, `Negative`, `Boundary`, and `Regression`. A `—` in the `Test` column means nothing implements the scenario yet, and every such row reappears in §7 with its reason. + +Phase 1 (the csi-addons machinery, §4, §5.1's three verbs, and §6's steady-state contract) and Phase 2 (§5.2's lifecycle verbs and P0-3) have both landed; their unit rows below are filled in. Phase 4 (§14, `VolumeGroupReplication`) is **Planned**: the driver VolumeGroup service, the group-handle Replication routing, and the backend group-replication engine (U-62 … U-69) are not built yet, and its live group relocate is E-08. The earlier operator-owned reconciler and validator (U-57 … U-61) are retired (design §13). The operator's preflight and coexistence controllers (peerClasses, `PVCReplicationController`) remain a separate, unbuilt subsystem, and the integration and E2E tiers wait on a test bed neither phase has built yet. + +--- + +## 1. Unit Tests + +The Replication service against a mock control plane, and the operator pieces against fake clients. No Kubernetes API server. Numbering runs continuously across the groups. + +### Replication Verbs: Enable, Disable, Info (design §5.1) + +File: `csi-driver/internal/csi/controller/replication_test.go` (planned) + +| # | Scenario | Type | Test | +|----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------------------------------------------------------| +| U-01 | Enable on an unattached volume: the attach call carries the class's policy, and the RPC succeeds | Positive | `TestEnableVolumeReplication` | +| U-02 | Enable on a volume already attached to the same policy: success, no second attach call (idempotency) | Boundary | `TestEnableVolumeReplicationRepeatedIsIdempotent` | +| U-03 | Enable on a volume attached to a different policy: `FAILED_PRECONDITION` naming both policies, no attach call | Negative | — | +| U-04 | Disable on an attached volume: the detach call is made, success | Positive | `TestDisableVolumeReplication` | +| U-05 | Disable on a non-attached volume: success without a backend call (idempotency) | Boundary | `TestDisableVolumeReplicationNotAttachedIsSuccess` | +| U-06 | Disable while a cutover is in flight (backend 409): `ABORTED`, retryable | Negative | `TestDisableVolumeReplicationDuringCutoverIsAborted` | +| U-07 | Info returns `lastSyncTime` from the status read (`lastSyncDuration` and `lastSyncBytes` are not part of `GetVolumeReplicationInfoResponse` in csi-addons/spec v0.2.0, the version this driver builds against) | Positive | `TestGetVolumeReplicationInfo` | +| U-08 | A malformed volume handle: `INVALID_ARGUMENT` before any backend call | Negative | `TestEnableVolumeReplicationMalformedVolumeHandle` | +| U-09 | A cluster ID with no entry in this deployment's secret: `UNAVAILABLE`, distinct from a backend-side refusal | Negative | `TestEnableVolumeReplicationUnknownCluster` | +| U-28 | Enable without the `VolumeReplicationClass` policy parameter: `INVALID_ARGUMENT` before any backend call | Negative | `TestEnableVolumeReplicationMissingPolicyParam` | +| U-29 | Enable the backend refuses (412): `FAILED_PRECONDITION` carrying the backend's own reason | Negative | `TestEnableVolumeReplicationBackendRefusal` | +| U-30 | Info on a volume that never replicated: a nil `lastSyncTime`, never a `NOT_FOUND` | Boundary | `TestGetVolumeReplicationInfoNeverReplicated` | +| U-31 | Info on a volume id the backend does not recognize: `NOT_FOUND` | Negative | `TestGetVolumeReplicationInfoUnknownVolume` | +| ~~U-32~~ | **Retired.** Was "the remaining verbs fall through to `UNIMPLEMENTED`"; moot now that Phase 2 implements all six `csi-addons/spec` v0.2.0 methods, so `TestUnimplementedReplicationVerbsAreUnimplemented` was deleted rather than kept red. | — | — | + +### Replication Verbs: Promote, Demote, Resync (design §5.2) + +File: `csi-driver/internal/csi/controller/replication_lifecycle_test.go` + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| U-10 | Forced promote: the failover endpoint is called without `planned=true`, ignoring demote state entirely | Positive | `TestPromoteVolumeForced` | +| U-11 | Forced promote repeated after completion: success without a second failover (backend idempotency honored) | Boundary | — | +| U-12 | Planned promote sends `planned=true` on the failover call | Positive | `TestPromoteVolumePlannedSendsThePlannedFlag` | +| U-13 | Planned promote with no demote ever requested: `FAILED_PRECONDITION`, the one case meant to let the vendored controller's own force-escalation take over | Negative | `TestPromoteVolumePlannedWithNoDemoteIsFailedPrecondition` | +| U-14 | Demote: the demote endpoint is called, and the RPC succeeds only once the backend confirms the final snapshot landed | Positive | `TestDemoteVolumeDone` | +| U-15 | Demote repeated on a demoted volume: success (idempotency) | Boundary | `sbcli: test_demote_is_idempotent_once_done` (backend tier; the driver's `TestDemoteVolumeDone` exercises the same success path) | +| U-16 | Resync: the failback endpoint is called, forwarding the `sourceClusterID` class parameter when given | Positive | `TestResyncVolume`, `TestResyncVolumeSendsTheSourceClusterParameter` | +| U-17 | Resync's `ready` field reflects lag against budget: false while lag exceeds it, true once caught up | Boundary | `TestResyncVolumeNotReadyWhileLagExceedsBudget`, `TestResyncVolumeReadyReflectsLag` | +| U-40 | Planned promote while demote is still converging: `ABORTED` (retryable) — never `FAILED_PRECONDITION`, which the vendored controller auto-escalates to a forced, lossy promote inline with no wait-and-retry grace period of its own | Negative | `TestPromoteVolumePlannedWhileDemoteConvergingIsAborted` | +| U-41 | Demote still converging: `ABORTED` (retryable), non-blocking — the RPC never waits out the backend's own convergence loop | Boundary | `TestDemoteVolumeNotYetDoneIsAborted` | +| U-42 | `demote_lvol` fences the source strictly before triggering the final snapshot, never after (a write landing in the gap would be silently lost) | Positive | `sbcli: test_demote_fences_before_triggering_the_final_snapshot` | +| U-43 | `demote_lvol` re-invoked while pending: checks the marker only, never re-fences or re-triggers | Boundary | `sbcli: test_demote_does_not_refence_or_retrigger_once_pending` | +| U-44 | `demote_lvol` completes once the triggered snapshot carries the replicated marker | Positive | `sbcli: test_demote_completes_once_the_snapshot_carries_the_replicated_marker` | +| U-45 | `demote_lvol` surfaces a snapshot-creation failure without recording pending state | Negative | `sbcli: test_demote_surfaces_a_snapshot_creation_failure` | +| U-46 | The `failover` route's three-way planned-gate branch: demoted proceeds, converging is 409, no relationship is 412 | Positive/Negative | `sbcli: test_planned_failover_proceeds_once_demoted`, `test_planned_failover_while_demote_is_converging_is_409`, `test_planned_failover_without_any_demote_is_412` | +| U-47 | Unplanned failover ignores demote state, unchanged from before P0-3 | Regression | `sbcli: test_unplanned_failover_ignores_demote_state` | +| U-48 | Every verb given the SOURCE side of an existing relationship resolves to the local replica (`TargetLvolId`) before acting — Ramen's S3-restore hands the destination cluster the original source's volumeHandle verbatim | Positive | `TestEnableVolumeReplicationResolvesToTargetWhenGivenTheSourceSideOfARelationship`, `TestDisableVolumeReplicationResolvesToTargetWhenGivenTheSourceSideOfARelationship`, `TestPromoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship`, `TestDemoteVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship` | +| U-49 | Resync resolves the foreign source handle even after the source lvol record itself was reaped (2026-09-24-resync-foreign-handle-404: the only verb without the resolution 404ed on every reconcile and stalled the relocate back) | Regression | `TestResyncVolumeResolvesToTargetWhenGivenTheSourceSideOfARelationship` | +| U-50 | The standalone info read resolves the foreign source handle likewise, so lastSyncTime keeps flowing after the source record is reaped (2026-09-24-resync-foreign-handle-404) | Regression | `TestGetVolumeReplicationInfoResolvesToTargetWhenGivenTheSourceSideOfARelationship` | +| U-51 | Promote while the fail-back's first reverse-replicated snapshot is still in flight: control-plane 409 (converging, retryable), never a stack-traced 500; the same backend message with no replication configured stays 500 | Regression | `sbcli: test_failover_while_the_first_replicated_snapshot_is_in_flight_is_409`, `test_failover_with_no_snapshot_and_no_replication_configured_stays_500` | +| U-52 | DeleteVolume carrying a reaped, foreign handle follows the relationship and removes the retired, non-active members (2026-09-24-delete-foreign-handle-leak); the `Relationship.ActiveLvolID` parse in atlas-lib is part of the same fix | Regression | `TestDeleteVolumeRemovesTheRetiredReplicaBehindAForeignHandle`, `atlas-lib: TestClientGetVolumeReplicationRelationshipParsesTheActiveVolume` | +| U-53 | DeleteVolume never deletes the pairing's ACTIVE volume through a dead source handle — the workload is running on it | Negative | `TestDeleteVolumeNeverDeletesTheActiveReplicaThroughADeadHandle` | +| U-54 | Resolution walks CHAINED pairings to the active volume: after a relocate round trip, verbs on the original handle land on the last hop's clone, never the retired middle hop (2026-09-24-chained-relationship-resolves-one-hop-short) | Regression | `TestPromoteVolumeResolvesAcrossAChainedRelationshipToTheActiveVolume` | +| U-55 | The node's deleted-volume redirect walks the same chain using each record's own target triple, never pairing active_lvol_id with another hop's cluster; the single-pairing migration case is unchanged | Regression | `node: TestRedirectToActiveVolumeWalksAChainedRelationship`, `TestRedirectToActiveVolumeSinglePairingIsUnchanged` | +| U-56 | Promote does NOT attach the class's policy to the promoted volume, even when handed one (2026-09-24-promote-must-not-attach-the-policy: an attach at promote time races the fail-back's re-pointing of the same volume's replication and wedged the relocate on a wrong-LVS clone build); re-protection belongs to the planned-cutover flow | Regression | `TestPromoteVolumeDoesNotAttachThePolicyItWasHanded` | + +### Condition Derivation (design §6.2) + +File: `csi-driver/internal/csi/controller/replication_conditions_test.go` (planned) + +| # | Scenario | Type | Test | +|------|-----------------------------------------------------------------------------------------------------------------|----------|------| +| U-18 | Status `in_sync` and `replicating`: `Completed=True, Degraded=False, Resyncing=False` (lag alone trips nothing) | Positive | — | +| U-19 | Status `degraded` (suspended tasks) and `error` (retries exhausted): `Degraded=True` with the failure detail | Negative | — | +| U-20 | Lag beyond the budget with healthy shipping: `Degraded=True` on the staleness rule | Boundary | — | +| U-21 | Status read reports `resyncing`: `Resyncing=True`; cleared on the next read without it | Positive | — | +| U-22 | Status `not_replicating`, role `none`: reported as disabled, no `Degraded` | Boundary | — | + +### Operator: Preflight and Coexistence (design §7.2, §8) + +Files: `operator/internal/controller/peerclasses_preflight_test.go`, `operator/internal/controller/pvcreplication_controller_test.go` (planned) + +| # | Scenario | Type | Test | +|------|-------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| U-23 | Same-named classes on both clusters with mutually pointing policies: `PeerClassesVerified` on the pair | Positive | — | +| U-24 | A replication-enabled class missing on the peer: `PeerClassesMismatch` naming the class and side | Negative | — | +| U-25 | Named policies exist but do not point at each other's clusters: `PeerClassesMismatch` | Negative | — | +| U-26 | `PVCAnnotationWatcher` skips a PVC whose volume has a `VolumeReplication`: no slot is created, a skip is recorded | Negative | — | +| U-27 | Annotation added and later a `VolumeReplication` appears: the existing slot is not deleted by the adapter, and the enable is refused per the one-owner rule | Boundary | — | + +### csi-addons Identity Service (design §4) + +File: `csi-driver/internal/csi/csiaddons/identity/identity_test.go` + +| # | Scenario | Type | Test | +|------|--------------------------------------------------------------------------------|----------|--------------------------------------------------| +| U-33 | `GetIdentity` returns this driver's name and version | Positive | `TestGetIdentity` | +| U-34 | `GetCapabilities` advertises `VOLUME_REPLICATION` and `CONTROLLER_SERVICE` | Positive | `TestGetCapabilitiesAdvertisesVolumeReplication` | +| U-35 | `Probe` reports ready (a nil `Ready` per the spec, not a stray `true`/`false`) | Positive | `TestProbeReportsReady` | + +### Operator: Sidecar Deployment and RBAC (design §4.1) + +File: `operator/internal/controllers/driver/workloads_test.go`, `operator/internal/controllers/driver/rbac_test.go` + +| # | Scenario | Type | Test | +|------|-----------------------------------------------------------------------------------------------------------------------------------------|----------|---------------------------------------------------------------| +| U-36 | The csi-addons sidecar is appended after the plugin container, never inserted, and is addressed at the plugin's own socket | Positive | `TestCSIAddonsSidecarIsAppliedAfterThePlugin` | +| U-37 | The sidecar advertises its own pod (IP, name, namespace, UID) through the downward API, since the StatefulSet runs on the host network | Positive | `TestCSIAddonsSidecarAdvertisesItsOwnPod` | +| U-38 | The sidecar's grant is a namespaced Role bound to the controller plugin's account, not a ClusterRole, since CSIAddonsNode is namespaced | Positive | `TestCSIAddonsRoleIsNamespacedAndBoundToTheControllerAccount` | +| U-39 | The namespaced Role's rules are scoped to the sidecar's own job: its CSIAddonsNode and its own leader-election Lease | Positive | `TestCSIAddonsRoleRulesAreScopedToItsOwnJob` | + +### VolumeGroupReplication: VolumeGroup service and group-wide replication (design §14, Phase 4 — Planned) + +The driver-driven, `external: false` model (design §14): the plugin's csi-addons VolumeGroup service maps a set of volume handles to the backend consistency group (§14.3), and the existing Replication verbs route a group handle to the new backend group-replication endpoints (§14.4). None of this is built yet, so every row's `Test` is `—` and reappears in §7. Files (planned): `csi-driver/internal/csi/controller/volumegroup_test.go`, `csi-driver/internal/csi/controller/replication_group_test.go`, and the backend tier in `sbcli`. + +U-57 … U-61 are **retired**: they covered the earlier operator-owned `VolumeGroupReplicationReconciler`/`VolumeGroupReplicationValidator`, which design §13 removes in favor of this path. Their Go tests are deleted with the reconciler. + +| # | Scenario | Type | Test | +|----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| ~~U-57~~ | **Retired** (operator fan-in aggregation; reconciler removed, design §13) | — | — | +| ~~U-58~~ | **Retired** (operator fan-in disjunction; reconciler removed) | — | — | +| ~~U-59~~ | **Retired** (operator oldest-`lastSyncTime`; reconciler removed) | — | — | +| ~~U-60~~ | **Retired** (operator webhook admits a whole group; validator removed) | — | — | +| ~~U-61~~ | **Retired** (operator webhook rejects a subset/cross-group selector; validator removed) | — | — | +| U-62 | `CreateVolumeGroup` over the label-formed, co-placed group resolves and returns the existing group's id (idempotent), making no second group | Positive | — | +| U-63 | `CreateVolumeGroup` with a handle that resolves to no consistency group: `FAILED_PRECONDITION` (or the group's own not-found), no group created | Negative | — | +| U-64 | `ModifyVolumeGroupMembership` adding a volume not co-placed on the group's node/LVS: refused (design §14.3, dynamic membership deferred) | Negative | — | +| U-65 | `DeleteVolumeGroup` dissolves the grouping and leaves every member volume intact; repeating it succeeds (idempotent) | Boundary | — | +| U-66 | A group handle on `PromoteVolume`/`DemoteVolume`/`ResyncVolume` routes to the group endpoints; a per-volume handle still takes the §5 path unchanged | Positive | — | +| U-67 | `GetVolumeReplicationInfo` on a group handle returns the group `lastSyncTime` (newest fully replicated group generation), never `NOT_FOUND` for a real group | Positive | — | +| U-68 | Backend group failover clones the last group-snapshot generation for every member atomically; if one member cannot clone, the whole group operation aborts (§14.4) | Boundary | — | +| U-69 | Backend group demote quiesces every member, ships one final group snapshot, confirms it landed, then fences every member | Positive | — | + +--- + +## 2. Integration Tests + +The csi-addons sidecar and controller-manager against the driver with a mock control plane, on a kind or envtest-backed cluster with the vendored CRDs installed. + +### VolumeReplication Lifecycle (design §4, §5) + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-01 | The sidecar probes the driver, advertises `VOLUME_REPLICATION`, and a `CSIAddonsNode` is published | Positive | — | +| I-02 | Creating a `VolumeReplication` with `replicationState: primary` enables replication and the conditions settle at `Completed=True, Degraded=False` | Positive | — | +| I-03 | Flipping to `secondary` drives demote, and back to `primary` drives promote, each waiting on `Completed` | Positive | — | +| I-04 | Deleting the `VolumeReplication` disables replication | Positive | — | +| I-05 | A `VolumeReplicationClass` naming a nonexistent policy: enable fails, the condition carries the message, and the object retries without flapping | Negative | — | +| I-06 | `status.lastSyncTime` advances across reconciles while the mock backend advances its newest replicated snapshot | Positive | — | +| I-07 | Driver restart mid-operation: the re-driven verb is idempotent and the conditions re-derive without manual repair | Boundary | — | + +--- + +## 3. E2E Tests + +Two live simplyblock clusters with the chart-deployed csi-addons machinery. The Ramen VRG rows are the Phase 2 acceptance gate. + +### Adapter Lifecycle (design §5, §6) + +| # | Scenario | Type | Test | +|------|-------------------------------------------------------------------------------------------------------------------------------------------|------------|------| +| E-01 | Enable through a `VolumeReplication`, write data, and `lastSyncTime` advances at the policy cadence | Positive | — | +| E-02 | Forced promote on the DR cluster: the clone serves with the source fenced, and data matches the last replicated generation | Positive | — | +| E-03 | Resync back after a forced promote, then a planned swap (demote, then planned promote): a hashed writer proves zero loss across the swap | Positive | — | +| E-04 | Planned promote refused while lagging, succeeds after `Resyncing` clears | Negative | — | +| E-05 | The typed status read never 404s across the whole lifecycle (enable through swap), and the slot controller's `lastReplicatedAt` tracks it | Regression | — | + +### Ramen-Driven (design §12, Phase 2 gate) + +| # | Scenario | Type | Test | +|------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|---------------------------------------------------------------------------------| +| E-06 | A Ramen `VolumeReplicationGroup` in async mode selects the class, creates one `VolumeReplication` per PVC, and `lastGroupSyncTime` populates | Positive | → [`test-plan-ramen-integration.md`](test-plan-ramen-integration.md) M-01 | +| E-07 | Ramen failover (`force`) and relocate (demote plus planned promote) both complete against a live workload | Positive | → [`test-plan-ramen-integration.md`](test-plan-ramen-integration.md) M-02, M-03 | +| E-08 | A Ramen VRG protects and relocates a multi-volume app through one `VolumeGroupReplication`: all members move together, none diverging, and the group's `status.lastSyncTime` tracks the oldest member (design §14) | Positive | → [`test-plan-ramen-integration.md`](test-plan-ramen-integration.md) M-05 | + +--- + +## 4. Manual Scenarios and Test Concepts + +### M-01 — Demote with writes in flight + +**Design reference:** §5.2, Open Question 1 + +**What to verify:** the demote verb's quiesce behavior when the workload is still writing at the moment of the fence, since Ramen ordinarily unmounts first but nothing guarantees it. + +**Test concept:** +1. Run a continuous writer against a replicated volume. +2. Issue demote without stopping the writer. +3. Verify the final flush lands on the peer, the writer's in-flight I/O fails cleanly (no acknowledged-but-lost write), and a subsequent planned promote on the peer serves every acknowledged write. + +### M-02 — Coexistence under concurrent claims + +**Design reference:** §8 + +**What to verify:** the one-owner rule under a race: the annotation and a `VolumeReplication` claiming the same volume in the same reconcile window must converge on one owner with the other visibly refused, never two attach calls. + +**Test concept:** +1. Apply the PVC annotation and create a `VolumeReplication` for the same PVC near-simultaneously. +2. Verify exactly one path attached, the other surfaced its refusal (event or condition), and the backend saw a single policy attach. + +--- + +## 5. Axis Coverage + +| Axis | Values covered | IDs | Not covered | +|-------------------|-----------------------------------------------------------------------------|-----------------------------------------------------------------------------|----------------------------------------------------------------------------------------------| +| Verb lifecycle | enable, disable, info, forced promote, planned promote, demote, resync | U-01, U-02, U-04 … U-10, U-12 … U-17, U-40 … U-56, I-02 … I-04, E-01 … E-04 | — | +| Idempotency | repeat enable, disable, demote; re-drive after restart | U-02, U-05, U-15, I-07 | repeated promote (U-11), repeated resync | +| Conditions | healthy, degraded, error, staleness, resyncing, disabled | U-18 … U-22, E-05 | condition behavior across backend upgrade | +| Coexistence | slot skip, one-owner refusal, concurrent claim | U-26, U-27, M-02 | migration of an annotated volume onto a VolumeReplication | +| peerClasses | verified, missing class, mispaired policies | U-23 … U-25 | drift after verification | +| Orchestrator | direct kubectl lifecycle, Ramen VRG async (per volume and per group) | I-02 … I-06, E-06, E-07, E-08 | Ramen hub failover of multiple apps | +| Group replication | VolumeGroup service, group-handle routing, backend group ops, live relocate | U-62 … U-69, E-08 | Global (multi-VRG) VGR; the controller-manager group loop at the integration tier (no I-row) | +| Cluster topology | two clusters, one relationship addressed from both sides | E-01 … E-08 | three-cluster (cascaded) topologies | + +--- + +## 6. Coverage Summary + +| Class | Scenarios | Covered | Not covered | +|-------------|-----------|---------|--------------------------------------| +| Unit | 54 | 34 | U-03, U-11, U-18 … U-27, U-62 … U-69 | +| Integration | 7 | 0 | I-01 … I-07 | +| E2E | 8 | 0 | E-01 … E-08 | +| Manual | 2 | 0 | M-01, M-02 | + +Phase 1 landed the driver's Replication and Identity services, the error classifier, and the operator's sidecar and RBAC wiring, covering every Phase 1 unit scenario except U-03 (§7). Phase 2 landed P0-3 (the demote endpoint and the planned gate on `failover`, in sbcli) and the driver's `PromoteVolume`/`DemoteVolume`/`ResyncVolume`, covering every Phase 2 unit scenario except U-11 (§7). Phase 4 (group replication, §14) is Planned: U-62 … U-69 (the driver VolumeGroup service, group-handle routing, and the backend group engine) and E-08 are uncovered, and the earlier operator reconciler's rows (U-57 … U-61) are retired with it. The operator's preflight and coexistence controllers (peerClasses, `PVCReplicationController`) remain a separate, unbuilt subsystem, and neither a sidecar-and-controller-manager integration suite nor a live two-cluster E2E bed exists yet, so those tiers remain fully uncovered. + +--- + +## 7. What Is Not Yet Covered + +| # | Gap | Reason | +|------------------|-------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| U-03 | Different-policy enable refused with `FAILED_PRECONDITION` | Not implemented: sbcli's `attach_policy` silently re-attaches onto the new policy rather than refusing, and no endpoint exposes the policy id a volume is currently attached to, for the driver to compare against before attaching (design §5.1 assumed this refusal exists; it does not against the current backend) | +| U-11 | Forced promote repeated after completion: success without a second failover | Not implemented: the backend's own idempotency (`_failover_volumes`' `already failed_over` skip) is real and exercised transitively, but no driver-level test asserts it directly for `PromoteVolume` | +| U-18 … U-22 | Condition derivation (`Completed`/`Degraded`/`Resyncing`) from the status read | Not implemented: `csi-addons/spec` v0.2.0's `GetVolumeReplicationInfoResponse` carries only `lastSyncTime`, with no per-condition field at all; deriving these needs either a newer spec version or belongs in the controller-manager's own reconcile, neither examined yet | +| U-23 … U-27 | Preflight (`peerClasses` verification) and coexistence (`PVCReplicationController`, the one-owner rule) | Out of Phase 1 and Phase 2's scope: the auto-adapter and preflight webhook are a separate, unbuilt subsystem | +| I-01 … I-07 | The sidecar and controller-manager loop | The driver's Replication and Identity services and the sidecar container now exist (Phase 1); no envtest/kind suite exercises them against the real kubernetes-csi-addons controller-manager yet | +| E-01 … E-05 | The live lifecycle (non-Ramen half) | Needs a two-cluster live test bed. `regression_test/21/` exercises the same lifecycle by hand but is not wired as an automated E2E suite. | +| E-06, E-07, E-08 | The Ramen-driven gate (per-volume and per-group) | Detailed scenario ownership moved to [`test-plan-ramen-integration.md`](test-plan-ramen-integration.md) M-01 … M-03 (per volume) and M-05 (the `VolumeGroupReplication` group relocate), blocked there on a live OCM hub with Ramen installed (see that document's Phase 0) | +| U-62 … U-69 | Group replication (§14): the driver VolumeGroup service, group-handle routing, and the backend group engine | Phase 4 is Planned, not built. P0-6/P0-7 (the `sbcli` group-replication engine and group policy) are the blocking backend work; the driver's VolumeGroup service and group-handle routing depend on them | +| — | Repeated resync, class drift after verification, annotated-volume migration onto the adapter, cascaded topologies | Beyond the first coverage pass, recorded so the gaps are explicit rather than assumed covered | +| M-01, M-02 | Demote under writes; concurrent ownership race | Need failure injection and precise timing a live two-cluster run does not automate yet | diff --git a/operator/docs/tests/test-plan-ramen-integration.md b/operator/docs/tests/test-plan-ramen-integration.md new file mode 100644 index 000000000..210376608 --- /dev/null +++ b/operator/docs/tests/test-plan-ramen-integration.md @@ -0,0 +1,120 @@ +# Test Plan: Ramen Integration + +Related design: [`designs/design-ramen-integration.md`](../designs/design-ramen-integration.md) +Harness: one class. Everything this document specifies needs a live two-cluster simplyblock deployment (`regression_test/21/`), plus an OCM hub with Ramen installed on the hub and both managed clusters (design §6.1). `VolumeGroupReplication` (design §4) is now a driver-and-backend feature specified in `design-csi-addons-replication.md` §14; its unit and backend scenarios live in that document's test plan (U-62 … U-69), not here. + +Scope: this document specifies whether a real Ramen `VolumeReplicationGroup`, driven through a real OCM hub, correctly drives the csi-addons surface `design-csi-addons-replication.md` implements, for both a single volume and a consistency group of them, as manual E2E scenarios (`M-`), since no smaller harness substitutes for a real Ramen reconcile loop against a real OCM-registered cluster pair. The manual scenarios close two rows already carried in [`test-plan-csi-addons-replication.md`](test-plan-csi-addons-replication.md): E-06 and E-07, both `—` in that plan's `Test` column since they were written. Design §3 (peerClasses) adds no scenarios of its own: it confirms that no operator-side code exists to test. + +--- + +## 1. Unit Scenarios + +### VolumeGroupReplication (design §4) + +`VolumeGroupReplication` is no longer an operator reconciler; it is the driver-and-backend, `external: false` path of `design-csi-addons-replication.md` §14. Its unit and backend scenarios are that document's test plan U-62 … U-69, and its live group relocate is that plan's E-08 (which points back to M-05 here). This plan owns no unit scenario for it. + +U-01 … U-05 are **retired**: they covered the operator-owned `VolumeGroupReplicationReconciler`/`VolumeGroupReplicationValidator`, which `design-csi-addons-replication.md` §13 removes. + +| # | Scenario | Type | Test | +|----------|-----------------------------------------------------------------------------------------|------|------| +| ~~U-01~~ | **Retired** (operator fan-in, all members healthy; reconciler removed) | — | — | +| ~~U-02~~ | **Retired** (operator fan-in, one member degraded; reconciler removed) | — | — | +| ~~U-03~~ | **Retired** (operator oldest-`lastSyncTime`; reconciler removed) | — | — | +| ~~U-04~~ | **Retired** (operator webhook admits a whole group; validator removed) | — | — | +| ~~U-05~~ | **Retired** (operator webhook rejects a subset/cross-group selector; validator removed) | — | — | + +--- + +## 2. Manual Scenarios and Test Concepts + +### M-01: Ramen protects a workload, and `VolumeReplication` reflects it + +**Design reference:** design §6.2 step 1. Closes `test-plan-csi-addons-replication.md` E-06. + +**What to verify:** A Ramen `VolumeReplicationGroup` in async mode, given a `DRPlacementControl` protecting a namespace, selects the `VolumeReplicationClass` by its `replicationClassSelector` and `provisioner`, creates exactly one `VolumeReplication` per protected PVC, and that object's `status.lastSyncTime` advances on the policy's ordinary cadence, not just when driven by hand (already proven, `design-csi-addons-replication.md` §12), but when Ramen itself is the caller. + +**Open question:** whether SiteMap authors the `DRPlacementControl` or a hand-authored stand-in does (design §8, Open Question 2). Record which was used. + +**Test concept:** +1. Stand up the topology in design §6.1: two managed clusters, one shared simplyblock control plane, an OCM hub with Ramen installed, a `DRPolicy` naming both clusters. +2. Deploy a workload with a PVC on cluster A under a labeled `StorageClass`/`VolumeReplicationClass` pair (`ramendr.openshift.io/storageid`/`replicationid`, `design-csi-addons-replication.md` §7.1). +3. Create the `DRPlacementControl` protecting the workload's namespace. +4. Assert: exactly one `VolumeReplication` exists, named and owned per Ramen's own convention. Its `status.state` reaches `Primary`. `status.lastSyncTime` is non-nil and advances across at least two policy intervals. The VRG's `DataProtected` condition is `True`. + +### M-02: Ramen planned relocate, demote then promote, driven by the VRG + +**Design reference:** design §6.2 step 2. Closes `test-plan-csi-addons-replication.md` E-07 (relocate half). + +**What to verify:** Ramen's `Relocate` action demotes the source VRG and promotes the target VRG with `force=false`, gated on the source's `PeerReady`, and the workload comes up on the target cluster with the data a hashed writer wrote before relocation: zero loss, the same guarantee `design-csi-addons-replication.md` §5.2 already proves by hand, now proven through Ramen's own reconcile loop. + +**Current behavior:** the demote → planned-promote sequence, the `Aborted`/`FailedPrecondition` split that keeps a converging demote from being force-escalated, and the source-health no-op on an already-settled promote are all implemented and unit/E2E-tested against a hand-driven `VolumeReplication` (`design-csi-addons-replication.md` §5.2, §12). Never yet exercised by Ramen's own controller issuing the calls. + +**Test concept:** +1. From M-01's protected state, write and hash a known data set to the workload's volume. +2. Trigger Ramen's `Relocate` action toward cluster B. +3. Assert, in order: cluster A's `VolumeReplication` reaches `Secondary` with `Completed=True` (demote confirmed). Cluster B's `VolumeReplication` reaches `Primary` with `Completed=True` and no force-promotion event fired (the planned path, not an escalation). The workload is schedulable and serving on cluster B. The hash taken before relocation matches the data read after. + +### M-03: Ramen unplanned failover, force promote, driven by the VRG + +**Design reference:** design §6.2 step 3. Closes `test-plan-csi-addons-replication.md` E-07 (failover half). + +**What to verify:** with cluster A unreachable (not merely demoted), Ramen's `Failover` action promotes cluster B with `force=true`, and the force-escalation behavior `design-csi-addons-replication.md` §5.2 documents (the vendored controller's own "no wait-and-retry grace period") still lands correctly when Ramen, not a test script, drives it. + +**Test concept:** +1. From a protected, healthy state (M-01), make cluster A's storage genuinely unreachable (network partition or node shutdown, not merely a demote). +2. Trigger Ramen's `Failover` action toward cluster B. +3. Assert: cluster B's `VolumeReplication` reaches `Primary`. The promote succeeded via the forced path (a force-promotion event is expected here, unlike M-02). The workload serves on cluster B. + +### M-04: Ramen-driven resync after recovery + +**Design reference:** design §6.2 step 4 (new scope this document adds, with no corresponding row in `test-plan-csi-addons-replication.md`). + +**What to verify:** once cluster A recovers after M-03's failover, Ramen reconciles the diverged copy via `ResyncVolume` without merging or re-triggering a full cutover, matching `design-csi-addons-replication.md` §5.2's "it never merges" guarantee. + +**Test concept:** +1. Restore cluster A's connectivity/node after M-03. +2. Confirm Ramen (not a manual `kubectl patch`) drives the recovered volume's `VolumeReplication` toward `Resyncing=True`, then `Completed=True` once caught up. +3. Assert the resync reconciled by delta, not a full copy (compare shipped bytes against the volume's total size), and that cluster B remains primary throughout. + +### M-05: Ramen protects and relocates a multi-volume app through `VolumeGroupReplication` + +**Design reference:** design §4 and `design-csi-addons-replication.md` §14 (the driver-and-backend group surface), exercised through the topology and test flow §6 defines for the per-volume case. New scope this document adds; it closes `test-plan-csi-addons-replication.md` E-08. It cannot run until Phase 4 (that plan's U-62 … U-69) is built. + +**What to verify:** a VRG whose PVCs share the consistency-group labels and a `StorageClass` carrying `ramendr.openshift.io/groupreplicationid` (and **not** `offloaded`) creates one `VolumeGroupReplication` with `spec.external: false`; the stock kubernetes-csi-addons controller-manager forms the backend group through the driver's `CreateVolumeGroup`, replicates it as one unit addressed by the group handle, and a planned relocate of the whole app moves every member together at one crash-consistent point, none diverging. + +**Test concept:** +1. Extend M-01's topology: a workload with three PVCs sharing one `storage.simplyblock.io/consistency-group` value (and Ramen's own `ramendr.openshift.io/consistency-group`), under a group `StorageClass` carrying `groupreplicationid` and a `VolumeGroupReplicationClass` matching it. +2. Confirm exactly one `VolumeGroupReplication` (`external: false`) and its `VolumeGroupReplicationContent` exist, that the backend consistency group holds all three members, and that the group is replicating (the group `lastSyncTime` advances). +3. Trigger Ramen's `Relocate` action, as in M-02. +4. Assert: the whole group promotes on cluster B and demotes on cluster A as one unit, each member's PVC restored and serving on cluster B, every member's pre-relocate data intact, and the group's `lastSyncTime` tracking one group generation throughout. No member is left behind mid-relocate. + +--- + +## 3. Axis Coverage + +| Axis | Values covered | IDs | Not covered | +|----------------------------------|----------------------------------------------------------------------------|--------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------| +| Group replication (unit/backend) | the driver VolumeGroup service and the backend group engine | (in `test-plan-csi-addons-replication.md` U-62 … U-69) | owned by the csi-addons plan now that group replication is driver-and-backend (design §4), not this plan | +| Orchestrator | Ramen VRG async, hub-driven | M-01 … M-05 | direct `kubectl` lifecycle (already covered in `test-plan-csi-addons-replication.md`) | +| Cluster topology | two managed clusters, one relationship, hub-mediated | M-01 … M-05 | three-cluster (cascaded) topologies. SiteMap-authored `DRPlacementControl` specifically (§8 Open Question 2 may leave this a hand-authored stand-in) | +| Failure mode | planned relocate, unplanned failover, post-recovery resync, group relocate | M-02, M-03, M-04, M-05 | a demote that stalls mid-convergence while Ramen-driven (covered by hand in `test-plan-csi-addons-replication.md` M-01/M-02, not yet by Ramen) | + +--- + +## 4. Coverage Summary + +| Class | Scenarios | Covered | Not covered | +|--------------|-----------|---------|-------------------------------------------------------------------------------------------------------------| +| Unit | 0 | 0 | — (U-01 … U-05 retired; group unit/backend scenarios are `test-plan-csi-addons-replication.md` U-62 … U-69) | +| Manual (E2E) | 5 | 0 | M-01 … M-05 | + +--- + +## 5. What Is Not Yet Covered + +| # | Gap | Reason | +|-------------|-------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| +| M-01 … M-05 | The entire Ramen-driven validation | Blocked on Phase 0 (design §Phase 0): a live OCM hub with Ramen installed across a registered two-cluster pair, not yet confirmed available (§8 Open Question 1). M-05 additionally needs Phase 4 built (`test-plan-csi-addons-replication.md` U-62 … U-69: the driver VolumeGroup service and the backend group engine) and the `VolumeGroupReplication` CRDs installed on both clusters. | +| — | Three-cluster / cascaded topologies | Out of scope for this document, and not part of the gap analysis's Appendix A either | +| — | SiteMap-authored (rather than hand-authored) `DRPlacementControl` | Depends on SiteMap's own availability (§8 Open Question 2) | +| — | Global VGR (multi-VRG consensus) | Out of scope for design §4.1, tracked as design §8 Open Question 4 | diff --git a/operator/docs/tests/test-plan-test-failover.md b/operator/docs/tests/test-plan-test-failover.md new file mode 100644 index 000000000..365467674 --- /dev/null +++ b/operator/docs/tests/test-plan-test-failover.md @@ -0,0 +1,280 @@ +# Test Plan: Non-Disruptive Test Failover + +Related design: [`designs/design-test-failover.md`](../designs/design-test-failover.md) +Harness: [`operator/internal/controller`](../../internal/controller) + +Scope: the operator, the CSI driver, and the Kubernetes surface of this +repository. Control-plane (`sbcli`), SPDK, and OCM behavior is a dependency, +faked at the boundary. See the `test-scenarios` skill. + +Scenario IDs are permanent: `U-` unit (no cluster, pure functions, fake +`client.Client`, mock HTTP), `I-` integration (full reconcile loop against +`envtest`, a mock backend, and a mock OCM), `E-` end-to-end (live clusters, real +data path), `M-` manual (needs failure injection or orchestration not yet +automated). Types are `Positive`, `Negative`, `Boundary`, `Regression`. The +`Test` column names the implementing function, or `—` when the scenario is not +yet covered. Every `—` also appears in §8. + +This design is `Draft` and nothing is implemented, so every `Test` cell is `—`. +The plan is the specification the implementation is written against, and §8 +carries the whole matrix as the gap list until the work lands. + +--- + +## 1. Unit Tests + +Pure helpers and controller methods in `testfailover_controller.go`, covered with +a fake `client.Client` and a mock control-plane HTTP server. Numbering runs +continuously across the groups. + +### Source resolution (design §4.1, §5.1) + +File: `internal/controller/testfailover_unit_test.go` + +| # | Scenario | Type | Test | +|------|------------------------------------------------------------------------------------------------------|----------|------| +| U-01 | `scope: Volume`: a `ManagedClusterView` projection of the source PVC and PV yields the volume handle | Positive | — | +| U-02 | `scope: Group`: `sourceRef` resolves to the group's member volumes, one recovered volume per member | Positive | — | +| U-03 | The view projects nothing for `sourceRef` → clean `Failed`, no panic | Negative | — | +| U-04 | `scope: Group` where the group has zero live members → `Failed`, nothing to recover | Boundary | — | + +### Recovery-point resolution (design §5.2) + +File: `internal/controller/testfailover_unit_test.go` + +| # | Scenario | Type | Test | +|------|------------------------------------------------------------------------------------------------------------------|----------|------| +| U-05 | `bubbleCluster` equals `sourceCluster` → `Failed`, rejected up front (recovery is onto a DIFFERENT cluster only) | Negative | — | +| U-06 | `bubbleCluster` is a DR target → latest replicated snapshot on that backend | Positive | — | +| U-07 | Group drill on a DR target → latest replicated generation, one snapshot per member | Positive | — | +| U-08 | DR target with no replicated point yet → `Failed` at `ResolvingPoint` | Negative | — | + +### Non-disruptiveness fingerprint (design §7.4) + +File: `internal/controller/testfailover_fingerprint_unit_test.go` + +| # | Scenario | Type | Test | +|------|------------------------------------------------------------------------------------------------------------|----------|------| +| U-09 | Source volume and its replication lag unchanged from `ResolvingSource` to `Ready` → `invariantsHeld: true` | Positive | — | +| U-10 | Source PVC rebound to a different volume between captures → drift detected | Negative | — | +| U-11 | A `ReplicationSlot` or VGR state change between captures → drift detected | Negative | — | + +### State machine and idempotency (design §6, §8) + +File: `internal/controller/testfailover_statemachine_unit_test.go` + +| # | Scenario | Type | Test | +|------|--------------------------------------------------------------------------------------|----------|------| +| U-12 | Each step advances to the next on its success condition | Positive | — | +| U-13 | A step whose deadline has passed moves the object to `Failed` | Boundary | — | +| U-14 | Deadline exactly at now is not yet expired, and now plus ε is (strict `>`) | Boundary | — | +| U-15 | Terminal `Ready` re-reconcile is a no-op | Positive | — | +| U-16 | Terminal `Failed` re-reconcile is a no-op | Positive | — | +| U-17 | The snapshot and clone idempotency key is `(test-id, source)`, stable across a retry | Positive | — | + +### Defaults and immutability (design §4.1, §9) + +File: `internal/controller/testfailover_unit_test.go` + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------|----------|------| +| U-18 | `bubbleNamespace` defaults to `bubble` when unset | Boundary | — | +| U-19 | `ttlSeconds` unset → no auto-teardown scheduled | Boundary | — | +| U-20 | `ttlSeconds` set → teardown scheduled at creation time plus the TTL | Positive | — | + +--- + +## 2. Integration Tests + +Run the full controller reconcile loop against a real Kubernetes API via +`envtest`, a mock control-plane HTTP server that records call counts, and a mock +OCM (`ManagedClusterView` and `ManifestWork` with status feedback). + +### Source read and DR-target drill (design §5.1–§5.5, §6) + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-01 | Create → `ResolvingSource` creates a `ManagedClusterView` on `sourceCluster`, reads the PVC handle, resolves the target's replicated snapshot (P0-4), clones it (P0-2), places PV and PVC, reaches `Ready` with a `BubbleReady` event | Positive | — | +| I-02 | `bubbleCluster` equals `sourceCluster` → `Failed` at `ResolvingSource`, message explains a same-cluster drill is unsupported | Negative | — | +| I-03 | `sourceCluster` is not a registered `ManagedCluster` → `Failed` at `ResolvingSource` | Negative | — | +| I-04 | The `ManagedClusterView` projects nothing for `sourceRef` → `Failed` at `ResolvingSource`, ref in `status.message` | Negative | — | +| I-05 | Clone returns 5xx → retried, no state advance, mock shows repeated calls | Negative | — | +| I-06 | Group drill: one group-consistent replicated generation → one PVC per member, all from that point | Positive | — | +| I-07 | Control plane unreachable (connection refused) → requeue with backoff, no partial state committed | Negative | — | + +### DR-target drill via OCM (design §5.5, §7.6) + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-08 | `bubbleCluster` set → the recovery point resolves to the target's latest replicated snapshot (P0-4), the clone is built on the target backend, and the PV and PVC are delivered as a `ManifestWork` to that cluster | Positive | — | +| I-09 | `ManifestWork` status feedback reports the PVC `Bound` → the drill reaches `Ready` | Positive | — | +| I-10 | `bubbleCluster` is not a registered `ManagedCluster` → `Failed` at `Placing` | Negative | — | +| I-11 | Phase 1 drill where replication has landed nothing on the target yet (P0-4 404) → `Failed` at `ResolvingPoint` | Negative | — | +| I-12 | `ManifestWork` never reports Bound before its deadline → `Failed` at `Placing`, clone recorded for reclaim | Negative | — | + +### Restart safety (design §6, §7.4) + +| # | Scenario | Type | Test | +|------|--------------------------------------------------------------------------------------------------------------|----------|------| +| I-13 | Restart at `ResolvingSource` → resumes, does not create a second `ManagedClusterView` | Negative | — | +| I-14 | Restart at `Cloning` after the clone was issued → resumes, clone call count is 1 | Negative | — | +| I-15 | Restart at `Placing` after the `ManifestWork` was created → resumes, does not create a second `ManifestWork` | Negative | — | +| I-16 | Restart at `Releasing` during teardown → resumes the reclaim, does not double-reclaim | Negative | — | + +### Non-disruptiveness guard (design §7.4) + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-17 | The source and its replication lag are unchanged before and after a `Ready` drill, and `invariantsHeld: true` | Positive | — | +| I-18 | Injected source or relationship change between start and `Ready` → `Failed`, `InvariantViolated` event, `invariantsHeld: false` | Negative | — | +| I-19 | The whole drill issues zero mutating calls against the source volume or its relationship | Positive | — | + +### Teardown and finalizer (design §5.7, §6) + +| # | Scenario | Type | Test | +|------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-20 | Delete a `Ready` drill → `TearingDown` deletes the `ManifestWork` and `ManagedClusterView`, reclaims the clone (P0-3), leaves the replicated recovery point alone, removes the finalizer | Positive | — | +| I-21 | A DR-target drill that only resolved a replicated snapshot → teardown reclaims the clone but does NOT delete that snapshot | Positive | — | +| I-22 | Delete a `Failed` drill → teardown still runs, finalizer removed on the failure path | Negative | — | +| I-23 | Reclaim returns non-success → holds `TearingDown`, finalizer retained, `ReclaimPending` event | Negative | — | +| I-24 | Reclaim of an already-gone clone or snapshot returns success (404-as-success), teardown completes | Negative | — | +| I-25 | After teardown, no object carrying the drill's `test-id` label remains | Positive | — | + +### Admission and concurrency (design §4.1, §7.3) + +| # | Scenario | Type | Test | +|------|------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| I-26 | Patch any immutable spec field (`scope`, `sourceCluster`, `sourceNamespace`, `sourceRef`, `bubbleCluster`, `bubbleNamespace`) after creation → admission rejects | Negative | — | +| I-27 | A second drill on the same `(scope, sourceCluster, sourceRef, bubbleCluster)` while the first is active → refused | Negative | — | +| I-28 | Two drills on different sources run independently, neither blocks the other | Positive | — | +| I-29 | RBAC sufficiency: the controller creates a `ManagedClusterView` and a `ManifestWork` without a forbidden verb, and needs no PV/PVC or snapshot-API permission | Positive | — | + +--- + +## 3. E2E Tests + +Run against a live two-cluster DR setup (a source cluster and a DR target). +Data-path rows assert recovered-data correctness (a marker written to the +source), not merely that a PVC bound. + +### Same-cluster rejection (design §2 Non-Goals, §5) + +| # | Scenario | Type | Test | +|------|---------------------------------------------------------------------------------------------------------------------|----------|------| +| E-01 | `bubbleCluster` = the source's own cluster → the drill fails fast at `ResolvingSource`, nothing is cloned or placed | Negative | — | + +### DR-target drill (design §5.5) + +| # | Scenario | Type | Test | +|------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------|------| +| E-03 | `bubbleCluster` = the DR target: the replicated snapshot there is cloned, the bubble PVC is placed on the target via `ManifestWork`, a pod boots on the target, and the recovered marker matches the source | Positive | — | +| E-04 | Across a DR-target drill, the source and the running source-to-target replication lag are unchanged | Positive | — | +| E-05 | Group drill onto the target: every member PVC recovers from one replicated generation and each carries the marker (crash-consistent set) | Positive | — | +| E-06 | Teardown of a DR-target drill deletes the OCM objects and reclaims the clone, and the target shows no leaked volume, and the replicated snapshot it resolved is left intact | Positive | — | +| E-07 | Operator restart mid-drill → a single view, snapshot, clone, and `ManifestWork`, the drill still reaches `Ready` | Negative | — | + +--- + +## 4. Ship-to-Non-Target — Phase 2 (Planned) + +Testable only once P0-6 (on-demand shipping to a backend with no copy) exists +(design §5.6). Type and Test are decided when the phase is scoped. + +| # | Scenario | +|---------|---------------------------------------------------------------------------------------------------------------------------------------| +| U-P2-01 | Recovery-point resolution routes through the `Shipping` step only when the bubble backend holds no copy | +| I-P2-01 | `bubbleCluster` has no replica → `Shipping` calls P0-6, polls the returned handle, then clones on that backend and places the PVC | +| I-P2-02 | Ship handle poll exceeds its deadline → `Failed` at `Shipping` | +| E-P2-01 | Drill onto a third cluster with no replica: the point is shipped, cloned, and a pod boots on the recovered PVC with the marker intact | + +--- + +## 5. Manual Scenarios and Test Concepts + +### M-01 — Non-disruptiveness under sustained production I/O + +**Design reference:** design §7.4, §2 (Goals) + +**What to verify:** a drill run while the source application is actively writing, +and while source-to-target replication is running, disturbs neither. No write is +lost and the replication lag does not regress. + +**Test concept:** +1. Run fio in verify mode against the source workload, with replication to the target active. +2. While it runs, create a `TestFailover` with `bubbleCluster` = the target and let it reach `Ready`. +3. Assert `status.report.invariantsHeld` is true, the source is bound to the same volume, and the replication lag is within normal variance. +4. Tear the drill down and confirm fio still verifies with no errors. + +### M-02 — Replicated recovery point pruned before the clone + +**Design reference:** design §5.2, §10 (Failure Modes) + +**What to verify:** a drill degrades cleanly if the replicated snapshot it +resolved is pruned by retention before the clone runs. + +**Test concept:** +1. Create a DR-target `TestFailover` and let it resolve the replicated recovery point. +2. Prune that snapshot on the target before the clone call. +3. Assert the drill reports `Failed` at `Cloning` with a not-found reason, and that no clone was created. + +### M-03 — Leftover proof with `LIST_VOLUMES` disabled + +**Design reference:** design §5.7, Open Question 3 + +**What to verify:** teardown leaves no leaked backend clone on the bubble +backend, even though the CSI driver does not advertise `LIST_VOLUMES`, so the +Kubernetes-side label enumeration cannot be cross-checked through CSI. + +**Open question:** whether the label enumeration on Kubernetes objects is +sufficient, or a backend enumeration is required (design Open Question 3). + +**Test concept:** +1. Run and tear down a DR-target drill. +2. Enumerate backend volumes and snapshots directly on the target's storage cluster (out of band) and assert none carry the drill's `test-id`. + +--- + +## 6. Axis Coverage + +| Axis | Values covered | IDs | Not covered | +|------------------------------|--------------------------------------------------------------------------------------|------------------------------------|--------------------------------------------------------| +| Where the bubble runs | DR target; same-cluster rejected | I-01, I-08, E-03; U-05, I-02, E-01 | non-target cluster (Phase 2: I-P2-01, E-P2-01) | +| Source location | read on a managed cluster via ManagedClusterView | U-01, I-01 | source on the hub's own self-managed cluster | +| Scope | Volume, Group | U-01, U-02, I-06, E-05 | — | +| Recovery point | replicated (volume and group) | U-06, U-07 | shipped (Phase 2) | +| Cross-cluster transport | ManagedClusterView read, ManifestWork write | I-01, I-08 | — | +| Lifecycle / restart | mid-step restart (each step), delete mid-drill, TTL teardown | I-13, I-15, I-20, U-20 | control-plane restart mid-call | +| Control-plane / OCM response | 404, 5xx, connection refused, ManifestWork timeout, idempotent retry, 404-as-success | I-04, I-05, I-07, I-12, I-14, I-24 | partial-write then crash | +| Concurrency | same source, different sources | I-27, I-28 | spec mutated mid-drill (blocked by immutability, I-26) | +| Data correctness | recovered marker, source untouched, crash-consistent group | E-03, E-04, E-05 | recovered-data checksum under load (M-01) | + +--- + +## 7. Coverage Summary + +| Class | Scenarios | Covered | Not covered | +|-------------------|-----------|---------|------------------------------------| +| Unit | 20 | 0 | U-01 … U-20 | +| Integration | 29 | 0 | I-01 … I-29 | +| E2E | 6 | 0 | E-01, E-03 … E-07 | +| Manual | 3 | 0 | M-01 … M-03 | +| Phase 2 (planned) | 4 | 0 | U-P2-01, I-P2-01, I-P2-02, E-P2-01 | + +Nothing is covered: the design is `Draft` and the CRD and controller are unbuilt. +Phase 1 has no unbuilt backend dependency, so its scenarios become implementable +as soon as the controller exists; Phase 2 waits on P0-6. + +--- + +## 8. What Is Not Yet Covered + +| # | Gap | Reason | +|------------------------------------|----------------------------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------| +| U-01 … U-20 | All unit scenarios | The `TestFailover` type and controller do not exist yet | +| I-01 … I-29 | All integration scenarios | The controller, its `envtest` suite, and the mock OCM are unwritten | +| E-01 … E-07 | All E2E scenarios | Needs a live two-cluster DR setup and the shipped feature | +| M-01 … M-03 | All manual scenarios | Need the shipped feature plus failure injection (snapshot delete, backend enumeration) | +| U-P2-01, I-P2-01, I-P2-02, E-P2-01 | Ship-to-non-target | Blocked on P0-6 (on-demand shipping to a backend with no copy), which does not exist | +| — | Source on the hub's own self-managed cluster | An edge of the topology axis. The primary path reads the source on a managed cluster, and the self-managed case is not exercised separately | +| — | Partial-write then crash on the backend | The backend verbs are the idempotency boundary, and asserting a mid-write crash needs backend fault injection this repository's harness lacks | +| — | Recovered-data checksum under sustained load | Covered as a manual concept (M-01), and not automatable without a live-cluster fio harness | diff --git a/operator/go.mod b/operator/go.mod index 1c8c8254a..f13ee1b86 100644 --- a/operator/go.mod +++ b/operator/go.mod @@ -30,6 +30,7 @@ require ( k8s.io/component-base v0.36.2 k8s.io/kube-openapi v0.0.0-20260317180543-43fb72c5454a k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2 + open-cluster-management.io/api v0.15.0 sigs.k8s.io/controller-runtime v0.24.1 sigs.k8s.io/yaml v1.6.0 ) diff --git a/operator/go.sum b/operator/go.sum index 229bc266f..460407f26 100644 --- a/operator/go.sum +++ b/operator/go.sum @@ -349,8 +349,9 @@ github.com/oasdiff/yaml3 v0.0.14/go.mod h1:csto2xfDjYccdUn/yw/bPjj/cYTdp6HtFA0J4 github.com/onsi/ginkgo v1.6.0/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE= github.com/onsi/ginkgo v1.10.2/go.mod h1:lLunBs/Ym6LB5Z9jYTR76FiuTmxDTDusOGeTQH+WWjE= github.com/onsi/ginkgo v1.12.1/go.mod h1:zj2OWP4+oCPe1qIXoGWkgMRwljMUYCdkwsT2108oapk= -github.com/onsi/ginkgo v1.16.4 h1:29JGrr5oVBm5ulCWet69zQkzWipVXIol6ygQUe/EzNc= github.com/onsi/ginkgo v1.16.4/go.mod h1:dX+/inL/fNMqNlz0e9LfyB9TswhZpCVdJM/Z6Vvnwo0= +github.com/onsi/ginkgo v1.16.5 h1:8xi0RTUf59SOSfEtZMvwTvXYMzG4gV23XVHOZiXNtnE= +github.com/onsi/ginkgo v1.16.5/go.mod h1:+E8gABHa3K6zRBolWtd+ROzc/U5bkGt0FwiG042wbpU= github.com/onsi/ginkgo/v2 v2.1.3/go.mod h1:vw5CSIxN1JObi/U8gcbwft7ZxR2dgaR70JSE3/PpL4c= github.com/onsi/ginkgo/v2 v2.28.1 h1:S4hj+HbZp40fNKuLUQOYLDgZLwNUVn19N3Atb98NCyI= github.com/onsi/ginkgo/v2 v2.28.1/go.mod h1:CLtbVInNckU3/+gC8LzkGUb9oF+e8W8TdUsxPwvdOgE= @@ -687,6 +688,8 @@ k8s.io/streaming v0.36.2 h1:NSKthPPg9UFSKsRauVJUVGH2Dvn8fhKmY4qrMkw/p98= k8s.io/streaming v0.36.2/go.mod h1:z6fV3D+NVkoeqRMtWwlUZK6U17SY/LqNzOxWL6GyR/s= k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2 h1:AZYQSJemyQB5eRxqcPky+/7EdBj0xi3g0ZcxxJ7vbWU= k8s.io/utils v0.0.0-20260210185600-b8788abfbbc2/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= +open-cluster-management.io/api v0.15.0 h1:lRee1KOlGHZb2scTA7ff9E9Fxt2hJc7jpkHnaCbvkOU= +open-cluster-management.io/api v0.15.0/go.mod h1:9erZEWEn4bEqh0nIX2wA7f/s3KCuFycQdBrPrRzi0QM= oras.land/oras-go/v2 v2.6.1 h1:bonOEkjLfp8tt6qXWRRWP6p1F+9octchOf2EqnWB4Zs= oras.land/oras-go/v2 v2.6.1/go.mod h1:dhtFrFOuZuDtAVeZ9FUnaa5zfzplG3ZnFX9/uH1J/Yk= sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.34.0 h1:hSfpvjjTQXQY2Fol2CS0QHMNs/WI1MOSGzCm1KhM5ec= diff --git a/operator/internal/autoplacement/logical_volume_selector.go b/operator/internal/autoplacement/logical_volume_selector.go index 2a07a174e..2ab3804f8 100644 --- a/operator/internal/autoplacement/logical_volume_selector.go +++ b/operator/internal/autoplacement/logical_volume_selector.go @@ -235,8 +235,15 @@ func (lvs *LogicalVolumeSelector) selectMigrationSet(ranked []RankedCandidate, m return out } +// consistencyGroupLabel marks a PVC whose volume is a consistency-group member. +const consistencyGroupLabel = "storage.simplyblock.io/consistency-group" + // BuildPinnedSet returns the set of volume UUIDs whose bound PVC carries a pin -// annotation (see kube.IsPinnedVolume). It scans all PersistentVolumes and +// annotation (see kube.IsPinnedVolume) or the consistency-group label. A +// group's members live on one node/LVS, so the rebalancer must not move one +// of them alone; it skips them (the control plane refuses such a migration +// too), and a group moves only as a whole through the group migration +// (co-location design §3). It scans all PersistentVolumes and // resolves the volume UUID from the CSI volume handle // ("::"). Pass an empty clusterUUID to include // volumes from all clusters. @@ -270,7 +277,7 @@ func (lvs *LogicalVolumeSelector) BuildPinnedSet(ctx context.Context, clusterUUI }, pvc); err != nil { continue } - if atlaskube.IsPinnedVolume(pvc.Annotations) { + if atlaskube.IsPinnedVolume(pvc.Annotations) || pvc.Labels[consistencyGroupLabel] != "" { pinned[lvolID] = true } } diff --git a/operator/internal/autoplacement/pinned_set_test.go b/operator/internal/autoplacement/pinned_set_test.go new file mode 100644 index 000000000..46be610ff --- /dev/null +++ b/operator/internal/autoplacement/pinned_set_test.go @@ -0,0 +1,55 @@ +package autoplacement + +import ( + "context" + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + atlaskube "github.com/simplyblock/atlas/kube" +) + +func boundPV(name, handle, claim string) *corev1.PersistentVolume { + pv := csiPV(name, handle, "sc") + pv.Spec.ClaimRef = &corev1.ObjectReference{Name: claim, Namespace: "app"} + return pv +} + +func claim(name string, labels, annotations map[string]string) *corev1.PersistentVolumeClaim { + return &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{ + Name: name, Namespace: "app", Labels: labels, Annotations: annotations, + }} +} + +// A consistency group's members live on one node/LVS: the rebalancer moving +// one of them alone would split the group, so a member is treated like a +// pinned volume and skipped. +func TestBuildPinnedSetSkipsConsistencyGroupMembers(t *testing.T) { + objects := []client.Object{ + boundPV("pv-member", "cluster-a:pool:vol-member", "member"), + boundPV("pv-plain", "cluster-a:pool:vol-plain", "plain"), + boundPV("pv-pinned", "cluster-a:pool:vol-pinned", "pinned"), + claim("member", map[string]string{consistencyGroupLabel: "db"}, nil), + claim("plain", nil, nil), + claim("pinned", nil, map[string]string{atlaskube.AnnoSelectedStorageNode: "node-1"}), + } + cl := fake.NewClientBuilder().WithScheme(namespacedTestScheme(t)).WithObjects(objects...).Build() + lvs := NewLogicalVolumeSelector(nil, cl, nil) + + got, err := lvs.BuildPinnedSet(context.Background(), "cluster-a") + if err != nil { + t.Fatalf("BuildPinnedSet: %v", err) + } + if !got["vol-member"] { + t.Error("a consistency-group member must be skipped by the rebalancer") + } + if !got["vol-pinned"] { + t.Error("a pinned volume must stay skipped") + } + if got["vol-plain"] { + t.Error("an ordinary volume must stay a rebalancing candidate") + } +} diff --git a/operator/internal/controller/replicationpair_controller.go b/operator/internal/controller/replicationpair_controller.go index 20b8909ce..92e2a8d25 100644 --- a/operator/internal/controller/replicationpair_controller.go +++ b/operator/internal/controller/replicationpair_controller.go @@ -23,6 +23,7 @@ import ( "net/http" "time" + corev1 "k8s.io/api/core/v1" "k8s.io/apimachinery/pkg/runtime" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" @@ -30,6 +31,7 @@ import ( logf "sigs.k8s.io/controller-runtime/pkg/log" simplyblockv1alpha1 "github.com/simplyblock/simplyblock-operator/api/v1alpha1" + "github.com/simplyblock/simplyblock-operator/internal/controllers/controlplane" "github.com/simplyblock/simplyblock-operator/internal/utils" "github.com/simplyblock/simplyblock-operator/internal/webapi" ) @@ -46,6 +48,52 @@ const ( type ReplicationPairReconciler struct { client.Client Scheme *runtime.Scheme + + // EndpointResolver answers where the control plane currently is, resolved + // per reconcile the same way internal/controllers/pool's StoragePoolReconciler + // does. Nil means the default/SIMPLYBLOCK_WEBAPI_BASE_URL client is the only + // one, which is what a standalone deployment and every pre-existing test + // still get. Without this, a ReplicationPair reconciled on a + // ControlPlane.spec.source.managed member cluster can never reach its + // control plane: webapi.NewClient() defaults to a Service that cluster + // never runs (confirmed live: cross-cluster ReplicationPair authoring + // failed with "dial tcp: lookup simplyblock-webappapi ... no such host"). + EndpointResolver controlplane.EndpointResolver +} + +// apiClient resolves the control-plane client for one reconcile call, through +// EndpointResolver when set, exactly as StoragePoolReconciler.apiClient does. +func (r *ReplicationPairReconciler) apiClient(ctx context.Context) *webapi.Client { + if r.EndpointResolver == nil { + return webapi.NewClient() + } + + if endpoint := r.EndpointResolver(ctx); endpoint != "" { + return webapi.NewClient(endpoint) + } + + return webapi.NewClient() +} + +// clusterSecret reads the credential StorageClusterReconciler.persist wrote +// for the named local StorageCluster, so a call authenticates as that +// cluster instead of as this operator's own Kubernetes identity -- the only +// way to reach a control plane a different Kubernetes cluster runs, since a +// TokenReview can never cross that boundary. Mirrors +// internal/controllers/pool's identically named method (and +// internal/controllers/cluster's, internal/controllers/node's). +func (r *ReplicationPairReconciler) clusterSecret( + ctx context.Context, namespace, clusterName string, +) (string, error) { + var secret corev1.Secret + key := client.ObjectKey{ + Name: fmt.Sprintf("simplyblock-cluster-%s", clusterName), + Namespace: namespace, + } + if err := r.Get(ctx, key, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil } // +kubebuilder:rbac:groups=storage.simplyblock.io,resources=replicationpairs,verbs=get;list;watch;create;update;patch;delete @@ -62,7 +110,11 @@ func (r *ReplicationPairReconciler) Reconcile(ctx context.Context, req ctrl.Requ return ctrl.Result{}, client.IgnoreNotFound(err) } - apiClient := webapi.NewClient() + if secret, err := r.clusterSecret(ctx, pair.Namespace, pair.Spec.SourceCluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + apiClient := r.apiClient(ctx) if !pair.DeletionTimestamp.IsZero() { return r.reconcileDelete(ctx, &pair, apiClient) diff --git a/operator/internal/controller/replicationpair_controller_unit_test.go b/operator/internal/controller/replicationpair_controller_unit_test.go index 2ec8d1fde..68fdf7cd2 100644 --- a/operator/internal/controller/replicationpair_controller_unit_test.go +++ b/operator/internal/controller/replicationpair_controller_unit_test.go @@ -159,6 +159,100 @@ func TestSitePair_CreatesBackendTarget(t *testing.T) { } } +// ---------- EndpointResolver overrides the default/env-var endpoint ---------- + +// TestSitePair_UsesEndpointResolver proves the reconciler reaches the control +// plane through EndpointResolver, the same mechanism StoragePoolReconciler +// already uses (controlplane.NewEndpointResolver, wired from +// ControlPlane.status.endpoint) -- required for a managed cluster whose own +// operator has no reachable simplyblock-webappapi Service, only the hub's +// externally-published one. SIMPLYBLOCK_WEBAPI_BASE_URL is deliberately left +// unset here so a pass proves the resolver path, not the pre-existing +// env-var fallback TestSitePair_CreatesBackendTarget already covers. +func TestSitePair_UsesEndpointResolver(t *testing.T) { + cluster1 := testCluster("default", "cluster1", "src-uuid") + cluster2 := testCluster("default", "cluster2", "tgt-uuid") + pair := newSitePair() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + if req.Method == http.MethodGet { + w.WriteHeader(http.StatusOK) + _, _ = w.Write([]byte(`[]`)) + return + } + if req.Method == http.MethodPost { + w.WriteHeader(http.StatusCreated) + resp, _ := json.Marshal(map[string]string{"id": "tgt-backend-uuid"}) + _, _ = w.Write(resp) + return + } + w.WriteHeader(http.StatusMethodNotAllowed) + }) + + r, cl := newSitePairReconciler(t, cluster1, cluster2, pair) + r.EndpointResolver = func(context.Context) string { return srv.URL } + + res, err := r.Reconcile(context.Background(), sitePairRequest("pair1")) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if res.RequeueAfter != replPairSyncInterval { + t.Errorf("RequeueAfter = %v, want %v", res.RequeueAfter, replPairSyncInterval) + } + + got := getSitePair(t, cl) + if !got.Status.Ready { + t.Errorf("pair.Status.Ready = false, want true (EndpointResolver should have been used)") + } + if got.Status.BackendTargetID != "tgt-backend-uuid" { + t.Errorf("BackendTargetID = %q, want tgt-backend-uuid", got.Status.BackendTargetID) + } +} + +// TestSitePair_AuthenticatesAsSourceClusterOnceSecretIsKnown mirrors +// internal/controllers/pool's TestAPoolAuthenticatesAsItsClusterOnceTheSecretIsKnown +// -- same fix, same reason: cluster B's operator authenticating as its own +// Kubernetes identity can never pass a TokenReview on the hub's cluster (the +// whole reason per-cluster secrets exist at all), confirmed live this session +// as the very next error once EndpointResolver alone let the request reach +// the right endpoint ("status 401: Invalid token"). +func TestSitePair_AuthenticatesAsSourceClusterOnceSecretIsKnown(t *testing.T) { + cluster1 := testCluster("default", "cluster1", "src-uuid") + cluster2 := testCluster("default", "cluster2", "tgt-uuid") + pair := newSitePair() + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{ + Name: "simplyblock-cluster-cluster1", + Namespace: "default", + }, + Data: map[string][]byte{"secret": []byte("cluster1-own-secret")}, + } + + var gotAuth string + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + gotAuth = req.Header.Get("Authorization") + if req.Method == http.MethodGet { + w.WriteHeader(http.StatusOK) + _, _ = w.Write([]byte(`[]`)) + return + } + w.WriteHeader(http.StatusCreated) + resp, _ := json.Marshal(map[string]string{"id": "tgt-backend-uuid"}) + _, _ = w.Write(resp) + }) + + r, _ := newSitePairReconciler(t, cluster1, cluster2, pair, secret) + r.EndpointResolver = func(context.Context) string { return srv.URL } + + if _, err := r.Reconcile(context.Background(), sitePairRequest("pair1")); err != nil { + t.Fatalf("unexpected error: %v", err) + } + + if gotAuth != "Bearer cluster1-own-secret" { + t.Errorf("authorization = %q, want the source cluster's own secret", gotAuth) + } +} + // ---------- backend target already exists → reuses it ---------- func TestSitePair_ReuseExistingTarget(t *testing.T) { diff --git a/operator/internal/controller/replicationpolicy_controller.go b/operator/internal/controller/replicationpolicy_controller.go index d355b8f56..e5ae43fac 100644 --- a/operator/internal/controller/replicationpolicy_controller.go +++ b/operator/internal/controller/replicationpolicy_controller.go @@ -24,6 +24,7 @@ import ( "net/http" "time" + corev1 "k8s.io/api/core/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/types" @@ -35,6 +36,7 @@ import ( "sigs.k8s.io/controller-runtime/pkg/reconcile" simplyblockv1alpha1 "github.com/simplyblock/simplyblock-operator/api/v1alpha1" + "github.com/simplyblock/simplyblock-operator/internal/controllers/controlplane" "github.com/simplyblock/simplyblock-operator/internal/utils" "github.com/simplyblock/simplyblock-operator/internal/webapi" ) @@ -63,6 +65,44 @@ type idResponse struct { type ReplicationPolicyReconciler struct { client.Client Scheme *runtime.Scheme + + // EndpointResolver answers where the control plane currently is -- see + // ReplicationPairReconciler's identical field (replicationpair_controller.go) + // for why this exists: without it, a ReplicationPolicy reconciled on a + // ControlPlane.spec.source.managed member cluster can never reach its + // control plane. + EndpointResolver controlplane.EndpointResolver +} + +// apiClient resolves the control-plane client for one reconcile call, through +// EndpointResolver when set. Mirrors ReplicationPairReconciler.apiClient. +func (r *ReplicationPolicyReconciler) apiClient(ctx context.Context) *webapi.Client { + if r.EndpointResolver == nil { + return webapi.NewClient() + } + + if endpoint := r.EndpointResolver(ctx); endpoint != "" { + return webapi.NewClient(endpoint) + } + + return webapi.NewClient() +} + +// clusterSecret reads the credential StorageClusterReconciler.persist wrote +// for the named local StorageCluster. Mirrors +// ReplicationPairReconciler.clusterSecret. +func (r *ReplicationPolicyReconciler) clusterSecret( + ctx context.Context, namespace, clusterName string, +) (string, error) { + var secret corev1.Secret + key := client.ObjectKey{ + Name: fmt.Sprintf("simplyblock-cluster-%s", clusterName), + Namespace: namespace, + } + if err := r.Get(ctx, key, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil } // +kubebuilder:rbac:groups=storage.simplyblock.io,resources=replicationpolicies,verbs=get;list;watch;create;update;patch;delete @@ -101,7 +141,11 @@ func (r *ReplicationPolicyReconciler) Reconcile(ctx context.Context, req ctrl.Re return ctrl.Result{RequeueAfter: 10 * time.Second}, nil } - apiClient := webapi.NewClient() + if secret, err := r.clusterSecret(ctx, policy.Namespace, pair.Spec.SourceCluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + apiClient := r.apiClient(ctx) if !policy.DeletionTimestamp.IsZero() { return r.reconcileDelete(ctx, &policy, apiClient, clusterUUID) diff --git a/operator/internal/controller/replicationpolicy_controller_unit_test.go b/operator/internal/controller/replicationpolicy_controller_unit_test.go index 41e07dc52..0ddd9a8b2 100644 --- a/operator/internal/controller/replicationpolicy_controller_unit_test.go +++ b/operator/internal/controller/replicationpolicy_controller_unit_test.go @@ -183,6 +183,95 @@ func TestPolicy_CreatesBackendPolicy_WhenAbsent(t *testing.T) { } } +// ---------- EndpointResolver overrides the default/env-var endpoint ---------- + +// TestPolicy_UsesEndpointResolver mirrors +// TestSitePair_UsesEndpointResolver (replicationpair_controller_unit_test.go): +// same bug, same fix, same reason -- a ReplicationPolicy reconciled on a +// ControlPlane.spec.source.managed member cluster needs the hub's externally +// published endpoint, not the in-cluster Service this reconciler otherwise +// defaults to. SIMPLYBLOCK_WEBAPI_BASE_URL is deliberately left unset so a +// pass proves the resolver path, not the env-var fallback +// TestPolicy_CreatesBackendPolicy_WhenAbsent already covers. +func TestPolicy_UsesEndpointResolver(t *testing.T) { + pair := readyPairForPolicy() + policy := &simplyblockv1alpha1.ReplicationPolicy{ + ObjectMeta: metav1.ObjectMeta{ + Name: "pol", Namespace: "default", + Finalizers: []string{utils.FinalizerReplicationPolicy}, + }, + Spec: simplyblockv1alpha1.ReplicationPolicySpec{PairRef: "pair1"}, + } + r, cl := newPolicyReconciler(t, pair, policy) + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + switch { + case req.Method == http.MethodGet && req.URL.Path == apiPathReplicationPolicies: + writeJSON(w, []interface{}{}) + case req.Method == http.MethodPost && req.URL.Path == apiPathReplicationPolicies: + writeJSON(w, map[string]string{"id": "pol-backend-uuid"}) + default: + w.WriteHeader(http.StatusOK) + } + }) + r.EndpointResolver = func(context.Context) string { return srv.URL } + + _, err := r.Reconcile(context.Background(), policyRequest("pol")) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + + got := getPolicy(t, cl) + if got.Status.BackendPolicyID != "pol-backend-uuid" { + t.Errorf("BackendPolicyID = %q, want pol-backend-uuid (EndpointResolver should have been used)", got.Status.BackendPolicyID) + } +} + +// TestPolicy_AuthenticatesAsSourceClusterOnceSecretIsKnown mirrors +// replicationpair_controller_unit_test.go's identically named test -- same +// bug, same fix, same reason: authenticating as this operator's own +// Kubernetes identity can never pass a TokenReview on the hub's cluster. +func TestPolicy_AuthenticatesAsSourceClusterOnceSecretIsKnown(t *testing.T) { + pair := readyPairForPolicy() + policy := &simplyblockv1alpha1.ReplicationPolicy{ + ObjectMeta: metav1.ObjectMeta{ + Name: "pol", Namespace: "default", + Finalizers: []string{utils.FinalizerReplicationPolicy}, + }, + Spec: simplyblockv1alpha1.ReplicationPolicySpec{PairRef: "pair1"}, + } + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{ + Name: "simplyblock-cluster-" + testClusterName, + Namespace: "default", + }, + Data: map[string][]byte{"secret": []byte("cluster-own-secret")}, + } + r, _ := newPolicyReconciler(t, pair, policy, secret) + + var gotAuth string + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + gotAuth = req.Header.Get("Authorization") + switch { + case req.Method == http.MethodGet && req.URL.Path == apiPathReplicationPolicies: + writeJSON(w, []interface{}{}) + case req.Method == http.MethodPost && req.URL.Path == apiPathReplicationPolicies: + writeJSON(w, map[string]string{"id": "pol-backend-uuid"}) + default: + w.WriteHeader(http.StatusOK) + } + }) + r.EndpointResolver = func(context.Context) string { return srv.URL } + + if _, err := r.Reconcile(context.Background(), policyRequest("pol")); err != nil { + t.Fatalf("unexpected error: %v", err) + } + + if gotAuth != "Bearer cluster-own-secret" { + t.Errorf("authorization = %q, want the source cluster's own secret", gotAuth) + } +} + // ---------- reuse existing backend policy ---------- func TestPolicy_ReusesExistingBackendPolicy(t *testing.T) { diff --git a/operator/internal/controller/storagesitedeployment_controller.go b/operator/internal/controller/storagesitedeployment_controller.go new file mode 100644 index 000000000..87108d4fc --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller.go @@ -0,0 +1,723 @@ +// The StorageSiteDeployment controller carries a managed site's storage +// deployment request from the hub to the site and projects the site's answer +// back. +// +// It never holds a site kubeconfig: every write to the site is a ManifestWork +// in the site's hub namespace, every read a ManagedClusterView there, the same +// two primitives the TestFailover controller uses. The work carries the +// OperatorOps discovery first; once the site has written a draft with nodes, it +// carries a server-side apply of the draft's sizing, and when the request is +// approved, the draft's approval. The views project the draft, the +// StorageCluster the approved draft expands into, and that cluster's nodes. +// +// The request withdraws nothing on deletion: the work is released with its +// resources orphaned, so a storage cluster is never torn down by deleting the +// request that asked for it. See docs/design/control-center-managed-discovery.md +// in the simplyblock-dr repository. + +package controller + +import ( + "context" + "crypto/sha256" + "encoding/json" + "fmt" + "sort" + "time" + + corev1 "k8s.io/api/core/v1" + apiequality "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/client-go/tools/events" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + logf "sigs.k8s.io/controller-runtime/pkg/log" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/pool" +) + +// storageSiteDeploymentIDLabel tags the work and the views of one request, so +// its release can enumerate them. +const storageSiteDeploymentIDLabel = "storage.simplyblock.io/site-deployment" + +// finalizerStorageSiteDeployment holds the request until its work and views +// are released. The work is released with its resources orphaned: the +// discovery, the draft and the storage cluster stay on the site. +const finalizerStorageSiteDeployment = "storage.simplyblock.io/storagesitedeployment-release" + +// hubDeployFieldManager is the field manager the work agent applies the +// draft's sizing and approval with, so the discovery's own fields on the draft +// are left to their owner. +const hubDeployFieldManager = "hub-deploy" + +// storageSiteDeploymentRequeue is how long the reconcile waits before reading +// the site's views again while the request is in progress. +const storageSiteDeploymentRequeue = 15 * time.Second + +// storageSiteDeploymentOnlineRequeue keeps an Online request's projection of +// the storage cluster fresh without polling the site hard. +const storageSiteDeploymentOnlineRequeue = 2 * time.Minute + +// maxStorageSiteNodeViews bounds the per-node views a request keeps: list +// views are not supported by ManagedClusterView, so there is one per node +// named in the draft's nodeRefs. +const maxStorageSiteNodeViews = 64 + +// annotationStorageClusterDefaultPool is the annotation the StorageCluster +// controller records the cluster's first pool under (controllers/cluster). +const annotationStorageClusterDefaultPool = "storage.simplyblock.io/default-pool" + +// Conditions of a request. +const ( + ConditionStorageSiteDelivered = "Delivered" + ConditionStorageSiteDiscovered = "Discovered" + ConditionStorageSiteApproved = "Approved" + ConditionStorageSiteReady = "Ready" +) + +// StorageSiteDeploymentReconciler reconciles a StorageSiteDeployment object. +type StorageSiteDeploymentReconciler struct { + client.Client + Scheme *runtime.Scheme + Recorder events.EventRecorder +} + +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/status,verbs=get;update;patch +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storagesitedeployments/finalizers,verbs=update +// +kubebuilder:rbac:groups=work.open-cluster-management.io,resources=manifestworks,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=view.open-cluster-management.io,resources=managedclusterviews,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=cluster.open-cluster-management.io,resources=managedclusters,verbs=get;list;watch +// +kubebuilder:rbac:groups=events.k8s.io,resources=events,verbs=create;patch + +// Reconcile carries the request to the site and projects the site's answer: +// it ensures the finalizer and the work, reads the draft, the storage cluster +// and its nodes through views, and derives the phase from what they report. +func (r *StorageSiteDeploymentReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + log := logf.FromContext(ctx) + + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := r.Get(ctx, req.NamespacedName, &sd); err != nil { + return ctrl.Result{}, client.IgnoreNotFound(err) + } + + if !sd.DeletionTimestamp.IsZero() { + return r.reconcileDeletion(ctx, &sd) + } + if !controllerutil.ContainsFinalizer(&sd, finalizerStorageSiteDeployment) { + controllerutil.AddFinalizer(&sd, finalizerStorageSiteDeployment) + if err := r.Update(ctx, &sd); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{Requeue: true}, nil + } + + work, err := r.ensureWork(ctx, &sd) + if err != nil { + log.Error(err, "ensure ManifestWork") + return ctrl.Result{}, err + } + return r.project(ctx, &sd, work) +} + +// ensureWork creates or updates the request's ManifestWork: the discovery +// always, and the draft's sizing and approval once the site has a draft with +// nodes to apply them to. +func (r *StorageSiteDeploymentReconciler) ensureWork(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + want, err := r.manifestWork(sd) + if err != nil { + return nil, err + } + var have workv1.ManifestWork + err = r.Get(ctx, client.ObjectKeyFromObject(want), &have) + if apierrors.IsNotFound(err) { + if err := r.Create(ctx, want); err != nil { + return nil, err + } + return want, nil + } + if err != nil { + return nil, err + } + if !apiequality.Semantic.DeepEqual(have.Spec, want.Spec) || !apiequality.Semantic.DeepEqual(have.Labels, want.Labels) { + have.Spec = want.Spec + have.Labels = want.Labels + if err := r.Update(ctx, &have); err != nil { + return nil, err + } + } + return &have, nil +} + +// manifestWork builds the request's work as it should be now. Its resources +// are orphaned on delete: the discovery, the draft and what it expanded into +// are the site's, and deleting the request must not take them away. +func (r *StorageSiteDeploymentReconciler) manifestWork(sd *simplyblockv1alpha2.StorageSiteDeployment) (*workv1.ManifestWork, error) { + ns := siteNamespace(sd) + draft := draftName(sd) + labels := map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)} + + ops := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "OperatorOps", + "metadata": map[string]any{"name": discoveryName(sd), "namespace": ns, "labels": labels}, + "spec": map[string]any{ + "action": string(simplyblockv1alpha2.OperatorOpsActionDiscover), + "discover": discoverSpec(sd), + }, + } + manifests := []workv1.Manifest{} + var configs []workv1.ManifestConfigOption + raw, err := json.Marshal(ops) + if err != nil { + return nil, fmt.Errorf("marshal discovery: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + + // The sizing and the approval are applied onto the draft the discovery + // wrote, never before it exists: an apply that created the draft would + // make a document with no nodes, which the site refuses. + if draftHasNodes(sd.Status.Draft) && (sd.Spec.Sizing != nil || sd.Spec.Approved) { + spec := map[string]any{"approved": sd.Spec.Approved} + if tpl := sizingTemplate(sd.Spec.Sizing); len(tpl) > 0 { + spec["cluster"] = tpl + } + cdc := map[string]any{ + "apiVersion": simplyblockv1alpha2.GroupVersion.String(), + "kind": "ClusterDeploymentConfig", + "metadata": map[string]any{"name": draft, "namespace": ns}, + "spec": spec, + } + raw, err := json.Marshal(cdc) + if err != nil { + return nil, fmt.Errorf("marshal draft apply: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + configs = append(configs, workv1.ManifestConfigOption{ + ResourceIdentifier: workv1.ResourceIdentifier{ + Group: simplyblockv1alpha2.GroupVersion.Group, Resource: "clusterdeploymentconfigs", Namespace: ns, Name: draft, + }, + UpdateStrategy: &workv1.UpdateStrategy{ + Type: workv1.UpdateStrategyTypeServerSideApply, + ServerSideApply: &workv1.ServerSideApplyConfig{ + Force: true, + FieldManager: hubDeployFieldManager, + }, + }, + FeedbackRules: []workv1.FeedbackRule{{ + Type: workv1.JSONPathsType, + JsonPaths: []workv1.JsonPath{ + {Name: "phase", Path: ".status.phase"}, + {Name: "approved", Path: ".spec.approved"}, + }, + }}, + }) + } + + return &workv1.ManifestWork{ + ObjectMeta: metav1.ObjectMeta{ + Name: workName(sd), + Namespace: sd.Spec.Cluster, + Labels: labels, + }, + Spec: workv1.ManifestWorkSpec{ + Workload: workv1.ManifestsTemplate{Manifests: manifests}, + ManifestConfigs: configs, + DeleteOption: &workv1.DeleteOption{PropagationPolicy: workv1.DeletePropagationPolicyTypeOrphan}, + }, + }, nil +} + +// discoverSpec is the OperatorOps discover block of the request. +func discoverSpec(sd *simplyblockv1alpha2.StorageSiteDeployment) map[string]any { + d := map[string]any{"configName": draftName(sd)} + if sd.Spec.Discover.EnableControlPlaneNodes != nil { + d["enableControlPlaneNodes"] = *sd.Spec.Discover.EnableControlPlaneNodes + } + if len(sd.Spec.Discover.Workers) > 0 { + d["workers"] = sd.Spec.Discover.Workers + } + if len(sd.Spec.Discover.NodeSelector) > 0 { + d["nodeSelector"] = sd.Spec.Discover.NodeSelector + } + return d +} + +// sizingTemplate is the draft's cluster template fields the request sets. +// Only the stated fields are applied, so what the discovery wrote stays. +func sizingTemplate(s *simplyblockv1alpha2.StorageSiteSizing) map[string]any { + tpl := map[string]any{} + if s == nil { + return tpl + } + if s.Name != "" { + tpl["name"] = s.Name + } + if s.VCPUCount != nil { + tpl["vcpuCount"] = *s.VCPUCount + } + if s.MinHugePagesSize != "" { + tpl["minHugePagesSize"] = s.MinHugePagesSize + } + if s.MaxSubsystemCount != nil { + tpl["maxSubsystemCount"] = *s.MaxSubsystemCount + } + if s.EnableDriveFormat != nil { + tpl["enableDriveFormat"] = *s.EnableDriveFormat + } + if s.EnableJournalDevice != nil { + tpl["enableJournalDevice"] = *s.EnableJournalDevice + } + if s.Stripe != nil { + stripe := map[string]any{} + if s.Stripe.DataChunks != nil { + stripe["dataChunks"] = *s.Stripe.DataChunks + } + if s.Stripe.ParityChunks != nil { + stripe["parityChunks"] = *s.Stripe.ParityChunks + } + tpl["stripe"] = stripe + } + return tpl +} + +// project reads the site's views and derives the request's phase. +func (r *StorageSiteDeploymentReconciler) project(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, work *workv1.ManifestWork) (ctrl.Result, error) { + delivered, deliveryMessage := workDelivery(work) + + draftObj, haveDraft, err := r.projected(ctx, sd, "draft", "clusterdeploymentconfigs", draftName(sd), siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + if !haveDraft { + if deliveryMessage != "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + setCondition(s, ConditionStorageSiteDelivered, false, "NotApplied", deliveryMessage) + }) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + fmt.Sprintf("waiting for site %s to write draft %s/%s", sd.Spec.Cluster, siteNamespace(sd), draftName(sd)), + func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + }) + } + + var cdc simplyblockv1alpha2.ClusterDeploymentConfig + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(draftObj, &cdc); err != nil { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("the site's draft could not be read: %v", err), nil) + } + draft := projectDraft(&cdc) + base := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + s.WorkName = work.Name + s.Draft = draft + setCondition(s, ConditionStorageSiteDelivered, delivered, deliveryReason(delivered), deliveryNote(delivered, deliveryMessage)) + setCondition(s, ConditionStorageSiteDiscovered, draftHasNodes(draft), "Nodes", fmt.Sprintf("%d node(s) in the draft", draftNodeCount(draft))) + setCondition(s, ConditionStorageSiteApproved, draft.Approved, "SiteDraft", fmt.Sprintf("the site's draft approved=%t", draft.Approved)) + } + + if !draftHasNodes(draft) { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering, + "the site's draft names no node yet: discovery is running", base) + } + if deliveryMessage != "" { + // The draft exists, so the message is about the sizing or the approval + // the work could not apply: the site refused it. + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, deliveryMessage, base) + } + switch { + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed: + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + "the site's draft failed: "+orDefault(cdc.Status.Message, "no message"), base) + case !cdc.Spec.Approved: + msg := "the draft awaits approval" + if sd.Spec.Approved { + msg = "approval requested; waiting for the site's draft to take it" + } else if sd.Spec.Sizing != nil && !sizingApplied(sd.Spec.Sizing, cdc.Spec.Cluster) { + msg = "the draft awaits approval; the sizing is being applied" + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted, msg, base) + } + + // Approved on the site: follow the StorageCluster it expands into. + clusterName := cdc.Status.ClusterRef + if clusterName == "" && cdc.Spec.Cluster != nil { + clusterName = cdc.Spec.Cluster.Name + } + if clusterName == "" { + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, + "the draft is approved; waiting for the site to name its StorageCluster", base) + } + scObj, haveSC, err := r.projected(ctx, sd, "cluster", "storageclusters", clusterName, siteNamespace(sd)) + if err != nil { + return ctrl.Result{}, err + } + sc := &simplyblockv1alpha2.StorageSiteCluster{Name: clusterName} + if haveSC { + var cluster simplyblockv1alpha2.StorageCluster + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(scObj, &cluster); err == nil { + sc.UUID = cluster.Status.UUID + sc.Phase = string(cluster.Status.Phase) + sc.Pool = cluster.Annotations[annotationStorageClusterDefaultPool] + } + } + if sc.Pool == "" { + sc.Pool = pool.DefaultPoolName(clusterName) + } + nodes, err := r.projectNodes(ctx, sd, draft.NodeRefs) + if err != nil { + return ctrl.Result{}, err + } + sc.Nodes = nodes + withCluster := func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + base(s) + s.StorageCluster = sc + } + + switch { + case sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseOnline) && sc.UUID != "": + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline, + fmt.Sprintf("StorageCluster %s is Online (%d node(s))", clusterName, len(nodes)), func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, true, "Online", "the StorageCluster is Online") + }) + case cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanded && sc.Phase == string(simplyblockv1alpha2.StorageClusterPhaseUnavailable): + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed, + fmt.Sprintf("StorageCluster %s is Unavailable after the expansion", clusterName), withCluster) + } + msg := fmt.Sprintf("draft %s, StorageCluster %s %s", orDefault(string(cdc.Status.Phase), "Expanding"), clusterName, orDefault(sc.Phase, "not reported yet")) + if step := cdc.Status.Step.State; step != "" && cdc.Status.Phase == simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding { + msg = fmt.Sprintf("draft Expanding (%s), StorageCluster %s %s", step, clusterName, orDefault(sc.Phase, "not reported yet")) + } + return r.setPhase(ctx, sd, simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying, msg, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + withCluster(s) + setCondition(s, ConditionStorageSiteReady, false, "Deploying", msg) + }) +} + +// projectNodes projects the draft's StorageNodes, one view each. +func (r *StorageSiteDeploymentReconciler) projectNodes(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, refs []string) ([]simplyblockv1alpha2.StorageSiteNode, error) { + sorted := append([]string(nil), refs...) + sort.Strings(sorted) + if len(sorted) > maxStorageSiteNodeViews { + sorted = sorted[:maxStorageSiteNodeViews] + } + nodes := make([]simplyblockv1alpha2.StorageSiteNode, 0, len(sorted)) + for i, name := range sorted { + obj, ok, err := r.projected(ctx, sd, fmt.Sprintf("node-%d", i), "storagenodes", name, siteNamespace(sd)) + if err != nil { + return nil, err + } + n := simplyblockv1alpha2.StorageSiteNode{Name: name} + if ok { + var sn simplyblockv1alpha2.StorageNode + if err := runtime.DefaultUnstructuredConverter.FromUnstructured(obj, &sn); err == nil { + n.Phase = string(sn.Status.Phase) + n.Hostname = sn.Status.Hostname + } + } + nodes = append(nodes, n) + } + return nodes, nil +} + +// projected reads one of the request's views, creating it when it is missing, +// and reports whether the site has projected the object yet. +func (r *StorageSiteDeploymentReconciler) projected(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix, resource, name, namespace string) (map[string]interface{}, bool, error) { + viewName := storageSiteViewName(sd, suffix) + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + getErr := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: viewName}, view) + if apierrors.IsNotFound(getErr) { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + if createErr := r.Create(ctx, want); createErr != nil { + return nil, false, createErr + } + return nil, false, nil + } + if getErr != nil { + return nil, false, getErr + } + // A view that names another object (the draft's cluster changed) is + // pointed at the right one. + scope, _, _ := unstructured.NestedMap(view.Object, "spec", "scope") + if scope["name"] != name || scope["resource"] != resource { + want := newStorageSiteView(sd, viewName, resource, name, namespace) + view.Object["spec"] = want.Object["spec"] + if err := r.Update(ctx, view); err != nil { + return nil, false, err + } + return nil, false, nil + } + result, found, nestedErr := unstructured.NestedMap(view.Object, "status", "result") + if nestedErr != nil || !found || len(result) == 0 { + return nil, false, nil + } + return result, true, nil +} + +// newStorageSiteView asks the site to project one object back to the hub. +func newStorageSiteView(sd *simplyblockv1alpha2.StorageSiteDeployment, name, resource, targetName, targetNamespace string) *unstructured.Unstructured { + scope := map[string]interface{}{"resource": resource, "name": targetName} + if targetNamespace != "" { + scope["namespace"] = targetNamespace + } + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + view.SetNamespace(sd.Spec.Cluster) + view.SetName(name) + view.SetLabels(map[string]string{storageSiteDeploymentIDLabel: string(sd.UID)}) + _ = unstructured.SetNestedMap(view.Object, scope, "spec", "scope") + return view +} + +// setPhase writes the phase, the message and the mutation, and records a +// phase change as an event. +func (r *StorageSiteDeploymentReconciler) setPhase(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, phase simplyblockv1alpha2.StorageSiteDeploymentPhase, message string, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) (ctrl.Result, error) { + previous := sd.Status.Phase + if err := r.patchStatus(ctx, sd, func(s *simplyblockv1alpha2.StorageSiteDeploymentStatus) { + if mutate != nil { + mutate(s) + } + s.Phase = phase + s.Message = message + }); err != nil { + return ctrl.Result{}, err + } + if previous != phase && r.Recorder != nil { + kind := corev1.EventTypeNormal + if phase == simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + kind = corev1.EventTypeWarning + } + r.Recorder.Eventf(sd, nil, kind, string(phase), string(phase), "%s", message) + } + switch phase { + case simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline: + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + case simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed: + // A failure on the site may clear (a node comes back, a draft is + // corrected on the site): keep reading at the slow cadence. + return ctrl.Result{RequeueAfter: storageSiteDeploymentOnlineRequeue}, nil + } + return ctrl.Result{RequeueAfter: storageSiteDeploymentRequeue}, nil +} + +// patchStatus applies mutate to the status and writes it with the generation +// it was computed from. +func (r *StorageSiteDeploymentReconciler) patchStatus(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment, mutate func(*simplyblockv1alpha2.StorageSiteDeploymentStatus)) error { + base := client.MergeFrom(sd.DeepCopy()) + mutate(&sd.Status) + sd.Status.ObservedGeneration = sd.Generation + return r.Status().Patch(ctx, sd, base) +} + +// reconcileDeletion releases the request's work and views and removes the +// finalizer. The work orphans its resources, so nothing on the site goes. +func (r *StorageSiteDeploymentReconciler) reconcileDeletion(ctx context.Context, sd *simplyblockv1alpha2.StorageSiteDeployment) (ctrl.Result, error) { + if !controllerutil.ContainsFinalizer(sd, finalizerStorageSiteDeployment) { + return ctrl.Result{}, nil + } + var work workv1.ManifestWork + err := r.Get(ctx, client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &work) + switch { + case apierrors.IsNotFound(err): + case err != nil: + return ctrl.Result{}, err + case work.DeletionTimestamp.IsZero(): + if err := client.IgnoreNotFound(r.Delete(ctx, &work)); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + default: + // Deleting: wait for the work agent to release it. + return ctrl.Result{RequeueAfter: 5 * time.Second}, nil + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + views.SetGroupVersionKind(listGVK) + if err := r.List(ctx, views, client.InNamespace(sd.Spec.Cluster), client.MatchingLabels{storageSiteDeploymentIDLabel: string(sd.UID)}); err != nil && !meta.IsNoMatchError(err) { + return ctrl.Result{}, err + } + for i := range views.Items { + if err := client.IgnoreNotFound(r.Delete(ctx, &views.Items[i])); err != nil { + return ctrl.Result{}, err + } + } + controllerutil.RemoveFinalizer(sd, finalizerStorageSiteDeployment) + return ctrl.Result{}, r.Update(ctx, sd) +} + +// workDelivery reads the work's status: whether every manifest is applied, +// and the first manifest's refusal when one is not. +func workDelivery(work *workv1.ManifestWork) (applied bool, message string) { + if work == nil { + return false, "" + } + for _, m := range work.Status.ResourceStatus.Manifests { + for _, c := range m.Conditions { + if c.Type == workv1.ManifestApplied && c.Status == metav1.ConditionFalse { + return false, fmt.Sprintf("the site did not apply %s %s: %s", m.ResourceMeta.Kind, m.ResourceMeta.Name, c.Message) + } + } + } + for _, c := range work.Status.Conditions { + if c.Type == workv1.WorkApplied { + return c.Status == metav1.ConditionTrue, "" + } + } + return false, "" +} + +func deliveryReason(delivered bool) string { + if delivered { + return "Applied" + } + return "Pending" +} + +func deliveryNote(delivered bool, message string) string { + switch { + case message != "": + return message + case delivered: + return "the work is applied on the site" + } + return "the work is not applied on the site yet" +} + +// projectDraft is the draft as the status carries it. +func projectDraft(cdc *simplyblockv1alpha2.ClusterDeploymentConfig) *simplyblockv1alpha2.StorageSiteDraft { + d := &simplyblockv1alpha2.StorageSiteDraft{ + Name: cdc.Name, + Phase: string(cdc.Status.Phase), + Message: cdc.Status.Message, + Approved: cdc.Spec.Approved, + NodeSets: cdc.Spec.NodeSets, + NodeRefs: cdc.Status.NodeRefs, + } + if cdc.Spec.Cluster != nil { + d.Cluster = cdc.Spec.Cluster.DeepCopy() + } + if d.Phase == "" { + d.Phase = string(simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft) + } + return d +} + +// draftHasNodes is whether the draft names at least one worker. +func draftHasNodes(d *simplyblockv1alpha2.StorageSiteDraft) bool { + return draftNodeCount(d) > 0 +} + +func draftNodeCount(d *simplyblockv1alpha2.StorageSiteDraft) int { + if d == nil { + return 0 + } + n := 0 + for _, set := range d.NodeSets { + for _, g := range set.Groups { + n += len(g.Workers) + } + } + return n +} + +// sizingApplied is whether the draft's template carries the request's sizing. +func sizingApplied(s *simplyblockv1alpha2.StorageSiteSizing, tpl *simplyblockv1alpha2.ClusterTemplate) bool { + if s == nil { + return true + } + if tpl == nil { + return false + } + eq32 := func(want, have *int32) bool { return want == nil || (have != nil && *have == *want) } + eqBool := func(want, have *bool) bool { return want == nil || (have != nil && *have == *want) } + if s.Name != "" && tpl.Name != s.Name { + return false + } + if s.MinHugePagesSize != "" && tpl.MinHugePagesSize != s.MinHugePagesSize { + return false + } + if !eq32(s.VCPUCount, tpl.VCPUCount) || !eq32(s.MaxSubsystemCount, tpl.MaxSubsystemCount) || + !eqBool(s.EnableDriveFormat, tpl.EnableDriveFormat) || !eqBool(s.EnableJournalDevice, tpl.EnableJournalDevice) { + return false + } + if s.Stripe != nil { + if tpl.Stripe == nil || !eq32(s.Stripe.DataChunks, tpl.Stripe.DataChunks) || !eq32(s.Stripe.ParityChunks, tpl.Stripe.ParityChunks) { + return false + } + } + return true +} + +func setCondition(s *simplyblockv1alpha2.StorageSiteDeploymentStatus, kind string, ok bool, reason, message string) { + status := metav1.ConditionFalse + if ok { + status = metav1.ConditionTrue + } + meta.SetStatusCondition(&s.Conditions, metav1.Condition{Type: kind, Status: status, Reason: reason, Message: message, ObservedGeneration: s.ObservedGeneration}) +} + +func orDefault(s, d string) string { + if s == "" { + return d + } + return s +} + +func siteNamespace(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.SiteNamespace, "simplyblock") +} + +func draftName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return orDefault(sd.Spec.DraftName, "site-draft") +} + +// storageSiteHash is a short, deterministic id of the request for the names +// of its work and views, which must be bounded and found again on a restart. +func storageSiteHash(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + h := sha256.Sum256([]byte(sd.Namespace + "/" + sd.Name)) + return fmt.Sprintf("%x", h[:6]) +} + +func workName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + return "sbsd-" + storageSiteHash(sd) +} + +func storageSiteViewName(sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string) string { + return "sbsd-" + storageSiteHash(sd) + "-" + suffix +} + +// discoveryName is the OperatorOps on the site. It carries a hash of the +// discovery's parameters, so a changed discovery is a new run rather than an +// edit of a finished one. +func discoveryName(sd *simplyblockv1alpha2.StorageSiteDeployment) string { + raw, _ := json.Marshal(sd.Spec.Discover) + h := sha256.Sum256(raw) + return fmt.Sprintf("hub-discover-%s-%x", draftName(sd), h[:3]) +} + +// SetupWithManager registers the controller. The work and the views live in +// the site's namespace on the hub, where an owner reference to the request +// cannot point, so the site's answers are read on the requeue cadence of the +// phase rather than through a watch. +func (r *StorageSiteDeploymentReconciler) SetupWithManager(mgr ctrl.Manager) error { + return ctrl.NewControllerManagedBy(mgr). + For(&simplyblockv1alpha2.StorageSiteDeployment{}). + Named("storagesitedeployment"). + Complete(r) +} + +// listKindSuffix turns a kind into its list kind (StorageCluster -> +// StorageClusterList) for an unstructured list read. +const listKindSuffix = "List" diff --git a/operator/internal/controller/storagesitedeployment_controller_unit_test.go b/operator/internal/controller/storagesitedeployment_controller_unit_test.go new file mode 100644 index 000000000..4bb445d6f --- /dev/null +++ b/operator/internal/controller/storagesitedeployment_controller_unit_test.go @@ -0,0 +1,335 @@ +package controller + +import ( + "context" + "encoding/json" + "testing" + + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/types" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +const ( + testSDNamespace = "simplyblock" + testSDName = "site-a" + testSDCluster = "site-a" +) + +func newStorageSiteDeploymentReconciler(t *testing.T, objects ...client.Object) (*StorageSiteDeploymentReconciler, client.Client) { + t.Helper() + scheme := newTestScheme(t) + scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) + if err := workv1.Install(scheme); err != nil { + t.Fatalf("register work/v1 scheme: %v", err) + } + cl := newTestClient(t, scheme, + []client.Object{&simplyblockv1alpha2.StorageSiteDeployment{}, &workv1.ManifestWork{}}, + objects...) + return &StorageSiteDeploymentReconciler{Client: cl, Scheme: scheme, Recorder: &fakeRecorder{}}, cl +} + +func newSiteDeployment(mutate func(*simplyblockv1alpha2.StorageSiteDeployment)) *simplyblockv1alpha2.StorageSiteDeployment { + enable := true + sd := &simplyblockv1alpha2.StorageSiteDeployment{ + ObjectMeta: metav1.ObjectMeta{Name: testSDName, Namespace: testSDNamespace, UID: types.UID("uid-site-a")}, + Spec: simplyblockv1alpha2.StorageSiteDeploymentSpec{ + Cluster: testSDCluster, + Discover: simplyblockv1alpha2.StorageSiteDiscovery{EnableControlPlaneNodes: &enable}, + }, + } + if mutate != nil { + mutate(sd) + } + return sd +} + +// reconcileSD runs the reconcile n times (the first adds the finalizer) and +// returns the request as stored. +func reconcileSD(t *testing.T, r *StorageSiteDeploymentReconciler, cl client.Client, n int) *simplyblockv1alpha2.StorageSiteDeployment { + t.Helper() + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + for i := 0; i < n; i++ { + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile %d: %v", i, err) + } + } + var sd simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), req.NamespacedName, &sd); err != nil { + t.Fatalf("get request: %v", err) + } + return &sd +} + +func getSiteWork(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment) *workv1.ManifestWork { + t.Helper() + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: sd.Spec.Cluster, Name: workName(sd)}, &w); err != nil { + t.Fatalf("get ManifestWork: %v", err) + } + return &w +} + +// workManifests decodes the work's manifests by kind. +func workManifests(t *testing.T, w *workv1.ManifestWork) map[string]map[string]any { + t.Helper() + out := map[string]map[string]any{} + for _, m := range w.Spec.Workload.Manifests { + obj := map[string]any{} + if err := json.Unmarshal(m.Raw, &obj); err != nil { + t.Fatalf("decode manifest: %v", err) + } + out[obj["kind"].(string)] = obj + } + return out +} + +// projectSiteObject writes what the site's view controller would project. +func projectSiteObject(t *testing.T, cl client.Client, sd *simplyblockv1alpha2.StorageSiteDeployment, suffix string, obj any) { + t.Helper() + v := getView(t, cl, sd.Spec.Cluster, storageSiteViewName(sd, suffix)) + result, err := runtime.DefaultUnstructuredConverter.ToUnstructured(obj) + if err != nil { + t.Fatalf("convert projection: %v", err) + } + setViewResult(t, cl, v, result) +} + +func siteDraft(approved bool, phase simplyblockv1alpha2.ClusterDeploymentConfigPhase, mutate func(*simplyblockv1alpha2.ClusterDeploymentConfig)) *simplyblockv1alpha2.ClusterDeploymentConfig { + cdc := &simplyblockv1alpha2.ClusterDeploymentConfig{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "ClusterDeploymentConfig"}, + ObjectMeta: metav1.ObjectMeta{Name: "site-draft", Namespace: "simplyblock"}, + Spec: simplyblockv1alpha2.ClusterDeploymentConfigSpec{ + Approved: approved, + NodeSets: []simplyblockv1alpha2.NodeSet{{ + Name: "default", + Groups: []simplyblockv1alpha2.NodeGroup{{Workers: []string{"n1", "n2", "n3"}}}, + }}, + }, + } + cdc.Status.Phase = phase + if mutate != nil { + mutate(cdc) + } + return cdc +} + +func TestStorageSiteDeploymentDeliversTheDiscoveryAndWaitsForTheDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + w := getSiteWork(t, cl, sd) + if w.Spec.DeleteOption == nil || w.Spec.DeleteOption.PropagationPolicy != workv1.DeletePropagationPolicyTypeOrphan { + t.Errorf("work delete option = %+v, want Orphan: a request must never take the site's storage away", w.Spec.DeleteOption) + } + ms := workManifests(t, w) + ops, ok := ms["OperatorOps"] + if !ok || len(ms) != 1 { + t.Fatalf("manifests = %v, want the discovery alone before the draft exists", keys(ms)) + } + discover := ops["spec"].(map[string]any)["discover"].(map[string]any) + if discover["configName"] != "site-draft" || discover["enableControlPlaneNodes"] != true { + t.Errorf("discover = %v, want configName site-draft and control-plane nodes enabled", discover) + } + // The draft view exists so the site can project the draft. + getView(t, cl, testSDCluster, storageSiteViewName(sd, "draft")) +} + +func TestStorageSiteDeploymentSizesTheDraftOnceItNamesNodes(t *testing.T) { + vcpu := int32(8) + data, parity := int32(1), int32(1) + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Sizing = &simplyblockv1alpha2.StorageSiteSizing{ + Name: "sb-site-a", VCPUCount: &vcpu, MinHugePagesSize: "8G", + Stripe: &simplyblockv1alpha2.StripeSpec{DataChunks: &data, ParityChunks: &parity}, + } + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + sd = reconcileSD(t, r, cl, 2) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q (%s), want Drafted", sd.Status.Phase, sd.Status.Message) + } + if got := draftNodeCount(sd.Status.Draft); got != 3 { + t.Errorf("projected draft nodes = %d, want 3", got) + } + ms := workManifests(t, getSiteWork(t, cl, sd)) + cdc, ok := ms["ClusterDeploymentConfig"] + if !ok { + t.Fatalf("manifests = %v, want the draft's sizing applied", keys(ms)) + } + spec := cdc["spec"].(map[string]any) + if spec["approved"] != false { + t.Errorf("approved = %v, want false before the request is approved", spec["approved"]) + } + cluster := spec["cluster"].(map[string]any) + if cluster["name"] != "sb-site-a" || cluster["vcpuCount"] != float64(8) || cluster["minHugePagesSize"] != "8G" { + t.Errorf("cluster template = %v, want the sizing", cluster) + } + if _, ok := spec["nodeSets"]; ok { + t.Error("the sizing apply must not carry nodeSets: they are the discovery's") + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteDiscovered) { + t.Error("Discovered condition is not True for a draft with nodes") + } +} + +func TestStorageSiteDeploymentApprovalFollowsTheClusterToOnline(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, nil)) + // One pass projects the draft, the next delivers the approval onto it. + sd = reconcileSD(t, r, cl, 2) + + ms := workManifests(t, getSiteWork(t, cl, sd)) + if ms["ClusterDeploymentConfig"]["spec"].(map[string]any)["approved"] != true { + t.Fatalf("the work does not carry the approval: %v", ms["ClusterDeploymentConfig"]) + } + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDrafted { + t.Fatalf("phase = %q, want Drafted until the site's draft takes the approval", sd.Status.Phase) + } + + // The site's draft takes it and starts expanding. + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseExpanding, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.ClusterRef = "sb-site-a" + c.Status.NodeRefs = []string{"sn-1", "sn-2", "sn-3"} + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDeploying { + t.Fatalf("phase = %q (%s), want Deploying", sd.Status.Phase, sd.Status.Message) + } + + // The cluster comes Online. + sc := &simplyblockv1alpha2.StorageCluster{ + TypeMeta: metav1.TypeMeta{APIVersion: simplyblockv1alpha2.GroupVersion.String(), Kind: "StorageCluster"}, + ObjectMeta: metav1.ObjectMeta{Name: "sb-site-a", Namespace: "simplyblock"}, + } + sc.Status.UUID = "8f8dd277-1544-4177-9a74-e0f66eb2672c" + sc.Status.Phase = simplyblockv1alpha2.StorageClusterPhaseOnline + projectSiteObject(t, cl, sd, "cluster", sc) + sd = reconcileSD(t, r, cl, 1) + + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseOnline { + t.Fatalf("phase = %q (%s), want Online", sd.Status.Phase, sd.Status.Message) + } + if sd.Status.StorageCluster == nil || sd.Status.StorageCluster.UUID != sc.Status.UUID { + t.Fatalf("storageCluster = %+v, want the site's uuid", sd.Status.StorageCluster) + } + if sd.Status.StorageCluster.Pool == "" { + t.Error("storageCluster.pool is empty: a StorageClass needs it") + } + if n := len(sd.Status.StorageCluster.Nodes); n != 3 { + t.Errorf("projected nodes = %d, want 3", n) + } + if !meta.IsStatusConditionTrue(sd.Status.Conditions, ConditionStorageSiteReady) { + t.Error("Ready condition is not True for an Online cluster") + } +} + +func TestStorageSiteDeploymentReportsAFailedDraft(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(true, simplyblockv1alpha2.ClusterDeploymentConfigPhaseFailed, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Status.Message = "node n2 has no free device" + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseFailed { + t.Fatalf("phase = %q, want Failed", sd.Status.Phase) + } + if sd.Status.Message != "the site's draft failed: node n2 has no free device" { + t.Errorf("message = %q, want the site's own reason", sd.Status.Message) + } +} + +func TestStorageSiteDeploymentADraftWithoutNodesIsStillDiscovering(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Approved = true + })) + sd := reconcileSD(t, r, cl, 2) + projectSiteObject(t, cl, sd, "draft", siteDraft(false, simplyblockv1alpha2.ClusterDeploymentConfigPhaseDraft, func(c *simplyblockv1alpha2.ClusterDeploymentConfig) { + c.Spec.NodeSets = nil + })) + sd = reconcileSD(t, r, cl, 1) + if sd.Status.Phase != simplyblockv1alpha2.StorageSiteDeploymentPhaseDiscovering { + t.Fatalf("phase = %q, want Discovering", sd.Status.Phase) + } + if _, ok := workManifests(t, getSiteWork(t, cl, sd))["ClusterDeploymentConfig"]; ok { + t.Error("the approval was delivered onto a draft without nodes") + } +} + +func TestStorageSiteDeploymentAChangedDiscoveryIsANewRun(t *testing.T) { + a := newSiteDeployment(nil) + b := newSiteDeployment(func(sd *simplyblockv1alpha2.StorageSiteDeployment) { + sd.Spec.Discover.Workers = []string{"n1"} + }) + if discoveryName(a) == discoveryName(b) { + t.Fatalf("discovery name %q did not change with the discovery's parameters", discoveryName(a)) + } + if discoveryName(a) != discoveryName(newSiteDeployment(nil)) { + t.Fatal("the discovery name is not stable for the same parameters") + } +} + +func TestStorageSiteDeploymentDeletionOrphansTheSitesStorage(t *testing.T) { + r, cl := newStorageSiteDeploymentReconciler(t, newSiteDeployment(nil)) + sd := reconcileSD(t, r, cl, 2) + if err := cl.Delete(context.Background(), sd); err != nil { + t.Fatalf("delete request: %v", err) + } + // First pass deletes the work; the fake client removes it at once. + reconcileOnce := func() { + req := ctrl.Request{NamespacedName: types.NamespacedName{Namespace: testSDNamespace, Name: testSDName}} + if _, err := r.Reconcile(context.Background(), req); err != nil { + t.Fatalf("reconcile: %v", err) + } + } + reconcileOnce() + reconcileOnce() + + var w workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: testSDCluster, Name: workName(sd)}, &w); err == nil { + t.Error("the work is still there after the request was deleted") + } + views := &unstructured.UnstructuredList{} + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + views.SetGroupVersionKind(listGVK) + if err := cl.List(context.Background(), views, client.InNamespace(testSDCluster)); err != nil { + t.Fatalf("list views: %v", err) + } + if len(views.Items) != 0 { + t.Errorf("%d view(s) left after the request was deleted", len(views.Items)) + } + var gone simplyblockv1alpha2.StorageSiteDeployment + if err := cl.Get(context.Background(), client.ObjectKeyFromObject(sd), &gone); err == nil { + t.Error("the request is still there: the finalizer was not released") + } +} + +func keys(m map[string]map[string]any) []string { + out := make([]string, 0, len(m)) + for k := range m { + out = append(out, k) + } + return out +} diff --git a/operator/internal/controller/testfailover_controller.go b/operator/internal/controller/testfailover_controller.go new file mode 100644 index 000000000..7d7b983b3 --- /dev/null +++ b/operator/internal/controller/testfailover_controller.go @@ -0,0 +1,1399 @@ +// The TestFailover controller drives one non-disruptive test-failover drill to a +// terminal phase and holds it there until the object is deleted. +// +// It coordinates from the hub: it reads the source on its cluster, resolves a +// recovery point on the recovery cluster's backend, clones it there, and places +// the clone as a bound PVC in an isolated namespace on the recovery cluster, +// without ever touching the source. Deleting the object reclaims what the drill +// created. See operator/docs/designs/design-test-failover.md. +// +// This file is built in slices: the state graph and the drill's lifecycle +// scaffolding land first, and each step's side effects (source read, snapshot, +// clone, placement, teardown) fill in behind the graph the reconcile already +// walks. + +package controller + +import ( + "context" + "crypto/sha256" + "encoding/json" + "fmt" + "net/http" + "strings" + "time" + + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/client-go/tools/events" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + logf "sigs.k8s.io/controller-runtime/pkg/log" + + "github.com/simplyblock/atlas/statemachine" + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/utils" + "github.com/simplyblock/simplyblock-operator/internal/webapi" +) + +// csiDriverName is the CSI driver the bubble's PersistentVolume adopts the clone +// through, on the recovery cluster. +const csiDriverName = "csi.simplyblock.io" + +// testFailoverIDLabel tags every object a drill creates, so teardown can +// enumerate and prove nothing was left behind. +const testFailoverIDLabel = "storage.simplyblock.io/test-id" + +// managedClusterViewGVK is the OCM read primitive the hub uses to project a +// managed cluster's object back to itself. It is driven unstructured to avoid a +// dependency on the multicloud-operators-foundation module that defines it. +var managedClusterViewGVK = schema.GroupVersionKind{ + Group: "view.open-cluster-management.io", + Version: "v1beta1", + Kind: "ManagedClusterView", +} + +// finalizerTestFailover holds the object until its drill is torn down, so a +// clone, a drill-taken snapshot, or a placed PVC is never orphaned by a delete +// that races the controller. +const finalizerTestFailover = "storage.simplyblock.io/testfailover-teardown" + +// testFailoverStepRequeue is how long the reconcile waits before re-entering a +// step that is still in progress. Named so the controller's cadence is tunable +// in one place (reconciler-patterns §7). +const testFailoverStepRequeue = 10 * time.Second + +// testFailoverStepTimeout bounds how long any one step may take. A step that +// blows it fails the drill rather than holding forever, and because the deadline +// lives in status it survives an operator restart (reconciler-patterns §3). +const testFailoverStepTimeout = 15 * time.Minute + +// testFailoverGraph is the drill's provisioning state graph: the ordered steps +// from reading the source to a placed, bound PVC. Every state arms a per-step +// deadline on entry. Teardown (Releasing) is not in this graph; it runs on the +// deletion path, off the finalizer, not as a forward transition. +func testFailoverGraph() statemachine.Config[simplyblockv1alpha2.TestFailoverStep] { + armDeadline := func(context.Context, simplyblockv1alpha2.TestFailoverStep, simplyblockv1alpha2.TestFailoverStep) (time.Duration, error) { + return testFailoverStepTimeout, nil + } + return statemachine.Config[simplyblockv1alpha2.TestFailoverStep]{ + Initial: simplyblockv1alpha2.TestFailoverStepResolvingSource, + States: map[simplyblockv1alpha2.TestFailoverStep]statemachine.StateDef[simplyblockv1alpha2.TestFailoverStep]{ + simplyblockv1alpha2.TestFailoverStepResolvingSource: { + To: []simplyblockv1alpha2.TestFailoverStep{simplyblockv1alpha2.TestFailoverStepResolvingPoint}, + OnEnter: armDeadline, + }, + simplyblockv1alpha2.TestFailoverStepResolvingPoint: { + To: []simplyblockv1alpha2.TestFailoverStep{ + simplyblockv1alpha2.TestFailoverStepShipping, + simplyblockv1alpha2.TestFailoverStepCloning, + }, + OnEnter: armDeadline, + }, + simplyblockv1alpha2.TestFailoverStepShipping: { + To: []simplyblockv1alpha2.TestFailoverStep{simplyblockv1alpha2.TestFailoverStepCloning}, + OnEnter: armDeadline, + }, + simplyblockv1alpha2.TestFailoverStepCloning: { + To: []simplyblockv1alpha2.TestFailoverStep{simplyblockv1alpha2.TestFailoverStepPlacing}, + OnEnter: armDeadline, + }, + // Placing is terminal in the provisioning graph: once the bubble PVC is + // bound, the drill's phase is Ready and it holds until deleted. + simplyblockv1alpha2.TestFailoverStepPlacing: {OnEnter: armDeadline}, + }, + } +} + +// TestFailoverReconciler reconciles a TestFailover object. +type TestFailoverReconciler struct { + client.Client + Scheme *runtime.Scheme + Recorder events.EventRecorder +} + +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=testfailovers,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=testfailovers/status,verbs=get;update;patch +// +kubebuilder:rbac:groups=storage.simplyblock.io,resources=testfailovers/finalizers,verbs=update +// +kubebuilder:rbac:groups=work.open-cluster-management.io,resources=manifestworks,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=view.open-cluster-management.io,resources=managedclusterviews,verbs=get;list;watch;create;update;patch;delete +// +kubebuilder:rbac:groups=events.k8s.io,resources=events,verbs=create;patch + +// Reconcile drives one drill: it ensures the finalizer, walks the provisioning +// graph to Ready, holds there, and tears the drill down on deletion. +func (r *TestFailoverReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctrl.Result, error) { + log := logf.FromContext(ctx) + + var tf simplyblockv1alpha2.TestFailover + if err := r.Get(ctx, req.NamespacedName, &tf); err != nil { + return ctrl.Result{}, client.IgnoreNotFound(err) + } + + if !tf.DeletionTimestamp.IsZero() { + return r.reconcileDeletion(ctx, &tf) + } + + // A drill that never got its finalizer gets it before any side effect, so + // teardown is guaranteed a chance to run. + if controllerutil.AddFinalizer(&tf, finalizerTestFailover) { + if err := r.Update(ctx, &tf); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{Requeue: true}, nil + } + + // A terminal phase re-reconciles to nothing. Ready holds until the object is + // deleted, and Failed stays as the record. The finalizer is retained in both, + // so teardown still runs on delete. + if tf.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseReady || + tf.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseFailed { + return ctrl.Result{}, nil + } + + log.V(1).Info("reconciling test-failover drill", "phase", tf.Status.Phase, "step", tf.Status.Step.State) + return r.advanceDrill(ctx, &tf) +} + +// advanceDrill walks the provisioning graph one step per reconcile. It builds +// the machine from status.step, so the position survives a restart, and never +// runs a step's side effect from construction. +func (r *TestFailoverReconciler) advanceDrill(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + machine, err := statemachine.NewFromSnapshot(ctx, testFailoverGraph(), + statemachine.FromKube[simplyblockv1alpha2.TestFailoverStep](tf.Status.Step)) + if err != nil { + return r.fail(ctx, tf, "invalid drill step: "+err.Error()) + } + defer machine.Close() + + // Nobody has reconciled this yet: enter the initial step. + if tf.Status.Step.State == "" { + return r.begin(ctx, tf, machine.CurrentState()) + } + + // A step that blew its deadline fails the drill rather than holding forever. + if machine.TimeoutReached() { + return r.fail(ctx, tf, "step "+string(machine.CurrentState())+" exceeded its deadline") + } + + switch machine.CurrentState() { + case simplyblockv1alpha2.TestFailoverStepResolvingSource: + return r.resolveSource(ctx, tf) + case simplyblockv1alpha2.TestFailoverStepResolvingPoint: + return r.resolvePoint(ctx, tf) + case simplyblockv1alpha2.TestFailoverStepCloning: + return r.cloneRecoveryPoint(ctx, tf) + case simplyblockv1alpha2.TestFailoverStepPlacing: + return r.placeBubble(ctx, tf) + default: + // Later slices implement the remaining steps; until then a step in + // progress holds rather than blocks, and re-enters on the requeue. + return r.hold(ctx, tf, "step "+string(machine.CurrentState())+" not yet implemented") + } +} + +// resolveSource reads the source on its cluster through ManagedClusterViews to +// learn the source volume's backend handle, then advances to ResolvingPoint. It +// is non-blocking: each view is created once and its result awaited across +// reconciles, so a restart re-enters rather than re-creates. +func (r *TestFailoverReconciler) resolveSource(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + // Test-failover recovers onto a DIFFERENT cluster (a DR target or another + // cluster), never in place: recovering a bubble into the source's own + // subsystem cannot be mounted alongside the live source, and the point is to + // rehearse the site that would take over. Reject a same-cluster drill up front. + if tf.Spec.BubbleCluster == tf.Spec.SourceCluster { + return r.fail(ctx, tf, "test-failover within the same cluster is not supported: bubbleCluster must differ from sourceCluster") + } + if tf.Spec.Scope == simplyblockv1alpha2.TestFailoverScopeGroup { + return r.resolveSourceGroup(ctx, tf) + } + + pvc, ready, err := r.projectedSource(ctx, tf, "src-pvc", "persistentvolumeclaims", tf.Spec.SourceRef, tf.Spec.SourceNamespace) + if err != nil { + return ctrl.Result{}, err + } + if !ready { + return r.hold(ctx, tf, "waiting for the source PVC projection from cluster "+tf.Spec.SourceCluster) + } + pvName, _, _ := unstructured.NestedString(pvc, "spec", "volumeName") + if pvName == "" { + return r.fail(ctx, tf, "source PVC "+tf.Spec.SourceRef+" is not bound to a volume") + } + + pv, ready, err := r.projectedSource(ctx, tf, "src-pv", "persistentvolumes", pvName, "") + if err != nil { + return ctrl.Result{}, err + } + if !ready { + return r.hold(ctx, tf, "waiting for the source PV projection from cluster "+tf.Spec.SourceCluster) + } + handle, _, _ := unstructured.NestedString(pv, "spec", "csi", "volumeHandle") + if handle == "" { + return r.fail(ctx, tf, "source PV "+pvName+" has no CSI volume handle") + } + srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") + bubbleVC := bubbleVolumeContext(srcAttrs) + fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") + + if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepResolvingPoint, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Clones = []simplyblockv1alpha2.TestFailoverClone{{ + SourceRef: tf.Spec.SourceRef, + SourceHandle: handle, + SourceVolumeContext: bubbleVC, + SourceFSType: fsType, + SourceVolumeMode: volumeMode, + }} + s.Message = "resolved the source volume; resolving the recovery point" + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "SourceResolved", "SourceResolved", + "resolved source volume %s on cluster %s", handle, tf.Spec.SourceCluster) + return ctrl.Result{Requeue: true}, nil +} + +// resolveSourceGroup resolves a consistency group's members into one clone slot +// each, then advances to ResolvingPoint. The members and their K8s identity come +// from the source cluster's backend (a group member carries only an lvol id); +// the shared class metadata (fsType and volumeAttributes, identical across +// members of one StorageClass) is read once from a representative member's PV +// through a ManagedClusterView. It is non-blocking: the representative views are +// created once and awaited across reconciles. +func (r *TestFailoverReconciler) resolveSourceGroup(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + srcUUID, err := r.resolveGroupSourceUUID(ctx, tf) + if err != nil { + return r.hold(ctx, tf, "resolving the source cluster's backend UUID: "+err.Error()) + } + + api := webapi.NewClient() + if secret, secErr := r.clusterSecret(ctx, tf.Namespace, tf.Spec.SourceCluster); secErr == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + group, err := api.GetConsistencyGroupByName(ctx, srcUUID, tf.Spec.SourceRef) + if err != nil { + return ctrl.Result{}, err + } + if group == nil { + return r.fail(ctx, tf, "consistency group "+tf.Spec.SourceRef+" not found on cluster "+tf.Spec.SourceCluster) + } + memberIDs, err := api.GetConsistencyGroupMembers(ctx, srcUUID, group.UUID) + if err != nil { + return ctrl.Result{}, err + } + if len(memberIDs) == 0 { + return r.fail(ctx, tf, "consistency group "+tf.Spec.SourceRef+" has no members") + } + memberVols, err := api.ResolveMemberVolumes(ctx, srcUUID, memberIDs) + if err != nil { + return ctrl.Result{}, err + } + if len(memberVols) != len(memberIDs) { + return r.hold(ctx, tf, fmt.Sprintf("resolved %d of %d group members' source volumes; retrying", len(memberVols), len(memberIDs))) + } + + // One member's PV carries the class metadata every member shares, so a single + // projection serves the whole group. + rep := memberVols[memberIDs[0]] + pvc, ready, err := r.projectedSource(ctx, tf, "src-pvc", "persistentvolumeclaims", rep.PVCName, rep.PVCNamespace) + if err != nil { + return ctrl.Result{}, err + } + if !ready { + return r.hold(ctx, tf, "waiting for the source PVC projection for group member "+rep.PVCName) + } + pvName, _, _ := unstructured.NestedString(pvc, "spec", "volumeName") + if pvName == "" { + return r.fail(ctx, tf, "group member PVC "+rep.PVCName+" is not bound to a volume") + } + pv, ready, err := r.projectedSource(ctx, tf, "src-pv", "persistentvolumes", pvName, "") + if err != nil { + return ctrl.Result{}, err + } + if !ready { + return r.hold(ctx, tf, "waiting for the source PV projection for group member "+rep.PVCName) + } + srcAttrs, _, _ := unstructured.NestedStringMap(pv, "spec", "csi", "volumeAttributes") + bubbleVC := bubbleVolumeContext(srcAttrs) + fsType, _, _ := unstructured.NestedString(pv, "spec", "csi", "fsType") + volumeMode, _, _ := unstructured.NestedString(pv, "spec", "volumeMode") + + clones := make([]simplyblockv1alpha2.TestFailoverClone, 0, len(memberIDs)) + for _, id := range memberIDs { + v := memberVols[id] + clones = append(clones, simplyblockv1alpha2.TestFailoverClone{ + SourceRef: v.PVCName, + SourceHandle: srcUUID + ":" + v.PoolID + ":" + v.LvolID, + SourceFSType: fsType, + SourceVolumeMode: volumeMode, + SourceVolumeContext: bubbleVC, + SizeBytes: v.Size, + }) + } + + if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepResolvingPoint, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Clones = clones + s.Message = fmt.Sprintf("resolved the consistency group's %d members; resolving the recovery point", len(clones)) + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "SourceResolved", "SourceResolved", + "resolved consistency group %s (%d members) on cluster %s", tf.Spec.SourceRef, len(clones), tf.Spec.SourceCluster) + return ctrl.Result{Requeue: true}, nil +} + +// resolveGroupSourceUUID returns the backend UUID of the source cluster. It +// prefers the canonical resolution (a raw UUID, or a local StorageCluster named +// like the cluster), and falls back to the sole local StorageCluster when the +// hub is colocated on the source cluster, where the OCM cluster name does not +// match the StorageCluster name. +func (r *TestFailoverReconciler) resolveGroupSourceUUID(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (string, error) { + if utils.IsUUID(tf.Spec.SourceCluster) { + return tf.Spec.SourceCluster, nil + } + // StorageClusters live in the operator's namespace, not the drill's, so list + // cluster-wide rather than in the CR's namespace. + var clusters simplyblockv1alpha2.StorageClusterList + if err := r.List(ctx, &clusters); err != nil { + return "", err + } + // Prefer a StorageCluster named like the source cluster. + for i := range clusters.Items { + if clusters.Items[i].Name == tf.Spec.SourceCluster && clusters.Items[i].Status.UUID != "" { + return clusters.Items[i].Status.UUID, nil + } + } + // Fall back to the sole StorageCluster with a UUID: on a colocated hub the OCM + // cluster name does not match the StorageCluster name, but there is one. + uuid, ready := "", 0 + for i := range clusters.Items { + if clusters.Items[i].Status.UUID != "" { + uuid = clusters.Items[i].Status.UUID + ready++ + } + } + if ready == 1 { + return uuid, nil + } + return "", fmt.Errorf("no unique backend UUID for source cluster %q (%d Storage Clusters with a UUID)", tf.Spec.SourceCluster, ready) +} + +// projectedSource ensures a ManagedClusterView for one source object exists on +// the source cluster and returns the projected object once the view controller +// has fetched it. ready is false while the projection is still pending, which is +// the reconcile's cue to hold. +func (r *TestFailoverReconciler) projectedSource(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, suffix, resourceKind, name, namespace string) (obj map[string]interface{}, ready bool, err error) { + viewName := testFailoverViewName(tf, suffix) + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + getErr := r.Get(ctx, client.ObjectKey{Namespace: tf.Spec.SourceCluster, Name: viewName}, view) + if apierrors.IsNotFound(getErr) { + if createErr := r.Create(ctx, newManagedClusterView(tf, viewName, resourceKind, name, namespace)); createErr != nil { + return nil, false, createErr + } + return nil, false, nil + } + if getErr != nil { + return nil, false, getErr + } + result, found, nestedErr := unstructured.NestedMap(view.Object, "status", "result") + if nestedErr != nil || !found || len(result) == 0 { + return nil, false, nil + } + return result, true, nil +} + +// newManagedClusterView builds a view that asks the source cluster to project one +// object back to the hub. It lives in the source cluster's namespace on the hub +// and is labeled with the drill's test-id for teardown enumeration. +func newManagedClusterView(tf *simplyblockv1alpha2.TestFailover, name, resourceKind, targetName, targetNamespace string) *unstructured.Unstructured { + scope := map[string]interface{}{"resource": resourceKind, "name": targetName} + if targetNamespace != "" { + scope["namespace"] = targetNamespace + } + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + view.SetNamespace(tf.Spec.SourceCluster) + view.SetName(name) + view.SetLabels(map[string]string{testFailoverIDLabel: string(tf.UID)}) + _ = unstructured.SetNestedMap(view.Object, scope, "spec", "scope") + return view +} + +// testFailoverViewName is a deterministic, bounded name for one of a drill's +// views, so a restart finds the existing view instead of creating a second. +func testFailoverViewName(tf *simplyblockv1alpha2.TestFailover, suffix string) string { + h := sha256.Sum256([]byte(tf.Namespace + "/" + tf.Name)) + return fmt.Sprintf("tfo-%x-%s", h[:6], suffix) +} + +// transitionTo validates the edge against the graph and persists the new step +// together with any status mutation. +func (r *TestFailoverReconciler) transitionTo(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, to simplyblockv1alpha2.TestFailoverStep, mutate func(*simplyblockv1alpha2.TestFailoverStatus)) error { + machine, err := statemachine.NewFromSnapshot(ctx, testFailoverGraph(), + statemachine.FromKube[simplyblockv1alpha2.TestFailoverStep](tf.Status.Step)) + if err != nil { + return err + } + defer machine.Close() + if err := machine.TransitionTo(ctx, to); err != nil { + return err + } + snap := statemachine.ToKube(machine.Snapshot()) + return r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Step = snap + if mutate != nil { + mutate(s) + } + }) +} + +// begin records that the drill has started and enters the initial step. It +// refuses to start a second active drill against the same source and bubble, so +// two drills cannot duplicate each other's snapshots and clones (design §7.3). +func (r *TestFailoverReconciler) begin(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, initial simplyblockv1alpha2.TestFailoverStep) (ctrl.Result, error) { + if other, err := r.conflictingActiveDrill(ctx, tf); err != nil { + return ctrl.Result{}, err + } else if other != "" { + return r.fail(ctx, tf, "another active drill "+other+" is running for the same source and bubble cluster") + } + + now := metav1.Now() + deadline := metav1.NewTime(now.Add(testFailoverStepTimeout)) + if err := r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + s.Step = statemachine.KubeSnapshot{State: string(initial), Deadline: &deadline} + s.Message = "resolving the source" + if s.StartedAt == nil { + s.StartedAt = &now + } + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "DrillStarted", "DrillStarted", "the test-failover drill started") + return ctrl.Result{Requeue: true}, nil +} + +// conflictingActiveDrill returns the name of another non-terminal drill against +// the same source and bubble, or empty when there is none. A drill that has +// Failed or is tearing down no longer holds resources and does not conflict. +func (r *TestFailoverReconciler) conflictingActiveDrill(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (string, error) { + var drills simplyblockv1alpha2.TestFailoverList + if err := r.List(ctx, &drills, client.InNamespace(tf.Namespace)); err != nil { + return "", err + } + for i := range drills.Items { + other := &drills.Items[i] + if other.UID == tf.UID || !other.DeletionTimestamp.IsZero() { + continue + } + if other.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseFailed || + other.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseTearingDown { + continue + } + if other.Spec.Scope == tf.Spec.Scope && + other.Spec.SourceRef == tf.Spec.SourceRef && + other.Spec.BubbleCluster == tf.Spec.BubbleCluster { + return other.Name, nil + } + } + return "", nil +} + +// hold keeps the drill on its current step and re-enters after the requeue. +func (r *TestFailoverReconciler) hold(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, message string) (ctrl.Result, error) { + if err := r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Message = message + }); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{RequeueAfter: testFailoverStepRequeue}, nil +} + +// fail records a terminal failure. The finalizer is retained, so teardown still +// runs on delete. +func (r *TestFailoverReconciler) fail(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, message string) (ctrl.Result, error) { + now := metav1.Now() + if err := r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Phase = simplyblockv1alpha2.TestFailoverPhaseFailed + s.Message = message + if s.CompletedAt == nil { + s.CompletedAt = &now + } + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeWarning, "DrillFailed", "DrillFailed", "%s", message) + return ctrl.Result{}, nil +} + +// reconcileDeletion tears the drill down and removes the finalizer only once +// every drill resource is confirmed gone. It reclaims in dependency order, and +// each step tolerates a not-found (a re-run after a partial teardown is safe). A +// reclaim that cannot be confirmed holds the object in TearingDown rather than +// clearing the finalizer and orphaning backend storage. +func (r *TestFailoverReconciler) reconcileDeletion(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + if !controllerutil.ContainsFinalizer(tf, finalizerTestFailover) { + return ctrl.Result{}, nil + } + if tf.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseTearingDown { + _ = r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Phase = simplyblockv1alpha2.TestFailoverPhaseTearingDown + s.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepReleasing)} + s.Message = "tearing down the bubble" + }) + } + + // The ManifestWork's removal garbage-collects the PV and PVC on the recovery + // cluster, so it goes first. + if err := r.deleteManifestWork(ctx, tf); err != nil { + return r.reclaimPending(ctx, tf, "remove the bubble placement", err) + } + // The namespace goes with the last drill placing into it. + if err := r.deleteBubbleNamespaceWork(ctx, tf); err != nil { + return r.reclaimPending(ctx, tf, "remove the bubble namespace", err) + } + + // Reclaim every clone slot (one for a volume drill, one per member for a + // group drill). Each reclaim tolerates a not-found, so a re-run after a + // partial teardown is safe. + for i := range tf.Status.Clones { + clone := tf.Status.Clones[i] + if clone.CloneID != "" { + if err := r.reclaimClone(ctx, tf, clone.CloneID); err != nil { + return r.reclaimPending(ctx, tf, "reclaim the clone for "+clone.SourceRef, err) + } + } + // The recovery point is the replicated snapshot already on the target, + // resolved rather than created by the drill, so it is left alone. + } + + // The read-side views cost nothing to leave, but teardown proves no test-id + // object remains, so they go too. + if err := r.deleteView(ctx, tf, "src-pvc"); err != nil { + return r.reclaimPending(ctx, tf, "remove the source view", err) + } + if err := r.deleteView(ctx, tf, "src-pv"); err != nil { + return r.reclaimPending(ctx, tf, "remove the source view", err) + } + + controllerutil.RemoveFinalizer(tf, finalizerTestFailover) + if err := r.Update(ctx, tf); err != nil { + return ctrl.Result{}, err + } + return ctrl.Result{}, nil +} + +// reclaimPending records that teardown is holding on a reclaim it could not +// confirm, keeping the finalizer so nothing is orphaned. +func (r *TestFailoverReconciler) reclaimPending(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, what string, cause error) (ctrl.Result, error) { + _ = r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Message = "teardown is holding: could not " + what + ": " + cause.Error() + }) + r.Recorder.Eventf(tf, nil, corev1.EventTypeWarning, "ReclaimPending", "ReclaimPending", + "teardown could not %s: %v", what, cause) + return ctrl.Result{}, cause +} + +// deleteManifestWork removes the bubble's ManifestWork, tolerating an already-gone +// or already-deleting one. +func (r *TestFailoverReconciler) deleteManifestWork(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) error { + var mw workv1.ManifestWork + err := r.Get(ctx, client.ObjectKey{Namespace: tf.Spec.BubbleCluster, Name: testFailoverManifestWorkName(tf)}, &mw) + if apierrors.IsNotFound(err) { + return nil + } + if err != nil { + return err + } + if !mw.DeletionTimestamp.IsZero() { + return nil + } + return client.IgnoreNotFound(r.Delete(ctx, &mw)) +} + +// deleteView removes one of the drill's ManagedClusterViews, tolerating a +// not-found. +func (r *TestFailoverReconciler) deleteView(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, suffix string) error { + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + view.SetNamespace(tf.Spec.SourceCluster) + view.SetName(testFailoverViewName(tf, suffix)) + return client.IgnoreNotFound(r.Delete(ctx, view)) +} + +// reclaimClone deletes the drill's clone from the recovery cluster's backend. +func (r *TestFailoverReconciler) reclaimClone(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, handle string) error { + cluster, pool, vol, ok := splitHandle(handle) + if !ok { + return nil + } + apiClient := webapi.NewClient() + if secret, err := r.clusterSecret(ctx, tf.Namespace, tf.Spec.BubbleCluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + return backendDelete(ctx, apiClient, "reclaim clone", + fmt.Sprintf("/api/v2/clusters/%s/storage-pools/%s/volumes/%s", cluster, pool, vol)) +} + +// backendDelete issues an idempotent DELETE, treating a not-found as success. +func backendDelete(ctx context.Context, api *webapi.Client, what, endpoint string) error { + body, status, err := api.Do(ctx, http.MethodDelete, endpoint, nil) + if err != nil { + return fmt.Errorf("%s: %w", what, err) + } + if status == http.StatusNotFound || status < 300 { + return nil + } + return fmt.Errorf("%s: status %d: %s", what, status, string(body)) +} + +// sourceUnchanged re-reads the source PV projection and reports whether the +// source is still the same backend volume the drill recovered from. It is +// best-effort: when the projection cannot be re-read it does not fail the drill, +// but a projection that shows a different handle is a genuine invariant breach. +func (r *TestFailoverReconciler) sourceUnchanged(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) bool { + if len(tf.Status.Clones) == 0 || tf.Status.Clones[0].SourceHandle == "" { + return true + } + pv, ok := r.readViewResult(ctx, tf, "src-pv") + if !ok { + return true + } + handle, _, _ := unstructured.NestedString(pv, "spec", "csi", "volumeHandle") + if handle == "" { + return true + } + return handle == tf.Status.Clones[0].SourceHandle +} + +// readViewResult reads an existing view's projected object without creating one. +func (r *TestFailoverReconciler) readViewResult(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, suffix string) (map[string]interface{}, bool) { + view := &unstructured.Unstructured{} + view.SetGroupVersionKind(managedClusterViewGVK) + if err := r.Get(ctx, client.ObjectKey{Namespace: tf.Spec.SourceCluster, Name: testFailoverViewName(tf, suffix)}, view); err != nil { + return nil, false + } + result, found, err := unstructured.NestedMap(view.Object, "status", "result") + if err != nil || !found || len(result) == 0 { + return nil, false + } + return result, true +} + +// patchStatus applies mutate to the status and writes it, recording +// observedGeneration so a stale status can be told from a current one. +func (r *TestFailoverReconciler) patchStatus(ctx context.Context, tf *simplyblockv1alpha2.TestFailover, mutate func(*simplyblockv1alpha2.TestFailoverStatus)) error { + base := client.MergeFrom(tf.DeepCopy()) + mutate(&tf.Status) + tf.Status.ObservedGeneration = tf.Generation + return r.Status().Patch(ctx, tf, base) +} + +// replicatedSnapshotResult is the control plane's ReplicatedSnapshotDTO for the +// latest replicated snapshot on a DR-target backend. +// LvolID is the replica volume on the target the snapshot belongs to. +type replicatedSnapshotResult struct { + SnapshotID string `json:"snapshot_id"` + ClusterID string `json:"cluster_id"` + PoolID string `json:"pool_id"` + LvolID string `json:"lvol_id"` + CreatedAt time.Time `json:"created_at"` +} + +// resolvePoint resolves the recovery point on the bubble cluster's backend and +// advances to Cloning. The bubble is a DR target holding the latest replicated +// snapshot already there, so no data moves and nothing is triggered. It records +// the point as a CSI snapshot handle so Cloning is self-contained. +func (r *TestFailoverReconciler) resolvePoint(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + if tf.Spec.Scope == simplyblockv1alpha2.TestFailoverScopeGroup { + return r.resolvePointGroup(ctx, tf) + } + if len(tf.Status.Clones) == 0 || tf.Status.Clones[0].SourceHandle == "" { + return r.fail(ctx, tf, "internal: the source was not resolved before ResolvingPoint") + } + srcCluster, srcPool, srcLvol, ok := splitHandle(tf.Status.Clones[0].SourceHandle) + if !ok { + return r.fail(ctx, tf, "source handle is malformed: "+tf.Status.Clones[0].SourceHandle) + } + + apiClient := webapi.NewClient() + if secret, err := r.clusterSecret(ctx, tf.Namespace, tf.Spec.SourceCluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + // The replicated snapshot is already on the DR target's backend. + dto, found, err := r.latestReplicatedSnapshot(ctx, apiClient, srcCluster, srcLvol) + if err != nil { + return ctrl.Result{}, err + } + if !found { + return r.fail(ctx, tf, "no replicated snapshot on the target for volume "+srcLvol+" yet") + } + snapCluster, snapPool, snapUUID := dto.ClusterID, dto.PoolID, dto.SnapshotID + var pointTime *metav1.Time + if !dto.CreatedAt.IsZero() { + pt := metav1.NewTime(dto.CreatedAt) + pointTime = &pt + } + if snapPool == "" { + snapPool = srcPool + } + pointHandle := snapCluster + ":" + snapPool + ":" + snapUUID + + if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepCloning, func(s *simplyblockv1alpha2.TestFailoverStatus) { + s.Clones[0].SnapshotID = pointHandle + if s.Report == nil { + s.Report = &simplyblockv1alpha2.TestFailoverReport{} + } + s.Report.BubbleCluster = tf.Spec.BubbleCluster + s.Report.RecoveryPoint = snapUUID + s.Report.RecoveryPointTime = pointTime + s.Message = "resolved the recovery point; cloning it into the bubble" + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "RecoveryPointResolved", "RecoveryPointResolved", + "recovery point %s on cluster %s", snapUUID, snapCluster) + return ctrl.Result{Requeue: true}, nil +} + +// resolvePointGroup resolves the group's one group-consistent recovery point on +// the target and records one snapshot handle per clone slot, then advances to +// Cloning. The point comes from the group's replication policy's latest +// generation, which the control plane returns only when every member has a +// snapshot at the same generation, so the recovered set is crash-consistent. +func (r *TestFailoverReconciler) resolvePointGroup(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + if len(tf.Status.Clones) == 0 { + return r.fail(ctx, tf, "internal: the group source was not resolved before ResolvingPoint") + } + srcUUID, err := r.resolveGroupSourceUUID(ctx, tf) + if err != nil { + return r.hold(ctx, tf, "resolving the source cluster's backend UUID: "+err.Error()) + } + + api := webapi.NewClient() + if secret, secErr := r.clusterSecret(ctx, tf.Namespace, tf.Spec.SourceCluster); secErr == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + group, err := api.GetConsistencyGroupByName(ctx, srcUUID, tf.Spec.SourceRef) + if err != nil { + return ctrl.Result{}, err + } + if group == nil { + return r.fail(ctx, tf, "consistency group "+tf.Spec.SourceRef+" not found on cluster "+tf.Spec.SourceCluster) + } + + // The policy is read off the group: attach_group_policy stores it on the group + // record, and a group-first attach leaves the policy's own placement empty, so + // the recovery point is keyed on group.policy_id, not the policy list. + if group.PolicyID == "" { + return r.fail(ctx, tf, "no replication policy attached to group "+tf.Spec.SourceRef) + } + + groupSeq, members, found, err := api.LatestReplicatedGeneration(ctx, srcUUID, group.PolicyID) + if err != nil { + return ctrl.Result{}, err + } + if !found { + return r.fail(ctx, tf, "no replicated generation on the target for group "+tf.Spec.SourceRef+" yet") + } + if len(members) != len(tf.Status.Clones) { + return r.fail(ctx, tf, fmt.Sprintf("the group generation has %d members but %d were resolved; group membership changed mid-drill", len(members), len(tf.Status.Clones))) + } + + // Each slot gets the snapshot of ITS member, matched by volume identity: + // one generation makes the set crash-consistent, but only the identity + // keeps one member's disk off another's PVC. + slots := make([]groupMemberSource, len(tf.Status.Clones)) + for i, c := range tf.Status.Clones { + _, _, lvol, ok := splitHandle(c.SourceHandle) + if !ok { + return r.fail(ctx, tf, "group member "+c.SourceRef+" has a malformed source handle: "+c.SourceHandle) + } + slots[i] = groupMemberSource{PVC: c.SourceRef, LvolID: lvol} + } + targetOf, err := r.replicaVolumes(ctx, api, srcUUID, slots, members) + if err != nil { + return ctrl.Result{}, err + } + assigned, err := assignGenerationMembers(slots, members, targetOf) + if err != nil { + return r.fail(ctx, tf, fmt.Sprintf("cannot pair generation %d with the group's PVCs: %v", groupSeq, err)) + } + + if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepCloning, func(s *simplyblockv1alpha2.TestFailoverStatus) { + for i, m := range assigned { + s.Clones[i].SnapshotID = members[m].ClusterID + ":" + members[m].PoolID + ":" + members[m].SnapshotID + } + if s.Report == nil { + s.Report = &simplyblockv1alpha2.TestFailoverReport{} + } + s.Report.BubbleCluster = tf.Spec.BubbleCluster + s.Report.RecoveryPoint = fmt.Sprintf("generation %d", groupSeq) + s.Message = fmt.Sprintf("resolved the group-consistent point (generation %d); cloning %d members", groupSeq, len(members)) + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "RecoveryPointResolved", "RecoveryPointResolved", + "group-consistent generation %d on cluster %s (%d members)", groupSeq, tf.Spec.BubbleCluster, len(members)) + return ctrl.Result{Requeue: true}, nil +} + +// replicaVolumes maps each slot's source lvol to its replica volume on the +// target, for a control plane whose generation members do not name their source +// volume. It returns nil, and reads nothing, when every member names it. +func (r *TestFailoverReconciler) replicaVolumes( + ctx context.Context, + api *webapi.Client, + sourceClusterUUID string, + slots []groupMemberSource, + members []webapi.ReplicatedGroupSnapshot, +) (map[string]string, error) { + complete := true + for i := range members { + if members[i].SourceLvolID == "" { + complete = false + break + } + } + if complete { + return nil, nil + } + targetOf := make(map[string]string, len(slots)) + for _, slot := range slots { + dto, found, err := r.latestReplicatedSnapshot(ctx, api, sourceClusterUUID, slot.LvolID) + if err != nil { + return nil, err + } + if found { + targetOf[slot.LvolID] = dto.LvolID + } + } + return targetOf, nil +} + +// latestReplicatedSnapshot reads the latest replicated snapshot for a source +// volume on its DR target. found is false when replication has landed nothing. +func (r *TestFailoverReconciler) latestReplicatedSnapshot(ctx context.Context, api *webapi.Client, sourceClusterUUID, sourceLvolUUID string) (replicatedSnapshotResult, bool, error) { + endpoint := fmt.Sprintf("/api/v2/clusters/%s/replication/relationships/%s/latest-snapshot", sourceClusterUUID, sourceLvolUUID) + body, status, err := api.Do(ctx, http.MethodGet, endpoint, nil) + if status == http.StatusNotFound { + return replicatedSnapshotResult{}, false, nil + } + if err != nil || status >= 300 { + return replicatedSnapshotResult{}, false, requestError("resolve latest replicated snapshot", body, status, err) + } + var dto replicatedSnapshotResult + if err := json.Unmarshal(body, &dto); err != nil { + return replicatedSnapshotResult{}, false, fmt.Errorf("decode latest-snapshot: %w", err) + } + return dto, dto.SnapshotID != "", nil +} + +// cloneRecoveryPoint clones the resolved recovery point into a writable volume on +// the recovery cluster's backend and advances to Placing. The clone is built on +// the same backend the point lives on, so no data crosses a cluster boundary. +func (r *TestFailoverReconciler) cloneRecoveryPoint(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + if len(tf.Status.Clones) == 0 { + return r.fail(ctx, tf, "internal: no recovery point was resolved before Cloning") + } + + apiClient := webapi.NewClient() + if secret, err := r.clusterSecret(ctx, tf.Namespace, tf.Spec.BubbleCluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + + // One clone per slot, on the backend the point lives on, so no data crosses a + // cluster boundary. ensureClone is idempotent, so re-entry reuses any clone + // already built. A volume drill has one slot; a group drill has one per member. + handles := make([]string, len(tf.Status.Clones)) + sizes := make([]int64, len(tf.Status.Clones)) + for i := range tf.Status.Clones { + c := tf.Status.Clones[i] + if c.SnapshotID == "" { + return r.fail(ctx, tf, "internal: the recovery point was not resolved for member "+c.SourceRef) + } + snapCluster, snapPool, snapUUID, ok := splitHandle(c.SnapshotID) + if !ok { + return r.fail(ctx, tf, "recovery point handle is malformed: "+c.SnapshotID) + } + name := testFailoverCloneName(tf) + if tf.Spec.Scope == simplyblockv1alpha2.TestFailoverScopeGroup { + name = testFailoverMemberName(tf, c.SourceRef, "clone") + } + cloneUUID, sizeBytes, err := r.ensureClone(ctx, apiClient, snapCluster, snapPool, snapUUID, name) + if err != nil { + return ctrl.Result{}, err + } + handles[i] = snapCluster + ":" + snapPool + ":" + cloneUUID + sizes[i] = sizeBytes + } + + if err := r.transitionTo(ctx, tf, simplyblockv1alpha2.TestFailoverStepPlacing, func(s *simplyblockv1alpha2.TestFailoverStatus) { + for i := range s.Clones { + s.Clones[i].CloneID = handles[i] + if sizes[i] > 0 { + s.Clones[i].SizeBytes = sizes[i] + } + } + s.Message = fmt.Sprintf("cloned %d recovery point(s); placing on %s", len(s.Clones), tf.Spec.BubbleCluster) + }); err != nil { + return ctrl.Result{}, err + } + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "CloneBuilt", "CloneBuilt", + "cloned %d recovery point(s) on cluster %s, source untouched", len(tf.Status.Clones), tf.Spec.BubbleCluster) + return ctrl.Result{Requeue: true}, nil +} + +// ensureClone returns the id of the clone named name, cloning the snapshot if it +// does not exist yet. Ask-then-act: it lists the pool's volumes first, so a retry +// after a crash between clone and status-write reuses the clone rather than +// building a second (reconciler-patterns §4). +func (r *TestFailoverReconciler) ensureClone(ctx context.Context, api *webapi.Client, clusterUUID, poolID, snapUUID, name string) (string, int64, error) { + listEndpoint := fmt.Sprintf("/api/v2/clusters/%s/storage-pools/%s/volumes", clusterUUID, poolID) + body, status, err := api.Do(ctx, http.MethodGet, listEndpoint, nil) + if err != nil || status >= 300 { + return "", 0, requestError("list volumes", body, status, err) + } + var existing []struct { + ID string `json:"id"` + Name string `json:"name"` + Size int64 `json:"size"` + } + _ = json.Unmarshal(body, &existing) + for _, v := range existing { + if v.Name == name { + return v.ID, v.Size, nil + } + } + + createEndpoint := fmt.Sprintf("/api/v2/clusters/%s/storage-pools/%s/volumes?response_format=full", clusterUUID, poolID) + body, status, err = api.Do(ctx, http.MethodPost, createEndpoint, + map[string]interface{}{"name": name, "snapshot_id": snapUUID}) + if err != nil || status >= 300 { + return "", 0, requestError("clone snapshot", body, status, err) + } + var created struct { + ID string `json:"id"` + Size int64 `json:"size"` + } + if err := json.Unmarshal(body, &created); err != nil || created.ID == "" { + return "", 0, fmt.Errorf("clone snapshot: no volume id in the response") + } + return created.ID, created.Size, nil +} + +// placeBubble delivers the bubble PV and PVC to the recovery cluster through an +// OCM ManifestWork and marks the drill Ready once the PVC binds. It is +// non-blocking: the ManifestWork is created once and its bind status awaited +// across reconciles via its status feedback. +func (r *TestFailoverReconciler) placeBubble(ctx context.Context, tf *simplyblockv1alpha2.TestFailover) (ctrl.Result, error) { + if len(tf.Status.Clones) == 0 || tf.Status.Clones[0].CloneID == "" { + return r.fail(ctx, tf, "internal: the clone was not built before Placing") + } + + if err := r.ensureBubbleNamespaceWork(ctx, tf); err != nil { + return ctrl.Result{}, err + } + var mw workv1.ManifestWork + err := r.Get(ctx, client.ObjectKey{Namespace: tf.Spec.BubbleCluster, Name: testFailoverManifestWorkName(tf)}, &mw) + if apierrors.IsNotFound(err) { + desired, buildErr := r.bubbleManifestWork(tf) + if buildErr != nil { + return ctrl.Result{}, buildErr + } + if createErr := r.Create(ctx, desired); createErr != nil { + return ctrl.Result{}, createErr + } + return r.hold(ctx, tf, "placing the bubble PVC on cluster "+tf.Spec.BubbleCluster) + } + if err != nil { + return ctrl.Result{}, err + } + + if bound := boundBubblePVCs(&mw); bound < len(tf.Status.Clones) { + return r.hold(ctx, tf, fmt.Sprintf("waiting for the bubble PVCs to bind on cluster %s (%d of %d bound)", tf.Spec.BubbleCluster, bound, len(tf.Status.Clones))) + } + + // The non-disruptiveness guard: the drill only ever read the source, so it + // must still be the same volume it started from. A drill that cannot prove + // this is a defect, not a pass. + held := r.sourceUnchanged(ctx, tf) + now := metav1.Now() + if err := r.patchStatus(ctx, tf, func(s *simplyblockv1alpha2.TestFailoverStatus) { + if s.Report == nil { + s.Report = &simplyblockv1alpha2.TestFailoverReport{} + } + s.Report.InvariantsHeld = held + if held { + s.Phase = simplyblockv1alpha2.TestFailoverPhaseReady + s.Message = "bubble ready: the recovered PVC is bound on cluster " + tf.Spec.BubbleCluster + if s.ReadyAt == nil { + s.ReadyAt = &now + } + } else { + s.Phase = simplyblockv1alpha2.TestFailoverPhaseFailed + s.Message = "the source changed during the drill; the test was not non-disruptive" + if s.CompletedAt == nil { + s.CompletedAt = &now + } + } + }); err != nil { + return ctrl.Result{}, err + } + if held { + r.Recorder.Eventf(tf, nil, corev1.EventTypeNormal, "BubbleReady", "BubbleReady", + "the bubble PVC is bound on cluster %s", tf.Spec.BubbleCluster) + } else { + r.Recorder.Eventf(tf, nil, corev1.EventTypeWarning, "InvariantViolated", "InvariantViolated", + "the source changed during the drill") + } + return ctrl.Result{}, nil +} + +// volumeModeOf is the bubble PV's and PVC's volumeMode: the source's, so a +// VM's Block disk stays a block device instead of being mounted as a +// filesystem; nil (the default, Filesystem) when the source did not say. +func volumeModeOf(clone simplyblockv1alpha2.TestFailoverClone) *corev1.PersistentVolumeMode { + if clone.SourceVolumeMode == "" { + return nil + } + mode := corev1.PersistentVolumeMode(clone.SourceVolumeMode) + return &mode +} + +// bubbleVolumeContextStripKeys are the source PV volumeAttributes that must NOT +// be carried onto the bubble PV: they identify the SOURCE volume and its NVMe-oF +// target. The node plugin re-resolves the clone's own identity from the clone +// handle at stage time, so these are redundant on success; on a failed clone +// lookup, a stale source NQN/connections here would silently point the mount back +// at the source (reachable across clusters on a flat network), so they are +// dropped and staging fails safe instead. +var bubbleVolumeContextStripKeys = map[string]struct{}{ + "cluster_id": {}, "pool_name": {}, "nqn": {}, "connections": {}, + "model": {}, "name": {}, "uuid": {}, "nsId": {}, "targetLvolID": {}, +} + +// bubbleVolumeContext copies the source PV's volumeAttributes minus the identity +// keys above and the provisioner-injected keys (csi.storage.k8s.io/*, +// storage.kubernetes.io/*), leaving the class-level parameters the node plugin +// needs. It returns nil when nothing survives, which the driver tolerates. +func bubbleVolumeContext(src map[string]string) map[string]string { + if len(src) == 0 { + return nil + } + out := make(map[string]string, len(src)) + for k, v := range src { + if _, strip := bubbleVolumeContextStripKeys[k]; strip { + continue + } + if strings.HasPrefix(k, "csi.storage.k8s.io/") || strings.HasPrefix(k, "storage.kubernetes.io/") { + continue + } + out[k] = v + } + if len(out) == 0 { + return nil + } + return out +} + +// bubbleManifestWork wraps the bubble namespace, a static PersistentVolume bound +// to the clone, and its PersistentVolumeClaim, in a ManifestWork addressed to the +// recovery cluster, with a feedback rule that reports the PVC's bind phase back to +// the hub. +func (r *TestFailoverReconciler) bubbleManifestWork(tf *simplyblockv1alpha2.TestFailover) (*workv1.ManifestWork, error) { + ns := tf.Spec.BubbleNamespace + labels := map[string]string{testFailoverIDLabel: string(tf.UID)} + scName := "" + group := tf.Spec.Scope == simplyblockv1alpha2.TestFailoverScopeGroup + + // A PV+PVC pair per clone slot. The bubble namespace is not in this work: + // every drill of one test shares it, so it has one owner, the bubble's + // namespace work (bubbleNamespaceWork). + manifests := make([]workv1.Manifest, 0, 2*len(tf.Status.Clones)) + + // One PV+PVC pair per clone slot, each reporting its own bind phase back to the + // hub. A volume drill has one; a group drill has one per member, all in the one + // bubble namespace so the recovered set is crash-consistent. + configs := make([]workv1.ManifestConfigOption, 0, len(tf.Status.Clones)) + for i := range tf.Status.Clones { + clone := tf.Status.Clones[i] + pvName := testFailoverPVName(tf) + pvcName := tf.Spec.SourceRef + if group { + pvName = testFailoverMemberName(tf, clone.SourceRef, "pv") + pvcName = clone.SourceRef + } + capacity := *resource.NewQuantity(clone.SizeBytes, resource.BinarySI) + + pv := &corev1.PersistentVolume{ + TypeMeta: metav1.TypeMeta{Kind: "PersistentVolume", APIVersion: "v1"}, + ObjectMeta: metav1.ObjectMeta{Name: pvName, Labels: labels}, + Spec: corev1.PersistentVolumeSpec{ + Capacity: corev1.ResourceList{corev1.ResourceStorage: capacity}, + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + PersistentVolumeReclaimPolicy: corev1.PersistentVolumeReclaimRetain, + StorageClassName: scName, + VolumeMode: volumeModeOf(clone), + ClaimRef: &corev1.ObjectReference{ + Kind: "PersistentVolumeClaim", APIVersion: "v1", Namespace: ns, Name: pvcName, + }, + PersistentVolumeSource: corev1.PersistentVolumeSource{ + CSI: &corev1.CSIPersistentVolumeSource{ + Driver: csiDriverName, + VolumeHandle: clone.CloneID, + FSType: clone.SourceFSType, + VolumeAttributes: clone.SourceVolumeContext, + }, + }, + }, + } + pvc := &corev1.PersistentVolumeClaim{ + TypeMeta: metav1.TypeMeta{Kind: "PersistentVolumeClaim", APIVersion: "v1"}, + ObjectMeta: metav1.ObjectMeta{Name: pvcName, Namespace: ns, Labels: labels}, + Spec: corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{Requests: corev1.ResourceList{corev1.ResourceStorage: capacity}}, + StorageClassName: &scName, + VolumeName: pvName, + VolumeMode: volumeModeOf(clone), + }, + } + for _, obj := range []client.Object{pv, pvc} { + raw, err := json.Marshal(obj) + if err != nil { + return nil, fmt.Errorf("marshal bubble manifest: %w", err) + } + manifests = append(manifests, workv1.Manifest{RawExtension: runtime.RawExtension{Raw: raw}}) + } + configs = append(configs, workv1.ManifestConfigOption{ + ResourceIdentifier: workv1.ResourceIdentifier{ + Group: "", Resource: "persistentvolumeclaims", Namespace: ns, Name: pvcName, + }, + FeedbackRules: []workv1.FeedbackRule{{ + Type: workv1.JSONPathsType, + JsonPaths: []workv1.JsonPath{{Name: "phase", Path: ".status.phase"}}, + }}, + UpdateStrategy: createOnly(), + }, workv1.ManifestConfigOption{ + ResourceIdentifier: workv1.ResourceIdentifier{Group: "", Resource: "persistentvolumes", Name: pvName}, + UpdateStrategy: createOnly(), + }) + } + + return &workv1.ManifestWork{ + ObjectMeta: metav1.ObjectMeta{ + Name: testFailoverManifestWorkName(tf), + Namespace: tf.Spec.BubbleCluster, + Labels: labels, + }, + Spec: workv1.ManifestWorkSpec{ + Workload: workv1.ManifestsTemplate{Manifests: manifests}, + ManifestConfigs: configs, + }, + }, nil +} + +// boundBubblePVCs counts the bubble PVCs the ManifestWork's status feedback +// reports Bound. Placing is complete only when every clone's PVC is bound. +func boundBubblePVCs(mw *workv1.ManifestWork) int { + bound := 0 + for _, m := range mw.Status.ResourceStatus.Manifests { + if m.ResourceMeta.Resource != "persistentvolumeclaims" { + continue + } + for _, v := range m.StatusFeedbacks.Values { + if v.Name == "phase" && v.Value.String != nil && *v.Value.String == string(corev1.ClaimBound) { + bound++ + break + } + } + } + return bound +} + +// testFailoverPVName is the deterministic name of the drill's static +// PersistentVolume on the recovery cluster. +func testFailoverPVName(tf *simplyblockv1alpha2.TestFailover) string { + h := sha256.Sum256([]byte(tf.Namespace + "/" + tf.Name)) + return fmt.Sprintf("tfo-%x-pv", h[:6]) +} + +// testFailoverManifestWorkName is the deterministic name of the drill's +// ManifestWork in the recovery cluster's namespace on the hub. +func testFailoverManifestWorkName(tf *simplyblockv1alpha2.TestFailover) string { + h := sha256.Sum256([]byte(tf.Namespace + "/" + tf.Name)) + return fmt.Sprintf("tfo-%x-bubble", h[:6]) +} + +// createOnly is the update strategy of every object a drill places: created once, +// never re-applied. The default (Update) re-applied the whole PersistentVolume and +// wiped spec.claimRef.uid, which the PV controller sets on binding; the PV fell +// back to Available and the PVC became Lost ("ClaimMisbound: Two claims are bound +// to the same volume", 2026-10-04). ServerSideApply does not help: claimRef is an +// atomic struct, so the work agent would still own all of it. Nothing a drill +// places changes after creation. +func createOnly() *workv1.UpdateStrategy { + return &workv1.UpdateStrategy{Type: workv1.UpdateStrategyTypeCreateOnly} +} + +// bubbleNamespaceWorkName is the name of the one ManifestWork that owns a bubble +// namespace on its recovery cluster, shared by every drill placing into it. +func bubbleNamespaceWorkName(cluster, namespace string) string { + h := sha256.Sum256([]byte(cluster + "/" + namespace)) + return fmt.Sprintf("tfo-ns-%x", h[:6]) +} + +// bubbleNamespaceWork is the ManifestWork holding only the bubble namespace. One +// test fail-over creates a drill per protected PVC, all in the same bubble +// namespace; with the namespace in each drill's own work, the works fought over +// its labels and the first drill torn down deleted the namespace under the +// others' PVCs. +func bubbleNamespaceWork(tf *simplyblockv1alpha2.TestFailover) (*workv1.ManifestWork, error) { + labels := map[string]string{testFailoverIDLabel: string(tf.UID)} + namespace := &corev1.Namespace{ + TypeMeta: metav1.TypeMeta{Kind: "Namespace", APIVersion: "v1"}, + ObjectMeta: metav1.ObjectMeta{Name: tf.Spec.BubbleNamespace, Labels: labels}, + } + raw, err := json.Marshal(namespace) + if err != nil { + return nil, fmt.Errorf("marshal bubble namespace: %w", err) + } + return &workv1.ManifestWork{ + ObjectMeta: metav1.ObjectMeta{ + Name: bubbleNamespaceWorkName(tf.Spec.BubbleCluster, tf.Spec.BubbleNamespace), + Namespace: tf.Spec.BubbleCluster, + Labels: labels, + }, + Spec: workv1.ManifestWorkSpec{ + Workload: workv1.ManifestsTemplate{ + Manifests: []workv1.Manifest{{RawExtension: runtime.RawExtension{Raw: raw}}}, + }, + ManifestConfigs: []workv1.ManifestConfigOption{{ + ResourceIdentifier: workv1.ResourceIdentifier{ + Group: "", Resource: "namespaces", Name: tf.Spec.BubbleNamespace, + }, + UpdateStrategy: createOnly(), + }}, + }, + }, nil +} + +// ensureBubbleNamespaceWork creates the bubble's namespace work unless another +// drill of the same bubble already did. +func (r *TestFailoverReconciler) ensureBubbleNamespaceWork( + ctx context.Context, tf *simplyblockv1alpha2.TestFailover, +) error { + desired, err := bubbleNamespaceWork(tf) + if err != nil { + return err + } + if err := r.Create(ctx, desired); err != nil && !apierrors.IsAlreadyExists(err) { + return err + } + return nil +} + +// liveDrillOnBubble reports whether a drill other than tf, not itself being +// deleted, places into the same bubble namespace on the same cluster. +func liveDrillOnBubble(items []simplyblockv1alpha2.TestFailover, tf *simplyblockv1alpha2.TestFailover) bool { + for i := range items { + o := &items[i] + if o.UID == tf.UID || !o.DeletionTimestamp.IsZero() { + continue + } + if o.Spec.BubbleCluster == tf.Spec.BubbleCluster && o.Spec.BubbleNamespace == tf.Spec.BubbleNamespace { + return true + } + } + return false +} + +// deleteBubbleNamespaceWork removes the bubble's namespace work once no other live +// drill places into it; the last drill torn down takes the namespace with it. +func (r *TestFailoverReconciler) deleteBubbleNamespaceWork( + ctx context.Context, tf *simplyblockv1alpha2.TestFailover, +) error { + var list simplyblockv1alpha2.TestFailoverList + if err := r.List(ctx, &list); err != nil { + return err + } + if liveDrillOnBubble(list.Items, tf) { + return nil + } + var mw workv1.ManifestWork + key := client.ObjectKey{ + Namespace: tf.Spec.BubbleCluster, + Name: bubbleNamespaceWorkName(tf.Spec.BubbleCluster, tf.Spec.BubbleNamespace), + } + if err := r.Get(ctx, key, &mw); err != nil { + return client.IgnoreNotFound(err) + } + if !mw.DeletionTimestamp.IsZero() { + return nil + } + return client.IgnoreNotFound(r.Delete(ctx, &mw)) +} + +// clusterSecret returns the cluster secret the control plane authenticates a +// per-cluster call with, from the hub Secret simplyblock-cluster-. +func (r *TestFailoverReconciler) clusterSecret(ctx context.Context, namespace, clusterName string) (string, error) { + var secret corev1.Secret + if err := r.Get(ctx, client.ObjectKey{Namespace: namespace, Name: "simplyblock-cluster-" + clusterName}, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil +} + +// testFailoverCloneName is the deterministic name of the clone a drill builds, so +// ask-then-act can find it on a retry. +func testFailoverCloneName(tf *simplyblockv1alpha2.TestFailover) string { + h := sha256.Sum256([]byte(tf.Namespace + "/" + tf.Name)) + return fmt.Sprintf("tfo-%x-clone", h[:6]) +} + +// testFailoverMemberName is the deterministic name of one group member's drill +// object (clone or PV), keyed by the drill and the member's source ref, so a +// group drill's members do not collide and each is re-findable on a retry. +func testFailoverMemberName(tf *simplyblockv1alpha2.TestFailover, memberRef, kind string) string { + h := sha256.Sum256([]byte(tf.Namespace + "/" + tf.Name)) + m := sha256.Sum256([]byte(memberRef)) + return fmt.Sprintf("tfo-%x-%x-%s", h[:6], m[:4], kind) +} + +// splitHandle splits a CSI handle "cluster:pool:uuid" into its three parts. +func splitHandle(handle string) (cluster, pool, uuid string, ok bool) { + parts := strings.Split(handle, ":") + if len(parts) != 3 || parts[0] == "" || parts[1] == "" || parts[2] == "" { + return "", "", "", false + } + return parts[0], parts[1], parts[2], true +} + +// requestError builds an error for a failed control-plane call. +func requestError(what string, body []byte, status int, err error) error { + if err != nil { + return fmt.Errorf("%s: %w", what, err) + } + return fmt.Errorf("%s: status %d: %s", what, status, string(body)) +} + +// SetupWithManager registers the controller. +func (r *TestFailoverReconciler) SetupWithManager(mgr ctrl.Manager) error { + return ctrl.NewControllerManagedBy(mgr). + For(&simplyblockv1alpha2.TestFailover{}). + Owns(&workv1.ManifestWork{}). + Named("testfailover"). + Complete(r) +} diff --git a/operator/internal/controller/testfailover_controller_unit_test.go b/operator/internal/controller/testfailover_controller_unit_test.go new file mode 100644 index 000000000..cc4af3d0a --- /dev/null +++ b/operator/internal/controller/testfailover_controller_unit_test.go @@ -0,0 +1,1469 @@ +package controller + +import ( + "context" + "encoding/json" + "net/http" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/types" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller/controllerutil" + + "github.com/simplyblock/atlas/statemachine" + workv1 "open-cluster-management.io/api/work/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +// The handles the drill fixtures resolve to at each step. The source lives on +// clusterA and the recovery bubble on clusterB, so bubble != source in the sample. +const ( + testSourceHandle = "clusterA:poolA:lvolX" + testSnapshotHandle = "clusterB:poolB:snapY" + testCloneHandle = "clusterB:poolB:cloneVol" + testFSTypeXFS = "xfs" + testFabricTCP = "tcp" + testSourceCluster = "ramen-cluster-a" +) + +// atResolvingPoint returns a drill seeded at ResolvingPoint with its source +// already resolved to testSourceHandle. +func atResolvingPoint() *simplyblockv1alpha2.TestFailover { + tf := atResolvingSource() + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingPoint)} + tf.Status.Clones = []simplyblockv1alpha2.TestFailoverClone{{ + SourceRef: tf.Spec.SourceRef, + SourceHandle: testSourceHandle, + }} + return tf +} + +func newTestFailoverReconciler(t *testing.T, objects ...client.Object) (*TestFailoverReconciler, client.Client) { + t.Helper() + scheme := newTestScheme(t) + // The ManagedClusterView is driven unstructured, so the fake client needs its + // GVK (and list GVK) registered to create and read it. + scheme.AddKnownTypeWithName(managedClusterViewGVK, &unstructured.Unstructured{}) + listGVK := managedClusterViewGVK + listGVK.Kind += listKindSuffix + scheme.AddKnownTypeWithName(listGVK, &unstructured.UnstructuredList{}) + if err := workv1.Install(scheme); err != nil { + t.Fatalf("register work/v1 scheme: %v", err) + } + cl := newTestClient(t, scheme, + []client.Object{&simplyblockv1alpha2.TestFailover{}, &workv1.ManifestWork{}}, + objects...) + return &TestFailoverReconciler{Client: cl, Scheme: scheme, Recorder: &fakeRecorder{}}, cl +} + +// getView reads a ManagedClusterView the controller created. +func getView(t *testing.T, cl client.Client, namespace, name string) *unstructured.Unstructured { + t.Helper() + v := &unstructured.Unstructured{} + v.SetGroupVersionKind(managedClusterViewGVK) + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: namespace, Name: name}, v); err != nil { + t.Fatalf("get ManagedClusterView %s/%s: %v", namespace, name, err) + } + return v +} + +// setViewResult simulates the OCM view controller having projected an object, +// by writing status.result onto the view. +func setViewResult(t *testing.T, cl client.Client, v *unstructured.Unstructured, result map[string]interface{}) { + t.Helper() + if err := unstructured.SetNestedMap(v.Object, result, "status", "result"); err != nil { + t.Fatalf("set status.result: %v", err) + } + if err := cl.Update(context.Background(), v); err != nil { + t.Fatalf("update view with result: %v", err) + } +} + +// atResolvingSource returns a drill seeded at the ResolvingSource step, the state +// the entry transition leaves it in. +func atResolvingSource() *simplyblockv1alpha2.TestFailover { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingSource)} + return tf +} + +func sampleTestFailover() *simplyblockv1alpha2.TestFailover { + return &simplyblockv1alpha2.TestFailover{ + ObjectMeta: metav1.ObjectMeta{Name: "drill-1", Namespace: "simplyblock"}, + Spec: simplyblockv1alpha2.TestFailoverSpec{ + Scope: simplyblockv1alpha2.TestFailoverScopeVolume, + SourceCluster: testSourceCluster, + SourceNamespace: "prod-app", + SourceRef: "postgres-data", + BubbleCluster: "ramen-cluster-b", + }, + } +} + +func testFailoverRequest(tf *simplyblockv1alpha2.TestFailover) ctrl.Request { + return ctrl.Request{NamespacedName: types.NamespacedName{Name: tf.Name, Namespace: tf.Namespace}} +} + +// TestFailoverAddsFinalizerThenEntersResolvingSource covers the entry into the +// drill: the first reconcile adds the teardown finalizer, and the next begins +// the drill at the initial step with the phase and bookkeeping set. +func TestFailoverAddsFinalizerThenEntersResolvingSource(t *testing.T) { + tf := sampleTestFailover() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + // First pass: the finalizer is added and nothing else has happened yet. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("first reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatalf("get after first reconcile: %v", err) + } + if !controllerutil.ContainsFinalizer(&got, finalizerTestFailover) { + t.Fatalf("finalizer %q was not added", finalizerTestFailover) + } + if got.Status.Phase != "" { + t.Errorf("phase = %q, want empty before the drill begins", got.Status.Phase) + } + + // Second pass: the drill begins at ResolvingSource. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("second reconcile: %v", err) + } + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatalf("get after second reconcile: %v", err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseProvisioning { + t.Errorf("phase = %q, want %q", got.Status.Phase, simplyblockv1alpha2.TestFailoverPhaseProvisioning) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepResolvingSource) { + t.Errorf("step = %q, want %q", got.Status.Step.State, simplyblockv1alpha2.TestFailoverStepResolvingSource) + } + if got.Status.ObservedGeneration != got.Generation { + t.Errorf("observedGeneration = %d, want %d", got.Status.ObservedGeneration, got.Generation) + } + if got.Status.StartedAt == nil { + t.Errorf("startedAt was not set") + } +} + +// TestFailoverDeletionClearsTheFinalizer covers that a deleted drill is torn +// down and does not hang on its finalizer. +func TestFailoverDeletionClearsTheFinalizer(t *testing.T) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + now := metav1.Now() + tf.DeletionTimestamp = &now + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile of a deleting drill: %v", err) + } + var got simplyblockv1alpha2.TestFailover + err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got) + if err == nil && controllerutil.ContainsFinalizer(&got, finalizerTestFailover) { + t.Fatalf("finalizer still present after deletion reconcile") + } +} + +// TestFailoverResolvingSourceResolvesHandleThenAdvances covers the full source +// read: the controller creates a ManagedClusterView for the PVC, then for its +// PV, and once both are projected it records the volume handle and advances to +// ResolvingPoint. Each projection arrives across reconciles, never blocking. +func TestFailoverResolvingSourceResolvesHandleThenAdvances(t *testing.T) { + tf := atResolvingSource() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + // Pass 1: the PVC view is created and the drill holds on its projection. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + pvcView := getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pvc")) + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepResolvingSource) { + t.Fatalf("step advanced before the source was projected: %q", got.Status.Step.State) + } + setViewResult(t, cl, pvcView, map[string]interface{}{ + "spec": map[string]interface{}{"volumeName": "pv-1"}, + }) + + // Pass 2: the PVC result is read, and the PV view is created. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + pvView := getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pv")) + setViewResult(t, cl, pvView, map[string]interface{}{ + "spec": map[string]interface{}{ + "csi": map[string]interface{}{"volumeHandle": "clusterA:pool:lvolX"}, + }, + }) + + // Pass 3: the handle is read and the drill advances to ResolvingPoint. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 3: %v", err) + } + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepResolvingPoint) { + t.Errorf("step = %q, want %q", got.Status.Step.State, simplyblockv1alpha2.TestFailoverStepResolvingPoint) + } + if len(got.Status.Clones) != 1 || got.Status.Clones[0].SourceHandle != "clusterA:pool:lvolX" { + t.Errorf("clones = %+v, want one with sourceHandle clusterA:pool:lvolX", got.Status.Clones) + } +} + +// Regression: 2026-09-29-testfailover-nil-volumecontext — the bubble PV must +// carry a VolumeContext or the node plugin panics staging it. The source PV's +// volumeAttributes are captured here, minus the identity and provisioner keys: +// the class params are needed to stage, but the identity keys would point a +// failed clone lookup back at the source, so they are dropped. +func TestFailoverResolvingSourceCapturesStrippedVolumeContext(t *testing.T) { + tf := atResolvingSource() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + setViewResult(t, cl, getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pvc")), + map[string]interface{}{"spec": map[string]interface{}{"volumeName": "pv-1"}}) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + setViewResult(t, cl, getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pv")), + map[string]interface{}{"spec": map[string]interface{}{"csi": map[string]interface{}{ + "volumeHandle": "clusterA:pool:lvolX", + "fsType": testFSTypeXFS, + "volumeAttributes": map[string]interface{}{ + // class params — kept + "tune2fs_reserved_blocks": "", + "fabric": testFabricTCP, + "qos_rw_iops": "0", + // identity — stripped (would mis-point a failed clone lookup) + "cluster_id": "clusterA", + "pool_name": "poolA", + "nqn": "nqn.source", + "connections": "[{\"ip\":\"10.0.0.1\",\"port\":4420}]", + "uuid": "lvolX", + "nsId": "1", + "model": "lvolX", + // provisioner-injected — stripped (stale source metadata) + "csi.storage.k8s.io/pv/name": "pv-1", + "storage.kubernetes.io/csiProvisionerIdentity": "x", + }, + }}}) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 3: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if len(got.Status.Clones) != 1 { + t.Fatalf("clones = %+v, want one", got.Status.Clones) + } + if got.Status.Clones[0].SourceFSType != testFSTypeXFS { + t.Errorf("sourceFSType = %q, want the source PV's xfs", got.Status.Clones[0].SourceFSType) + } + vc := got.Status.Clones[0].SourceVolumeContext + for _, k := range []string{"tune2fs_reserved_blocks", "fabric", "qos_rw_iops"} { + if _, ok := vc[k]; !ok { + t.Errorf("class param %q was dropped from the bubble VolumeContext: %+v", k, vc) + } + } + for _, k := range []string{ + "cluster_id", "pool_name", "nqn", "connections", "uuid", "nsId", "model", + "csi.storage.k8s.io/pv/name", "storage.kubernetes.io/csiProvisionerIdentity", + } { + if _, ok := vc[k]; ok { + t.Errorf("identity/provisioner key %q leaked into the bubble VolumeContext: %+v", k, vc) + } + } +} + +// TestFailoverResolvingSourceGroupResolvesMembers covers the group source path: +// the members come from the source cluster's backend (each carries only an lvol +// id, resolved to its PVC), one representative PV supplies the shared class +// metadata, and the drill advances to ResolvingPoint with one clone slot per +// member. +func TestFailoverResolvingSourceGroupResolvesMembers(t *testing.T) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Spec.Scope = simplyblockv1alpha2.TestFailoverScopeGroup + tf.Spec.SourceRef = "cg" + tf.Spec.SourceCluster = testSourceCluster // not a UUID: exercises the sole-StorageCluster fallback + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingSource)} + + sc := &simplyblockv1alpha2.StorageCluster{ + ObjectMeta: metav1.ObjectMeta{Name: "local-sc", Namespace: tf.Namespace}, + Status: simplyblockv1alpha2.StorageClusterStatus{UUID: "C"}, + } + r, cl := newTestFailoverReconciler(t, tf, sc) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + w.Header().Set("Content-Type", "application/json") + p := req.URL.Path + switch { + case strings.HasSuffix(p, "/consistency-groups/") && req.URL.Query().Get("name") == "cg": + _, _ = w.Write([]byte(`[{"id":"g1","name":"cg","lvs_name":"lvs-a","node_id":"node-a"}]`)) + case strings.HasSuffix(p, "/consistency-groups/g1/members"): + _, _ = w.Write([]byte(`[{"lvol_id":"lvol-a"},{"lvol_id":"lvol-b"}]`)) + case strings.HasSuffix(p, "/storage-pools/"): + _, _ = w.Write([]byte(`[{"id":"pool-1"}]`)) + case strings.HasSuffix(p, "/storage-pools/pool-1/volumes"): + _, _ = w.Write([]byte(`[{"id":"lvol-a","pvc_name":"app/data-1","namespace":"nvme-ns","pool_id":null,"size":1073741824},` + + `{"id":"lvol-b","pvc_name":"app/data-2","namespace":"nvme-ns","pool_id":null,"size":1073741824}]`)) + default: + t.Errorf("unexpected request %s %s", req.Method, p) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + // Pass 1: backend resolution done, the representative PVC view is created. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + setViewResult(t, cl, getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pvc")), + map[string]interface{}{"spec": map[string]interface{}{"volumeName": "pv-1"}}) + + // Pass 2: the representative PV view is created. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + setViewResult(t, cl, getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pv")), + map[string]interface{}{"spec": map[string]interface{}{"csi": map[string]interface{}{ + "volumeHandle": "C:pool-1:lvol-a", + "fsType": testFSTypeXFS, + "volumeAttributes": map[string]interface{}{ + "fabric": testFabricTCP, + "nqn": "nqn.source", // identity: must be stripped + }, + }}}) + + // Pass 3: the clones are built and the drill advances. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 3: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepResolvingPoint) { + t.Fatalf("step = %q, want ResolvingPoint", got.Status.Step.State) + } + if len(got.Status.Clones) != 2 { + t.Fatalf("clones = %+v, want one per member (2)", got.Status.Clones) + } + want := map[string]string{"data-1": "C:pool-1:lvol-a", "data-2": "C:pool-1:lvol-b"} + for _, c := range got.Status.Clones { + if want[c.SourceRef] != c.SourceHandle { + t.Errorf("clone %q handle = %q, want %q", c.SourceRef, c.SourceHandle, want[c.SourceRef]) + } + if c.SourceFSType != testFSTypeXFS { + t.Errorf("clone %q fsType = %q, want xfs", c.SourceRef, c.SourceFSType) + } + if c.SourceVolumeContext["fabric"] != testFabricTCP { + t.Errorf("clone %q did not carry the shared class attrs: %+v", c.SourceRef, c.SourceVolumeContext) + } + if _, leaked := c.SourceVolumeContext["nqn"]; leaked { + t.Errorf("clone %q leaked the identity key nqn", c.SourceRef) + } + if c.SizeBytes != 1073741824 { + t.Errorf("clone %q size = %d, want 1Gi", c.SourceRef, c.SizeBytes) + } + } +} + +// TestFailoverResolvingSourceHoldsWithoutAProjection covers that a pending view +// holds the drill on its step rather than advancing or failing. +func TestFailoverResolvingSourceHoldsWithoutAProjection(t *testing.T) { + tf := atResolvingSource() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + res, err := r.Reconcile(ctx, testFailoverRequest(tf)) + if err != nil { + t.Fatalf("reconcile: %v", err) + } + if res.RequeueAfter == 0 { + t.Errorf("expected a requeue while waiting for the projection") + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("drill failed while merely waiting for a projection") + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepResolvingSource) { + t.Errorf("step moved off ResolvingSource while waiting: %q", got.Status.Step.State) + } +} + +// TestFailoverResolvingSourceReuseViewOnRestart covers restart safety: a second +// reconcile before the projection arrives finds the existing view rather than +// creating a duplicate. +func TestFailoverResolvingSourceReuseViewOnRestart(t *testing.T) { + tf := atResolvingSource() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2 (restart): %v", err) + } + + list := &unstructured.UnstructuredList{} + gvk := managedClusterViewGVK + gvk.Kind += listKindSuffix + list.SetGroupVersionKind(gvk) + if err := cl.List(ctx, list, client.InNamespace(tf.Spec.SourceCluster)); err != nil { + t.Fatalf("list views: %v", err) + } + if len(list.Items) != 1 { + t.Errorf("got %d ManagedClusterViews, want exactly 1 (no duplicate on restart)", len(list.Items)) + } +} + +// TestFailoverResolvingSourceUnboundPVCFails covers that a source PVC bound to no +// volume is a terminal failure, not an endless hold. +func TestFailoverResolvingSourceUnboundPVCFails(t *testing.T) { + tf := atResolvingSource() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + pvcView := getView(t, cl, tf.Spec.SourceCluster, testFailoverViewName(tf, "src-pvc")) + setViewResult(t, cl, pvcView, map[string]interface{}{ + "spec": map[string]interface{}{}, // no volumeName: unbound + }) + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("phase = %q, want Failed for an unbound source PVC", got.Status.Phase) + } +} + +// TestFailoverResolvingPointDRTargetUsesReplicatedSnapshot covers the DR-target +// path: the recovery point is the latest replicated snapshot already on the +// target backend, read (never taken), and the drill advances to Cloning. +func TestFailoverResolvingPointDRTargetUsesReplicatedSnapshot(t *testing.T) { + tf := atResolvingPoint() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + if req.Method == http.MethodGet && strings.HasSuffix(req.URL.Path, "/relationships/lvolX/latest-snapshot") { + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]interface{}{ + "snapshot_id": "snapY", "cluster_id": "clusterB", "pool_id": "poolB", + "created_at": "2026-09-29T00:00:00Z", + }) + return + } + t.Errorf("unexpected request %s %s", req.Method, req.URL.Path) + w.WriteHeader(http.StatusInternalServerError) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepCloning) { + t.Errorf("step = %q, want Cloning", got.Status.Step.State) + } + if got.Status.Clones[0].SnapshotID != "clusterB:poolB:snapY" { + t.Errorf("snapshot handle = %q, want clusterB:poolB:snapY", got.Status.Clones[0].SnapshotID) + } + if got.Status.Report == nil || got.Status.Report.RecoveryPoint != "snapY" { + t.Errorf("report.recoveryPoint not set to snapY: %+v", got.Status.Report) + } +} + +// TestFailoverResolvingPointGroupResolvesGeneration covers the group recovery +// point: the drill resolves the group's replication policy, reads its latest +// group-consistent generation, and records one target snapshot handle per clone +// slot before advancing to Cloning. +// +// Regression (2026-09-30): the drill inferred the policy from the policies list +// by matching the group's placement (group_lvs_name/group_node_id) and a +// consistency_group flag. A group attached with attach_group_policy sets +// group.policy_id and leaves both empty, so the heuristic matched nothing and the +// group drill failed at ResolvingPoint with "no consistency-group replication +// policy found." The policy id is read off the group DTO, which now carries it. +func TestFailoverResolvingPointGroupResolvesGeneration(t *testing.T) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Spec.Scope = simplyblockv1alpha2.TestFailoverScopeGroup + tf.Spec.SourceRef = "cg" + tf.Spec.SourceCluster = testSourceCluster + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingPoint)} + tf.Status.Clones = []simplyblockv1alpha2.TestFailoverClone{ + {SourceRef: "data-1", SourceHandle: "C:pool-1:lvol-a", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + {SourceRef: "data-2", SourceHandle: "C:pool-1:lvol-b", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + } + + sc := &simplyblockv1alpha2.StorageCluster{ + ObjectMeta: metav1.ObjectMeta{Name: "local-sc", Namespace: tf.Namespace}, + Status: simplyblockv1alpha2.StorageClusterStatus{UUID: "C"}, + } + r, cl := newTestFailoverReconciler(t, tf, sc) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + w.Header().Set("Content-Type", "application/json") + p := req.URL.Path + switch { + // The live group-first shape: the group carries its policy_id, and the + // policy itself exposes no placement and no consistency_group flag, so the + // drill must read the policy off the group rather than the policies list. + case strings.HasSuffix(p, "/consistency-groups/") && req.URL.Query().Get("name") == "cg": + _, _ = w.Write([]byte(`[{"id":"g1","name":"cg","lvs_name":"lvs-a","node_id":"node-a","policy_id":"p1"}]`)) + case strings.HasSuffix(p, "/replication/policies/p1/latest-generation"): + // Listed in the reverse order of the drill's PVCs: the pairing must be + // by source volume, not by position. + _, _ = w.Write([]byte(`{"group_seq":7,"members":[` + + `{"snapshot_id":"s2","cluster_id":"B","pool_id":"pb","lvol_id":"t2","source_lvol_id":"lvol-b",` + + `"size":1073741824,"group_seq":7},` + + `{"snapshot_id":"s1","cluster_id":"B","pool_id":"pb","lvol_id":"t1","source_lvol_id":"lvol-a",` + + `"size":1073741824,"group_seq":7}]}`)) + default: + t.Errorf("unexpected request %s %s", req.Method, p) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepCloning) { + t.Fatalf("step = %q, want Cloning", got.Status.Step.State) + } + if len(got.Status.Clones) != 2 || got.Status.Clones[0].SnapshotID != "B:pb:s1" || got.Status.Clones[1].SnapshotID != "B:pb:s2" { + t.Errorf("clone snapshot handles = %+v, want B:pb:s1 and B:pb:s2", got.Status.Clones) + } + if got.Status.Report == nil || got.Status.Report.RecoveryPoint != "generation 7" { + t.Errorf("report.recoveryPoint = %+v, want 'generation 7'", got.Status.Report) + } +} + +// groupAtResolvingPoint returns a two-member group drill seeded at +// ResolvingPoint, and the StorageCluster that resolves its source UUID. +func groupAtResolvingPoint() (*simplyblockv1alpha2.TestFailover, *simplyblockv1alpha2.StorageCluster) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Spec.Scope = simplyblockv1alpha2.TestFailoverScopeGroup + tf.Spec.SourceRef = "cg" + tf.Spec.SourceCluster = testSourceCluster + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingPoint)} + tf.Status.Clones = []simplyblockv1alpha2.TestFailoverClone{ + {SourceRef: "data-1", SourceHandle: "C:pool-1:lvol-a", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + {SourceRef: "data-2", SourceHandle: "C:pool-1:lvol-b", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + } + sc := &simplyblockv1alpha2.StorageCluster{ + ObjectMeta: metav1.ObjectMeta{Name: "local-sc", Namespace: tf.Namespace}, + Status: simplyblockv1alpha2.StorageClusterStatus{UUID: "C"}, + } + return tf, sc +} + +// TestFailoverResolvingPointGroupPairsByReplicaVolume covers a control plane +// whose generation names only the target replica volume: the drill reads each +// member's replica volume from its per-volume latest snapshot and pairs on it. +func TestFailoverResolvingPointGroupPairsByReplicaVolume(t *testing.T) { + tf, sc := groupAtResolvingPoint() + r, cl := newTestFailoverReconciler(t, tf, sc) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + w.Header().Set("Content-Type", "application/json") + p := req.URL.Path + switch { + case strings.HasSuffix(p, "/consistency-groups/") && req.URL.Query().Get("name") == "cg": + _, _ = w.Write([]byte(`[{"id":"g1","name":"cg","policy_id":"p1"}]`)) + case strings.HasSuffix(p, "/replication/policies/p1/latest-generation"): + _, _ = w.Write([]byte(`{"group_seq":7,"members":[` + + `{"snapshot_id":"s2","cluster_id":"B","pool_id":"pb","lvol_id":"t2","group_seq":7},` + + `{"snapshot_id":"s1","cluster_id":"B","pool_id":"pb","lvol_id":"t1","group_seq":7}]}`)) + case strings.HasSuffix(p, "/relationships/lvol-a/latest-snapshot"): + _, _ = w.Write([]byte(`{"snapshot_id":"s1","cluster_id":"B","pool_id":"pb","lvol_id":"t1"}`)) + case strings.HasSuffix(p, "/relationships/lvol-b/latest-snapshot"): + _, _ = w.Write([]byte(`{"snapshot_id":"s2","cluster_id":"B","pool_id":"pb","lvol_id":"t2"}`)) + default: + t.Errorf("unexpected request %s %s", req.Method, p) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepCloning) { + t.Fatalf("step = %q, want Cloning; message=%q", got.Status.Step.State, got.Status.Message) + } + if got.Status.Clones[0].SnapshotID != "B:pb:s1" || got.Status.Clones[1].SnapshotID != "B:pb:s2" { + t.Errorf("clone snapshot handles = %+v, want data-1=B:pb:s1, data-2=B:pb:s2", got.Status.Clones) + } +} + +// TestFailoverResolvingPointGroupMissingMemberFails covers a generation that +// lacks one PVC's member: the drill fails naming the PVC, never assigning +// another member's snapshot by position. +func TestFailoverResolvingPointGroupMissingMemberFails(t *testing.T) { + tf, sc := groupAtResolvingPoint() + r, cl := newTestFailoverReconciler(t, tf, sc) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + w.Header().Set("Content-Type", "application/json") + p := req.URL.Path + switch { + case strings.HasSuffix(p, "/consistency-groups/") && req.URL.Query().Get("name") == "cg": + _, _ = w.Write([]byte(`[{"id":"g1","name":"cg","policy_id":"p1"}]`)) + case strings.HasSuffix(p, "/replication/policies/p1/latest-generation"): + _, _ = w.Write([]byte(`{"group_seq":7,"members":[` + + `{"snapshot_id":"s1","cluster_id":"B","pool_id":"pb","lvol_id":"t1","source_lvol_id":"lvol-a"},` + + `{"snapshot_id":"s9","cluster_id":"B","pool_id":"pb","lvol_id":"t9","source_lvol_id":"lvol-z"}]}`)) + default: + t.Errorf("unexpected request %s %s", req.Method, p) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Fatalf("phase = %q, want Failed; message=%q", got.Status.Phase, got.Status.Message) + } + if !strings.Contains(got.Status.Message, "no snapshot for data-2") { + t.Errorf("message = %q, want it to name data-2", got.Status.Message) + } + for _, c := range got.Status.Clones { + if c.SnapshotID != "" { + t.Errorf("clone %s got snapshot %q despite the failed pairing", c.SourceRef, c.SnapshotID) + } + } +} + +// TestFailoverResolvingPointDRTargetNoReplicaFails covers that a target with no +// replicated point yet is a terminal failure, not an endless hold. +func TestFailoverResolvingPointDRTargetNoReplicaFails(t *testing.T) { + tf := atResolvingPoint() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusNotFound) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("phase = %q, want Failed when no replicated point exists", got.Status.Phase) + } +} + +// TestFailoverSameClusterIsRejected covers that a drill whose bubble is the +// source's own cluster fails immediately: test-failover recovers onto a DIFFERENT +// cluster, never in place. +func TestFailoverSameClusterIsRejected(t *testing.T) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Spec.BubbleCluster = tf.Spec.SourceCluster + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepResolvingSource)} + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Fatalf("phase = %q, want Failed for a same-cluster drill; message=%q", got.Status.Phase, got.Status.Message) + } + if !strings.Contains(got.Status.Message, "same cluster") { + t.Errorf("message = %q, want it to explain same-cluster is unsupported", got.Status.Message) + } +} + +// atCloning returns a drill seeded at Cloning with its recovery point resolved. +func atCloning() *simplyblockv1alpha2.TestFailover { + tf := atResolvingPoint() + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepCloning)} + tf.Status.Clones[0].SnapshotID = testSnapshotHandle + return tf +} + +// TestFailoverCloningClonesThenAdvances covers cloning the recovery point into a +// writable volume and advancing to Placing with the clone handle recorded. +func TestFailoverCloningClonesThenAdvances(t *testing.T) { + tf := atCloning() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + switch { + case req.Method == http.MethodGet && strings.HasSuffix(req.URL.Path, "/storage-pools/poolB/volumes"): + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte("[]")) + case req.Method == http.MethodPost && strings.HasSuffix(req.URL.Path, "/storage-pools/poolB/volumes"): + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]interface{}{"id": "cloneVol", "size": 1073741824}) + default: + t.Errorf("unexpected request %s %s", req.Method, req.URL.Path) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepPlacing) { + t.Errorf("step = %q, want Placing", got.Status.Step.State) + } + if got.Status.Clones[0].CloneID != "clusterB:poolB:cloneVol" { + t.Errorf("clone handle = %q, want clusterB:poolB:cloneVol", got.Status.Clones[0].CloneID) + } + if got.Status.Clones[0].SizeBytes != 1073741824 { + t.Errorf("clone size = %d, want 1073741824", got.Status.Clones[0].SizeBytes) + } +} + +// TestFailoverCloningReusesExistingClone covers ask-then-act idempotency: an +// existing clone with the drill's name is reused, and no second clone is built. +func TestFailoverCloningReusesExistingClone(t *testing.T) { + tf := atCloning() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + wantName := testFailoverCloneName(tf) + + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + if req.Method == http.MethodGet && strings.HasSuffix(req.URL.Path, "/storage-pools/poolB/volumes") { + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode([]map[string]interface{}{ + {"id": "cloneExisting", "name": wantName, "size": 2048}, + }) + return + } + if req.Method == http.MethodPost { + t.Errorf("a second clone was built though one already existed: %s", req.URL.Path) + } + w.WriteHeader(http.StatusInternalServerError) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Clones[0].CloneID != "clusterB:poolB:cloneExisting" { + t.Errorf("clone handle = %q, want clusterB:poolB:cloneExisting", got.Status.Clones[0].CloneID) + } +} + +// TestFailoverCloningRetriesOnServerError covers that a transient control-plane +// error is retried (error returned, no state advance), not swallowed. +func TestFailoverCloningRetriesOnServerError(t *testing.T) { + tf := atCloning() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusInternalServerError) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err == nil { + t.Fatalf("expected an error to trigger a retry on a 5xx") + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Step.State != string(simplyblockv1alpha2.TestFailoverStepCloning) { + t.Errorf("step = %q, want it to stay Cloning after a transient error", got.Status.Step.State) + } +} + +// atPlacing returns a drill seeded at Placing with its clone built. +func atPlacing() *simplyblockv1alpha2.TestFailover { + tf := atCloning() + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepPlacing)} + tf.Status.Clones[0].CloneID = testCloneHandle + tf.Status.Clones[0].SizeBytes = 1073741824 + return tf +} + +func getManifestWork(t *testing.T, cl client.Client, namespace, name string) *workv1.ManifestWork { + t.Helper() + var mw workv1.ManifestWork + if err := cl.Get(context.Background(), client.ObjectKey{Namespace: namespace, Name: name}, &mw); err != nil { + t.Fatalf("get ManifestWork %s/%s: %v", namespace, name, err) + } + return &mw +} + +// markManifestWorkPVCBound simulates the recovery cluster's work-agent reporting +// the bubble PVC bound through the ManifestWork status feedback. +func markManifestWorkPVCBound(t *testing.T, cl client.Client, mw *workv1.ManifestWork) { + t.Helper() + bound := string(corev1.ClaimBound) + mw.Status.ResourceStatus.Manifests = []workv1.ManifestCondition{{ + ResourceMeta: workv1.ManifestResourceMeta{Resource: "persistentvolumeclaims"}, + StatusFeedbacks: workv1.StatusFeedbackResult{Values: []workv1.FeedbackValue{{ + Name: "phase", + Value: workv1.FieldValue{Type: workv1.String, String: &bound}, + }}}, + }} + if err := cl.Status().Update(context.Background(), mw); err != nil { + t.Fatalf("update ManifestWork status: %v", err) + } +} + +// TestFailoverPlacingDeliversManifestWorkThenReady covers the placement step: a +// ManifestWork carrying the bubble PV and PVC (the namespace has its own work) is delivered to the +// recovery cluster, and the drill reaches Ready once the PVC binds. +func TestFailoverPlacingDeliversManifestWorkThenReady(t *testing.T) { + tf := atPlacing() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + // Pass 1: the ManifestWork is created and the drill holds on the bind. + res, err := r.Reconcile(ctx, testFailoverRequest(tf)) + if err != nil { + t.Fatalf("pass 1: %v", err) + } + if res.RequeueAfter == 0 { + t.Errorf("expected a requeue while waiting for the bubble PVC to bind") + } + mw := getManifestWork(t, cl, tf.Spec.BubbleCluster, testFailoverManifestWorkName(tf)) + if len(mw.Spec.Workload.Manifests) != 2 { + t.Errorf("ManifestWork carries %d manifests, want 2 (PV, PVC)", len(mw.Spec.Workload.Manifests)) + } + if len(mw.Spec.ManifestConfigs) != 2 || mw.Spec.ManifestConfigs[0].ResourceIdentifier.Name != tf.Spec.SourceRef { + t.Errorf("feedback rule not set on the bubble PVC %q: %+v", tf.Spec.SourceRef, mw.Spec.ManifestConfigs) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("drill reached Ready before the PVC was reported bound") + } + + // The work-agent reports the PVC bound. + markManifestWorkPVCBound(t, cl, mw) + + // Pass 2: the drill reaches Ready. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("phase = %q, want Ready", got.Status.Phase) + } + if got.Status.ReadyAt == nil { + t.Errorf("readyAt was not set") + } +} + +// Regression: 2026-09-29-testfailover-nil-volumecontext — the bubble PV must +// carry the source's class-level VolumeContext (so the node plugin has a non-nil +// context to stage) while still pointing at the clone by handle. +func TestFailoverPlacingPVCarriesSourceVolumeContext(t *testing.T) { + tf := atPlacing() + tf.Status.Clones[0].SourceVolumeContext = map[string]string{ + "tune2fs_reserved_blocks": "", + "fabric": testFabricTCP, + } + tf.Status.Clones[0].SourceFSType = testFSTypeXFS + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + mw := getManifestWork(t, cl, tf.Spec.BubbleCluster, testFailoverManifestWorkName(tf)) + // manifests are [PV, PVC]; decode the PV. + if len(mw.Spec.Workload.Manifests) != 2 { + t.Fatalf("ManifestWork carries %d manifests, want 2", len(mw.Spec.Workload.Manifests)) + } + var pv corev1.PersistentVolume + if err := json.Unmarshal(mw.Spec.Workload.Manifests[0].Raw, &pv); err != nil { + t.Fatalf("decode bubble PV manifest: %v", err) + } + if pv.Spec.CSI == nil { + t.Fatal("bubble PV has no CSI source") + } + if pv.Spec.CSI.VolumeHandle != tf.Status.Clones[0].CloneID { + t.Errorf("bubble PV points at %q, want the clone handle %q", pv.Spec.CSI.VolumeHandle, tf.Status.Clones[0].CloneID) + } + if pv.Spec.CSI.VolumeAttributes["fabric"] != testFabricTCP { + t.Errorf("bubble PV VolumeAttributes did not carry the source class params: %+v", pv.Spec.CSI.VolumeAttributes) + } + // The clone carries the source's filesystem; without this the node plugin + // defaults to ext4 and refuses to mount the XFS volume. + if pv.Spec.CSI.FSType != testFSTypeXFS { + t.Errorf("bubble PV fsType = %q, want the source's xfs", pv.Spec.CSI.FSType) + } +} + +// TestFailoverPlacingGroupDeliversAllMembersThenReady covers the group placement: +// one ManifestWork carries the namespace and a PV+PVC pair per member, and the +// drill reaches Ready only once every member's PVC binds. +func TestFailoverPlacingGroupDeliversAllMembersThenReady(t *testing.T) { + tf := sampleTestFailover() + tf.Finalizers = []string{finalizerTestFailover} + tf.Spec.Scope = simplyblockv1alpha2.TestFailoverScopeGroup + tf.Spec.SourceRef = "cg" + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseProvisioning + tf.Status.Step = statemachine.KubeSnapshot{State: string(simplyblockv1alpha2.TestFailoverStepPlacing)} + tf.Status.Clones = []simplyblockv1alpha2.TestFailoverClone{ + {SourceRef: "data-1", SourceHandle: "C:pool-1:lvol-a", SnapshotID: "B:pb:s1", CloneID: "B:pb:c1", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + {SourceRef: "data-2", SourceHandle: "C:pool-1:lvol-b", SnapshotID: "B:pb:s2", CloneID: "B:pb:c2", SourceFSType: testFSTypeXFS, SizeBytes: 1073741824}, + } + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + key := testFailoverRequest(tf).NamespacedName + + // Pass 1: the ManifestWork is created carrying 2*(PV,PVC) = 4 manifests and + // two configs per member (PVC feedback, PV strategy); the drill holds until both bind. + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + mw := getManifestWork(t, cl, tf.Spec.BubbleCluster, testFailoverManifestWorkName(tf)) + if len(mw.Spec.Workload.Manifests) != 4 { + t.Errorf("ManifestWork carries %d manifests, want 4 (2*(PV,PVC))", len(mw.Spec.Workload.Manifests)) + } + if len(mw.Spec.ManifestConfigs) != 4 { + t.Errorf("ManifestWork has %d configs, want two per member (4)", len(mw.Spec.ManifestConfigs)) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("drill reached Ready before any PVC bound") + } + + // Only one member bound: still not Ready. + bound := string(corev1.ClaimBound) + pvcBound := func(name string) workv1.ManifestCondition { + return workv1.ManifestCondition{ + ResourceMeta: workv1.ManifestResourceMeta{Resource: "persistentvolumeclaims", Name: name}, + StatusFeedbacks: workv1.StatusFeedbackResult{Values: []workv1.FeedbackValue{{Name: "phase", Value: workv1.FieldValue{Type: workv1.String, String: &bound}}}}, + } + } + mw.Status.ResourceStatus.Manifests = []workv1.ManifestCondition{pvcBound("data-1")} + if err := cl.Status().Update(ctx, mw); err != nil { + t.Fatalf("update MW status (one bound): %v", err) + } + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2: %v", err) + } + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase == simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("drill reached Ready with only one of two member PVCs bound") + } + + // Both bound: Ready. + mw.Status.ResourceStatus.Manifests = []workv1.ManifestCondition{pvcBound("data-1"), pvcBound("data-2")} + if err := cl.Status().Update(ctx, mw); err != nil { + t.Fatalf("update MW status (both bound): %v", err) + } + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 3: %v", err) + } + if err := cl.Get(ctx, key, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("phase = %q, want Ready once both member PVCs are bound", got.Status.Phase) + } +} + +// TestFailoverPlacingReuseManifestWorkOnRestart covers restart safety: a second +// reconcile before the PVC binds finds the existing ManifestWork, not a duplicate. +func TestFailoverPlacingReuseManifestWorkOnRestart(t *testing.T) { + tf := atPlacing() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 1: %v", err) + } + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("pass 2 (restart): %v", err) + } + var list workv1.ManifestWorkList + if err := cl.List(ctx, &list, client.InNamespace(tf.Spec.BubbleCluster)); err != nil { + t.Fatalf("list ManifestWorks: %v", err) + } + // The drill's own work plus the bubble's namespace work; neither duplicated. + if len(list.Items) != 2 { + t.Errorf("got %d ManifestWorks, want exactly 2 (drill + namespace, no duplicate on restart)", len(list.Items)) + } +} + +// deletingReadyDrill returns a Ready drill with a clone recorded, being deleted. +func deletingReadyDrill() *simplyblockv1alpha2.TestFailover { + tf := atPlacing() + tf.Status.Phase = simplyblockv1alpha2.TestFailoverPhaseReady + tf.Status.Clones[0].SnapshotID = "clusterA:poolA:snapS" + now := metav1.Now() + tf.DeletionTimestamp = &now + return tf +} + +func srcPVView(tf *simplyblockv1alpha2.TestFailover, handle string) *unstructured.Unstructured { + v := &unstructured.Unstructured{} + v.SetGroupVersionKind(managedClusterViewGVK) + v.SetNamespace(tf.Spec.SourceCluster) + v.SetName(testFailoverViewName(tf, "src-pv")) + _ = unstructured.SetNestedMap(v.Object, map[string]interface{}{ + "spec": map[string]interface{}{"csi": map[string]interface{}{"volumeHandle": handle}}, + }, "status", "result") + return v +} + +// TestFailoverTeardownReclaimsThenClearsFinalizer covers that deleting a drill +// reclaims the clone, removes the ManifestWork, and only then clears the +// finalizer. The recovery point is a replicated snapshot the drill only resolved, +// so it is left alone. +func TestFailoverTeardownReclaimsThenClearsFinalizer(t *testing.T) { + tf := deletingReadyDrill() + mw := &workv1.ManifestWork{ObjectMeta: metav1.ObjectMeta{ + Name: testFailoverManifestWorkName(tf), Namespace: tf.Spec.BubbleCluster, + }} + r, cl := newTestFailoverReconciler(t, tf, mw) + ctx := context.Background() + + var reclaimedClone bool + srv := newAPIServer(t, func(w http.ResponseWriter, req *http.Request) { + switch { + case req.Method == http.MethodDelete && strings.Contains(req.URL.Path, "/volumes/cloneVol"): + reclaimedClone = true + w.WriteHeader(http.StatusNoContent) + case req.Method == http.MethodDelete && strings.Contains(req.URL.Path, "/snapshots/"): + t.Errorf("teardown deleted a snapshot it only resolved: %s", req.URL.Path) + w.WriteHeader(http.StatusInternalServerError) + default: + t.Errorf("unexpected request %s %s", req.Method, req.URL.Path) + w.WriteHeader(http.StatusInternalServerError) + } + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("teardown reconcile: %v", err) + } + if !reclaimedClone { + t.Errorf("the clone was not reclaimed") + } + err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &simplyblockv1alpha2.TestFailover{}) + if err == nil || !apierrors.IsNotFound(err) { + t.Errorf("drill still present after teardown (finalizer not cleared): %v", err) + } + if err := cl.Get(ctx, client.ObjectKey{Namespace: tf.Spec.BubbleCluster, Name: mw.Name}, &workv1.ManifestWork{}); !apierrors.IsNotFound(err) { + t.Errorf("ManifestWork still present after teardown: %v", err) + } +} + +// TestFailoverTeardownTolersatesAlreadyGoneResources covers idempotency: a 404 +// from every reclaim is treated as success, so a re-run after a partial teardown +// still completes. +func TestFailoverTeardownToleratesAlreadyGone(t *testing.T) { + tf := deletingReadyDrill() + r, cl := newTestFailoverReconciler(t, tf) // no ManifestWork seeded + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusNotFound) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("teardown reconcile: %v", err) + } + err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &simplyblockv1alpha2.TestFailover{}) + if err == nil || !apierrors.IsNotFound(err) { + t.Errorf("drill still present after teardown of already-gone resources: %v", err) + } +} + +// TestFailoverTeardownHoldsWhenReclaimFails covers that a reclaim that cannot be +// confirmed holds the object with its finalizer rather than orphaning storage. +func TestFailoverTeardownHoldsWhenReclaimFails(t *testing.T) { + tf := deletingReadyDrill() + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + srv := newAPIServer(t, func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusInternalServerError) + }) + t.Setenv("SIMPLYBLOCK_WEBAPI_BASE_URL", srv.URL) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err == nil { + t.Fatalf("expected an error to hold teardown when a reclaim fails") + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatalf("drill was removed despite a failed reclaim: %v", err) + } + if !controllerutil.ContainsFinalizer(&got, finalizerTestFailover) { + t.Errorf("finalizer was cleared despite a failed reclaim") + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseTearingDown { + t.Errorf("phase = %q, want TearingDown while holding", got.Status.Phase) + } +} + +// TestFailoverPlacingFailsWhenSourceChanged covers the non-disruptiveness guard: +// if the source's projected volume handle differs at Ready, the drill fails +// rather than reporting a passing, non-disruptive test. +func TestFailoverPlacingFailsWhenSourceChanged(t *testing.T) { + tf := atPlacing() // SourceHandle is clusterA:poolA:lvolX + view := srcPVView(tf, "clusterA:poolA:DIFFERENT") + mw := &workv1.ManifestWork{ObjectMeta: metav1.ObjectMeta{ + Name: testFailoverManifestWorkName(tf), Namespace: tf.Spec.BubbleCluster, + }} + r, cl := newTestFailoverReconciler(t, tf, view, mw) + ctx := context.Background() + + markManifestWorkPVCBound(t, cl, getManifestWork(t, cl, tf.Spec.BubbleCluster, mw.Name)) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("phase = %q, want Failed when the source changed", got.Status.Phase) + } + if got.Status.Report == nil || got.Status.Report.InvariantsHeld { + t.Errorf("invariantsHeld = true, want false when the source changed") + } +} + +// TestFailoverPlacingConfirmsInvariantHeld covers the passing guard: an unchanged +// source projection yields Ready with invariantsHeld true. +func TestFailoverPlacingConfirmsInvariantHeld(t *testing.T) { + tf := atPlacing() + view := srcPVView(tf, "clusterA:poolA:lvolX") // same as SourceHandle + mw := &workv1.ManifestWork{ObjectMeta: metav1.ObjectMeta{ + Name: testFailoverManifestWorkName(tf), Namespace: tf.Spec.BubbleCluster, + }} + r, cl := newTestFailoverReconciler(t, tf, view, mw) + ctx := context.Background() + + markManifestWorkPVCBound(t, cl, getManifestWork(t, cl, tf.Spec.BubbleCluster, mw.Name)) + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseReady { + t.Errorf("phase = %q, want Ready", got.Status.Phase) + } + if got.Status.Report == nil || !got.Status.Report.InvariantsHeld { + t.Errorf("invariantsHeld = false, want true for an unchanged source") + } +} + +// TestFailoverRefusesSecondDrillForSameSource covers the concurrency guard: a +// second drill against the same source and bubble as an active one is refused. +func TestFailoverRefusesSecondDrillForSameSource(t *testing.T) { + existing := atResolvingSource() // active (Provisioning) on the sample source/bubble + existing.Name = "drill-existing" + existing.UID = "uid-existing" + + second := sampleTestFailover() // same source/bubble, not yet started + second.Name = "drill-second" + second.UID = "uid-second" + second.Finalizers = []string{finalizerTestFailover} + + r, cl := newTestFailoverReconciler(t, existing, second) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(second)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(second).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("phase = %q, want Failed for a conflicting second drill", got.Status.Phase) + } +} + +// TestFailoverStepDeadlineFailsTheDrill covers that a step that blew its deadline +// fails the drill rather than holding forever. +func TestFailoverStepDeadlineFailsTheDrill(t *testing.T) { + tf := atResolvingSource() + past := metav1.NewTime(time.Now().Add(-time.Hour)) + tf.Status.Step = statemachine.KubeSnapshot{ + State: string(simplyblockv1alpha2.TestFailoverStepResolvingSource), + Deadline: &past, + } + r, cl := newTestFailoverReconciler(t, tf) + ctx := context.Background() + + if _, err := r.Reconcile(ctx, testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + var got simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(tf).NamespacedName, &got); err != nil { + t.Fatal(err) + } + if got.Status.Phase != simplyblockv1alpha2.TestFailoverPhaseFailed { + t.Errorf("phase = %q, want Failed after the step deadline passed", got.Status.Phase) + } + if !strings.Contains(got.Status.Message, "deadline") { + t.Errorf("message = %q, want it to mention the deadline", got.Status.Message) + } +} + +// Regression: 2026-10-04 bubble PVC Lost. The work agent re-applied the whole PV +// under the default update strategy and wiped spec.claimRef.uid; every object a +// drill places is CreateOnly now, and the namespace is in its own work. +func TestFailoverPlacingPlacesEverythingCreateOnly(t *testing.T) { + tf := atPlacing() + tf.Spec.BubbleNamespace = bubbleNS + r, cl := newTestFailoverReconciler(t, tf) + if _, err := r.Reconcile(context.Background(), testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile: %v", err) + } + mw := getManifestWork(t, cl, tf.Spec.BubbleCluster, testFailoverManifestWorkName(tf)) + seen := map[string]bool{} + for _, c := range mw.Spec.ManifestConfigs { + if c.UpdateStrategy == nil || c.UpdateStrategy.Type != workv1.UpdateStrategyTypeCreateOnly { + t.Errorf("%s/%s: update strategy %+v, want CreateOnly", + c.ResourceIdentifier.Resource, c.ResourceIdentifier.Name, c.UpdateStrategy) + } + seen[c.ResourceIdentifier.Resource] = true + } + if !seen["persistentvolumes"] || !seen["persistentvolumeclaims"] { + t.Errorf("configs cover %v, want both the PV and the PVC", seen) + } + for _, m := range mw.Spec.Workload.Manifests { + var obj metav1.TypeMeta + if err := json.Unmarshal(m.Raw, &obj); err != nil { + t.Fatal(err) + } + if obj.Kind == "Namespace" { + t.Errorf("the drill's work carries the bubble namespace; it belongs to the namespace work") + } + } + nsw := getManifestWork(t, cl, tf.Spec.BubbleCluster, + bubbleNamespaceWorkName(tf.Spec.BubbleCluster, tf.Spec.BubbleNamespace)) + if len(nsw.Spec.Workload.Manifests) != 1 || len(nsw.Spec.ManifestConfigs) != 1 || + nsw.Spec.ManifestConfigs[0].UpdateStrategy.Type != workv1.UpdateStrategyTypeCreateOnly { + t.Errorf("namespace work = %+v, want the one namespace, CreateOnly", nsw.Spec) + } +} + +// Two drills of one test share the bubble namespace: the second finds the +// namespace work and does not fail; one work owns the namespace. +func TestFailoverPlacingSharesOneNamespaceWork(t *testing.T) { + a := atPlacing() + a.Spec.BubbleNamespace = bubbleNS + b := atPlacing() + b.Name, b.UID = "drill-2", "uid-2" + b.Spec.BubbleNamespace = bubbleNS + b.Spec.SourceRef = "other-data" + r, cl := newTestFailoverReconciler(t, a, b) + for _, tf := range []*simplyblockv1alpha2.TestFailover{a, b} { + if _, err := r.Reconcile(context.Background(), testFailoverRequest(tf)); err != nil { + t.Fatalf("reconcile %s: %v", tf.Name, err) + } + } + var works workv1.ManifestWorkList + if err := cl.List(context.Background(), &works, client.InNamespace(a.Spec.BubbleCluster)); err != nil { + t.Fatal(err) + } + owners := 0 + for i := range works.Items { + for _, m := range works.Items[i].Spec.Workload.Manifests { + var obj metav1.TypeMeta + if err := json.Unmarshal(m.Raw, &obj); err != nil { + t.Fatal(err) + } + if obj.Kind == "Namespace" { + owners++ + } + } + } + if owners != 1 { + t.Errorf("%d works carry the bubble namespace, want exactly one", owners) + } +} + +func TestLiveDrillOnBubble(t *testing.T) { + self := sampleTestFailover() + self.UID = "self" + self.Spec.BubbleNamespace = "ns-1" + other := func(uid, cluster, ns string, deleting bool) simplyblockv1alpha2.TestFailover { + o := *sampleTestFailover() + o.UID = types.UID(uid) + o.Spec.BubbleCluster, o.Spec.BubbleNamespace = cluster, ns + if deleting { + now := metav1.Now() + o.DeletionTimestamp = &now + } + return o + } + cases := []struct { + name string + items []simplyblockv1alpha2.TestFailover + want bool + }{ + {"only itself", []simplyblockv1alpha2.TestFailover{*self}, false}, + {"another live drill on the bubble", []simplyblockv1alpha2.TestFailover{*self, + other("o1", self.Spec.BubbleCluster, "ns-1", false)}, true}, + {"the other is being deleted", []simplyblockv1alpha2.TestFailover{*self, + other("o1", self.Spec.BubbleCluster, "ns-1", true)}, false}, + {"another bubble namespace", []simplyblockv1alpha2.TestFailover{*self, + other("o1", self.Spec.BubbleCluster, "ns-2", false)}, false}, + {"another cluster", []simplyblockv1alpha2.TestFailover{*self, + other("o1", "elsewhere", "ns-1", false)}, false}, + } + for _, c := range cases { + if got := liveDrillOnBubble(c.items, self); got != c.want { + t.Errorf("%s: got %v, want %v", c.name, got, c.want) + } + } +} + +// Teardown keeps the namespace work while another live drill places into it, and +// removes it with the last one. +func TestFailoverDeletionRemovesTheNamespaceWorkWithTheLastDrill(t *testing.T) { + mk := func(name, uid string, deleting bool) *simplyblockv1alpha2.TestFailover { + tf := sampleTestFailover() + tf.Name, tf.UID = name, types.UID(uid) + tf.Spec.BubbleNamespace = bubbleNS + tf.Finalizers = []string{finalizerTestFailover} + if deleting { + now := metav1.Now() + tf.DeletionTimestamp = &now + } + return tf + } + first, second := mk("drill-1", "u1", true), mk("drill-2", "u2", false) + nsw, err := bubbleNamespaceWork(first) + if err != nil { + t.Fatal(err) + } + r, cl := newTestFailoverReconciler(t, first, second, nsw) + ctx := context.Background() + key := client.ObjectKey{Namespace: nsw.Namespace, Name: nsw.Name} + + if _, err := r.Reconcile(ctx, testFailoverRequest(first)); err != nil { + t.Fatalf("teardown of drill-1: %v", err) + } + if err := cl.Get(ctx, key, &workv1.ManifestWork{}); err != nil { + t.Fatalf("namespace work removed while drill-2 still places into it: %v", err) + } + + var live simplyblockv1alpha2.TestFailover + if err := cl.Get(ctx, testFailoverRequest(second).NamespacedName, &live); err != nil { + t.Fatal(err) + } + if err := cl.Delete(ctx, &live); err != nil { + t.Fatal(err) + } + if _, err := r.Reconcile(ctx, testFailoverRequest(second)); err != nil { + t.Fatalf("teardown of drill-2: %v", err) + } + if err := cl.Get(ctx, key, &workv1.ManifestWork{}); !apierrors.IsNotFound(err) { + t.Errorf("namespace work still present after the last drill: %v", err) + } +} + +// bubbleNS is the bubble namespace the binding tests place their clones in. +const bubbleNS = "app-drtest-1" diff --git a/operator/internal/controller/testfailover_group_mapping.go b/operator/internal/controller/testfailover_group_mapping.go new file mode 100644 index 000000000..723f1d986 --- /dev/null +++ b/operator/internal/controller/testfailover_group_mapping.go @@ -0,0 +1,99 @@ +package controller + +import ( + "fmt" + "strings" + + "github.com/simplyblock/simplyblock-operator/internal/webapi" +) + +// groupMemberSource is one clone slot of a group drill as the mapping sees it: +// the PVC it recovers and the source lvol it was resolved from. +type groupMemberSource struct { + PVC string + LvolID string +} + +// assignGenerationMembers pairs every clone slot with the generation member that +// holds THAT slot's data, returning the member index per slot. The pairing is by +// identity, never by position: the control plane lists a generation's members in +// no particular order, so a positional pairing can restore one member's disk onto +// another's PVC (the database disk onto the web volume). +// +// The key is the member's source lvol (source_lvol_id) when every member carries +// it. A control plane that does not report it yet returns only the replica volume +// on the target (lvol_id); then targetOf maps each slot's source lvol to its +// replica volume, read from the per-volume latest-snapshot, and the pairing is on +// the replica volume. Anything short of a one-to-one pairing is an error: a slot +// without a member, a member no slot claims, an empty or a duplicate key. +func assignGenerationMembers( + slots []groupMemberSource, + members []webapi.ReplicatedGroupSnapshot, + targetOf map[string]string, +) ([]int, error) { + bySource := true + for i := range members { + if members[i].SourceLvolID == "" { + bySource = false + break + } + } + + memberKey := func(m webapi.ReplicatedGroupSnapshot) string { + if bySource { + return m.SourceLvolID + } + return m.LvolID + } + index := make(map[string]int, len(members)) + for i := range members { + k := memberKey(members[i]) + if k == "" { + return nil, fmt.Errorf("generation member snapshot %s names no volume; cannot tell whose data it holds", + members[i].SnapshotID) + } + if j, dup := index[k]; dup { + return nil, fmt.Errorf("generation members %s and %s both belong to volume %s", + members[j].SnapshotID, members[i].SnapshotID, k) + } + index[k] = i + } + + assigned := make([]int, len(slots)) + used := make(map[int]bool, len(members)) + var missing []string + for s, slot := range slots { + k := slot.LvolID + if !bySource { + k = targetOf[slot.LvolID] + if k == "" { + missing = append(missing, fmt.Sprintf("%s (source volume %s has no replica volume on the target)", + slot.PVC, slot.LvolID)) + continue + } + } + i, ok := index[k] + if !ok { + missing = append(missing, fmt.Sprintf("%s (volume %s)", slot.PVC, k)) + continue + } + if used[i] { + return nil, fmt.Errorf("generation member snapshot %s matches more than one PVC", members[i].SnapshotID) + } + used[i] = true + assigned[s] = i + } + if len(missing) > 0 { + return nil, fmt.Errorf("the generation holds no snapshot for %s", strings.Join(missing, ", ")) + } + var stray []string + for i := range members { + if !used[i] { + stray = append(stray, fmt.Sprintf("%s (volume %s)", members[i].SnapshotID, memberKey(members[i]))) + } + } + if len(stray) > 0 { + return nil, fmt.Errorf("generation snapshots %s match no PVC of the drill", strings.Join(stray, ", ")) + } + return assigned, nil +} diff --git a/operator/internal/controller/testfailover_group_mapping_test.go b/operator/internal/controller/testfailover_group_mapping_test.go new file mode 100644 index 000000000..35887c129 --- /dev/null +++ b/operator/internal/controller/testfailover_group_mapping_test.go @@ -0,0 +1,123 @@ +package controller + +import ( + "strings" + "testing" + + "github.com/simplyblock/simplyblock-operator/internal/webapi" +) + +const ( + mapSnapWeb = "snap-web" + mapSnapDB = "snap-db" + mapTgtWeb = "tgt-web" + mapTgtDB = "tgt-db" + mapSrcWeb = "src-web" + mapSrcDB = "src-db" +) + +// The two members of a web+db group, as the drill resolved them. +var mappingSlots = []groupMemberSource{ + {PVC: "web-disk", LvolID: mapSrcWeb}, + {PVC: "db-disk", LvolID: mapSrcDB}, +} + +// TestAssignGenerationMembersBySourceShuffled covers the regression: the control +// plane lists the generation's members in its own order (db first here), and each +// PVC must still get its own member's snapshot, not the one at its position. +func TestAssignGenerationMembersBySourceShuffled(t *testing.T) { + members := []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapDB, LvolID: mapTgtDB, SourceLvolID: mapSrcDB}, + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb, SourceLvolID: mapSrcWeb}, + } + got, err := assignGenerationMembers(mappingSlots, members, nil) + if err != nil { + t.Fatal(err) + } + if members[got[0]].SnapshotID != mapSnapWeb || members[got[1]].SnapshotID != mapSnapDB { + t.Errorf("web-disk got %s, db-disk got %s; want snap-web, snap-db", + members[got[0]].SnapshotID, members[got[1]].SnapshotID) + } +} + +// TestAssignGenerationMembersByReplicaVolume covers a control plane that names +// only the replica volume on the target: the pairing goes through the source to +// replica map and is still order-independent. +func TestAssignGenerationMembersByReplicaVolume(t *testing.T) { + members := []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapDB, LvolID: mapTgtDB}, + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb}, + } + targetOf := map[string]string{mapSrcWeb: mapTgtWeb, mapSrcDB: mapTgtDB} + got, err := assignGenerationMembers(mappingSlots, members, targetOf) + if err != nil { + t.Fatal(err) + } + if members[got[0]].SnapshotID != mapSnapWeb || members[got[1]].SnapshotID != mapSnapDB { + t.Errorf("web-disk got %s, db-disk got %s; want snap-web, snap-db", + members[got[0]].SnapshotID, members[got[1]].SnapshotID) + } +} + +// TestAssignGenerationMembersRefusesMismatch covers every way the pairing can be +// incomplete: none of them falls back to position. +func TestAssignGenerationMembersRefusesMismatch(t *testing.T) { + cases := []struct { + name string + members []webapi.ReplicatedGroupSnapshot + targetOf map[string]string + want string + }{ + { + name: "missing member", + members: []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb, SourceLvolID: mapSrcWeb}, + {SnapshotID: "snap-other", LvolID: "tgt-other", SourceLvolID: "src-other"}, + }, + want: "no snapshot for db-disk", + }, + { + name: "stray member", + members: []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb, SourceLvolID: mapSrcWeb}, + {SnapshotID: mapSnapDB, LvolID: mapTgtDB, SourceLvolID: mapSrcDB}, + {SnapshotID: "snap-x", LvolID: "tgt-x", SourceLvolID: "src-x"}, + }, + want: "snap-x (volume src-x) match no PVC", + }, + { + name: "duplicate member", + members: []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: "snap-a", LvolID: mapTgtWeb, SourceLvolID: mapSrcWeb}, + {SnapshotID: "snap-b", LvolID: mapTgtWeb, SourceLvolID: mapSrcWeb}, + }, + want: "both belong to volume src-web", + }, + { + name: "no replica volume known", + members: []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb}, + {SnapshotID: mapSnapDB, LvolID: mapTgtDB}, + }, + targetOf: map[string]string{mapSrcWeb: mapTgtWeb}, + want: "db-disk (source volume src-db has no replica volume", + }, + { + name: "member without a volume", + members: []webapi.ReplicatedGroupSnapshot{ + {SnapshotID: mapSnapWeb, LvolID: mapTgtWeb}, + {SnapshotID: mapSnapDB}, + }, + targetOf: map[string]string{mapSrcWeb: mapTgtWeb, mapSrcDB: mapTgtDB}, + want: "snap-db names no volume", + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + _, err := assignGenerationMembers(mappingSlots, tc.members, tc.targetOf) + if err == nil || !strings.Contains(err.Error(), tc.want) { + t.Errorf("err = %v, want it to contain %q", err, tc.want) + } + }) + } +} diff --git a/operator/internal/controllers/cluster/controlplane.go b/operator/internal/controllers/cluster/controlplane.go index 0c0b79179..67470e1f7 100644 --- a/operator/internal/controllers/cluster/controlplane.go +++ b/operator/internal/controllers/cluster/controlplane.go @@ -53,6 +53,9 @@ type ControlPlane interface { // reports false rather than an error when there is none, because "no such // cluster" is the ordinary answer on the creation path. ClusterByName(ctx context.Context, name string) (utils.ClusterListEntry, bool, error) + // Clusters lists every cluster of the control plane, as ClusterByName + // reads them. + Clusters(ctx context.Context) ([]utils.ClusterListEntry, error) // DeleteCluster is retried until it succeeds, and the finalizer is not // removed before it does. @@ -81,6 +84,14 @@ type ControlPlane interface { // outcome being asked for reached by another route, so the implementation // reads a 404 as success. CancelTask(ctx context.Context, clusterID, taskID string) error + + // Endpoint is the base URL this call would currently reach the control + // plane at, resolved the same way every other call resolves it (§3.3). + // upsertCSICredentials carries it into the CSI driver's aggregate Secret, + // since the CSI driver dials it directly rather than through this + // reconciler, and a hardcoded in-cluster address is wrong the moment the + // control plane is a ControlPlane.spec.source.managed one. + Endpoint(ctx context.Context) string } // httpControlPlane is the ControlPlane the operator runs with: one method per @@ -94,6 +105,12 @@ type httpControlPlane struct { // startup client is the only one. resolve controlplane.EndpointResolver + // credential answers what a managed control plane's CreateCluster and + // ClusterByName calls authenticate with (see adminContext). Nil, or a + // resolver that answers false, means neither call adds a bearer token + // beyond whatever webapi.Client already carries. + credential controlplane.CredentialResolver + // mu guards resolved, which is rebuilt when the published endpoint changes. mu sync.Mutex resolved *webapi.Client @@ -101,12 +118,17 @@ type httpControlPlane struct { // NewControlPlane returns the HTTP-backed control-plane surface. // -// The resolver may be nil, which is what a test passes: calls then go to the -// startup client and nothing reads a ControlPlane object. -func NewControlPlane(resolve controlplane.EndpointResolver) ControlPlane { - return &httpControlPlane{client: webapi.NewClient(), resolve: resolve} +// Either resolver may be nil, which is what a test passes: calls then go to +// the startup client, unauthenticated beyond its own saToken, and nothing +// reads a ControlPlane object. +func NewControlPlane( + resolve controlplane.EndpointResolver, credential controlplane.CredentialResolver, +) ControlPlane { + return &httpControlPlane{client: webapi.NewClient(), resolve: resolve, credential: credential} } +func (c *httpControlPlane) Endpoint(ctx context.Context) string { return c.clientFor(ctx).BaseURL } + func (c *httpControlPlane) Ready(ctx context.Context) error { _, err := c.call(ctx, http.MethodGet, "/api/v2/_meta/ready", nil) return err @@ -115,7 +137,7 @@ func (c *httpControlPlane) Ready(ctx context.Context) error { func (c *httpControlPlane) CreateCluster( ctx context.Context, params utils.ClusterAddParams, ) (webapi.ClusterResponse, error) { - body, err := c.call(ctx, http.MethodPost, "/api/v2/clusters/", params) + body, err := c.call(c.adminContext(ctx), http.MethodPost, "/api/v2/clusters/", params) if err != nil { return webapi.ClusterResponse{}, err } @@ -132,17 +154,25 @@ func (c *httpControlPlane) Cluster( return webapi.ParseClusterResponse(body) } +func (c *httpControlPlane) Clusters(ctx context.Context) ([]utils.ClusterListEntry, error) { + body, err := c.call(c.adminContext(ctx), http.MethodGet, "/api/v2/clusters/", nil) + if err != nil { + return nil, err + } + var entries []utils.ClusterListEntry + if err := json.Unmarshal(body, &entries); err != nil { + return nil, fmt.Errorf("read the cluster list: %w", err) + } + return entries, nil +} + func (c *httpControlPlane) ClusterByName( ctx context.Context, name string, ) (utils.ClusterListEntry, bool, error) { - body, err := c.call(ctx, http.MethodGet, "/api/v2/clusters/", nil) + entries, err := c.Clusters(ctx) if err != nil { return utils.ClusterListEntry{}, false, err } - var entries []utils.ClusterListEntry - if err := json.Unmarshal(body, &entries); err != nil { - return utils.ClusterListEntry{}, false, fmt.Errorf("read the cluster list: %w", err) - } for _, entry := range entries { if entry.Name == name { return entry, true, nil @@ -296,3 +326,27 @@ func (c *httpControlPlane) clientFor(ctx context.Context) *webapi.Client { } return c.resolved } + +// adminContext is what CreateCluster and ClusterByName call with, instead of +// ctx directly. +// +// Both run before any cluster secret exists -- there is no cluster yet to +// have one, and no adoption has happened either -- so a managed control +// plane's admin credential is the only thing that can authenticate them, and +// nothing else in this package has a stronger claim to the context's bearer +// token slot. A ctx that already carries one (a caller more specific than +// this method wins, though none exists on this interface today) is left +// alone. +func (c *httpControlPlane) adminContext(ctx context.Context) context.Context { + if c.credential == nil { + return ctx + } + if _, ok := webapi.BearerTokenFromContext(ctx); ok { + return ctx + } + token, ok := c.credential(ctx) + if !ok { + return ctx + } + return webapi.WithBearerToken(ctx, token) +} diff --git a/operator/internal/controllers/cluster/controlplane_admin_test.go b/operator/internal/controllers/cluster/controlplane_admin_test.go new file mode 100644 index 000000000..a4f19e35e --- /dev/null +++ b/operator/internal/controllers/cluster/controlplane_admin_test.go @@ -0,0 +1,108 @@ +// Whether CreateCluster and ClusterByName carry a managed control plane's +// admin credential. +// +// These two are the only calls made before any cluster secret exists to +// authenticate with -- there is no cluster yet to have one, and no adoption +// has happened yet either. Without the credential reaching them, a +// StorageCluster CR applied against a managed control plane could never +// create its backend identity there at all, which is the gap this closes. +// Every other call on this interface keeps authenticating however it already +// did: as the cluster it was adopted or created as, via the context wrap the +// reconciler applies at the call site (storagecluster_controller.go). + +package cluster + +import ( + "context" + "net/http" + "testing" + + "github.com/simplyblock/simplyblock-operator/internal/utils" + webapimock "github.com/simplyblock/simplyblock-operator/internal/webapi/mock" +) + +const ( + clusterAdminSpecPath = "../../../../shared/openapi.json" + testAdminToken = "admin-token" + testAdminAuthHeader = "Bearer " + testAdminToken +) + +func TestCreateClusterAuthenticatesWithTheManagedAdminCredentialWhenOneResolves(t *testing.T) { + mock := webapimock.NewSpecServerFromFile(t, clusterAdminSpecPath, false) + defer mock.Close() + + mock.Register(http.MethodPost, "/api/v2/clusters/", webapimock.RouteResponse{ + Status: http.StatusCreated, + Body: `{"id":"cluster-uuid","secret":"cluster-secret"}`, + }) + + resolveEndpoint := func(context.Context) string { return mock.URL() } + resolveCredential := func(context.Context) (string, bool) { return testAdminToken, true } + + api := NewControlPlane(resolveEndpoint, resolveCredential) + if _, err := api.CreateCluster(context.Background(), utils.ClusterAddParams{Name: "b"}); err != nil { + t.Fatalf("CreateCluster: %v", err) + } + + reqs := mock.Requests() + if len(reqs) != 1 { + t.Fatalf("expected one request, got %d", len(reqs)) + } + if got := reqs[0].Headers["Authorization"]; got != testAdminAuthHeader { + t.Errorf("authorization header = %q, want the managed admin credential", got) + } +} + +func TestClusterByNameAuthenticatesWithTheManagedAdminCredentialWhenOneResolves(t *testing.T) { + mock := webapimock.NewSpecServerFromFile(t, clusterAdminSpecPath, false) + defer mock.Close() + + mock.Register(http.MethodGet, "/api/v2/clusters/", webapimock.RouteResponse{ + Status: http.StatusOK, Body: `[]`, + }) + + resolveEndpoint := func(context.Context) string { return mock.URL() } + resolveCredential := func(context.Context) (string, bool) { return testAdminToken, true } + + api := NewControlPlane(resolveEndpoint, resolveCredential) + if _, _, err := api.ClusterByName(context.Background(), "b"); err != nil { + t.Fatalf("ClusterByName: %v", err) + } + + reqs := mock.Requests() + if len(reqs) != 1 { + t.Fatalf("expected one request, got %d", len(reqs)) + } + if got := reqs[0].Headers["Authorization"]; got != testAdminAuthHeader { + t.Errorf("authorization header = %q, want the managed admin credential", got) + } +} + +// A local control plane's resolver names no credential, so CreateCluster +// authenticates exactly as it always has: the client's own (empty) saToken, +// not the admin credential. +func TestCreateClusterCarriesNoAdminCredentialWhenNoneResolves(t *testing.T) { + mock := webapimock.NewSpecServerFromFile(t, clusterAdminSpecPath, false) + defer mock.Close() + + mock.Register(http.MethodPost, "/api/v2/clusters/", webapimock.RouteResponse{ + Status: http.StatusCreated, + Body: `{"id":"cluster-uuid","secret":"cluster-secret"}`, + }) + + resolveEndpoint := func(context.Context) string { return mock.URL() } + resolveCredential := func(context.Context) (string, bool) { return "", false } + + api := NewControlPlane(resolveEndpoint, resolveCredential) + if _, err := api.CreateCluster(context.Background(), utils.ClusterAddParams{Name: "b"}); err != nil { + t.Fatalf("CreateCluster: %v", err) + } + + reqs := mock.Requests() + if len(reqs) != 1 { + t.Fatalf("expected one request, got %d", len(reqs)) + } + if got := reqs[0].Headers["Authorization"]; got == testAdminAuthHeader { + t.Errorf("authorization header = %q, want the client's own (empty) token, not the admin credential", got) + } +} diff --git a/operator/internal/controllers/cluster/csicredentials_test.go b/operator/internal/controllers/cluster/csicredentials_test.go new file mode 100644 index 000000000..8e30bd80b --- /dev/null +++ b/operator/internal/controllers/cluster/csicredentials_test.go @@ -0,0 +1,64 @@ +package cluster + +import ( + "testing" + + "github.com/simplyblock/simplyblock-operator/internal/utils" +) + +func entryIDs(c CSICredentials) map[string]CSIClusterEntry { + out := map[string]CSIClusterEntry{} + for _, e := range c.Clusters { + out[e.ClusterID] = e + } + return out +} + +func TestMergeCSICredentialsRegistersEveryClusterOfTheControlPlane(t *testing.T) { + // Site A's operator manages cluster A; the control plane also runs + // cluster B (site B). A volume failed over from B to A arrives under a + // PV whose handle names B: the driver on A must reach B's cluster too. + creds := CSICredentials{} + own := CSIClusterEntry{ClusterID: "A", ClusterEndpoint: "http://cp:5000", ClusterSecret: "sa", Local: true} + peers := []utils.ClusterListEntry{{UUID: "A", Secret: "sa"}, {UUID: "B", Secret: "sb"}} + mergeCSICredentials(&creds, own, peers, true) + got := entryIDs(creds) + if len(got) != 2 || !got["A"].Local || got["B"].Local || got["B"].ClusterSecret != "sb" || got["B"].ClusterEndpoint != "http://cp:5000" { + t.Fatalf("entries %+v", creds.Clusters) + } +} + +func TestMergeCSICredentialsKeepsAnotherLocalEntryAndPrunesStaleForeignOnes(t *testing.T) { + creds := CSICredentials{Clusters: []CSIClusterEntry{ + {ClusterID: "A2", ClusterEndpoint: "http://cp:5000", ClusterSecret: "sa2", Local: true}, // another operator here + {ClusterID: "OLD", ClusterEndpoint: "http://cp:5000", ClusterSecret: "x"}, // a cluster since removed + {ClusterID: "B", ClusterEndpoint: "http://cp:5000", ClusterSecret: "kept"}, + }} + own := CSIClusterEntry{ClusterID: "A", ClusterEndpoint: "http://cp:5000", ClusterSecret: "sa", Local: true} + // The list withholds B's secret: the recorded one stays. + peers := []utils.ClusterListEntry{{UUID: "A"}, {UUID: "A2"}, {UUID: "B"}} + mergeCSICredentials(&creds, own, peers, true) + got := entryIDs(creds) + if _, stale := got["OLD"]; stale { + t.Fatal("a cluster the control plane no longer lists stays registered") + } + if !got["A2"].Local || got["A2"].ClusterSecret != "sa2" { + t.Fatalf("the other operator's local entry changed: %+v", got["A2"]) + } + if got["B"].Local || got["B"].ClusterSecret != "kept" { + t.Fatalf("B: %+v", got["B"]) + } + if !got["A"].Local { + t.Fatalf("own: %+v", got["A"]) + } +} + +func TestMergeCSICredentialsWithoutTheListOnlyWritesTheOwnEntry(t *testing.T) { + creds := CSICredentials{Clusters: []CSIClusterEntry{{ClusterID: "B", ClusterSecret: "sb"}}} + own := CSIClusterEntry{ClusterID: "A", ClusterSecret: "sa", Local: true} + mergeCSICredentials(&creds, own, nil, false) + got := entryIDs(creds) + if len(got) != 2 || got["B"].ClusterSecret != "sb" || !got["A"].Local { + t.Fatalf("entries %+v", creds.Clusters) + } +} diff --git a/operator/internal/controllers/cluster/helpers_test.go b/operator/internal/controllers/cluster/helpers_test.go index 4c7ba4d5e..6a2e7bc68 100644 --- a/operator/internal/controllers/cluster/helpers_test.go +++ b/operator/internal/controllers/cluster/helpers_test.go @@ -198,10 +198,22 @@ var _ events.EventRecorder = (*recorder)(nil) type fakeControlPlane struct { t *testing.T + // endpoint is what Endpoint() returns verbatim -- a pure config read, not + // an action the test scripts, so a zero value is a legitimate "unset" + // rather than an unexpected call. + endpoint string + + // clusterCtx, when set, is handed the context Cluster() was called with -- + // how a test observes a bearer-token override a caller attached + // (webapi.BearerTokenFromContext), since the closures below only see the + // clusterID. + clusterCtx func(context.Context) + ready func() error create func(utils.ClusterAddParams) (webapi.ClusterResponse, error) cluster func(string) (webapi.ClusterResponse, error) byName func(string) (utils.ClusterListEntry, bool, error) + clusters func() ([]utils.ClusterListEntry, error) deleteCall func(string) error activate func(string) error expand func(string) error @@ -227,6 +239,8 @@ type fakeControlPlane struct { cancelTaskCalls int } +func (f *fakeControlPlane) Endpoint(context.Context) string { return f.endpoint } + func (f *fakeControlPlane) Ready(context.Context) error { if f.ready == nil { return nil @@ -245,8 +259,11 @@ func (f *fakeControlPlane) CreateCluster( } func (f *fakeControlPlane) Cluster( - _ context.Context, clusterID string, + ctx context.Context, clusterID string, ) (webapi.ClusterResponse, error) { + if f.clusterCtx != nil { + f.clusterCtx(ctx) + } if f.cluster == nil { f.t.Fatal("the control plane was asked for a cluster and the test did not script it") } @@ -262,6 +279,13 @@ func (f *fakeControlPlane) ClusterByName( return f.byName(name) } +func (f *fakeControlPlane) Clusters(context.Context) ([]utils.ClusterListEntry, error) { + if f.clusters == nil { + return nil, nil + } + return f.clusters() +} + func (f *fakeControlPlane) DeleteCluster(_ context.Context, clusterID string) error { f.deleteCalls++ if f.deleteCall == nil { diff --git a/operator/internal/controllers/cluster/storagecluster_controller.go b/operator/internal/controllers/cluster/storagecluster_controller.go index 1037e4391..81e297b95 100644 --- a/operator/internal/controllers/cluster/storagecluster_controller.go +++ b/operator/internal/controllers/cluster/storagecluster_controller.go @@ -46,6 +46,7 @@ import ( "github.com/simplyblock/simplyblock-operator/internal/cpinformer" "github.com/simplyblock/simplyblock-operator/internal/cpinformer/subscriptions" "github.com/simplyblock/simplyblock-operator/internal/utils" + "github.com/simplyblock/simplyblock-operator/internal/webapi" ) const ( @@ -200,6 +201,14 @@ type CSIClusterEntry struct { ClusterID string `json:"cluster_id"` ClusterEndpoint string `json:"cluster_endpoint"` ClusterSecret string `json:"cluster_secret"` + // Local marks a cluster this operator manages, i.e. the storage of the + // Kubernetes cluster the driver runs on, as opposed to an entry another + // site registered so that a failed-over volume's handle still resolves. + // The driver's csi-addons Replication RPCs act on the LOCAL member of a + // replication chain: a volume's PV keeps the handle it was created with + // across fail-overs, and the chain of relationships behind it alternates + // between the sites. + Local bool `json:"local,omitempty"` } // +kubebuilder:rbac:groups=storage.simplyblock.io,resources=storageclusters,verbs=get;list;watch;create;update;patch;delete @@ -530,7 +539,13 @@ func (r *StorageClusterReconciler) upgradeClaim( return adoption{}, false, nil } - found, err := r.API.Cluster(ctx, uuid) + // This read authenticates as the cluster itself, using the secret the + // upgrade Secret names, rather than as this operator's own Kubernetes + // identity: the control plane this cluster belongs to may be a + // ControlPlane.spec.source.managed one, on a different Kubernetes + // cluster, where a Kubernetes TokenReview of this operator's own + // service-account token can never succeed. + found, err := r.API.Cluster(webapi.WithBearerToken(ctx, clusterSecret), uuid) if err != nil { // The Secret names a cluster the control plane does not have. That is // worth retrying rather than failing: the control plane may be @@ -666,7 +681,8 @@ func (r *StorageClusterReconciler) sync( ) (ctrl.Result, error) { log := logf.FromContext(ctx) - if secret, err := r.clusterSecret(ctx, cluster); err == nil && secret != "" { + secret, err := r.clusterSecret(ctx, cluster) + if err == nil && secret != "" { if err := r.upsertCSICredentials(ctx, cluster.Status.UUID, secret); err != nil { log.Error(err, "the CSI credentials entry could not be restored", "cluster", cluster.Name) @@ -674,13 +690,24 @@ func (r *StorageClusterReconciler) sync( } } - reading, err := r.reading(ctx, cluster.Status.UUID) + // Every read below authenticates as this cluster, using its own recorded + // secret, rather than as this operator's own Kubernetes identity: the + // control plane this cluster belongs to may be a + // ControlPlane.spec.source.managed one, on a different Kubernetes + // cluster, where a Kubernetes TokenReview of this operator's own + // service-account token can never succeed. + readCtx := ctx + if secret != "" { + readCtx = webapi.WithBearerToken(ctx, secret) + } + + reading, err := r.reading(readCtx, cluster.Status.UUID) if err != nil { log.Error(err, "the cluster could not be read", "cluster", cluster.Name) return ctrl.Result{RequeueAfter: clusterResync}, nil } - tasks := r.readTasks(ctx, cluster) + tasks := r.readTasks(readCtx, cluster) ftt := int32(reading.MaxFaultTolerance) //nolint:gosec // a fault tolerance is a small count err = r.writeStatus(ctx, cluster, func(status *simplyblockv1alpha2.StorageClusterStatus) { @@ -875,7 +902,15 @@ func (r *StorageClusterReconciler) teardown( if cluster.Status.UUID != "" { r.closeStreams(cluster.Status.UUID) - if err := r.API.DeleteCluster(ctx, cluster.Status.UUID); err != nil { + // Authenticates as this cluster, using its own recorded secret, for + // the same reason sync() does: the control plane this cluster + // belongs to may be a ControlPlane.spec.source.managed one, on a + // different Kubernetes cluster than this operator. + deleteCtx := ctx + if secret, err := r.clusterSecret(ctx, cluster); err == nil && secret != "" { + deleteCtx = webapi.WithBearerToken(ctx, secret) + } + if err := r.API.DeleteCluster(deleteCtx, cluster.Status.UUID); err != nil { log.Error(err, "the cluster could not be deleted; retrying", "cluster", cluster.Name, "uuid", cluster.Status.UUID) return ctrl.Result{RequeueAfter: clusterRetry}, nil @@ -1092,24 +1127,91 @@ func (r *StorageClusterReconciler) clusterSecret( } // upsertCSICredentials adds or replaces this cluster's entry in the aggregate -// Secret the CSI driver reads. +// Secret the CSI driver reads, and registers every other cluster of the +// same control plane beside it. +// +// A volume replicated to another site is promoted there under a PV that +// keeps the handle it was created with -- the handle names the cluster the +// volume came from -- so that site's driver must be able to reach the +// control plane for that cluster too. Each operator therefore writes, next +// to its own cluster (Local), one entry per other cluster the control plane +// lists, with the secret the list carries (empty when the control plane +// withholds it: the driver then authenticates with its API token). Entries +// another operator on this Kubernetes cluster marked Local are left alone; +// a foreign entry whose cluster the control plane no longer lists is +// dropped. Until this, the cross-registration was a manual merge of the +// sites' secrets (realbed deploy.sh csi). func (r *StorageClusterReconciler) upsertCSICredentials( ctx context.Context, clusterID, clusterSecret string, ) error { + endpoint := r.API.Endpoint(ctx) + peers, err := r.API.Clusters(ctx) + if err != nil { + // The own entry must never wait on the peer list: the driver reaches + // this cluster through it. Peers are registered on the next sync. + logf.FromContext(ctx).Info("the control plane's cluster list could not be read; "+ + "other clusters are registered with the CSI driver on the next sync", "error", err.Error()) + peers = nil + } return r.editCSICredentials(ctx, func(creds *CSICredentials) { - entry := CSIClusterEntry{ + mergeCSICredentials(creds, CSIClusterEntry{ ClusterID: clusterID, - ClusterEndpoint: utils.ENDPOINT, + ClusterEndpoint: endpoint, ClusterSecret: clusterSecret, + Local: true, + }, peers, err == nil) + }) +} + +// mergeCSICredentials writes own into creds and, when the control plane's +// cluster list was read, one non-local entry per other cluster of that +// list, pruning non-local entries the list no longer has. +func mergeCSICredentials(creds *CSICredentials, own CSIClusterEntry, peers []utils.ClusterListEntry, listed bool) { + replaced := false + for i := range creds.Clusters { + if creds.Clusters[i].ClusterID == own.ClusterID { + creds.Clusters[i] = own + replaced = true + } + } + if !replaced { + creds.Clusters = append(creds.Clusters, own) + } + if !listed { + return + } + known := map[string]bool{} + for _, p := range peers { + known[p.UUID] = true + if p.UUID == own.ClusterID { + continue } + entry := CSIClusterEntry{ClusterID: p.UUID, ClusterEndpoint: own.ClusterEndpoint, ClusterSecret: p.Secret} + found := false for i := range creds.Clusters { - if creds.Clusters[i].ClusterID == clusterID { - creds.Clusters[i] = entry - return + if creds.Clusters[i].ClusterID != p.UUID { + continue + } + found = true + if creds.Clusters[i].Local { + break // another operator here manages it; its entry stands + } + if entry.ClusterSecret == "" { + entry.ClusterSecret = creds.Clusters[i].ClusterSecret } + creds.Clusters[i] = entry } - creds.Clusters = append(creds.Clusters, entry) - }) + if !found { + creds.Clusters = append(creds.Clusters, entry) + } + } + kept := creds.Clusters[:0] + for _, e := range creds.Clusters { + if e.Local || known[e.ClusterID] { + kept = append(kept, e) + } + } + creds.Clusters = kept } // removeCSICredentials drops this cluster's entry from it. diff --git a/operator/internal/controllers/cluster/storagecluster_controller_test.go b/operator/internal/controllers/cluster/storagecluster_controller_test.go index 96c5f85a4..299a019c7 100644 --- a/operator/internal/controllers/cluster/storagecluster_controller_test.go +++ b/operator/internal/controllers/cluster/storagecluster_controller_test.go @@ -15,6 +15,7 @@ package cluster import ( "context" + "encoding/json" "errors" "fmt" "net/http" @@ -130,6 +131,42 @@ func TestACreatedClusterReachesSteadyState(t *testing.T) { } } +// The CSI driver dials ClusterEndpoint directly rather than through this +// reconciler, so a control plane a ControlPlane.spec.source.managed points at +// elsewhere needs that entry to carry the endpoint this reconciler actually +// resolved, not a hardcoded address that only resolves inside its own +// cluster. +func TestCSICredentialsCarryTheControlPlanesResolvedEndpoint(t *testing.T) { + const hubEndpoint = "http://simplyblock-webappapi.hub.example:31500" + api := &fakeControlPlane{ + endpoint: hubEndpoint, + create: func(utils.ClusterAddParams) (webapi.ClusterResponse, error) { + reading := activeCluster() + reading.Secret = testClusterSecret + return reading, nil + }, + cluster: func(string) (webapi.ClusterResponse, error) { return activeCluster(), nil }, + } + r := newClusterReconciler(t, api, &recorder{}, newUncreatedCluster()) + reconcileCluster(t, r, 6) + + var secret corev1.Secret + key := types.NamespacedName{Namespace: testNamespace, Name: csiCredentialsSecret} + if err := r.Get(context.Background(), key, &secret); err != nil { + t.Fatalf("read the CSI credentials secret: %v", err) + } + var creds CSICredentials + if err := json.Unmarshal(secret.Data["secret.json"], &creds); err != nil { + t.Fatalf("unmarshal secret.json: %v", err) + } + if len(creds.Clusters) != 1 { + t.Fatalf("clusters = %d entries, want 1", len(creds.Clusters)) + } + if got := creds.Clusters[0].ClusterEndpoint; got != hubEndpoint { + t.Errorf("clusterEndpoint = %q, want the resolved endpoint %q", got, hubEndpoint) + } +} + // spec.deviceClass is the CRD's spelling of what sbcli's cluster-create wire // format calls `device_mode`, so the two must map onto each other rather than // the field simply passing through unmapped. @@ -263,6 +300,102 @@ func TestAnUpgradeSecretAdoptsRatherThanCreating(t *testing.T) { } } +// An adopted cluster's control plane may be a ControlPlane.spec.source.managed +// one, on a different Kubernetes cluster, where a Kubernetes TokenReview of +// this operator's own service-account token can never succeed. The read that +// confirms the adoption must authenticate as the cluster itself instead, +// using the secret the upgrade Secret names. +func TestAnUpgradeSecretAuthenticatesTheAdoptionReadAsTheAdoptedCluster(t *testing.T) { + // Every Cluster() call across the run is checked, not just the last: a + // later, correctly-authenticated steady-state read would otherwise + // overwrite a single captured value and hide a regression in the + // adoption read specifically. + var calls []struct { + token string + ok bool + } + api := &fakeControlPlane{ + cluster: func(string) (webapi.ClusterResponse, error) { + return activeCluster(), nil + }, + clusterCtx: func(ctx context.Context) { + token, ok := webapi.BearerTokenFromContext(ctx) + calls = append(calls, struct { + token string + ok bool + }{token, ok}) + }, + } + upgrade := &corev1.Secret{ + ObjectMeta: objectMeta("simplyblock-" + testClusterName + "-upgrade"), + Data: map[string][]byte{ + "uuid": []byte(testClusterUUID), + "secret": []byte(testClusterSecret), + }, + } + r := newClusterReconciler(t, api, &recorder{}, newUncreatedCluster(), upgrade) + reconcileCluster(t, r, 6) + + if len(calls) == 0 { + t.Fatal("the control plane was never asked for the cluster") + } + for i, call := range calls { + if !call.ok { + t.Errorf("call %d: carried no bearer-token override", i) + continue + } + if call.token != testClusterSecret { + t.Errorf("call %d: bearer token = %q, want the cluster's own secret %q", + i, call.token, testClusterSecret) + } + } +} + +// Once a cluster is created and its secret is on record, every later +// steady-state read of it must keep authenticating as that cluster -- the +// same reasoning as the adoption read above, just for the read that runs on +// every reconcile after. +func TestStorageClusterSyncAuthenticatesAsTheClusterOnceItsSecretIsKnown(t *testing.T) { + var calls []struct { + token string + ok bool + } + api := &fakeControlPlane{ + create: func(utils.ClusterAddParams) (webapi.ClusterResponse, error) { + reading := activeCluster() + reading.Secret = testClusterSecret + return reading, nil + }, + cluster: func(string) (webapi.ClusterResponse, error) { return activeCluster(), nil }, + clusterCtx: func(ctx context.Context) { + token, ok := webapi.BearerTokenFromContext(ctx) + calls = append(calls, struct { + token string + ok bool + }{token, ok}) + }, + } + r := newClusterReconciler(t, api, &recorder{}, newUncreatedCluster()) + // The creation machine takes 6 passes to reach steady state + // (TestACreatedClusterReachesSteadyState); one more pass is the first + // steady-state sync, which is the read this test is about. + reconcileCluster(t, r, 7) + + if len(calls) == 0 { + t.Fatal("the control plane was never asked for the cluster") + } + for i, call := range calls { + if !call.ok { + t.Errorf("call %d: carried no bearer-token override", i) + continue + } + if call.token != testClusterSecret { + t.Errorf("call %d: bearer token = %q, want the cluster's own recorded secret %q", + i, call.token, testClusterSecret) + } + } +} + // The second route: a POST that failed against a cluster which already exists. // That covers two reconciles that both passed the claim on different // resourceVersions, and a response lost after the backend committed. diff --git a/operator/internal/controllers/cluster/storageclusterops_controller.go b/operator/internal/controllers/cluster/storageclusterops_controller.go index 04d9a60e5..bb38f118a 100644 --- a/operator/internal/controllers/cluster/storageclusterops_controller.go +++ b/operator/internal/controllers/cluster/storageclusterops_controller.go @@ -249,6 +249,15 @@ func (r *StorageClusterOpsReconciler) Reconcile( func (r *StorageClusterOpsReconciler) advance( ctx context.Context, ops *simplyblockv1alpha2.StorageClusterOps, ) (ctrl.Result, error) { + // Every control-plane call this step and everything downstream of it + // makes (perform, advanceWalk, and everything under them) authenticates + // as this operation's own cluster when its secret is known, rather than + // as this operator's Kubernetes identity -- the only way to reach a + // control plane a different Kubernetes cluster runs (a + // ControlPlane.spec.source.managed one), since a Kubernetes TokenReview + // can never cross a cluster boundary. + ctx = r.authenticatedContext(ctx, ops) + graph := action(ops.Spec.Action) machine, err := graphs().FromSnapshot(ctx, graph, statemachine.FromKube[step](ops.Status.Step)) @@ -947,6 +956,41 @@ func (r *StorageClusterOpsReconciler) clusterReading( }, nil } +// clusterSecret reads the secret StorageClusterReconciler.persist wrote for +// this operation's cluster, keyed by the StorageCluster's Kubernetes name (not +// its backend UUID, which this reconciler is not always given yet at the point +// it needs the credential). It reports the empty string when there is none. +func (r *StorageClusterOpsReconciler) clusterSecret( + ctx context.Context, ops *simplyblockv1alpha2.StorageClusterOps, +) (string, error) { + var secret corev1.Secret + key := types.NamespacedName{ + Name: fmt.Sprintf("simplyblock-cluster-%s", ops.Spec.ClusterRef), + Namespace: ops.Namespace, + } + if err := r.Get(ctx, key, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil +} + +// authenticatedContext attaches this operation's cluster's own credential to +// ctx when one is known, so every control-plane call the operation makes +// authenticates as that cluster instead of as this operator's Kubernetes +// identity -- the only way to reach a control plane a different Kubernetes +// cluster runs (a ControlPlane.spec.source.managed one), since a Kubernetes +// TokenReview can never cross a cluster boundary. See StorageClusterReconciler's +// identically-named method. +func (r *StorageClusterOpsReconciler) authenticatedContext( + ctx context.Context, ops *simplyblockv1alpha2.StorageClusterOps, +) context.Context { + secret, err := r.clusterSecret(ctx, ops) + if err != nil || secret == "" { + return ctx + } + return webapi.WithBearerToken(ctx, secret) +} + // effectiveConcurrentRestarts is min(specVal, FTT), defaulting to 1 when the // spec says nothing. Both inputs may be absent, because the fault tolerance // comes from the control plane. diff --git a/operator/internal/controllers/cluster/storageclusterops_controller_test.go b/operator/internal/controllers/cluster/storageclusterops_controller_test.go index a27846f27..03d8bf5d0 100644 --- a/operator/internal/controllers/cluster/storageclusterops_controller_test.go +++ b/operator/internal/controllers/cluster/storageclusterops_controller_test.go @@ -21,6 +21,7 @@ import ( "testing" "time" + corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/types" ctrl "sigs.k8s.io/controller-runtime" @@ -247,7 +248,7 @@ func TestAShutdownIssuesOneCallAndWaitsForTheCluster(t *testing.T) { cluster: func(string) (webapi.ClusterResponse, error) { reading := activeCluster() if !active { - reading.Status = "suspended" + reading.Status = statusSuspended } return reading, nil }, @@ -265,6 +266,60 @@ func TestAShutdownIssuesOneCallAndWaitsForTheCluster(t *testing.T) { } } +// An operation against an already-adopted cluster may reach a control plane +// on a different Kubernetes cluster (a ControlPlane.spec.source.managed one), +// the same as StorageClusterReconciler: it authenticates as the cluster's own +// recorded secret instead of as this operator's Kubernetes identity, since a +// Kubernetes TokenReview can never cross a cluster boundary. Every Cluster() +// call across the run is checked, not just the last, so a later correctly- +// authenticated read can't hide a regression in an earlier one. +func TestAnOperationAuthenticatesAsItsClusterOnceItsSecretIsKnown(t *testing.T) { + var calls []struct { + token string + ok bool + } + active := true + api := &fakeControlPlane{ + cluster: func(string) (webapi.ClusterResponse, error) { + reading := activeCluster() + if !active { + reading.Status = statusSuspended + } + return reading, nil + }, + clusterCtx: func(ctx context.Context) { + token, ok := webapi.BearerTokenFromContext(ctx) + calls = append(calls, struct { + token string + ok bool + }{token, ok}) + }, + shutdown: func(string) error { active = false; return nil }, + } + secret := &corev1.Secret{ + ObjectMeta: objectMeta("simplyblock-cluster-" + testClusterName), + Data: map[string][]byte{"secret": []byte(testClusterSecret)}, + } + r := newOpsReconciler(t, api, &recorder{}, + newTestCluster(), newTestOps(simplyblockv1alpha2.StorageClusterOpsActionShutdown), secret) + + reconcileOps(t, r, 6) + + if len(calls) == 0 { + t.Fatal("the control plane was never asked for the cluster") + } + for i, call := range calls { + if !call.ok { + t.Errorf("call %d: carried no bearer-token override", i) + continue + } + if call.token != testClusterSecret { + t.Errorf("call %d: bearer token = %q, want the cluster's own recorded secret %q", + i, call.token, testClusterSecret) + } + } +} + // Restart is the one action with two side effects, because the control plane // has no restart endpoint of its own. func TestARestartShutsDownThenStarts(t *testing.T) { @@ -275,7 +330,7 @@ func TestARestartShutsDownThenStarts(t *testing.T) { reading.Status = status return reading, nil }, - shutdown: func(string) error { status = "suspended"; return nil }, + shutdown: func(string) error { status = statusSuspended; return nil }, start: func(string) error { status = utils.ClusterStatusActive; return nil }, } r := newOpsReconciler(t, api, &recorder{}, diff --git a/operator/internal/controllers/controlplane/adminaccounts_test.go b/operator/internal/controllers/controlplane/adminaccounts_test.go deleted file mode 100644 index 129914ed7..000000000 --- a/operator/internal/controllers/controlplane/adminaccounts_test.go +++ /dev/null @@ -1,24 +0,0 @@ -package controlplane - -import "testing" - -// The Control Center reads every storage cluster through the management API -// with its own service account; the chart names it in the operator's -// environment and the management API trusts it next to the operator. -func TestExtraAdminServiceAccountsAreAppendedOnceAndValidated(t *testing.T) { - t.Setenv(extraAdminAccountsEnv, - " system:serviceaccount:simplyblock:console , not-an-account,system:serviceaccount:a:b:c,"+ - "system:serviceaccount:simplyblock:console") - want := "system:serviceaccount:sb:simplyblock-operator,system:serviceaccount:simplyblock:console" - if got := adminServiceAccounts("sb"); got != want { - t.Errorf("adminServiceAccounts = %q, want %q", got, want) - } -} - -// Without the operator's variable the management API trusts the operator alone. -func TestNoExtraAdminServiceAccountsMeansTheOperatorAlone(t *testing.T) { - t.Setenv(extraAdminAccountsEnv, "") - if got, want := adminServiceAccounts("sb"), "system:serviceaccount:sb:simplyblock-operator"; got != want { - t.Errorf("adminServiceAccounts = %q, want %q", got, want) - } -} diff --git a/operator/internal/controllers/controlplane/credential.go b/operator/internal/controllers/controlplane/credential.go new file mode 100644 index 000000000..3afe53bf3 --- /dev/null +++ b/operator/internal/controllers/controlplane/credential.go @@ -0,0 +1,80 @@ +// What the rest of the operator authenticates a call to a managed control +// plane with. +// +// This mirrors NewEndpointResolver (endpoint.go, resolver.go): the same +// singleton, the same cache interval, and the same "unreadable is a transient +// miss, not an error" answer, because a caller with no credential falls back +// to whatever it already carries -- its own cluster-secret authentication, or +// none at all for a local control plane -- and that fallback is exactly what an +// admitting deployment looked like before this existed. + +package controlplane + +import ( + "context" + "strings" + "sync" + "time" + + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +// CredentialResolver answers what a call to the control plane authenticates +// with, per call. The second return is false when the ControlPlane names no +// managed credential -- the caller then keeps whatever it already had. +type CredentialResolver func(ctx context.Context) (token string, ok bool) + +// NewCredentialResolver returns a resolver reading the ControlPlane +// singleton's spec.source.managed.credentialsSecretRef, in the operator's +// namespace. +func NewCredentialResolver(reader client.Reader, namespace string) CredentialResolver { + var ( + mu sync.Mutex + cached string + cachedOK bool + cachedAt time.Time + ) + + return func(ctx context.Context) (string, bool) { + mu.Lock() + defer mu.Unlock() + + if !cachedAt.IsZero() && time.Since(cachedAt) < endpointCacheTTL { + return cached, cachedOK + } + + var cp simplyblockv1alpha2.ControlPlane + key := client.ObjectKey{Namespace: namespace, Name: SingletonName} + if err := reader.Get(ctx, key, &cp); err != nil { + // Not cached, for the same reason NewEndpointResolver does not cache + // this case: an unreadable singleton is transient, and caching the + // negative answer would hold every caller on no credential for the + // rest of the interval. + return "", false + } + + managed := cp.Spec.Source.Managed + if managed == nil || managed.CredentialsSecretRef == nil || managed.CredentialsSecretRef.Name == "" { + cached, cachedOK, cachedAt = "", false, time.Now() + return cached, cachedOK + } + + var secret corev1.Secret + secretKey := client.ObjectKey{Namespace: namespace, Name: managed.CredentialsSecretRef.Name} + if err := reader.Get(ctx, secretKey, &secret); err != nil { + return "", false + } + + for _, k := range credentialKeys { + if value := strings.TrimSpace(string(secret.Data[k])); value != "" { + cached, cachedOK, cachedAt = value, true, time.Now() + return cached, cachedOK + } + } + cached, cachedOK, cachedAt = "", false, time.Now() + return cached, cachedOK + } +} diff --git a/operator/internal/controllers/controlplane/credential_test.go b/operator/internal/controllers/controlplane/credential_test.go new file mode 100644 index 000000000..8aa0f06e9 --- /dev/null +++ b/operator/internal/controllers/controlplane/credential_test.go @@ -0,0 +1,121 @@ +// The credential resolver: what the rest of the operator authenticates a call +// to a managed control plane with. +// +// It follows NewEndpointResolver's shape deliberately: a ControlPlane or +// Secret this reads cannot answer right now is the same, to every caller, as +// one that names no credential, and both mean "fall back to whatever this +// caller already had." Nothing here is required to reach a control plane the +// deployment itself installed, which is what makes it additive. + +package controlplane + +import ( + "context" + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" +) + +func credentialSecret(token string) *corev1.Secret { + return &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: "cp-token", Namespace: testNamespace}, + Data: map[string][]byte{"token": []byte(token)}, + } +} + +// A managed control plane naming a credential resolves to it, which is what +// lets a call carry the token the remote control plane's admin auth expects. +func TestTheCredentialResolverAnswersWithTheManagedToken(t *testing.T) { + cp := managedControlPlane("https://sb-control.example.com:5000") + secret := credentialSecret("a-bearer-token") + + resolve := NewCredentialResolver(newClient(t, cp, secret), testNamespace) + + token, ok := resolve(context.Background()) + if !ok || token != "a-bearer-token" { + t.Errorf("resolved (%q, %v), want (%q, true)", token, ok, "a-bearer-token") + } +} + +// A local control plane names no managed credential at all, so the resolver +// says so rather than inventing one. +func TestALocalControlPlaneResolvesToNoCredential(t *testing.T) { + cp := localControlPlane() + + resolve := NewCredentialResolver(newClient(t, cp), testNamespace) + + if token, ok := resolve(context.Background()); ok { + t.Errorf("resolved (%q, true) from a local control plane, want no credential", token) + } +} + +// A managed control plane that names no credentialsSecretRef is the in-cluster +// case the chart writes when it installs the control plane itself -- no token +// to carry. +func TestAManagedControlPlaneWithNoCredentialsRefResolvesToNoCredential(t *testing.T) { + cp := managedControlPlane("https://sb-control.example.com:5000") + cp.Spec.Source.Managed.CredentialsSecretRef = nil + + resolve := NewCredentialResolver(newClient(t, cp), testNamespace) + + if token, ok := resolve(context.Background()); ok { + t.Errorf("resolved (%q, true) with no ref set, want no credential", token) + } +} + +// A singleton that cannot be read resolves to no credential rather than to an +// error, for the same reason the endpoint resolver does: an unreadable object +// is a transient miss, not a failed call. +func TestAnAbsentControlPlaneResolvesToNoCredential(t *testing.T) { + resolve := NewCredentialResolver(newClient(t), testNamespace) + + if token, ok := resolve(context.Background()); ok { + t.Errorf("resolved (%q, true) against a cluster with no ControlPlane", token) + } +} + +// A Secret the ref names but that does not exist is the same as no credential +// to the caller -- there is nothing to authenticate with either way. +func TestAMissingCredentialsSecretResolvesToNoCredential(t *testing.T) { + cp := managedControlPlane("https://sb-control.example.com:5000") + + resolve := NewCredentialResolver(newClient(t, cp), testNamespace) + + if token, ok := resolve(context.Background()); ok { + t.Errorf("resolved (%q, true) with the named Secret absent", token) + } +} + +// A change to the credential reaches the next caller, which is the property +// that makes this a resolver rather than a value captured once at startup. +func TestAChangedCredentialReachesTheNextCaller(t *testing.T) { + ctx := context.Background() + + cp := managedControlPlane("https://sb-control.example.com:5000") + secret := credentialSecret("first-token") + c := newClient(t, cp, secret) + + resolve := NewCredentialResolver(c, testNamespace) + if token, ok := resolve(ctx); !ok || token != "first-token" { + t.Fatalf("resolved (%q, %v) before the change", token, ok) + } + + var current corev1.Secret + key := client.ObjectKey{Name: "cp-token", Namespace: testNamespace} + if err := c.Get(ctx, key, ¤t); err != nil { + t.Fatalf("read the secret: %v", err) + } + current.Data["token"] = []byte("second-token") + if err := c.Update(ctx, ¤t); err != nil { + t.Fatalf("publish the new token: %v", err) + } + + // The resolver caches for a short interval, so a fresh one stands in for the + // interval passing, exactly as TestAChangedEndpointReachesTheNextCaller does. + resolve = NewCredentialResolver(c, testNamespace) + if token, ok := resolve(ctx); !ok || token != "second-token" { + t.Errorf("resolved (%q, %v), want the token the Secret now carries", token, ok) + } +} diff --git a/operator/internal/controllers/controlplane/helpers_test.go b/operator/internal/controllers/controlplane/helpers_test.go index 08e6f2441..dad4510d8 100644 --- a/operator/internal/controllers/controlplane/helpers_test.go +++ b/operator/internal/controllers/controlplane/helpers_test.go @@ -172,6 +172,20 @@ func findDeployment(t *testing.T, objects []client.Object, name string) *appsv1. return d } +// findEnvVar locates a container env entry by name in a Deployment's first +// container, failing rather than returning a zero value so a missing entry +// reads as the assertion it is instead of a nil-field panic later. +func findEnvVar(t *testing.T, d *appsv1.Deployment, name string) corev1.EnvVar { + t.Helper() + for _, e := range d.Spec.Template.Spec.Containers[0].Env { + if e.Name == name { + return e + } + } + t.Fatalf("no %q env var on %s", name, d.Name) + return corev1.EnvVar{} +} + func findClusterRole(t *testing.T, objects []client.Object, name string) *rbacv1.ClusterRole { t.Helper() obj := findObject(objects, name) diff --git a/operator/internal/controllers/controlplane/managementapi.go b/operator/internal/controllers/controlplane/managementapi.go index 9e07720a7..814d5a115 100644 --- a/operator/internal/controllers/controlplane/managementapi.go +++ b/operator/internal/controllers/controlplane/managementapi.go @@ -279,6 +279,21 @@ func webAPIDeployment(cp *simplyblockv1alpha2.ControlPlane) *appsv1.Deployment { {Name: "SB_K8S_METRICS_SERVICE_ACCOUNTS", Value: "system:serviceaccount:" + cp.Namespace + ":simplyblock-prometheus"}, } + if ref := managed.AdminTokenSecretRef; ref != nil && ref.Name != "" { + // Sourced from the Secret directly rather than read and copied in here, + // so this operator never itself holds the plaintext -- the same reason + // resolveManaged reads ManagedControlPlane.CredentialsSecretRef on the + // other side only to attach it to a request, never to log or store it. + env = append(env, corev1.EnvVar{ + Name: "SB_ADMIN_TOKENS", + ValueFrom: &corev1.EnvVarSource{ + SecretKeyRef: &corev1.SecretKeySelector{ + LocalObjectReference: *ref, + Key: "token", + }, + }, + }) + } env = append(env, prometheusEnv()...) env = append(env, monitoringSecretEnv(managed)...) env = append(env, tlsEnv(managed)...) diff --git a/operator/internal/controllers/controlplane/workloads_test.go b/operator/internal/controllers/controlplane/workloads_test.go index aa16c0aab..aa24ea213 100644 --- a/operator/internal/controllers/controlplane/workloads_test.go +++ b/operator/internal/controllers/controlplane/workloads_test.go @@ -88,6 +88,42 @@ func TestASingleManagementAPIInstanceStaysExpressible(t *testing.T) { } } +// AdminTokenSecretRef reaches the management API as SB_ADMIN_TOKENS, sourced +// via secretKeyRef rather than a literal value, so this operator never itself +// reads the plaintext. It is what lets a cluster this control plane manages +// remotely authenticate a CreateCluster call. +func TestAnAdminTokenSecretRefReachesTheManagementAPIsEnvironment(t *testing.T) { + cp := localControlPlane() + cp.Spec.Source.Local.AdminTokenSecretRef = &corev1.LocalObjectReference{Name: "hub-admin-token"} + + api := findDeployment(t, managementAPIObjects(cp), ComponentWebAPI) + env := findEnvVar(t, api, "SB_ADMIN_TOKENS") + + if env.ValueFrom == nil || env.ValueFrom.SecretKeyRef == nil { + t.Fatalf("SB_ADMIN_TOKENS is not sourced from a Secret: %#v", env) + } + if env.ValueFrom.SecretKeyRef.Name != "hub-admin-token" { + t.Errorf("secretKeyRef.name = %q, want %q", env.ValueFrom.SecretKeyRef.Name, "hub-admin-token") + } + if env.ValueFrom.SecretKeyRef.Key != "token" { + t.Errorf("secretKeyRef.key = %q, want %q", env.ValueFrom.SecretKeyRef.Key, "token") + } +} + +// Absent names no additional credential: the management API runs exactly as +// it always has, authenticating only this operator's own service account. +func TestNoAdminTokenSecretRefMeansNoExtraEnvVar(t *testing.T) { + cp := localControlPlane() + + api := findDeployment(t, managementAPIObjects(cp), ComponentWebAPI) + + for _, e := range api.Spec.Template.Spec.Containers[0].Env { + if e.Name == "SB_ADMIN_TOKENS" { + t.Fatalf("SB_ADMIN_TOKENS set with no adminTokenSecretRef: %#v", e) + } + } +} + // Every workload built from the control plane's own image runs it, so an upgrade // that writes one image onto the entity moves all of them. func TestEveryWorkloadOfTheControlPlaneRunsTheSpecsImage(t *testing.T) { @@ -414,6 +450,32 @@ func TestTheServicePoolsRunWhatTheyDeclare(t *testing.T) { } } +// Regression: the tasks-runner-backup-merge container must name the module the +// control-plane image ships, or it crash-loops ("python3: can't open file") and +// pins the whole tasks pod. sbcli main's task-runner rework (#1226) ships it as +// backup_merge_service.py; the older integrate_csi_addons_p0 image named it +// tasks_runner_backup_merge.py (2026-10-01). The control plane now runs sbcli +// main (integrate_csi_addons_p0 merged into it on 2026-10-05). A .py-suffix +// check does not catch a wrong name, so pin it. +func TestTheBackupMergeRunnerNamesItsRealModule(t *testing.T) { + cp := localControlPlane() + d := findDeployment(t, managementAPIObjects(cp), ComponentTasks) + const want = "simplyblock_core/services/backup_merge_service.py" + found := false + for _, container := range d.Spec.Template.Spec.Containers { + if container.Name != "tasks-runner-backup-merge" { + continue + } + found = true + if len(container.Command) != 2 || container.Command[1] != want { + t.Errorf("tasks-runner-backup-merge runs %v, want python3 %q", container.Command, want) + } + } + if !found { + t.Fatal("no tasks-runner-backup-merge container in the tasks deployment") + } +} + // The control plane's account is granted exec on pods, which is the strongest // thing in its role and the one an audit has to be able to find. Losing it would // stop the control plane driving the storage nodes' processes, which is not a @@ -537,3 +599,34 @@ func podSpecOf(obj client.Object) *corev1.PodSpec { return nil } } + +// The Control Center reads every storage cluster through the management API +// with its own service account; the chart names it in the operator's +// environment and the management API trusts it next to the operator. +func TestExtraAdminServiceAccountsReachTheManagementAPI(t *testing.T) { + t.Setenv(extraAdminAccountsEnv, + " system:serviceaccount:simplyblock:console , not-an-account,system:serviceaccount:a:b:c,"+ + "system:serviceaccount:simplyblock:console") + cp := localControlPlane() + + api := findDeployment(t, managementAPIObjects(cp), ComponentWebAPI) + env := findEnvVar(t, api, "SB_K8S_ADMIN_SERVICE_ACCOUNTS") + + want := "system:serviceaccount:" + cp.Namespace + ":simplyblock-operator,system:serviceaccount:simplyblock:console" + if env.Value != want { + t.Errorf("SB_K8S_ADMIN_SERVICE_ACCOUNTS = %q, want %q", env.Value, want) + } +} + +// Without the operator's variable the management API trusts the operator alone. +func TestNoExtraAdminServiceAccountsMeansTheOperatorAlone(t *testing.T) { + t.Setenv(extraAdminAccountsEnv, "") + cp := localControlPlane() + + api := findDeployment(t, managementAPIObjects(cp), ComponentWebAPI) + env := findEnvVar(t, api, "SB_K8S_ADMIN_SERVICE_ACCOUNTS") + + if want := "system:serviceaccount:" + cp.Namespace + ":simplyblock-operator"; env.Value != want { + t.Errorf("SB_K8S_ADMIN_SERVICE_ACCOUNTS = %q, want %q", env.Value, want) + } +} diff --git a/operator/internal/controllers/deployment/cluster_disambiguation_test.go b/operator/internal/controllers/deployment/cluster_disambiguation_test.go new file mode 100644 index 000000000..d7b032346 --- /dev/null +++ b/operator/internal/controllers/deployment/cluster_disambiguation_test.go @@ -0,0 +1,110 @@ +// Whether the cluster name a discovery run proposes carries something that +// distinguishes this Kubernetes cluster from another one pointed at the same +// control plane. +// +// InitialDiscoveryName is deliberately identical on every install, and that is +// fine for the OperatorOps and the ClusterDeploymentConfig discovery writes -- +// both live in this Kubernetes cluster's own API server, where the name +// collides with nothing else. The StorageCluster the document proposes does +// not stay local: it is registered on the control plane by name +// (StorageClusterReconciler.creationParams), and two separate Kubernetes +// clusters pointed at the same control plane (ControlPlane.spec.source.managed) +// otherwise propose the identical one. StorageClusterReconciler.postCluster's +// create-conflict fallback exists to resume a retried create of the SAME +// cluster and cannot tell that apart from a name that belongs to an entirely +// different Kubernetes cluster's own -- so it silently adopts the other one. + +package deployment + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" +) + +// kubeSystemNamespace is the fixture that gives a fake cluster its own +// identity, the same as the real kube-system Namespace every Kubernetes +// cluster carries from the moment its API server first comes up. +func kubeSystemNamespace(uid types.UID) *corev1.Namespace { + return &corev1.Namespace{ + ObjectMeta: metav1.ObjectMeta{Name: metav1.NamespaceSystem, UID: uid}, + } +} + +// writtenClusterName drives a discovery run to completion and returns the +// cluster name its document proposes. +func writtenClusterName(t *testing.T, r *runner) string { + t.Helper() + r.step() // start + r.step() // inspect + r.step() // probing: creates the Jobs + + if err := r.client.Create(t.Context(), + reportConfigMap(t, "worker-1", "0000:5e:00.0")); err != nil { + t.Fatalf("write a report: %v", err) + } + + r.step() // probing: sees the report, moves to Writing + r.step() // writing + + configs := r.configs() + if len(configs) != 1 { + t.Fatalf("wrote %d documents, want 1", len(configs)) + } + if configs[0].Spec.Cluster == nil { + t.Fatalf("the document names no cluster template") + } + return configs[0].Spec.Cluster.Name +} + +// Two Kubernetes clusters running the identical, unmodified bootstrap +// discovery -- same OperatorOps name, same fleet shape -- propose different +// cluster names once each has its own kube-system identity. This is the +// property that keeps a second cluster's discovery from ever being mistaken, +// on the shared control plane, for the first cluster's own. +func TestTwoKubernetesClustersProposeDifferentClusterNames(t *testing.T) { + clusterA := newRunner(t, + discoverRun(nil), worker("worker-1"), kubeSystemNamespace("11111111-1111-1111-1111-111111111111")) + clusterB := newRunner(t, + discoverRun(nil), worker("worker-1"), kubeSystemNamespace("22222222-2222-2222-2222-222222222222")) + + nameA := writtenClusterName(t, clusterA) + nameB := writtenClusterName(t, clusterB) + + if nameA == nameB { + t.Fatalf("both clusters proposed %q, which is exactly the collision this fixes", nameA) + } +} + +// The same kube-system identity proposes the same cluster name on a second +// run: the suffix is derived, not random, which is what keeps the "create is +// idempotent by name" property bootstrap.go documents. +func TestTheSameKubernetesClusterProposesTheSameNameAcrossRuns(t *testing.T) { + const uid = types.UID("33333333-3333-3333-3333-333333333333") + + first := newRunner(t, discoverRun(nil), worker("worker-1"), kubeSystemNamespace(uid)) + second := newRunner(t, discoverRun(nil), worker("worker-1"), kubeSystemNamespace(uid)) + + nameFirst := writtenClusterName(t, first) + nameSecond := writtenClusterName(t, second) + + if nameFirst != nameSecond { + t.Errorf("proposed %q then %q for the same kube-system identity", nameFirst, nameSecond) + } +} + +// Without a readable kube-system Namespace -- every fixture in this package +// before this file, and any real cluster whose RBAC has not yet caught up -- +// the proposed name is exactly what it always was. This is what keeps the fix +// additive. +func TestWithNoKubeSystemNamespaceTheNameIsUnchanged(t *testing.T) { + r := newRunner(t, discoverRun(nil), worker("worker-1")) + + got := writtenClusterName(t, r) + + if got != configNamePrefix+opsName+clusterNameSuffix { + t.Errorf("proposed %q, want %q", got, configNamePrefix+opsName+clusterNameSuffix) + } +} diff --git a/operator/internal/controllers/deployment/draftseed_test.go b/operator/internal/controllers/deployment/draftseed_test.go index c6ad3c6d6..265b9866e 100644 --- a/operator/internal/controllers/deployment/draftseed_test.go +++ b/operator/internal/controllers/deployment/draftseed_test.go @@ -15,6 +15,7 @@ package deployment import ( + "context" "strings" "testing" @@ -73,7 +74,7 @@ func noteMentioning(notes []string, want string) bool { func TestTheStatedLayoutSeedsTheInitialRunsDraft(t *testing.T) { r := &OperatorOpsReconciler{} - draft, notes, err := r.draftFor(initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, + draft, notes, err := r.draftFor(context.Background(), initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, discoverypkg.Plan{}, statedLayout()) if err != nil { t.Fatalf("the fleet was refused: %v", err) @@ -120,7 +121,7 @@ func TestTheStatedLayoutSeedsTheInitialRunsDraft(t *testing.T) { func TestTheStatedDraftFieldsReachTheDocument(t *testing.T) { r := &OperatorOpsReconciler{} - draft, _, err := r.draftFor(initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, + draft, _, err := r.draftFor(context.Background(), initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, discoverypkg.Plan{}, statedLayout()) if err != nil { t.Fatalf("the fleet was refused: %v", err) @@ -157,7 +158,7 @@ func TestARunNobodyLabeledGetsNoSeed(t *testing.T) { ObjectMeta: metav1.ObjectMeta{Name: "discover-again"}, } - draft, _, err := r.draftFor(theirs, &simplyblockv1alpha2.DiscoverSpec{}, + draft, _, err := r.draftFor(context.Background(), theirs, &simplyblockv1alpha2.DiscoverSpec{}, discoverypkg.Plan{}, statedLayout()) if err != nil { t.Fatalf("the fleet was refused: %v", err) @@ -198,7 +199,7 @@ func TestAnUnstatedFieldStaysDerived(t *testing.T) { Cluster: bootstrap.ClusterConfig{EnableChecksumValidation: ptr.To(true)}, }} - draft, _, err := r.draftFor(initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, + draft, _, err := r.draftFor(context.Background(), initialRun(), &simplyblockv1alpha2.DiscoverSpec{}, discoverypkg.Plan{}, partial) if err != nil { t.Fatalf("the fleet was refused: %v", err) @@ -230,7 +231,7 @@ func TestAGrowthDraftIsNotSeeded(t *testing.T) { r := &OperatorOpsReconciler{} spec := &simplyblockv1alpha2.DiscoverSpec{ClusterRef: "simplyblock-cluster"} - draft, _, err := r.draftFor(initialRun(), spec, discoverypkg.Plan{}, statedLayout()) + draft, _, err := r.draftFor(context.Background(), initialRun(), spec, discoverypkg.Plan{}, statedLayout()) if err != nil { t.Fatalf("the fleet was refused: %v", err) } diff --git a/operator/internal/controllers/deployment/operatorops_controller.go b/operator/internal/controllers/deployment/operatorops_controller.go index 59622fba6..881da2b04 100644 --- a/operator/internal/controllers/deployment/operatorops_controller.go +++ b/operator/internal/controllers/deployment/operatorops_controller.go @@ -37,6 +37,7 @@ import ( "errors" "fmt" "slices" + "strings" "time" batchv1 "k8s.io/api/batch/v1" @@ -79,6 +80,19 @@ const ( // when the caller named nothing. configNamePrefix = "discovered-" clusterNameSuffix = "-cluster" + + // clusterNameLimit is the longest name a StorageCluster (and so the + // backend cluster it registers under) can carry. Restated here because + // draftFor enforces it directly rather than relying on the apiserver to + // refuse an over-length name later, which is what CreatingCluster would do + // with no way to say why. + clusterNameLimit = 63 + + // disambiguatorLength is how much of the kube-system Namespace's UID + // clusterDisambiguator keeps. Long enough that two Kubernetes clusters + // collide by chance only astronomically rarely, short enough that it + // barely touches clusterNameLimit. + disambiguatorLength = 8 ) // The reasons a discovery run emits. They are constants rather than literals at @@ -157,6 +171,7 @@ type OperatorOpsReconciler struct { // +kubebuilder:rbac:groups=batch,resources=jobs,verbs=get;list;watch;create // +kubebuilder:rbac:groups="",resources=nodes,verbs=get;list;watch // +kubebuilder:rbac:groups="",resources=configmaps,verbs=get;list;watch +// +kubebuilder:rbac:groups="",resources=namespaces,verbs=get // +kubebuilder:rbac:groups="",resources=events,verbs=create;patch // Reconcile advances one operator operation by one step. @@ -637,7 +652,7 @@ func (r *OperatorOpsReconciler) write( "could not be parsed; the draft states what this run found") } - config, notes, err := r.draftFor(ops, spec, plan, installation) + config, notes, err := r.draftFor(ctx, ops, spec, plan, installation) if err != nil { // A fleet the run read and cannot draft a document for. The reason names // the worker and the shape of its disks, and the run's own message is one @@ -702,9 +717,50 @@ func refuseUnreadableFilter(spec *simplyblockv1alpha2.DiscoverSpec) error { return nil } +// clusterDisambiguator is a short, stable-per-Kubernetes-cluster suffix for the +// cluster name a discovery run proposes. +// +// InitialDiscoveryName is deliberately identical on every install, which is +// fine for the OperatorOps and the ClusterDeploymentConfig it writes -- both +// live in this Kubernetes cluster's own API server, where the name collides +// with nothing else. The StorageCluster the document proposes does not stay +// local, though: it is registered on the control plane by name +// (StorageClusterReconciler.creationParams), and a second Kubernetes cluster +// pointed at the same one (ControlPlane.spec.source.managed) would otherwise +// propose the identical name. StorageClusterReconciler.postCluster's +// create-conflict fallback exists to resume a retried create of the SAME +// cluster and cannot tell that apart from a name that belongs to an entirely +// different Kubernetes cluster's own -- so it silently adopts the other one. +// +// The kube-system Namespace's UID is the closest thing a Kubernetes cluster has +// to its own fixed identity: present from the moment its API server first +// comes up, and never reissued afterward. An unreadable namespace answers the +// empty string rather than an error -- a cluster this cannot be read from is +// not going to succeed at registering on the control plane a moment later +// either, and the caller falls back to the name exactly as it was before this +// existed. +func (r *OperatorOpsReconciler) clusterDisambiguator(ctx context.Context) string { + // No client to read kube-system from is the same as an unreadable one: no + // disambiguator, and the name is left as it was. + if r.Client == nil { + return "" + } + var ns corev1.Namespace + key := client.ObjectKey{Name: metav1.NamespaceSystem} + if err := r.Get(ctx, key, &ns); err != nil || ns.UID == "" { + return "" + } + id := strings.ReplaceAll(string(ns.UID), "-", "") + if len(id) > disambiguatorLength { + id = id[:disambiguatorLength] + } + return id +} + // draftFor builds the document, and the notes explaining the numbers in it that // were not read off the hardware. func (r *OperatorOpsReconciler) draftFor( + ctx context.Context, ops *simplyblockv1alpha2.OperatorOps, spec *simplyblockv1alpha2.DiscoverSpec, plan discoverypkg.Plan, @@ -773,6 +829,18 @@ func (r *OperatorOpsReconciler) draftFor( // and cannot be changed, so a stated one is taken over the one derived // from the draft's own name. clusterName := name + clusterNameSuffix + if disambiguator := r.clusterDisambiguator(ctx); disambiguator != "" { + base := name + // Truncated so the disambiguated name still fits, the same way a + // name too long for the limit already does not: this does not newly + // break a base name that already did not fit, only keeps this suffix + // from being the reason a borderline one no longer does. + room := clusterNameLimit - len(clusterNameSuffix) - len(disambiguator) - 1 + if room > 0 && len(base) > room { + base = base[:room] + } + clusterName = base + "-" + disambiguator + clusterNameSuffix + } if seed != nil && seed.Name != "" { clusterName = seed.Name } diff --git a/operator/internal/controllers/deployment/operatorops_unit_test.go b/operator/internal/controllers/deployment/operatorops_unit_test.go index 00b20ed02..9295f4489 100644 --- a/operator/internal/controllers/deployment/operatorops_unit_test.go +++ b/operator/internal/controllers/deployment/operatorops_unit_test.go @@ -815,7 +815,10 @@ func TestARunThatOutlivesItsDeadlineSaysWhereTheEvidenceIs(t *testing.T) { // approving it sees what the cluster will be told and the expansion has // something to spend on ubuntuHost. func TestTheDraftStatesTheHostOSTheProbesRead(t *testing.T) { - r := &OperatorOpsReconciler{} + // draftFor reads kube-system's UID to disambiguate the cluster name; a fake + // client with no such namespace returns NotFound, so the disambiguator falls + // back cleanly and this test still asserts only the host OS. + r := &OperatorOpsReconciler{Client: fake.NewClientBuilder().WithScheme(opsScheme(t)).Build()} ops := &simplyblockv1alpha2.OperatorOps{ObjectMeta: metav1.ObjectMeta{Name: "discover-1"}} plan := discoverypkg.Plan{Workers: []discoverypkg.Worker{{ Name: "worker-01", @@ -824,7 +827,7 @@ func TestTheDraftStatesTheHostOSTheProbesRead(t *testing.T) { }}, }}} - config, notes, err := r.draftFor(ops, &simplyblockv1alpha2.DiscoverSpec{}, plan, nil) + config, notes, err := r.draftFor(context.Background(), ops, &simplyblockv1alpha2.DiscoverSpec{}, plan, nil) if err != nil { t.Fatalf("the fleet was refused: %v", err) } @@ -848,7 +851,7 @@ func TestTheDraftStatesTheHostOSTheProbesRead(t *testing.T) { // A fleet whose workers run different distributions gets no host OS and a note // naming the split, because one document becomes one DaemonSet with one flag. func TestTheDraftStatesNoHostOSForAFleetThatDisagrees(t *testing.T) { - r := &OperatorOpsReconciler{} + r := &OperatorOpsReconciler{Client: fake.NewClientBuilder().WithScheme(opsScheme(t)).Build()} ops := &simplyblockv1alpha2.OperatorOps{ObjectMeta: metav1.ObjectMeta{Name: "discover-1"}} worker := func(name, distro string) discoverypkg.Worker { return discoverypkg.Worker{ @@ -861,7 +864,7 @@ func TestTheDraftStatesNoHostOSForAFleetThatDisagrees(t *testing.T) { worker("worker-02", "rocky"), }} - config, notes, err := r.draftFor(ops, &simplyblockv1alpha2.DiscoverSpec{}, plan, nil) + config, notes, err := r.draftFor(context.Background(), ops, &simplyblockv1alpha2.DiscoverSpec{}, plan, nil) if err != nil { t.Fatalf("the fleet was refused: %v", err) } @@ -909,11 +912,11 @@ func TestDiscoverProbesTolerateWhatTheRunWasToldTo(t *testing.T) { // drafts has to live with, so the draft states them rather than leaving a // reviewer to work out that the DaemonSet will schedule nowhere. func TestTheDraftCarriesTheTolerationsTheRunProbedWith(t *testing.T) { - r := &OperatorOpsReconciler{} + r := &OperatorOpsReconciler{Client: fake.NewClientBuilder().WithScheme(opsScheme(t)).Build()} ops := &simplyblockv1alpha2.OperatorOps{ObjectMeta: metav1.ObjectMeta{Name: "discover-1"}} spec := &simplyblockv1alpha2.DiscoverSpec{Tolerations: storagePlaneTaint} - config, notes, err := r.draftFor(ops, spec, discoverypkg.Plan{}, nil) + config, notes, err := r.draftFor(context.Background(), ops, spec, discoverypkg.Plan{}, nil) if err != nil { t.Fatalf("the fleet was refused: %v", err) } @@ -939,7 +942,7 @@ func TestAGrowthDraftCarriesNoTolerations(t *testing.T) { Tolerations: storagePlaneTaint, } - config, _, err := r.draftFor(ops, spec, discoverypkg.Plan{}, nil) + config, _, err := r.draftFor(context.Background(), ops, spec, discoverypkg.Plan{}, nil) if err != nil { t.Fatalf("the fleet was refused: %v", err) } diff --git a/operator/internal/controllers/driver/names.go b/operator/internal/controllers/driver/names.go index 10c08fdf1..3794b4c57 100644 --- a/operator/internal/controllers/driver/names.go +++ b/operator/internal/controllers/driver/names.go @@ -26,6 +26,13 @@ var clusterRoleComponents = []string{ // rather than the controller plugin's. const nodeComponent = "node" +// csiAddonsComponent is not a clusterRoleComponents entry: CSIAddonsNode is a +// namespaced kind (helm-charts' csiaddons.openshift.io_csiaddonsnodes.yaml +// sets scope: Namespaced), so its sidecar gets a namespaced Role instead of a +// sixth ClusterRole, per rbac-hardening's preference for the narrowest scope +// that works. See rbac.go's csiAddonsRoleRules. +const csiAddonsComponent = "csi-addons" + // objectNames is the whole naming surface of one deployment. type objectNames struct { prefix string @@ -81,6 +88,17 @@ func (n objectNames) clusterRoleBinding(component string) string { return n.prefix + component + "-binding" } +// role and roleBinding name the namespaced counterparts. Only csiAddonsComponent +// uses these today; the suffixes match clusterRole/clusterRoleBinding because +// the two are different Kinds and do not share a namespace with each other. +func (n objectNames) role(component string) string { + return n.prefix + component + "-role" +} + +func (n objectNames) roleBinding(component string) string { + return n.prefix + component + "-binding" +} + // driverName is spec.driverName with the CRD's default applied, so that code // reading it does not have to care whether admission had a chance to default it. func driverName(d *simplyblockv1alpha2.SimplyblockDriver) string { diff --git a/operator/internal/controllers/driver/rbac.go b/operator/internal/controllers/driver/rbac.go index d76f2e337..a4a940f2f 100644 --- a/operator/internal/controllers/driver/rbac.go +++ b/operator/internal/controllers/driver/rbac.go @@ -1,5 +1,7 @@ -// The RBAC the two plugins need: a ServiceAccount each, and the five ClusterRole -// and ClusterRoleBinding pairs behind them. +// The RBAC the two plugins need: a ServiceAccount each, the five ClusterRole and +// ClusterRoleBinding pairs behind the sidecars that watch cluster-scoped kinds, +// and one namespaced Role and RoleBinding pair for the csi-addons sidecar, whose +// CSIAddonsNode is namespaced. // // The rules are the ones the chart applies, because adoption reconciles toward // the state that is running and a rule this file widens or narrows is a @@ -8,7 +10,7 @@ // runs a second set of its own and each set says which sidecar needs what. // // Specified by operator/docs/designs/crd-redesign/design-simplyblockdriver.md -// §4.1 and §4.3. +// §4.1 and §4.3, and operator/docs/designs/design-csi-addons-replication.md §4.1. package driver @@ -30,6 +32,9 @@ var ( storage = []string{"storage.k8s.io"} snapshot = []string{"snapshot.storage.k8s.io"} groupsnapshot = []string{"groupsnapshot.storage.k8s.io"} + csiaddons = []string{"csiaddons.openshift.io"} + simplyblock = []string{"storage.simplyblock.io"} + coordination = []string{"coordination.k8s.io"} ) // clusterRoleRules is the rule set of each of the five roles, keyed by the @@ -68,6 +73,12 @@ var clusterRoleRules = map[string][]rbacv1.PolicyRule{ rule(storage, []string{"csinodes"}, "get", "list", "watch"), rule(core, []string{"nodes"}, "get", "list", "watch"), rule(storage, []string{"volumeattachments"}, "get", "list", "watch"), + // The consistency-group watcher's pre-join live migration: a volume + // labeled into a group while it lives off the group's pinned node is + // moved there first through a VolumeMigration, which the operator + // runs (co-location design §5). It creates and reads its own + // requests; it never updates or deletes one. + rule(simplyblock, []string{"volumemigrations"}, "get", "create"), }, "attacher": { rule(core, []string{"persistentvolumes"}, "get", "list", "watch", "update", "patch"), @@ -93,6 +104,84 @@ var clusterRoleRules = map[string][]rbacv1.PolicyRule{ }, } +// csiAddonsRoleRules is the csi-addons sidecar's rule set, granted as a +// namespaced Role rather than added to clusterRoleRules: the sidecar only ever +// touches its own CSIAddonsNode and its own leader-election Lease, both in +// this deployment's namespace, and neither kind justifies a cluster-wide grant. +var csiAddonsRoleRules = []rbacv1.PolicyRule{ + // rbac-justified: the sidecar publishes and maintains exactly one + // CSIAddonsNode, naming itself, so the kubernetes-csi-addons + // controller-manager (design-csi-addons-replication.md §4.1) can find its + // endpoint. It does not read any other driver's CSIAddonsNode. + rule(csiaddons, []string{"csiaddonsnodes"}, "get", "list", "watch", "create", "update", "delete"), + rule(csiaddons, []string{"csiaddonsnodes/status"}, "get", "update", "patch"), + // rbac-justified: only one replica of the controller StatefulSet serves + // CONTROLLER_SERVICE requests at a time; the Lease is how the sidecar + // replicas elect that one, in this namespace only. + rule(coordination, []string{"leases"}, "get", "list", "watch", "create", "update", "delete"), + rule(core, []string{"events"}, "create", "patch"), +} + +func csiAddonsRole(d *simplyblockv1alpha2.SimplyblockDriver) *rbacv1.Role { + n := names(d) + return &rbacv1.Role{ + ObjectMeta: metav1.ObjectMeta{Name: n.role(csiAddonsComponent), Namespace: d.Namespace}, + Rules: csiAddonsRoleRules, + } +} + +func csiAddonsRoleBinding(d *simplyblockv1alpha2.SimplyblockDriver) *rbacv1.RoleBinding { + n := names(d) + return &rbacv1.RoleBinding{ + ObjectMeta: metav1.ObjectMeta{Name: n.roleBinding(csiAddonsComponent), Namespace: d.Namespace}, + Subjects: []rbacv1.Subject{{ + Kind: rbacv1.ServiceAccountKind, + Name: n.controllerServiceAccount, + Namespace: d.Namespace, + }}, + RoleRef: rbacv1.RoleRef{ + APIGroup: rbacv1.GroupName, + Kind: "Role", + Name: n.role(csiAddonsComponent), + }, + } +} + +// authDelegatorClusterRole is the well-known, built-in ClusterRole every +// component that validates bearer tokens via TokenReview binds to, rather +// than each defining its own copy of the same two-verb rule. +const authDelegatorClusterRole = "system:auth-delegator" + +// csiAddonsAuthDelegatorBinding grants the controller plugin's account +// tokenreviews.authentication.k8s.io:create, cluster-scoped since TokenReview +// has no namespaced form. Required, not optional: the csi-addons sidecar's +// gRPC server authenticates every incoming call from the controller-manager +// by reviewing its bearer token (internal/kubernetes/token/grpc.go, +// --enable-auth defaults to true) — confirmed against a live cluster, where +// omitting this left every connection attempt failing with "failed to +// review token ... is forbidden ... at the cluster scope". Binding to the +// built-in role rather than a hand-rolled ClusterRole needs no new marker on +// the operator's own ClusterRole: the operator already holds `bind` on +// every ClusterRole unconditionally (rbac.go's clusterroles;clusterrolebindings +// marker), which is what Kubernetes' escalation prevention checks for +// referencing an existing role instead of granting its permissions directly. +func csiAddonsAuthDelegatorBinding(d *simplyblockv1alpha2.SimplyblockDriver) *rbacv1.ClusterRoleBinding { + n := names(d) + return &rbacv1.ClusterRoleBinding{ + ObjectMeta: metav1.ObjectMeta{Name: n.clusterRoleBinding("csi-addons-auth-delegator")}, + Subjects: []rbacv1.Subject{{ + Kind: rbacv1.ServiceAccountKind, + Name: n.controllerServiceAccount, + Namespace: d.Namespace, + }}, + RoleRef: rbacv1.RoleRef{ + APIGroup: rbacv1.GroupName, + Kind: "ClusterRole", + Name: authDelegatorClusterRole, + }, + } +} + // serviceAccountFor names the account each role is bound to. The node plugin has // its own, and the controller plugin's sidecars share one. func serviceAccountFor(n objectNames, component string) string { diff --git a/operator/internal/controllers/driver/rbac_test.go b/operator/internal/controllers/driver/rbac_test.go index 7434fe61d..413a3220e 100644 --- a/operator/internal/controllers/driver/rbac_test.go +++ b/operator/internal/controllers/driver/rbac_test.go @@ -105,6 +105,7 @@ func TestRulesMatchTheChart(t *testing.T) { {nodeComponent, "", "events", []string{"create", "patch"}}, {"provisioner", "snapshot.storage.k8s.io", "volumesnapshotcontents/status", []string{"get", "update", "patch"}}, {"provisioner", "", "persistentvolumes", []string{"get", "list", "watch", "create", "delete", "patch"}}, + {"provisioner", "storage.simplyblock.io", "volumemigrations", []string{"get", "create"}}, {"attacher", "storage.k8s.io", "volumeattachments/status", []string{"patch"}}, {"resizer", "", "persistentvolumeclaims/status", []string{"patch"}}, {"health-monitor", "", "events", []string{"get", "list", "watch", "create", "patch"}}, @@ -181,6 +182,94 @@ func TestNoRuleIsAWildcard(t *testing.T) { } } +// The csi-addons sidecar is the one component whose grant is a namespaced Role: +// CSIAddonsNode is namespaced (helm-charts' +// csiaddons.openshift.io_csiaddonsnodes.yaml sets scope: Namespaced), so a +// ClusterRole would be wider than the sidecar's own job. +func TestCSIAddonsRoleIsNamespacedAndBoundToTheControllerAccount(t *testing.T) { + d := testDriver("simplyblock") + n := names(d) + + role := csiAddonsRole(d) + if role.Namespace != d.Namespace { + t.Errorf("role namespace = %q, want %q", role.Namespace, d.Namespace) + } + if len(role.Rules) == 0 { + t.Error("csi-addons role has no rules, so its sidecar can do nothing") + } + + binding := csiAddonsRoleBinding(d) + if binding.Namespace != d.Namespace { + t.Errorf("binding namespace = %q, want %q", binding.Namespace, d.Namespace) + } + if len(binding.Subjects) != 1 || binding.Subjects[0].Name != n.controllerServiceAccount || + binding.Subjects[0].Namespace != d.Namespace { + t.Errorf("binding subject = %+v, want the controller account in %q", + binding.Subjects, d.Namespace) + } + if binding.RoleRef.Kind != "Role" || binding.RoleRef.Name != role.Name { + t.Errorf("roleRef = %+v, want Role %q", binding.RoleRef, role.Name) + } +} + +// The csi-addons sidecar's gRPC server authenticates every incoming call +// from the controller-manager via TokenReview (internal/kubernetes/token/grpc.go, +// --enable-auth defaults to true), which is cluster-scoped and so cannot be +// granted by the namespaced Role above -- confirmed against a live cluster, +// where omitting this left every connection failing with "failed to review +// token ... is forbidden ... at the cluster scope". +func TestCSIAddonsSidecarCanAuthenticateIncomingCalls(t *testing.T) { + d := testDriver("simplyblock") + n := names(d) + + binding := csiAddonsAuthDelegatorBinding(d) + if binding.Namespace != "" { + t.Errorf("namespace = %q, want \"\" (ClusterRoleBinding is cluster-scoped)", binding.Namespace) + } + if len(binding.Subjects) != 1 || binding.Subjects[0].Name != n.controllerServiceAccount || + binding.Subjects[0].Namespace != d.Namespace { + t.Errorf("binding subject = %+v, want the controller account in %q", + binding.Subjects, d.Namespace) + } + if binding.RoleRef.Kind != "ClusterRole" || binding.RoleRef.Name != "system:auth-delegator" { + t.Errorf("roleRef = %+v, want the built-in ClusterRole system:auth-delegator", binding.RoleRef) + } +} + +// The rule set is exactly what the sidecar's own job needs: its CSIAddonsNode +// and its leader-election Lease, both scoped to this namespace. +func TestCSIAddonsRoleRulesAreScopedToItsOwnJob(t *testing.T) { + tests := []struct { + group string + resource string + verbs []string + }{ + {"csiaddons.openshift.io", "csiaddonsnodes", []string{"get", "list", "watch", "create", "update", "delete"}}, + {"csiaddons.openshift.io", "csiaddonsnodes/status", []string{"get", "update", "patch"}}, + {"coordination.k8s.io", "leases", []string{"get", "list", "watch", "create", "update", "delete"}}, + {"", "events", []string{"create", "patch"}}, + } + + for _, tc := range tests { + t.Run(tc.resource, func(t *testing.T) { + var found *rbacv1.PolicyRule + for i, r := range csiAddonsRoleRules { + if len(r.APIGroups) == 1 && r.APIGroups[0] == tc.group && + len(r.Resources) == 1 && r.Resources[0] == tc.resource { + found = &csiAddonsRoleRules[i] + break + } + } + if found == nil { + t.Fatalf("no rule for %s in the csi-addons role", tc.resource) + } + if !slices.Equal(found.Verbs, tc.verbs) { + t.Errorf("verbs = %v, want %v", found.Verbs, tc.verbs) + } + }) + } +} + // The csi-snapshotter sidecar watches VolumeGroupSnapshotContent and drives // the GroupController when the CSIVolumeGroupSnapshot gate is on // (design-consistency-groups.md §9, P0-4). The chart granted these on the diff --git a/operator/internal/controllers/driver/registration_test.go b/operator/internal/controllers/driver/registration_test.go index 32da742d6..4456ec910 100644 --- a/operator/internal/controllers/driver/registration_test.go +++ b/operator/internal/controllers/driver/registration_test.go @@ -135,12 +135,26 @@ func TestSidecarsDefaultToTheOperatorsRelease(t *testing.T) { snapshotter: defaultSnapshotterImage, healthMonitor: defaultHealthMonitorImage, nodeDriverRegistrar: defaultNodeDriverRegistrarImage, + csiAddons: defaultCSIAddonsImage, } if got != want { t.Errorf("sidecars = %+v, want %+v", got, want) } } +// The csi-addons sidecar's default names the real upstream image +// (quay.io/csiaddons/k8s-sidecar) rather than a quay.io/simplyblock-io mirror +// that does not exist yet -- unlike the other six sidecars, which do have one. +// TestSidecarsDefaultToTheOperatorsRelease checks defaultCSIAddonsImage +// against itself, so a wrong constant would still pass it; this pins the +// literal a pod actually pulls. +func TestTheCSIAddonsSidecarDefaultsToTheRealUpstreamImage(t *testing.T) { + const wantImage = "quay.io/csiaddons/k8s-sidecar:v0.15.0" + if got := sidecars(testDriver("simplyblock")).csiAddons; got != wantImage { + t.Errorf("csiAddons default = %q, want the real upstream image %q", got, wantImage) + } +} + // U-86: one override reaches its own sidecar and no other, which is the property // that makes a pin survivable without freezing the rest of the deployment. func TestOneSidecarOverrideReachesOnlyItsOwn(t *testing.T) { diff --git a/operator/internal/controllers/driver/sidecars.go b/operator/internal/controllers/driver/sidecars.go index 0c65dc244..582bb50c2 100644 --- a/operator/internal/controllers/driver/sidecars.go +++ b/operator/internal/controllers/driver/sidecars.go @@ -1,4 +1,4 @@ -// The six CSI sidecar images, and which of them a deployment gets. +// The seven CSI sidecar images, and which of them a deployment gets. // // The versions below are this operator release's, meaning the combination it was // tested against, and spec.sidecarImages overrides one at a time. The field @@ -25,6 +25,13 @@ const ( defaultSnapshotterImage = "quay.io/simplyblock-io/csi-snapshotter:v8.2.0" defaultHealthMonitorImage = "quay.io/simplyblock-io/csi-external-health-monitor-controller:v0.14.0" defaultNodeDriverRegistrarImage = "quay.io/simplyblock-io/csi-node-driver-registrar:v2.12.0" + // defaultCSIAddonsImage is the kubernetes-csi-addons sidecar, pinned at the + // same v0.15.0 the chart's controller-manager runs (design P0-5). The real + // upstream image (quay.io/csiaddons/k8s-sidecar) rather than a + // quay.io/simplyblock-io mirror: that mirror does not exist yet, and + // mirroring it is a release task, not something to assume has already + // happened. + defaultCSIAddonsImage = "quay.io/csiaddons/k8s-sidecar:v0.15.0" ) // resolvedSidecars is the image each sidecar runs, after the overrides. @@ -35,6 +42,7 @@ type resolvedSidecars struct { snapshotter string healthMonitor string nodeDriverRegistrar string + csiAddons string } func sidecars(d *simplyblockv1alpha2.SimplyblockDriver) resolvedSidecars { @@ -46,6 +54,7 @@ func sidecars(d *simplyblockv1alpha2.SimplyblockDriver) resolvedSidecars { snapshotter: orDefault(o.Snapshotter, defaultSnapshotterImage), healthMonitor: orDefault(o.HealthMonitor, defaultHealthMonitorImage), nodeDriverRegistrar: orDefault(o.NodeDriverRegistrar, defaultNodeDriverRegistrarImage), + csiAddons: orDefault(o.CSIAddons, defaultCSIAddonsImage), } } diff --git a/operator/internal/controllers/driver/simplyblockdriver_controller.go b/operator/internal/controllers/driver/simplyblockdriver_controller.go index 265b18f53..b5860b92c 100644 --- a/operator/internal/controllers/driver/simplyblockdriver_controller.go +++ b/operator/internal/controllers/driver/simplyblockdriver_controller.go @@ -67,8 +67,24 @@ const ( // +kubebuilder:rbac:groups=apps,resources=daemonsets;statefulsets,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups="",resources=serviceaccounts;configmaps,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups=rbac.authorization.k8s.io,resources=clusterroles;clusterrolebindings,verbs=get;list;watch;create;update;patch;delete;escalate;bind +// rbac-justified: rbac.go's csiAddonsRole/csiAddonsRoleBinding are a +// namespaced Role, not a ClusterRole (CSIAddonsNode is namespaced) -- this +// mirrors the ClusterRole marker above for the same reason: the manager +// creates RBAC on behalf of the sidecars it deploys, and escalate/bind is +// what Kubernetes' escalation prevention requires to do that, capped by the +// manager's own role, which is why the resources this grants are named +// individually below rather than left open-ended. +// +kubebuilder:rbac:groups=rbac.authorization.k8s.io,resources=roles;rolebindings,verbs=get;list;watch;create;update;patch;delete;escalate;bind // +kubebuilder:rbac:groups=storage.k8s.io,resources=csidrivers,verbs=get;list;watch;create;update;patch;delete // +kubebuilder:rbac:groups=snapshot.storage.k8s.io,resources=volumesnapshotclasses,verbs=get;list;watch;create;update;patch;delete +// rbac-justified: the csi-addons sidecar (rbac.go's csiAddonsRole) needs to +// publish and update its own CSIAddonsNode and hold its own leader-election +// Lease, and the manager creates that namespaced Role on the sidecar's +// behalf -- RBAC escalation prevention requires the manager to already hold +// what it grants, capped by holding only these same resources itself. +// +kubebuilder:rbac:groups=csiaddons.openshift.io,resources=csiaddonsnodes,verbs=get;list;watch;create;update;delete +// +kubebuilder:rbac:groups=csiaddons.openshift.io,resources=csiaddonsnodes/status,verbs=get;update;patch +// +kubebuilder:rbac:groups=coordination.k8s.io,resources=leases,verbs=get;list;watch;create;update;delete // SimplyblockDriverReconciler applies the CSI driver deployment. type SimplyblockDriverReconciler struct { @@ -281,6 +297,7 @@ func (r *SimplyblockDriverReconciler) desired( for _, crb := range clusterRoleBindings(d) { objects = append(objects, crb) } + objects = append(objects, csiAddonsRole(d), csiAddonsRoleBinding(d), csiAddonsAuthDelegatorBinding(d)) objects = append(objects, nodeDaemonSet(d, image), controllerStatefulSet(d, image), csiDriver(d)) // The class is this deployment's and is applied wherever the kinds exist, diff --git a/operator/internal/controllers/driver/simplyblockdriver_controller_test.go b/operator/internal/controllers/driver/simplyblockdriver_controller_test.go index ae5441e02..859e9e7d6 100644 --- a/operator/internal/controllers/driver/simplyblockdriver_controller_test.go +++ b/operator/internal/controllers/driver/simplyblockdriver_controller_test.go @@ -59,6 +59,10 @@ func TestDesiredCoversTheWholeObjectSet(t *testing.T) { counts["role"]++ case *rbacv1.ClusterRoleBinding: counts["binding"]++ + case *rbacv1.Role: + counts["namespacedRole"]++ + case *rbacv1.RoleBinding: + counts["namespacedBinding"]++ case *appsv1.DaemonSet: counts["ds"]++ case *appsv1.StatefulSet: @@ -71,7 +75,8 @@ func TestDesiredCoversTheWholeObjectSet(t *testing.T) { } want := map[string]int{ - "sa": 2, "cm": 2, "role": 5, "binding": 5, + "sa": 2, "cm": 2, "role": 5, "binding": 6, + "namespacedRole": 1, "namespacedBinding": 1, "ds": 1, "sts": 1, "csidriver": 1, // the VolumeSnapshotClass, which is unstructured "other": 1, diff --git a/operator/internal/controllers/driver/workloads.go b/operator/internal/controllers/driver/workloads.go index 1ac218374..804253469 100644 --- a/operator/internal/controllers/driver/workloads.go +++ b/operator/internal/controllers/driver/workloads.go @@ -14,6 +14,7 @@ package driver import ( "slices" + "strconv" appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" @@ -258,6 +259,7 @@ func controllerStatefulSet(d *simplyblockv1alpha2.SimplyblockDriver, image strin "--leader-election=false", }), controllerPluginContainer(d, image), + csiAddonsSidecarContainer(d, s.csiAddons, sidecarMount), } // The snapshotter runs privileged in the chart, and the health monitor // exposes a port. Both are properties of the container rather than of the @@ -321,6 +323,70 @@ func controllerSidecar( } } +// csiAddonsControllerPort is where this sidecar's own gRPC server listens for +// the kubernetes-csi-addons controller-manager (design +// design-csi-addons-replication.md §4.1). The manager never hardcodes it: it +// reads the endpoint the sidecar published on its own CSIAddonsNode, so this +// number only has to be free and to agree with the container port below it. +const csiAddonsControllerPort int32 = 9070 + +// csiAddonsSidecarContainer runs the kubernetes-csi-addons sidecar (upstream +// quay.io/csiaddons/k8s-sidecar), which is a separate binary from the +// controller-manager P0-5 already vendored: it connects to this pod's own CSI +// socket to probe the csi-addons Identity/Replication services this driver +// now serves alongside its CSI ones (driver.go), publishes a CSIAddonsNode +// naming this pod so the manager can find it, and leader-elects across +// replicas of this StatefulSet before serving controller requests. +// +// The endpoint the sidecar advertises MUST be the pod://. +// form, never a bare ip:port: the controller-manager's own resolveEndpoint +// (internal/controller/csiaddons/csiaddonsnode_controller.go, v0.15.0) +// hard-requires url.Parse's Scheme to equal "pod" and rejects everything +// else with "endpoint scheme %q not supported" -- confirmed against a live +// cluster, where passing --controller-ip produced the OTHER branch of the +// sidecar's own BuildEndpointURL (a bare ":", no scheme at all) +// and the controller-manager failed every connection attempt with "first +// path segment in URL cannot contain colon" until it gave up and deleted +// the CSIAddonsNode. Omitting --controller-ip is the fix, not a missing +// feature: the manager resolves the pod's CURRENT IP itself via a live API +// read at connection time (see resolveEndpoint), which is more robust than +// baking a static IP into the object anyway -- it survives the pod +// restarting with a new IP without anyone having to update anything. +func csiAddonsSidecarContainer( + d *simplyblockv1alpha2.SimplyblockDriver, image string, mounts []corev1.VolumeMount, +) corev1.Container { + return corev1.Container{ + Name: "csi-addons", + Image: image, + ImagePullPolicy: pullPolicy(d), + Args: []string{ + verbosity, + "--csi-addons-address=" + controllerSocketPath, + "--node-id=$(NODE_ID)", + "--controller-port=" + strconv.Itoa(int(csiAddonsControllerPort)), + "--pod=$(POD_NAME)", + "--namespace=$(POD_NAMESPACE)", + "--pod-uid=$(POD_UID)", + "--leader-election-namespace=$(POD_NAMESPACE)", + }, + Env: []corev1.EnvVar{ + // Required, not cosmetic: csiaddonsnode.Manager.Node ("the + // hostname of the system where the sidecar is running") rejects + // an empty value with "invalid configuration: missing node" + // before it ever creates the CSIAddonsNode object. + fieldRefEnv("NODE_ID", "spec.nodeName"), + fieldRefEnv("POD_NAME", "metadata.name"), + fieldRefEnv("POD_NAMESPACE", "metadata.namespace"), + fieldRefEnv("POD_UID", "metadata.uid"), + }, + Ports: []corev1.ContainerPort{ + {ContainerPort: csiAddonsControllerPort, Name: "csi-addons", Protocol: corev1.ProtocolTCP}, + }, + Resources: d.Spec.ControllerResources, + VolumeMounts: mounts, + } +} + func controllerPluginContainer(d *simplyblockv1alpha2.SimplyblockDriver, image string) corev1.Container { return corev1.Container{ Name: "csi-controller", diff --git a/operator/internal/controllers/driver/workloads_test.go b/operator/internal/controllers/driver/workloads_test.go index ad3c21a49..9ced3f5a8 100644 --- a/operator/internal/controllers/driver/workloads_test.go +++ b/operator/internal/controllers/driver/workloads_test.go @@ -96,6 +96,76 @@ func TestSnapshotterSidecarIsAppliedRegardlessOfTheToggle(t *testing.T) { } } +// The csi-addons sidecar is appended after the plugin container (index 6), +// never inserted: containerStatefulSet's positional tweaks (containers[1]'s +// SecurityContext, containers[4]'s Ports) only stay pointed at the snapshotter +// and health-monitor if nothing ahead of them shifts. +func TestCSIAddonsSidecarIsAppliedAfterThePlugin(t *testing.T) { + d := testDriver("simplyblock") + containers := controllerStatefulSet(d, testImage).Spec.Template.Spec.Containers + + if len(containers) != 7 { + t.Fatalf("got %d containers, want 7", len(containers)) + } + if containers[5].Name != "csi-controller" || containers[6].Name != "csi-addons" { + t.Errorf("containers[5:7] = %q, %q, want csi-controller, csi-addons", + containers[5].Name, containers[6].Name) + } + + addr, ok := argValue(&containers[6], "--csi-addons-address") + if !ok || addr != controllerSocketPath { + t.Errorf("csi-addons --csi-addons-address = %q, want %q", addr, controllerSocketPath) + } +} + +// The sidecar advertises itself as pod://., never a bare +// ip:port: the controller-manager's own resolveEndpoint (v0.15.0) hard-requires +// url.Parse's Scheme to equal "pod" and rejects everything else with +// "endpoint scheme %q not supported" -- confirmed against a live cluster, +// where passing --controller-ip produced a bare ":" (no scheme) +// that the controller-manager could never parse, so it kept failing every +// connection attempt and deleting the CSIAddonsNode in a tight loop. +func TestCSIAddonsSidecarAdvertisesItsOwnPod(t *testing.T) { + d := testDriver("simplyblock") + c := containerNamed(controllerStatefulSet(d, testImage).Spec.Template.Spec.Containers, "csi-addons") + if c == nil { + t.Fatal("csi-addons is not applied") + } + + wantFieldPaths := map[string]string{ + "NODE_ID": "spec.nodeName", "POD_NAME": "metadata.name", + "POD_NAMESPACE": "metadata.namespace", "POD_UID": "metadata.uid", + } + for _, e := range c.Env { + want, known := wantFieldPaths[e.Name] + if !known { + continue + } + delete(wantFieldPaths, e.Name) + if e.ValueFrom == nil || e.ValueFrom.FieldRef == nil || e.ValueFrom.FieldRef.FieldPath != want { + t.Errorf("%s field path = %+v, want %q", e.Name, e.ValueFrom, want) + } + } + for name := range wantFieldPaths { + t.Errorf("no %s env var", name) + } + + // Required, not cosmetic: csiaddonsnode.Manager.Node rejects an empty + // value with "invalid configuration: missing node" before it ever + // creates the CSIAddonsNode object (confirmed against a live cluster). + if nodeID, ok := argValue(c, "--node-id"); !ok || nodeID != "$(NODE_ID)" { + t.Errorf("--node-id = %q, want $(NODE_ID)", nodeID) + } + // --controller-ip must NEVER be set: it takes BuildEndpointURL's bare + // ip:port branch, which this controller-manager version cannot parse. + if _, ok := argValue(c, "--controller-ip"); ok { + t.Error("--controller-ip is set; the pod:// addressing scheme requires omitting it") + } + if ns, ok := argValue(c, "--leader-election-namespace"); !ok || ns != "$(POD_NAMESPACE)" { + t.Errorf("--leader-election-namespace = %q, want $(POD_NAMESPACE)", ns) + } +} + // U-33 and U-34: the driver name reaches the kubelet registration path and the // hostPath the node plugin mounts, which is what design §9 Q3 says the chart // writes literally. diff --git a/operator/internal/controllers/node/auth.go b/operator/internal/controllers/node/auth.go new file mode 100644 index 000000000..87dbb58d8 --- /dev/null +++ b/operator/internal/controllers/node/auth.go @@ -0,0 +1,65 @@ +// Cross-cluster authentication for this package's two reconcilers. +// +// A node's or node operation's control plane may be a +// ControlPlane.spec.source.managed one, on a different Kubernetes cluster +// than this operator, where a Kubernetes TokenReview of this operator's own +// service-account token can never succeed. What does cross that boundary is a +// cluster's own backend secret -- a plain credential the control plane's +// database matches by value, not a Kubernetes identity -- so every call +// scoped to an already-adopted cluster authenticates with that instead, once +// it is known. See internal/controllers/cluster's identically-motivated fix. + +package node + +import ( + "context" + "fmt" + + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/types" + "sigs.k8s.io/controller-runtime/pkg/client" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/webapi" +) + +// clusterSecretByName reads the secret StorageClusterReconciler.persist wrote +// for the named StorageCluster, and reports the empty string when there is +// none. +func clusterSecretByName( + ctx context.Context, c client.Client, namespace, clusterName string, +) (string, error) { + var secret corev1.Secret + key := types.NamespacedName{ + Name: fmt.Sprintf("simplyblock-cluster-%s", clusterName), + Namespace: namespace, + } + if err := c.Get(ctx, key, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil +} + +// clusterSecretForNode reads the same secret for the cluster a named +// StorageNode belongs to, for a caller that only has the node's name (a +// StorageNodeOps names its target node, not the node's cluster directly). +func clusterSecretForNode( + ctx context.Context, c client.Client, namespace, nodeRef string, +) (string, error) { + var node simplyblockv1alpha2.StorageNode + key := types.NamespacedName{Name: nodeRef, Namespace: namespace} + if err := c.Get(ctx, key, &node); err != nil { + return "", err + } + return clusterSecretByName(ctx, c, namespace, node.Spec.ClusterRef) +} + +// authenticatedContext attaches a cluster's own credential to ctx when one is +// known, so a call scoped to it authenticates as that cluster instead of as +// this operator's Kubernetes identity. +func authenticatedContext(ctx context.Context, secret string, err error) context.Context { + if err != nil || secret == "" { + return ctx + } + return webapi.WithBearerToken(ctx, secret) +} diff --git a/operator/internal/controllers/node/auth_test.go b/operator/internal/controllers/node/auth_test.go new file mode 100644 index 000000000..72c37a606 --- /dev/null +++ b/operator/internal/controllers/node/auth_test.go @@ -0,0 +1,80 @@ +// A node's or node operation's control plane may be a +// ControlPlane.spec.source.managed one, on a different Kubernetes cluster +// than this operator, where a Kubernetes TokenReview of this operator's own +// service-account token can never succeed. Every call scoped to an +// already-adopted cluster must authenticate as that cluster's own recorded +// secret instead. See internal/controllers/cluster's identically-motivated +// tests. + +package node + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" +) + +// authTestClusterSecret is the cluster's own recorded credential, the one +// every assertion below wants to see instead of an operator SA token. +const authTestClusterSecret = "the-clusters-own-secret" + +// aClusterSecret is the Secret StorageClusterReconciler.persist writes for +// anOpsCluster(), keyed by its Kubernetes name. +func aClusterSecret() *corev1.Secret { + return &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{ + Name: "simplyblock-cluster-" + opsCluster, + Namespace: opsNamespace, + }, + Data: map[string][]byte{"secret": []byte(authTestClusterSecret)}, + } +} + +// checkBearerTokens fails the test on any call the control plane recorded +// that did not carry authTestClusterSecret as its bearer-token override. +func checkBearerTokens(t *testing.T, api *scriptedControlPlane) { + t.Helper() + if len(api.calls) == 0 { + t.Fatal("the control plane was never asked anything") + } + for i, obs := range api.bearerTokens { + if !obs.ok { + t.Errorf("call %d (%s): carried no bearer-token override", i, api.calls[i]) + continue + } + if obs.token != authTestClusterSecret { + t.Errorf("call %d (%s): bearer token = %q, want the cluster's own secret %q", + i, api.calls[i], obs.token, authTestClusterSecret) + } + } +} + +// The entity reconciler's steady-state read (StorageNode()) authenticates as +// the node's cluster once that cluster's secret is on record. +func TestANodeReadAuthenticatesAsItsClusterOnceItsSecretIsKnown(t *testing.T) { + api := aControlPlane().reporting(nodeStatusInCreation) + r, _ := aSteadyNode(t, api, aClusterSecret()) + // Forces the direct StorageNode() read rather than the stream's cache. + r.Nodes = &deliveredNodes{synced: false} + + settle(t, r) + + checkBearerTokens(t, api) +} + +// A node operation's calls -- the ones actions.go, classify.go, +// hostmaintenance.go, migrate.go, and remove.go make -- authenticate the same +// way, resolved from the operation's own target node's cluster. +func TestANodeOperationAuthenticatesAsItsClusterOnceItsSecretIsKnown(t *testing.T) { + api := aControlPlane().reporting(nodeStatusOnline) + ops := anAdvancingOperation( + "a-shutdown", simplyblockv1alpha2.StorageNodeOpsActionShutdown, stepRequesting) + r, _ := anOpsWorld(t, api, ops, aClusterSecret()) + + pass(t, r, "a-shutdown") + + checkBearerTokens(t, api) +} diff --git a/operator/internal/controllers/node/awaitworker_test.go b/operator/internal/controllers/node/awaitworker_test.go index ffb3ac496..5f6d45cb9 100644 --- a/operator/internal/controllers/node/awaitworker_test.go +++ b/operator/internal/controllers/node/awaitworker_test.go @@ -72,6 +72,7 @@ func aProvisioner(t *testing.T, worker *corev1.Node, node *simplyblockv1alpha2.S Scheme: scheme, Recorder: events.NewFakeRecorder(64), API: countingBackend{adds: &adds}, + Workload: &Workload{Client: apiClient}, }, cluster, apiClient, &adds } diff --git a/operator/internal/controllers/node/fixtures_test.go b/operator/internal/controllers/node/fixtures_test.go index c271b864a..ee65ec727 100644 --- a/operator/internal/controllers/node/fixtures_test.go +++ b/operator/internal/controllers/node/fixtures_test.go @@ -91,6 +91,13 @@ type scriptedControlPlane struct { // once reads. calls []string + // bearerTokens parallels calls: bearerTokens[i] is what + // webapi.BearerTokenFromContext reported for calls[i]'s own context, so a + // test can check that a cluster-scoped call authenticated as the cluster + // it was actually asked about rather than as this operator's own + // Kubernetes identity. + bearerTokens []bearerObservation + // restarts carries the parameters of each restart, because three actions // issue one and they differ precisely in what they fill in. restarts []RestartParams @@ -106,6 +113,12 @@ type scriptedControlPlane struct { admissionAbsent bool } +// bearerObservation is one entry of scriptedControlPlane.bearerTokens. +type bearerObservation struct { + token string + ok bool +} + // aControlPlane reports one online node and nothing else. func aControlPlane() *scriptedControlPlane { return &scriptedControlPlane{ @@ -163,15 +176,17 @@ func (c *scriptedControlPlane) asked(method string) int { return count } -func (c *scriptedControlPlane) record(method, argument string) error { +func (c *scriptedControlPlane) record(ctx context.Context, method, argument string) error { + token, ok := webapi.BearerTokenFromContext(ctx) c.calls = append(c.calls, method+":"+argument) + c.bearerTokens = append(c.bearerTokens, bearerObservation{token, ok}) return c.refuse[method] } func (c *scriptedControlPlane) RemovalAdmission( - _ context.Context, _, nodeID string, + ctx context.Context, _, nodeID string, ) (RemovalAdmission, bool, error) { - if err := c.record("RemovalAdmission", nodeID); err != nil { + if err := c.record(ctx, "RemovalAdmission", nodeID); err != nil { return RemovalAdmission{}, false, err } if c.admissionAbsent { @@ -184,9 +199,9 @@ func (c *scriptedControlPlane) RemovalAdmission( } func (c *scriptedControlPlane) StorageNode( - _ context.Context, _, nodeID string, + ctx context.Context, _, nodeID string, ) (NodeReading, bool, error) { - if err := c.record("StorageNode", nodeID); err != nil { + if err := c.record(ctx, "StorageNode", nodeID); err != nil { return NodeReading{}, false, err } reading, found := c.nodes[nodeID] @@ -194,9 +209,9 @@ func (c *scriptedControlPlane) StorageNode( } func (c *scriptedControlPlane) StorageNodes( - _ context.Context, clusterID string, + ctx context.Context, clusterID string, ) ([]NodeReading, error) { - if err := c.record("StorageNodes", clusterID); err != nil { + if err := c.record(ctx, "StorageNodes", clusterID); err != nil { return nil, err } readings := make([]NodeReading, 0, len(c.nodes)) @@ -207,84 +222,84 @@ func (c *scriptedControlPlane) StorageNodes( } func (c *scriptedControlPlane) AddNode( - _ context.Context, clusterID string, _ utils.StorageNodeSetAddParams, + ctx context.Context, clusterID string, _ utils.StorageNodeSetAddParams, ) (string, error) { - return theAddTask, c.record("AddNode", clusterID) + return theAddTask, c.record(ctx, "AddNode", clusterID) } -func (c *scriptedControlPlane) Task(_ context.Context, _, taskID string) (TaskReading, error) { - return TaskReading{}, c.record("Task", taskID) +func (c *scriptedControlPlane) Task(ctx context.Context, _, taskID string) (TaskReading, error) { + return TaskReading{}, c.record(ctx, "Task", taskID) } -func (c *scriptedControlPlane) Suspend(_ context.Context, _, nodeID string) error { - return c.record("Suspend", nodeID) +func (c *scriptedControlPlane) Suspend(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "Suspend", nodeID) } -func (c *scriptedControlPlane) Resume(_ context.Context, _, nodeID string) error { - return c.record("Resume", nodeID) +func (c *scriptedControlPlane) Resume(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "Resume", nodeID) } -func (c *scriptedControlPlane) ShutdownNode(_ context.Context, _, nodeID string) error { - return c.record("ShutdownNode", nodeID) +func (c *scriptedControlPlane) ShutdownNode(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "ShutdownNode", nodeID) } func (c *scriptedControlPlane) RestartNode( - _ context.Context, _, nodeID string, params RestartParams, + ctx context.Context, _, nodeID string, params RestartParams, ) error { c.restarts = append(c.restarts, params) - return c.record("RestartNode", nodeID) + return c.record(ctx, "RestartNode", nodeID) } -func (c *scriptedControlPlane) Promote(_ context.Context, _, nodeID string) error { - return c.record("Promote", nodeID) +func (c *scriptedControlPlane) Promote(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "Promote", nodeID) } -func (c *scriptedControlPlane) PrepareRemoval(_ context.Context, _, nodeID string) error { - return c.record("PrepareRemoval", nodeID) +func (c *scriptedControlPlane) PrepareRemoval(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "PrepareRemoval", nodeID) } func (c *scriptedControlPlane) RemovalProgress( - _ context.Context, _, nodeID string, + ctx context.Context, _, nodeID string, ) (RemovalProgress, error) { - if err := c.record("RemovalProgress", nodeID); err != nil { + if err := c.record(ctx, "RemovalProgress", nodeID); err != nil { return RemovalProgress{}, err } return c.progress, nil } func (c *scriptedControlPlane) VerifyDrained( - _ context.Context, _, nodeID string, + ctx context.Context, _, nodeID string, ) (DrainVerification, error) { - if err := c.record("VerifyDrained", nodeID); err != nil { + if err := c.record(ctx, "VerifyDrained", nodeID); err != nil { return DrainVerification{}, err } return c.verification, nil } -func (c *scriptedControlPlane) RemoveNode(_ context.Context, _, nodeID string) error { - return c.record("RemoveNode", nodeID) +func (c *scriptedControlPlane) RemoveNode(ctx context.Context, _, nodeID string) error { + return c.record(ctx, "RemoveNode", nodeID) } func (c *scriptedControlPlane) StoragePools( - _ context.Context, clusterID string, + ctx context.Context, clusterID string, ) ([]webapi.StoragePoolInfo, error) { - if err := c.record("StoragePools", clusterID); err != nil { + if err := c.record(ctx, "StoragePools", clusterID); err != nil { return nil, err } return c.pools, nil } func (c *scriptedControlPlane) PoolVolumes( - _ context.Context, _, poolID string, + ctx context.Context, _, poolID string, ) ([]webapi.VolumeInfo, error) { - if err := c.record("PoolVolumes", poolID); err != nil { + if err := c.record(ctx, "PoolVolumes", poolID); err != nil { return nil, err } return c.volumes[poolID], nil } -func (c *scriptedControlPlane) DeleteVolume(_ context.Context, _, _, volumeID string) error { - return c.record("DeleteVolume", volumeID) +func (c *scriptedControlPlane) DeleteVolume(ctx context.Context, _, _, volumeID string) error { + return c.record(ctx, "DeleteVolume", volumeID) } // anOpsNode is the node every operation in these suites targets: provisioned, diff --git a/operator/internal/controllers/node/managed_node_address_test.go b/operator/internal/controllers/node/managed_node_address_test.go new file mode 100644 index 000000000..3fd97d3af --- /dev/null +++ b/operator/internal/controllers/node/managed_node_address_test.go @@ -0,0 +1,123 @@ +// Whether NodeAddress gives the control plane something it can actually +// reach: the per-pod Service DNS name a control plane on this Kubernetes +// cluster resolves itself, or the worker's own real address when the control +// plane runs on a different one entirely and could never resolve that DNS. + +package node + +import ( + "context" + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/testsupport" + "github.com/simplyblock/simplyblock-operator/internal/utils" +) + +func managedControlPlaneSingleton() *simplyblockv1alpha2.ControlPlane { + return &simplyblockv1alpha2.ControlPlane{ + ObjectMeta: metav1.ObjectMeta{Name: SingletonControlPlaneName, Namespace: opsNamespace}, + Spec: simplyblockv1alpha2.ControlPlaneSpec{ + Source: simplyblockv1alpha2.ControlPlaneSource{ + Managed: &simplyblockv1alpha2.ManagedControlPlane{ + Endpoint: "https://hub.example.com:5000", + }, + }, + }, + } +} + +func localControlPlaneSingleton() *simplyblockv1alpha2.ControlPlane { + return &simplyblockv1alpha2.ControlPlane{ + ObjectMeta: metav1.ObjectMeta{Name: SingletonControlPlaneName, Namespace: opsNamespace}, + Spec: simplyblockv1alpha2.ControlPlaneSpec{ + Source: simplyblockv1alpha2.ControlPlaneSource{ + Local: &simplyblockv1alpha2.LocalControlPlane{Image: "docker.io/simplyblock/simplyblock:1.0"}, + }, + }, + } +} + +func workerNode(name, internalIP string) *corev1.Node { + return &corev1.Node{ + ObjectMeta: metav1.ObjectMeta{Name: name}, + Status: corev1.NodeStatus{ + Addresses: []corev1.NodeAddress{ + {Type: corev1.NodeInternalIP, Address: internalIP}, + }, + }, + } +} + +// A managed control plane runs on a Kubernetes cluster that can never resolve +// this cluster's own Service DNS, so it is given the worker's real address -- +// one the storage-node-api pod already answers on directly, since it runs +// with hostNetwork. +func TestNodeAddressUsesTheWorkersRealIPWhenTheControlPlaneIsManaged(t *testing.T) { + scheme := testsupport.NewScheme(t, corev1.AddToScheme) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(managedControlPlaneSingleton(), workerNode(opsWorker, "192.168.10.112")). + Build() + w := &Workload{Client: c} + + got := w.NodeAddress(context.Background(), opsWorker, opsNamespace) + + if want := "192.168.10.112:5000"; got != want { + t.Errorf("NodeAddress = %q, want %q", got, want) + } +} + +// A local control plane resolves the per-pod DNS name itself, and this is the +// existing, well-tested precondition for that: nothing about a same-cluster +// deployment changes. +func TestNodeAddressKeepsTheDNSNameForALocalControlPlane(t *testing.T) { + scheme := testsupport.NewScheme(t, corev1.AddToScheme) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(localControlPlaneSingleton(), workerNode(opsWorker, "192.168.10.112")). + Build() + w := &Workload{Client: c} + + got := w.NodeAddress(context.Background(), opsWorker, opsNamespace) + + if want := utils.StorageNodeSetAPIAddress(opsWorker, opsNamespace); got != want { + t.Errorf("NodeAddress = %q, want the per-pod DNS name %q", got, want) + } +} + +// No ControlPlane singleton at all -- every fixture in this package before +// this file -- falls back to exactly the address it always resolved to. This +// is what keeps the fix additive. +func TestNodeAddressFallsBackToDNSWhenTheControlPlaneCannotBeRead(t *testing.T) { + scheme := testsupport.NewScheme(t, corev1.AddToScheme) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(workerNode(opsWorker, "192.168.10.112")). + Build() + w := &Workload{Client: c} + + got := w.NodeAddress(context.Background(), opsWorker, opsNamespace) + + if want := utils.StorageNodeSetAPIAddress(opsWorker, opsNamespace); got != want { + t.Errorf("NodeAddress = %q, want the per-pod DNS name %q", got, want) + } +} + +// A managed control plane but a worker Node this reader cannot find (a stale +// cache, a name that does not match) falls back the same way, rather than +// handing the control plane an empty or malformed address. +func TestNodeAddressFallsBackToDNSWhenTheWorkerNodeCannotBeRead(t *testing.T) { + scheme := testsupport.NewScheme(t, corev1.AddToScheme) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(managedControlPlaneSingleton()). + Build() + w := &Workload{Client: c} + + got := w.NodeAddress(context.Background(), opsWorker, opsNamespace) + + if want := utils.StorageNodeSetAPIAddress(opsWorker, opsNamespace); got != want { + t.Errorf("NodeAddress = %q, want the per-pod DNS name %q", got, want) + } +} diff --git a/operator/internal/controllers/node/migrate.go b/operator/internal/controllers/node/migrate.go index 64e98c5c5..2ed36fd9e 100644 --- a/operator/internal/controllers/node/migrate.go +++ b/operator/internal/controllers/node/migrate.go @@ -160,7 +160,7 @@ func (r *StorageNodeOpsReconciler) migrateRelocate( force = *ops.Spec.Force } params := RestartParams{ - NodeAddress: r.Workload.NodeAddress(target, node.Namespace), + NodeAddress: r.Workload.NodeAddress(ctx, target, node.Namespace), Force: force, ReattachVolume: boolValue(ops.Spec.ReattachVolume), NewSsdPcie: ops.Spec.MigrateParams().NewSsdPcie, diff --git a/operator/internal/controllers/node/provisioning_test.go b/operator/internal/controllers/node/provisioning_test.go index 30cfe247f..b23c5db58 100644 --- a/operator/internal/controllers/node/provisioning_test.go +++ b/operator/internal/controllers/node/provisioning_test.go @@ -144,7 +144,7 @@ func TestTheAddCarriesWhatTheNodeSaysAboutItself(t *testing.T) { } r, _ := aSteadyNode(t, aControlPlane()) - params := r.addParams(node, cluster) + params := r.addParams(context.Background(), node, cluster) if params.SPDKImage != "example.test/spdk:v1" || params.SPDKProxyImage != "example.test/spdk-proxy:v1" || @@ -181,7 +181,7 @@ func TestTheAddCarriesWhatTheNodeSaysAboutItself(t *testing.T) { func TestAnUnstatedJournalShareIsTheDefaultAndTheCountIsNot(t *testing.T) { r, _ := aSteadyNode(t, aControlPlane()) - params := r.addParams(anUnprovisionedNode(stepPosting), anOpsCluster()) + params := r.addParams(context.Background(), anUnprovisionedNode(stepPosting), anOpsCluster()) if params.JMPercent != 3 { t.Errorf("the journal share is %d%%, want the default 3", params.JMPercent) @@ -206,30 +206,30 @@ func TestAFaultGroupIsSentAsTheIndexTheClusterMapsItTo(t *testing.T) { named := anUnprovisionedNode(stepPosting) named.Spec.Config.FailureDomain = "rack-1" - if index := r.addParams(named, cluster).FailureDomain; index == nil || *index != 3 { + if index := r.addParams(context.Background(), named, cluster).FailureDomain; index == nil || *index != 3 { t.Errorf("failureDomain = %v, want the index the cluster maps rack-1 to", index) } mappedNumber := anUnprovisionedNode(stepPosting) mappedNumber.Spec.Config.FailureDomain = "2" - if index := r.addParams(mappedNumber, cluster).FailureDomain; index == nil || *index != 5 { + if index := r.addParams(context.Background(), mappedNumber, cluster).FailureDomain; index == nil || *index != 5 { t.Errorf("failureDomain = %v, want the mapped index over the label's number", index) } unmappedNumber := anUnprovisionedNode(stepPosting) unmappedNumber.Spec.Config.FailureDomain = "4" - if index := r.addParams(unmappedNumber, cluster).FailureDomain; index == nil || *index != 4 { + if index := r.addParams(context.Background(), unmappedNumber, cluster).FailureDomain; index == nil || *index != 4 { t.Errorf("failureDomain = %v, want the number an unmapped numeric label spells", index) } unmappedName := anUnprovisionedNode(stepPosting) unmappedName.Spec.Config.FailureDomain = "rack-9" - if index := r.addParams(unmappedName, cluster).FailureDomain; index != nil { + if index := r.addParams(context.Background(), unmappedName, cluster).FailureDomain; index != nil { t.Errorf("failureDomain = %v, want none for a name the cluster has no index for", index) } - if index := r.addParams(anUnprovisionedNode(stepPosting), cluster).FailureDomain; index != nil { + if index := r.addParams(context.Background(), anUnprovisionedNode(stepPosting), cluster).FailureDomain; index != nil { t.Errorf("failureDomain = %v, want none for a node that declares no group", index) } } diff --git a/operator/internal/controllers/node/storagenode_controller.go b/operator/internal/controllers/node/storagenode_controller.go index b26641839..37b34a8f3 100644 --- a/operator/internal/controllers/node/storagenode_controller.go +++ b/operator/internal/controllers/node/storagenode_controller.go @@ -240,6 +240,18 @@ func (r *StorageNodeReconciler) Reconcile( if err != nil { return ctrl.Result{}, err } + // Every control-plane call below authenticates as this node's cluster, + // using its own recorded secret, rather than as this operator's own + // Kubernetes identity -- the only way to reach a control plane a + // different Kubernetes cluster runs (a ControlPlane.spec.source.managed + // one), since a Kubernetes TokenReview can never cross a cluster + // boundary. cluster is nil for one the object outlived (§3.4), which + // leaves ctx unauthenticated the same as before this fix: nothing below + // reaches the control plane for a node whose cluster is gone. + if cluster != nil { + secret, err := clusterSecretByName(ctx, r.Client, cluster.Namespace, cluster.Name) + ctx = authenticatedContext(ctx, secret, err) + } if !node.DeletionTimestamp.IsZero() { return r.teardown(ctx, &node) @@ -763,7 +775,7 @@ func (r *StorageNodeReconciler) postNode( node *simplyblockv1alpha2.StorageNode, cluster *simplyblockv1alpha2.StorageCluster, ) error { - params := r.addParams(node, cluster) + params := r.addParams(ctx, node, cluster) taskID, err := r.API.AddNode(ctx, cluster.Status.UUID, params) if err != nil { return fmt.Errorf("add node %s on worker %s: %w", @@ -1556,6 +1568,7 @@ func (r *StorageNodeReconciler) upgradeAdoption( // addParams is what the node-add call carries. The node describes itself, so every // value but the subsystem cap comes from its own spec.config (§3.1). func (r *StorageNodeReconciler) addParams( + ctx context.Context, node *simplyblockv1alpha2.StorageNode, cluster *simplyblockv1alpha2.StorageCluster, ) utils.StorageNodeSetAddParams { @@ -1566,7 +1579,7 @@ func (r *StorageNodeReconciler) addParams( } params := utils.StorageNodeSetAddParams{ - NodeAddress: r.Workload.NodeAddress(node.Spec.WorkerNode, node.Namespace), + NodeAddress: r.Workload.NodeAddress(ctx, node.Spec.WorkerNode, node.Namespace), InterfaceName: workload.MgmtInterface, SPDKImage: config.SpdkImage, SPDKProxyImage: config.SpdkProxyImage, diff --git a/operator/internal/controllers/node/storagenodeops_controller.go b/operator/internal/controllers/node/storagenodeops_controller.go index e57ab8427..c66d06cff 100644 --- a/operator/internal/controllers/node/storagenodeops_controller.go +++ b/operator/internal/controllers/node/storagenodeops_controller.go @@ -318,6 +318,15 @@ func (r *StorageNodeOpsReconciler) Reconcile( func (r *StorageNodeOpsReconciler) advance( ctx context.Context, ops *simplyblockv1alpha2.StorageNodeOps, ) (ctrl.Result, error) { + // Every control-plane call this step and everything downstream of it + // makes authenticates as this operation's target node's cluster when its + // secret is known, rather than as this operator's own Kubernetes + // identity -- the only way to reach a control plane a different + // Kubernetes cluster runs (a ControlPlane.spec.source.managed one), + // since a Kubernetes TokenReview can never cross a cluster boundary. + secret, secretErr := clusterSecretForNode(ctx, r.Client, ops.Namespace, ops.Spec.NodeRef) + ctx = authenticatedContext(ctx, secret, secretErr) + machine, err := graphs().FromSnapshot(ctx, action(ops.Spec.Action), statemachine.FromKube[step](ops.Status.Step)) if err != nil { diff --git a/operator/internal/controllers/node/workload.go b/operator/internal/controllers/node/workload.go index 215c175c4..fe053c0dd 100644 --- a/operator/internal/controllers/node/workload.go +++ b/operator/internal/controllers/node/workload.go @@ -90,16 +90,52 @@ type Workload struct { ManagerNode string } -// NodeAddress is the per-pod DNS name the control plane is given as node_address -// when a node is added or restarted. +// NodeAddress is what the control plane is given as node_address when a node +// is added or restarted. // -// It is the precondition for both: a restart issued against a name that does not -// yet resolve fails name resolution inside the control plane, and the control -// plane's response to that is to reset the node to offline (§5.4). -func (w *Workload) NodeAddress(worker, namespace string) string { +// A local control plane resolves the per-pod DNS name itself, which is the +// existing precondition for both: a restart issued against a name that does +// not yet resolve fails name resolution inside the control plane, and the +// control plane's response to that is to reset the node to offline (§5.4). +// +// A managed control plane runs on a different Kubernetes cluster and can +// never resolve this cluster's own Service DNS, so it is given the worker's +// real, routable address instead -- one the storage-node-api pod already +// answers on directly, since it runs with hostNetwork (BuildStorageNodeDaemonSet). +// Reading the worker Node's own reported address is what keeps this additive: +// a deployment with no managed control plane takes exactly the path it always +// did. +func (w *Workload) NodeAddress(ctx context.Context, worker, namespace string) string { + if address, ok := w.managedNodeAddress(ctx, worker, namespace); ok { + return address + } return utils.StorageNodeSetAPIAddress(worker, namespace) } +// managedNodeAddress answers the worker's real address when the singleton +// ControlPlane names a managed control plane, and false otherwise -- including +// when the singleton or the worker Node cannot be read, since an operator that +// cannot tell falls back to the address that has always worked for a control +// plane this cluster hosts. +func (w *Workload) managedNodeAddress(ctx context.Context, worker, namespace string) (string, bool) { + var cp simplyblockv1alpha2.ControlPlane + key := client.ObjectKey{Namespace: namespace, Name: SingletonControlPlaneName} + if err := w.Get(ctx, key, &cp); err != nil || cp.Spec.Source.Managed == nil { + return "", false + } + + var node corev1.Node + if err := w.Get(ctx, client.ObjectKey{Name: worker}, &node); err != nil { + return "", false + } + for _, addr := range node.Status.Addresses { + if addr.Type == corev1.NodeInternalIP && addr.Address != "" { + return fmt.Sprintf("%s:5000", addr.Address), true + } + } + return "", false +} + // LabelWorker puts one worker into a cluster's storage plane and rewrites the // per-slot labels of every node on it. // diff --git a/operator/internal/controllers/node/workload_controller.go b/operator/internal/controllers/node/workload_controller.go index c8c4487da..f1a1d9231 100644 --- a/operator/internal/controllers/node/workload_controller.go +++ b/operator/internal/controllers/node/workload_controller.go @@ -370,8 +370,15 @@ func (r *StorageNodeWorkloadReconciler) image( "spec.storageNodes.image is unset and ControlPlane %s cannot be read: %w", SingletonControlPlaneName, err) } - if managed := controlPlane.Spec.Source.Local; managed != nil && managed.Image != "" { - return managed.Image, nil + if local := controlPlane.Spec.Source.Local; local != nil && local.Image != "" { + return local.Image, nil + } + // A managed control plane is a different Kubernetes cluster's install and + // carries no image of its own here to fall back to -- spec.source.managed + // says where the control plane is, not what this cluster's storage nodes + // should run. StorageNodeImage is the only source of a default left. + if managed := controlPlane.Spec.Source.Managed; managed != nil && managed.StorageNodeImage != "" { + return managed.StorageNodeImage, nil } return "", fmt.Errorf( "spec.storageNodes.image is unset and ControlPlane %s states no managed image", diff --git a/operator/internal/controllers/node/workload_controller_test.go b/operator/internal/controllers/node/workload_controller_test.go new file mode 100644 index 000000000..44bfc26b0 --- /dev/null +++ b/operator/internal/controllers/node/workload_controller_test.go @@ -0,0 +1,103 @@ +// What a StorageCluster's storage-node DaemonSet defaults its image to when +// spec.storageNodes.image is unset. +// +// A local control plane's own image doubles as the default, since a +// self-hosted deployment's control plane and its storage nodes are one +// release. A managed control plane is a different Kubernetes cluster's +// install and says nothing about this one's storage nodes, so +// ManagedControlPlane.StorageNodeImage is the only source of a default there +// -- without it, the workload can never be built at all. + +package node + +import ( + "context" + "testing" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/testsupport" +) + +func aStorageCluster() *simplyblockv1alpha2.StorageCluster { + return &simplyblockv1alpha2.StorageCluster{ + ObjectMeta: metav1.ObjectMeta{Name: opsCluster, Namespace: opsNamespace}, + } +} + +// An explicit spec.storageNodes.image always wins, whatever the ControlPlane +// says. +func TestImagePrefersTheClustersOwnSpec(t *testing.T) { + scheme := testsupport.NewScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(localControlPlaneSingleton()). + Build() + r := &StorageNodeWorkloadReconciler{Client: c, Namespace: opsNamespace} + + cluster := aStorageCluster() + cluster.Spec.StorageNodes = &simplyblockv1alpha2.StorageNodesSpec{ + Image: "docker.io/simplyblock/simplyblock:from-the-spec", + } + + got, err := r.image(context.Background(), cluster) + if err != nil { + t.Fatalf("image: %v", err) + } + if want := "docker.io/simplyblock/simplyblock:from-the-spec"; got != want { + t.Errorf("image = %q, want %q", got, want) + } +} + +// A local control plane's own image is the default, exactly as it always was. +func TestImageDefaultsToTheLocalControlPlanesImage(t *testing.T) { + scheme := testsupport.NewScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(localControlPlaneSingleton()). + Build() + r := &StorageNodeWorkloadReconciler{Client: c, Namespace: opsNamespace} + + got, err := r.image(context.Background(), aStorageCluster()) + if err != nil { + t.Fatalf("image: %v", err) + } + if want := "docker.io/simplyblock/simplyblock:1.0"; got != want { + t.Errorf("image = %q, want the local control plane's own %q", got, want) + } +} + +// A managed control plane names a storage-node image of its own, since its +// own spec.source.managed carries no image at all -- it is a different +// Kubernetes cluster's install. +func TestImageDefaultsToTheManagedControlPlanesStorageNodeImage(t *testing.T) { + cp := managedControlPlaneSingleton() + cp.Spec.Source.Managed.StorageNodeImage = "docker.io/simplyblock/simplyblock:from-managed" + + scheme := testsupport.NewScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(cp).Build() + r := &StorageNodeWorkloadReconciler{Client: c, Namespace: opsNamespace} + + got, err := r.image(context.Background(), aStorageCluster()) + if err != nil { + t.Fatalf("image: %v", err) + } + if want := "docker.io/simplyblock/simplyblock:from-managed"; got != want { + t.Errorf("image = %q, want the managed control plane's storage-node image %q", got, want) + } +} + +// A managed control plane that names no storage-node image either is a +// deployment nothing can build a DaemonSet for, and that has to fail loudly +// rather than build one with an empty image. +func TestImageErrorsWhenTheManagedControlPlaneNamesNoStorageNodeImage(t *testing.T) { + scheme := testsupport.NewScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme). + WithObjects(managedControlPlaneSingleton()). + Build() + r := &StorageNodeWorkloadReconciler{Client: c, Namespace: opsNamespace} + + if _, err := r.image(context.Background(), aStorageCluster()); err == nil { + t.Error("image returned no error for a managed control plane naming no storage-node image") + } +} diff --git a/operator/internal/controllers/pool/controlplane_endpoint_test.go b/operator/internal/controllers/pool/controlplane_endpoint_test.go new file mode 100644 index 000000000..99a877817 --- /dev/null +++ b/operator/internal/controllers/pool/controlplane_endpoint_test.go @@ -0,0 +1,92 @@ +// Whether StoragePoolReconciler reaches the control plane the ControlPlane +// object resolves to, and authenticates as its cluster once that cluster's +// secret is known -- rather than always the hardcoded in-cluster Service +// webapi.NewClient() defaults to. Before EndpointResolver existed on this +// reconciler, a pool on a ControlPlane.spec.source.managed deployment could +// never reach its control plane at all. See internal/controllers/cluster's +// identically-motivated fix (storagecluster_controller.go's clusterSecret and +// controlplane.go's clientFor). + +package pool + +import ( + "context" + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +// A pool's create reaches whatever endpoint the resolver publishes, not the +// package's built-in default. Without this, the request would try to dial +// the hardcoded in-cluster Service and never reach the stub at all -- the +// stub's own posts count is the proof, since the reconciler has no other way +// to answer testPoolUUID. +func TestAPoolReachesTheEndpointTheControlPlaneResolverPublishes(t *testing.T) { + cp, rec := newControlPlane(t), &recorder{} + c := newClient(t, newCluster(testClusterUUID), newPool("tenant-a")) + r := &StoragePoolReconciler{ + Client: c, + Scheme: testScheme(t), + Recorder: rec, + EndpointResolver: func(context.Context) string { return cp.server.URL }, + } + + p, _ := reconcileSettled(t, r, "tenant-a") + + if p.Status.UUID != testPoolUUID { + t.Fatalf("status.uuid = %q, want %q -- the create never reached the resolved endpoint", + p.Status.UUID, testPoolUUID) + } + if cp.posts != 1 { + t.Errorf("the control plane was asked to create the pool %d times, want 1", cp.posts) + } +} + +// Once the pool's cluster has its own recorded secret, the create +// authenticates as that cluster instead of as this operator's own Kubernetes +// identity -- the only way to reach a control plane a different Kubernetes +// cluster runs, since a TokenReview can never cross that boundary. +func TestAPoolAuthenticatesAsItsClusterOnceTheSecretIsKnown(t *testing.T) { + cp, rec := newControlPlane(t), &recorder{} + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{ + Name: "simplyblock-cluster-" + testCluster, + Namespace: testNamespace, + }, + Data: map[string][]byte{"secret": []byte("cluster-own-secret")}, + } + c := newClient(t, newCluster(testClusterUUID), newPool("tenant-a"), secret) + r := &StoragePoolReconciler{ + Client: c, + Scheme: testScheme(t), + Recorder: rec, + EndpointResolver: func(context.Context) string { return cp.server.URL }, + } + + reconcileSettled(t, r, "tenant-a") + + if cp.lastAuth != "Bearer cluster-own-secret" { + t.Errorf("authorization = %q, want the cluster's own secret", cp.lastAuth) + } +} + +// With no secret recorded yet (a cluster still being created), the call still +// goes out rather than being held -- the operator's own identity is what it +// falls back to, the same as before this fix. +func TestAPoolCallsWithNoClusterSecretDoesNotBlock(t *testing.T) { + cp, rec := newControlPlane(t), &recorder{} + c := newClient(t, newCluster(testClusterUUID), newPool("tenant-a")) + r := &StoragePoolReconciler{ + Client: c, + Scheme: testScheme(t), + Recorder: rec, + EndpointResolver: func(context.Context) string { return cp.server.URL }, + } + + p, _ := reconcileSettled(t, r, "tenant-a") + + if p.Status.UUID != testPoolUUID { + t.Errorf("status.uuid = %q, want %q", p.Status.UUID, testPoolUUID) + } +} diff --git a/operator/internal/controllers/pool/helpers_test.go b/operator/internal/controllers/pool/helpers_test.go index 67bb29dea..314f1cdfd 100644 --- a/operator/internal/controllers/pool/helpers_test.go +++ b/operator/internal/controllers/pool/helpers_test.go @@ -130,6 +130,10 @@ type controlPlane struct { deletes int hosts []string + // lastAuth is the Authorization header of the most recently handled + // request, which is what a test checks a call authenticated with. + lastAuth string + server *httptest.Server } @@ -150,6 +154,7 @@ func (cp *controlPlane) client() func() *webapi.Client { } func (cp *controlPlane) handle(w http.ResponseWriter, r *http.Request) { + cp.lastAuth = r.Header.Get("Authorization") switch { case r.Method == http.MethodPost && hasSuffix(r.URL.Path, "/host"): cp.hosts = append(cp.hosts, "added") diff --git a/operator/internal/controllers/pool/storagepool_controller.go b/operator/internal/controllers/pool/storagepool_controller.go index 3e5f59a0a..74fe8c1f8 100644 --- a/operator/internal/controllers/pool/storagepool_controller.go +++ b/operator/internal/controllers/pool/storagepool_controller.go @@ -51,6 +51,7 @@ import ( "github.com/simplyblock/atlas/ptr" simplyblockv1alpha2 "github.com/simplyblock/simplyblock-operator/api/v1alpha2" + "github.com/simplyblock/simplyblock-operator/internal/controllers/controlplane" "github.com/simplyblock/simplyblock-operator/internal/cpinformer" "github.com/simplyblock/simplyblock-operator/internal/utils" "github.com/simplyblock/simplyblock-operator/internal/webapi" @@ -83,15 +84,33 @@ type StoragePoolReconciler struct { VolumeScopes *cpinformer.ScopeSet // NewAPIClient builds the control-plane client. It is a field so a test can - // point the reconciler at a mock server; nil selects the real one. + // point the reconciler at a mock server; nil selects the real one, resolved + // through EndpointResolver. NewAPIClient func() *webapi.Client + // EndpointResolver answers where the control plane currently is, resolved + // per reconcile the same way internal/controllers/cluster and + // internal/controllers/node do (controlplane.NewEndpointResolver). Nil, or a + // resolver that answers nothing, means the startup client -- this cluster's + // own in-cluster address, or SIMPLYBLOCK_WEBAPI_BASE_URL -- is the only one, + // which is what a standalone deployment and every pre-existing test still + // get. Without this, a pool on a ControlPlane.spec.source.managed deployment + // could never reach its control plane at all: webapi.NewClient() defaults to + // a Service this Kubernetes cluster never runs. + EndpointResolver controlplane.EndpointResolver + // reportedMissingNodes remembers which unresolved spec.allowedNodes entries // have already been announced, so a name left behind by a removed node is // one event rather than one per reconcile forever. The authored list is // deliberately not pruned, so without this the event would repeat for the // life of the pool. reportedMissingNodes sync.Map + + // mu guards startupClient and resolvedClient, which clientFor rebuilds when + // the resolved endpoint changes. + mu sync.Mutex + startupClient *webapi.Client + resolvedClient *webapi.Client } // poolDTO is the control plane's storage-pool response. @@ -212,7 +231,14 @@ func (r *StoragePoolReconciler) Reconcile(ctx context.Context, req ctrl.Request) return ctrl.Result{}, err } - api := r.apiClient() + // Authenticates as this cluster, using its own recorded secret, for the + // same reason internal/controllers/cluster's sync() does: the control + // plane this cluster belongs to may be a ControlPlane.spec.source.managed + // one, on a different Kubernetes cluster than this operator. + if secret, err := r.clusterSecret(ctx, cluster); err == nil && secret != "" { + ctx = webapi.WithBearerToken(ctx, secret) + } + api := r.apiClient(ctx) if !p.DeletionTimestamp.IsZero() { return r.reconcileDeletion(ctx, p, api, clusterUUID) @@ -806,11 +832,52 @@ func (r *StoragePoolReconciler) event( r.Recorder.Eventf(object, nil, eventType, reason, reason, format, args...) } -func (r *StoragePoolReconciler) apiClient() *webapi.Client { +// apiClient is the client one reconcile call uses, resolved through +// EndpointResolver the same way internal/controllers/cluster's +// httpControlPlane.clientFor is. +func (r *StoragePoolReconciler) apiClient(ctx context.Context) *webapi.Client { if r.NewAPIClient != nil { return r.NewAPIClient() } - return webapi.NewClient() + + r.mu.Lock() + defer r.mu.Unlock() + + if r.startupClient == nil { + r.startupClient = webapi.NewClient() + } + if r.EndpointResolver == nil { + return r.startupClient + } + endpoint := r.EndpointResolver(ctx) + if endpoint == "" || endpoint == r.startupClient.BaseURL { + return r.startupClient + } + if r.resolvedClient == nil || r.resolvedClient.BaseURL != endpoint { + r.resolvedClient = webapi.NewClient(endpoint) + } + return r.resolvedClient +} + +// clusterSecret reads the credential StorageClusterReconciler.persist wrote +// for the pool's cluster, so a call scoped to it authenticates as that +// cluster instead of as this operator's own Kubernetes identity -- the only +// way to reach a control plane a different Kubernetes cluster runs, since a +// TokenReview can never cross that boundary. Mirrors +// internal/controllers/cluster's identically named method and +// internal/controllers/node's clusterSecretByName. +func (r *StoragePoolReconciler) clusterSecret( + ctx context.Context, cluster *simplyblockv1alpha2.StorageCluster, +) (string, error) { + var secret corev1.Secret + key := client.ObjectKey{ + Name: fmt.Sprintf("simplyblock-cluster-%s", cluster.Name), + Namespace: cluster.Namespace, + } + if err := r.Get(ctx, key, &secret); err != nil { + return "", err + } + return string(secret.Data["secret"]), nil } // SetupWithManager registers the reconciler and the one watch that is not on the diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_controlplanes.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_controlplanes.yaml index 94b0cad35..c426685ee 100644 --- a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_controlplanes.yaml +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_controlplanes.yaml @@ -163,6 +163,32 @@ spec: local: description: Local is a control plane the operator installs. properties: + adminTokenSecretRef: + description: |- + AdminTokenSecretRef names a Secret in this namespace holding a static + admin bearer token this control plane accepts, under the `token` key, in + addition to this deployment's own Kubernetes identity + (SB_K8S_ADMIN_SERVICE_ACCOUNTS). It is what lets a cluster this control + plane manages remotely (spec.source.managed there, + ManagedControlPlane.CredentialsSecretRef naming the same value) + authenticate a CreateCluster call, since a Kubernetes TokenReview can + never cross a cluster boundary. + + The Secret is projected into the management API container's environment + with secretKeyRef, so this operator never itself reads the plaintext. + Absent grants no credential beyond the operator's own service account. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic foundationDB: description: FoundationDB sizes the FoundationDB the management API stores its state in. properties: @@ -513,6 +539,20 @@ spec: so a loopback or link-local address is rejected. pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ type: string + storageNodeImage: + description: |- + StorageNodeImage is the storage-node image a StorageCluster on this + Kubernetes cluster defaults to when its own spec.storageNodes.image is + unset. A local control plane's own spec.source.local.image doubles as + this default (StorageNodeWorkloadReconciler.image), because a + self-hosted deployment's control plane and its storage nodes are one + release. A managed one is a different Kubernetes cluster's install and + says nothing about what this cluster's storage nodes should run, so + there is no equivalent to fall back to without this field -- every + StorageCluster on a managed deployment must get an image from here or + from its own spec. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string required: - endpoint type: object diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_simplyblockdrivers.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_simplyblockdrivers.yaml index 94d87edeb..e4a7d6c6b 100644 --- a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_simplyblockdrivers.yaml +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_simplyblockdrivers.yaml @@ -299,13 +299,29 @@ spec: type: object sidecarImages: description: |- - SidecarImages overrides the six CSI sidecars, one field each. Unset takes - the version this operator release ships. + SidecarImages overrides the seven CSI sidecars, one field each. Unset + takes the version this operator release ships. properties: attacher: description: Attacher is csi-attacher, on the controller plugin. pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ type: string + csiAddons: + description: |- + CSIAddons is the kubernetes-csi-addons sidecar, on the controller + plugin. It connects to the plugin's socket, probes the csi-addons + Identity service for capabilities, and publishes a CSIAddonsNode so the + kubernetes-csi-addons controller-manager (design + design-csi-addons-replication.md §4.1) can reach the Replication + service this driver serves. + + Unlike the other sidecars above, this one's upstream home is the + csi-addons project's own registry, not simplyblock's: the allowlist + carries quay.io/csiaddons alongside the simplyblock registries so a + deployment can run the stock kubernetes-csi-addons sidecar image + directly, ahead of (or instead of) a quay.io/simplyblock-io mirror. + pattern: ^($|(quay\.io/simplyblock-io|docker\.io/simplyblock|public\.ecr\.aws/simply-block|quay\.io/csiaddons)/[a-z0-9][a-z0-9._-]*:[a-zA-Z0-9][a-zA-Z0-9._-]*(@sha256:[a-f0-9]{64})?)$ + type: string healthMonitor: description: |- HealthMonitor is csi-external-health-monitor-controller, on the diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml new file mode 100644 index 000000000..393fd46a5 --- /dev/null +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_storagesitedeployments.yaml @@ -0,0 +1,1053 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: storagesitedeployments.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: StorageSiteDeployment + listKind: StorageSiteDeploymentList + plural: storagesitedeployments + shortNames: + - sbsd + singular: storagesitedeployment + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.cluster + name: Cluster + type: string + - jsonPath: .spec.approved + name: Approved + type: boolean + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.draft.phase + name: Draft + type: string + - jsonPath: .status.storageCluster.phase + name: Storage + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + StorageSiteDeployment requests a managed site's storage cluster from the hub: + a discovery on the site, the sizing of the draft it writes, and the approval + that expands the draft into a StorageCluster. The hub carries the request + through OCM and projects the site's draft and cluster into the status. + Deleting the request leaves the storage cluster alone. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: StorageSiteDeploymentSpec is the request for one site's storage + cluster. + properties: + approved: + default: false + description: |- + Approved is the review gate, delivered to the draft on the site. One-way, + as the draft's own gate is. + type: boolean + cluster: + description: |- + Cluster is the OCM ManagedCluster the storage is deployed on. The request's + ManifestWork and views live in its namespace on the hub. Immutable. + maxLength: 63 + minLength: 1 + type: string + x-kubernetes-validations: + - message: cluster is immutable + rule: self == oldSelf + discover: + description: |- + Discover is the discovery the site runs first. Changing it runs another + discovery, which rewrites the draft. + properties: + enableControlPlaneNodes: + description: |- + EnableControlPlaneNodes lets the discovery consider the nodes that run the + API server. Every server of a small distribution is one, so a three-node + site has no storage without it. + type: boolean + nodeSelector: + additionalProperties: + type: string + description: NodeSelector limits the discovery to the nodes carrying + these labels. + type: object + workers: + description: Workers limits the discovery to these nodes. Empty + is every worker. + items: + type: string + type: array + x-kubernetes-list-type: set + type: object + draftName: + default: site-draft + description: |- + DraftName is the ClusterDeploymentConfig the discovery writes on the site + and the request sizes and approves. Immutable. + maxLength: 63 + type: string + x-kubernetes-validations: + - message: draftName is immutable + rule: self == oldSelf + siteNamespace: + default: simplyblock + description: |- + SiteNamespace is the simplyblock operator's namespace on the site, where + the discovery and the draft live. + maxLength: 63 + type: string + sizing: + description: |- + Sizing is written onto the draft's cluster template once the draft exists, + so the reviewer sees the sized draft before approving it. + properties: + enableDriveFormat: + description: EnableDriveFormat lets the deployment format the + devices it takes. + type: boolean + enableJournalDevice: + description: EnableJournalDevice dedicates one device per node + to the journal. + type: boolean + maxSubsystemCount: + description: MaxSubsystemCount is the number of NVMe-oF subsystems + each node serves. + format: int32 + minimum: 1 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the hugepage memory each storage node takes, as a + quantity ("8G"). + type: string + name: + description: Name is the StorageCluster's name on the site. + maxLength: 63 + type: string + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per stripe + (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, 1], + [1, 2], [2, 2], [4, 2]]' + vcpuCount: + description: VCPUCount is the number of vCPUs each storage node + takes. + format: int32 + minimum: 1 + type: integer + type: object + required: + - cluster + type: object + x-kubernetes-validations: + - message: 'approval is one-way: an approved deployment cannot be un-approved' + rule: '!has(oldSelf.approved) || !oldSelf.approved || self.approved' + status: + description: StorageSiteDeploymentStatus is what the site reports back, + projected. + properties: + conditions: + description: |- + Conditions: Delivered (the work is applied on the site), Discovered (the + draft names nodes), Approved (the site's draft is approved), Ready (the + StorageCluster is Online). + items: + description: Condition contains details for one aspect of the current + state of this API Resource. + properties: + lastTransitionTime: + description: |- + lastTransitionTime is the last time the condition transitioned from one status to another. + This should be when the underlying condition changed. If that is not known, then using the time when the API field changed is acceptable. + format: date-time + type: string + message: + description: |- + message is a human readable message indicating details about the transition. + This may be an empty string. + maxLength: 32768 + type: string + observedGeneration: + description: |- + observedGeneration represents the .metadata.generation that the condition was set based upon. + For instance, if .metadata.generation is currently 12, but the .status.conditions[x].observedGeneration is 9, the condition is out of date + with respect to the current state of the instance. + format: int64 + minimum: 0 + type: integer + reason: + description: |- + reason contains a programmatic identifier indicating the reason for the condition's last transition. + Producers of specific condition types may define expected values and meanings for this field, + and whether the values are considered a guaranteed API. + The value should be a CamelCase string. + This field may not be empty. + maxLength: 1024 + minLength: 1 + pattern: ^[A-Za-z]([A-Za-z0-9_,:]*[A-Za-z0-9_])?$ + type: string + status: + description: status of the condition, one of True, False, Unknown. + enum: + - "True" + - "False" + - Unknown + type: string + type: + description: type of condition in CamelCase or in foo.example.com/CamelCase. + maxLength: 316 + pattern: ^([a-z0-9]([-a-z0-9]*[a-z0-9])?(\.[a-z0-9]([-a-z0-9]*[a-z0-9])?)*/)?(([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9])$ + type: string + required: + - lastTransitionTime + - message + - reason + - status + - type + type: object + type: array + x-kubernetes-list-map-keys: + - type + x-kubernetes-list-type: map + draft: + description: Draft is the draft as the site reports it. + properties: + approved: + description: Approved is whether the draft is approved on the + site. + type: boolean + cluster: + description: Cluster is the draft's cluster template, with the + sizing applied. + properties: + backup: + description: |- + Backup is where this cluster's backups live, and it expands into + StorageCluster.spec.backup unchanged. + + It is here for the reason KMS is: a store stated on the document is + present when the cluster is created rather than patched in afterward by + whoever remembers. Unlike most of what this template carries, the field it + fills is mutable, so a document that states none costs nothing permanent. + A cluster can be given a store whenever there is one to give. + + The Secret it names is not resolved at admission. It is a core object a + deployment legitimately creates alongside the document or after it, and + the cluster's own creation is where its absence is reported. + properties: + bucket: + description: Bucket is the bucket backups are written + to and read from. + type: string + credentialsSecretRef: + description: |- + CredentialsSecretRef names the Secret holding the access key and the + secret key. It is a reference rather than the values, because a spec is + readable by anybody who can read the object. + properties: + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + type: object + x-kubernetes-map-type: atomic + endpoint: + description: Endpoint is the S3 endpoint, for example, + https://s3.example.com. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + region: + description: Region is the bucket's region, for endpoints + that do not imply one. + type: string + required: + - bucket + - credentialsSecretRef + - endpoint + type: object + containerResources: + description: |- + ContainerResources sizes the storage-node container, and expands into the + cluster's own spec.storageNodes.containerResources. + + The container it sizes is the node's management API rather than SPDK, + which runs in a pod of its own: what outgrows the default is a node + answering for many subsystems, not a node moving more data. It is on the + document because a deployment is where a fleet's sizing is decided, and + a cluster written from a document that could not say so had to be edited + afterward on a field the document owns everywhere else. + + Stating either half replaces both. The defaults apply to a cluster that + states neither requests nor limits, so a document stating requests alone + produces a container with no limits rather than one with the default + limits, and a memory limit is what has the kubelet evict a leaking agent + rather than losing the worker. + + It is a pointer because a resource block is a struct, and a struct with + omitempty is serialized whether or not anything is in it: as a value, + every document a discovery run writes would carry an empty + containerResources that says nothing and that a reviewer has to decide + about. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + enableAtomicity4K: + description: |- + EnableAtomicity4K enforces 4K write atomicity on every device this + deployment names, which is what lets checksum validation run on devices + whose logical block size is under the data plane's 4K minimum. + + It is the route to checked I/O on a device that cannot be reformatted: a + logical block device's block size is fixed by the drive, and some NVMe + devices offer no 4K format either. Where a device can be reformatted, + EnableDriveFormat is the other route and this is unnecessary. + + It is an enforcement because the question is often unanswerable. A SATA + drive presenting 512-byte logical blocks over a 4K physical sector reports + 512 and nothing more, and a kernel older than 6.11 publishes no atomic + write attributes at all. Where a device does answer, the storage node's + report carries it, and a reviewer approves this against that rather than + against a vendor's datasheet -- because enforcing a guarantee the hardware + does not keep is how a torn write becomes a checksum that silently + disagrees with it. + + It means nothing unless EnableChecksumValidation is set, which is the + cluster's own rule and is left to the cluster to enforce. + type: boolean + enableChecksumValidation: + description: |- + EnableChecksumValidation turns on inline CRC validation of every I/O, for + silent-data-error protection. + + It is on the document because it is immutable on the cluster it lands on: + the backend bakes the checksum method into each device when the cluster is + created and never re-applies it, so a cluster created without this is one + nobody can turn it on for. A deployment that wants its data checked has to + say so here or not at all. + type: boolean + enableDriveFormat: + description: |- + EnableDriveFormat formats every device the document names before a storage + node takes it, which is how a drive carrying anything already is made + usable. + + It says what is wanted rather than how, because the how differs by device + class: an NVMe device is formatted to a 4K block size, and a logical block + device has its signatures wiped. One field covers both, so a document does + not have to know which class the expansion will resolve it to. + + It is on the document rather than defaulted further down because it is + destructive and the document is what somebody approves. A reviewer reading + a draft has to see that the drives it lists will be formatted, and be able + to strike it before approving; the cluster's own field is immutable once + the cluster exists, so a default nobody saw could not be undone either. + type: boolean + enableFailureDomains: + description: |- + EnableFailureDomains opts the cluster into failure-domain mode, in which + every group must label the fault group its workers belong to. + type: boolean + enableJournalDevice: + description: |- + EnableJournalDevice dedicates the smallest NVMe device on each of this + deployment's workers to the journal manager, instead of carving a journal + partition out of every device. + + It is here rather than on a node set because it is immutable on the cluster + it lands on, for the reason SocketsToUse is: the on-disk layout a fleet was + built with is not one a later document can vary. It also costs a drive of + capacity per node, which is a trade a reviewer approves rather than one a + default makes for them. + type: boolean + enableNodeAffinity: + description: |- + EnableNodeAffinity has the data plane serve an erasure-coded volume's I/O + from the local node's own devices where it can, before crossing the + network. + + It is not Kubernetes affinity, and the name is the one place this API + invites that reading: nothing about it schedules a pod, labels a worker, + or places a volume's primary node. The control plane carries it into the + cluster map it pushes to each node, where it sets the local node's index, + and what changes is which copy of a chunk is read. + Co-locating a workload with the primary node of its volume is a separate + mechanism and is not configured here. + + It is on the document because it is immutable on the cluster: the control + plane takes it at cluster create and never re-applies it, so this is the + only moment it can be set at all. + type: boolean + fabricType: + description: FabricType is the storage fabric. + maxLength: 32 + type: string + initContainerResources: + description: |- + InitContainerResources sizes both of the storage node's init containers, + and expands into the cluster's own spec.storageNodes.initContainerResources. + + They are sized apart from the container because they do a different job + and are gone before it starts: one writes the node's env file and the + other runs node_configure.py once, so what they need is a short burst + rather than the footprint of a process that runs for the node's life. + + Stating either half replaces both, as with containerResources, and it is + a pointer for the same reason. + properties: + claims: + description: |- + Claims lists the names of resources, defined in spec.resourceClaims, + that are used by this container. + + This field depends on the + DynamicResourceAllocation feature gate. + + This field is immutable. It can only be set for containers. + items: + description: ResourceClaim references one entry in PodSpec.ResourceClaims. + properties: + name: + description: |- + Name must match the name of one entry in pod.spec.resourceClaims of + the Pod where this field is used. It makes that resource available + inside a container. + type: string + request: + description: |- + Request is the name chosen for a request in the referenced claim. + If empty, everything from the claim is made available, otherwise + only the result of this request. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + limits: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Limits describes the maximum amount of compute resources allowed. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + requests: + additionalProperties: + anyOf: + - type: integer + - type: string + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + description: |- + Requests describes the minimum amount of compute resources required. + If Requests is omitted for a container, it defaults to Limits if that is explicitly specified, + otherwise to an implementation-defined value. Requests cannot exceed Limits. + More info: https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/ + type: object + type: object + kms: + description: |- + KMS selects where the cluster stores volume encryption keys. Stating it on + the document is what makes it present when the cluster is created, where + setting it on the StorageCluster afterward races with that creation. + properties: + vault: + description: Vault stores keys in HashiCorp Vault. + properties: + endpoint: + description: |- + Endpoint is the Vault endpoint, for example, https://vault.example.com:8200. + Rejected unless it resolves to an external address. + pattern: ^https?://[a-zA-Z0-9.-]+(:[0-9]{1,5})?(/.*)?$ + type: string + required: + - endpoint + type: object + type: object + maxSubsystemCount: + description: |- + MaxSubsystemCount is the maximum number of NVMe-oF subsystems each storage + node of this cluster serves. Required, because the StorageCluster's own + field is, and no StorageNode carries a copy of it. + format: int32 + maximum: 75 + minimum: 10 + type: integer + minHugePagesSize: + description: |- + MinHugePagesSize is the smallest huge-page allocation each storage node of + this cluster makes: 100G or 1T, where a bare number is gigabytes. Like + VCPUCount it is the cluster's and is copied onto every node the expansion + writes. Omitted, each node uses the computed minimum. + maxLength: 32 + type: string + name: + description: |- + Name is the StorageCluster's name, and is therefore held to what such a + name may be rather than to what an object name may be. A longer value is a + document the API server accepts and a CreatingCluster step that can never + succeed, since the cluster it would write is one the API server refuses. + maxLength: 63 + type: string + nodeProvisioningBudget: + description: |- + NodeProvisioningBudget is how many workers the expansion may have in the + node-add process at once. It expands into the cluster's own + spec.storageNodes.nodeProvisioningBudget, whose meaning it shares: the cap + is counted by distinct worker, so a two-socket host spends one of the + budget, and a worker hosting a FoundationDB pod is sequential whatever the + budget says. + + It is on the document because a document is what states the size of a + deployment, and a deployment of thirty workers added one at a time is the + difference between an afternoon and a week. Omitted, the cluster's default + of one applies, which is the serial behavior. + format: int32 + minimum: 1 + type: integer + nodesPerSocket: + description: |- + NodesPerSocket is how many storage nodes run per NUMA socket. See + SocketsToUse, which it multiplies. + format: int32 + maximum: 8 + minimum: 1 + type: integer + openshift: + description: |- + OpenShift is what this deployment states because it runs on OpenShift. It + expands into StorageCluster.spec.storageNodes.openshift, whose shape it + shares, and it is read only for a document whose environment is + OpenShift: the environment is what says which distribution this is, and + the block is what that distribution needs said beyond it. + properties: + machineConfigPool: + default: worker + description: |- + MachineConfigPool names a machine-config role the storage nodes' own pool + inherits from, beyond the worker role it always inherits. + + It is not the pool the nodes end up in, which the description it carried + before said and which cost a reader the reboot they were trying to avoid. + Adding a node creates a pool of its own, storage-, and moves the + node into it; a node belongs to exactly one custom pool, so whatever + machine configuration its previous pool carried is lost unless that + pool's role is named here for the new one to select as well. The default + is the role every pool already selects, which is what makes it a no-op + for a fleet whose workers are ordinary workers. + maxLength: 253 + pattern: ^[a-z0-9]([-a-z0-9]*[a-z0-9])?$ + type: string + type: object + ports: + description: |- + Ports are where this cluster's storage nodes listen. Unstated, and for + each member left unstated, the cluster's own defaults decide. + properties: + nodeAgent: + default: 50001 + description: |- + NodeAgent is the port each node's agent API listens on. It expands into + StorageCluster.spec.snodeApiPort, and it is named for the component + rather than for that field: the agent is what spec.images.nodeAgent pins + and what the storage-node DaemonSet runs. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + nvmf: + default: 4420 + description: |- + NVMf is the base of the NVMe-oF port range every node binds. It expands + into StorageCluster.spec.nvmfBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + rpc: + default: 8080 + description: |- + Rpc is the base of the RPC port range every node binds. It expands into + StorageCluster.spec.rpcBasePort. + format: int32 + maximum: 65535 + minimum: 1024 + type: integer + type: object + socketsToUse: + description: |- + SocketsToUse restricts the deployment to selected NUMA sockets, and empty + means socket 0 alone. With NodesPerSocket it decides how many storage nodes + each worker runs, so a group of two workers on a two-socket layout expands + to four nodes. + + It is here rather than on a node set because it is immutable on the cluster + it lands on: the layout a fleet was built with is not one a later document + can vary, and a reviewer should see it before the cluster exists. + items: + maxLength: 16 + type: string + maxItems: 16 + type: array + x-kubernetes-list-type: set + stripe: + description: Stripe is the erasure-coding layout. + properties: + dataChunks: + description: DataChunks is the number of data chunks per + stripe (ndcs). + format: int32 + minimum: 1 + type: integer + parityChunks: + description: |- + ParityChunks is the number of parity chunks per stripe (npcs), and + therefore how many chunk losses a stripe survives. + format: int32 + minimum: 0 + type: integer + type: object + x-kubernetes-validations: + - message: the erasure-coding scheme must be one of 1+0, 1+1, + 2+1, 4+1, 1+2, 2+2, or 4+2, written as dataChunks+parityChunks, + and an unstated half is 1 + rule: '[has(self.dataChunks) ? self.dataChunks : 1, has(self.parityChunks) + ? self.parityChunks : 1] in [[1, 0], [1, 1], [2, 1], [4, + 1], [1, 2], [2, 2], [4, 2]]' + tolerations: + description: |- + Tolerations are what the storage-node pods tolerate, and they expand into + the cluster's own spec.storageNodes.tolerations. + + A fleet that dedicates machines to storage taints them, which is what + keeps everything else off. The DaemonSet that lands on those machines has + to tolerate the taint or it schedules nowhere, and a document that could + not say so described a deployment that does not start: the correction was + an edit to the cluster the document had just created, on a field the + document owns everywhere else. + + A growth document states none. It names a cluster rather than describing + one, and that cluster already carries what its storage nodes tolerate. + items: + description: |- + The pod this Toleration is attached to tolerates any taint that matches + the triple using the matching operator . + properties: + effect: + description: |- + Effect indicates the taint effect to match. Empty means match all taint effects. + When specified, allowed values are NoSchedule, PreferNoSchedule and NoExecute. + type: string + key: + description: |- + Key is the taint key that the toleration applies to. Empty means match all taint keys. + If the key is empty, operator must be Exists; this combination means to match all values and all keys. + type: string + operator: + description: |- + Operator represents a key's relationship to the value. + Valid operators are Exists, Equal, Lt, and Gt. Defaults to Equal. + Exists is equivalent to wildcard for value, so that a pod can + tolerate all taints of a particular category. + Lt and Gt perform numeric comparisons (requires feature gate TaintTolerationComparisonOperators). + type: string + tolerationSeconds: + description: |- + TolerationSeconds represents the period of time the toleration (which must be + of effect NoExecute, otherwise this field is ignored) tolerates the taint. By default, + it is not set, which means tolerate the taint forever (do not evict). Zero and + negative values will be treated as 0 (evict immediately) by the system. + format: int64 + type: integer + value: + description: |- + Value is the taint value the toleration matches to. + If the operator is Exists, the value should be empty, otherwise just a regular string. + type: string + type: object + maxItems: 32 + type: array + vcpuCount: + description: |- + VCPUCount is the number of vCPUs allocated to SPDK on each storage node of + this cluster. It is stated here and nowhere below, because the control + plane assumes it uniform across a cluster's nodes; CreatingNodes copies it + into every StorageNode.spec.config.sizing it writes. Required, because the + StorageCluster's own field is. + The floor is 4 rather than a hardware limit: a node must carry one core + beyond this budget for the system, and the control plane's core layout + assigns no NVMe-oF poller core at all for a 2-vCPU budget. + format: int32 + minimum: 4 + type: integer + required: + - maxSubsystemCount + - name + - vcpuCount + type: object + message: + description: |- + Message is what the site says about the draft: validation findings while + it is a draft, the expansion's step afterwards. + type: string + name: + description: Name is the ClusterDeploymentConfig on the site. + type: string + nodeRefs: + description: NodeRefs are the StorageNode objects the expansion + created. + items: + type: string + type: array + x-kubernetes-list-type: set + nodeSets: + description: NodeSets are the nodes and devices the discovery + found, for review. + items: + description: |- + NodeSet is the organizational grouping of a deployment, usually a rack: the + workers a document adds or grows together. It carries no sizing, because sizing + is uniform across a cluster and is stated once in ClusterTemplate. + properties: + groups: + description: Groups are the sets of workers sharing one + configuration. + items: + description: |- + NodeGroup is a set of workers that share one configuration, which is what + makes ten identical machines one entry rather than ten. + properties: + dataInterfaces: + description: DataInterfaces are the data-plane network + interfaces. + items: + maxLength: 63 + type: string + maxItems: 32 + type: array + devices: + description: Devices selects the storage devices every + worker in the group uses. + properties: + block: + description: |- + Block names logical block devices by path ("/dev/sdb"). It expands into the + same config.deviceNames as NVMe, which takes a PCI address and a device + path in one list. It is the alternative to NVMe rather than a companion of + it: the two classes are not mixed within a cluster. + items: + maxLength: 255 + pattern: ^/dev/[a-zA-Z0-9._/-]+$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + nvme: + description: NVMe names NVMe devices by PCI address + ("0000:5e:00.0"). + items: + maxLength: 32 + pattern: ^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$ + type: string + maxItems: 128 + type: array + x-kubernetes-list-type: set + type: object + x-kubernetes-validations: + - message: a device selection names NVMe addresses + or block devices, not both + rule: has(self.nvme) != has(self.block) + failureDomain: + description: |- + FailureDomain is the label of the fault group every worker in this group + belongs to ("rack-b"), which is usually the name of the rack, zone, or + power feed they share. Discovery seeds it from topology.kubernetes.io/zone + and leaves it unset where the Kubernetes API carries no topology, which + holds provisioning with a clear reason rather than guessing. It expands + into StorageNode.spec.config.failureDomain, whose shape it shares. + maxLength: 63 + pattern: ^[a-zA-Z0-9]([-_.a-zA-Z0-9]*[a-zA-Z0-9])?$ + type: string + journalManager: + description: JournalManager tunes the journal managers + on these nodes. + properties: + count: + description: |- + Count is the number of journal managers to configure. The control plane + requires at least 3. + format: int32 + minimum: 3 + type: integer + percentPerDevice: + description: PercentPerDevice is the share of + each device given to the journal. + format: int32 + maximum: 100 + minimum: 1 + type: integer + type: object + mgmtInterface: + description: MgmtInterface is the management network + interface the storage nodes bind. + maxLength: 63 + type: string + name: + description: |- + Name identifies the group within its node set, for a reader and for the + events a validation failure emits. + maxLength: 253 + type: string + reservedSystemCPU: + description: |- + ReservedSystemCPU is the CPU set held back from SPDK for the system on + these nodes, as a core list such as 0,1 or 0-3. + + It is a group's rather than the cluster's because it names core ids, and a + group is what a document calls the workers that share their hardware: 0,1 + on a sixteen-core worker and 0,1 on a ninety-six-core worker are different + fractions of the machine. It expands into + StorageNode.spec.config.reservedSystemCPU, whose shape it shares, and a + group that states none leaves the cluster's fleet-wide value to decide. + + On OpenShift it reaches the kubelet through a KubeletConfig for the + machine config pool, which is the cluster's, so groups that disagree there + are writing over one another's pool configuration. + maxLength: 63 + pattern: ^[0-9]+(-[0-9]+)?(,[0-9]+(-[0-9]+)?)*$ + type: string + spdkSystemMemory: + description: |- + SpdkSystemMemory is the memory the control plane starts SPDK with on these + nodes. + maxLength: 32 + pattern: ^[0-9]+(G|GI|GB|GiB|M|MI|MB|MiB|g|gi|gb|gib|m|mi|mb|mib)?$ + type: string + workers: + description: Workers are the Kubernetes worker hostnames + in this group. + items: + maxLength: 253 + type: string + maxItems: 200 + minItems: 1 + type: array + x-kubernetes-list-type: set + required: + - name + - workers + type: object + maxItems: 64 + minItems: 1 + type: array + name: + description: |- + Name is the node set's name. It is copied to StorageNode.spec.nodeSet, so + that a node can be traced back to the part of the document that produced + it. + maxLength: 253 + type: string + required: + - groups + - name + type: object + type: array + phase: + description: |- + Phase is the draft's own phase on the site (Draft, Expanding, Expanded, + Failed). + type: string + required: + - name + type: object + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as + the request moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from. + format: int64 + type: integer + phase: + description: Phase is the request's own progress. + enum: + - Pending + - Discovering + - Drafted + - Deploying + - Online + - Failed + type: string + storageCluster: + description: StorageCluster is the cluster the approved draft produced. + properties: + name: + description: Name is the StorageCluster object on the site. + type: string + nodes: + description: Nodes are the cluster's storage nodes. + items: + description: |- + StorageSiteNode is one storage node of the deployed cluster, as the site + reports it. + properties: + hostname: + description: Hostname is the Kubernetes node it runs on. + type: string + name: + description: Name is the StorageNode object on the site. + type: string + phase: + description: Phase is the node's phase on the site. + type: string + required: + - name + type: object + type: array + x-kubernetes-list-map-keys: + - name + x-kubernetes-list-type: map + phase: + description: Phase is the StorageCluster's phase on the site. + type: string + pool: + description: |- + Pool is the pool the cluster was created with, which a StorageClass names + in pool_name. + type: string + uuid: + description: |- + UUID is the storage cluster's id in the control plane, which a + StorageClass names in cluster_id. + type: string + required: + - name + type: object + workName: + description: WorkName is the ManifestWork carrying the request to + the site. + type: string + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml new file mode 100644 index 000000000..d9117bb4c --- /dev/null +++ b/operator/internal/upgrade/crds/manifests/storage.simplyblock.io_testfailovers.yaml @@ -0,0 +1,336 @@ +--- +apiVersion: apiextensions.k8s.io/v1 +kind: CustomResourceDefinition +metadata: + annotations: + controller-gen.kubebuilder.io/version: v0.21.0 + name: testfailovers.storage.simplyblock.io +spec: + group: storage.simplyblock.io + names: + kind: TestFailover + listKind: TestFailoverList + plural: testfailovers + shortNames: + - tfo + singular: testfailover + scope: Namespaced + versions: + - additionalPrinterColumns: + - jsonPath: .spec.scope + name: Scope + type: string + - jsonPath: .spec.sourceRef + name: Source + type: string + - jsonPath: .spec.sourceCluster + name: "On" + type: string + - jsonPath: .spec.bubbleCluster + name: Bubble + type: string + - jsonPath: .status.phase + name: Phase + type: string + - jsonPath: .status.step.state + name: Step + type: string + - jsonPath: .status.message + name: Message + priority: 1 + type: string + - jsonPath: .metadata.creationTimestamp + name: Age + type: date + name: v1alpha2 + schema: + openAPIV3Schema: + description: |- + TestFailover is a one-way, non-disruptive test-failover drill. It recovers a + source volume, or a consistency group, from a snapshot into an isolated + namespace on a chosen cluster as bound PVCs, without touching the source. The + hub reads the source on its cluster and places the bubble on the recovery + cluster through OCM. It runs to a terminal phase, or holds Ready until it is + deleted, and deletion reclaims the clones and any snapshots the drill took. + properties: + apiVersion: + description: |- + APIVersion defines the versioned schema of this representation of an object. + Servers should convert recognized schemas to the latest internal value, and + may reject unrecognized values. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#resources + type: string + kind: + description: |- + Kind is a string value representing the REST resource this object represents. + Servers may infer this from the endpoint the client submits requests to. + Cannot be updated. + In CamelCase. + More info: https://git.k8s.io/community/contributors/devel/sig-architecture/api-conventions.md#types-kinds + type: string + metadata: + type: object + spec: + description: |- + TestFailoverSpec is the request for one non-disruptive test-failover drill. + + The source is named by where it runs and what it is, so the hub can find it + without anyone extracting a backend handle by hand. SourceNamespace is + required for a Volume drill, where the source is a PVC, and unused for a Group + drill, where SourceRef names a consistency group. + properties: + bubbleCluster: + description: |- + BubbleCluster is the OCM ManagedCluster to recover onto: a DR target holding + the replicated point, or another cluster. It must differ from SourceCluster; + test-failover recovers onto a different cluster, never in place. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + bubbleNamespace: + default: bubble + description: |- + BubbleNamespace is the namespace on the bubble cluster where the recovered + PVCs are created. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + scope: + description: Scope selects what the drill recovers. Immutable. + enum: + - Volume + - Group + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceCluster: + description: |- + SourceCluster is the OCM ManagedCluster the source runs on. The hub reads + the source there through a ManagedClusterView. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceNamespace: + description: |- + SourceNamespace is the namespace of the source PVC on SourceCluster. + Required for scope=Volume. Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + sourceRef: + description: |- + SourceRef names the source on SourceCluster: a PersistentVolumeClaim in + SourceNamespace (scope=Volume), or a consistency group (scope=Group). + Immutable. + type: string + x-kubernetes-validations: + - message: field is immutable + rule: self == oldSelf + ttlSeconds: + description: |- + TTLSeconds is an optional maximum lifetime: the drill is torn down after it + even without a delete, so a forgotten drill cannot hold a clone forever. + format: int64 + minimum: 0 + type: integer + required: + - bubbleCluster + - scope + - sourceCluster + - sourceRef + type: object + x-kubernetes-validations: + - message: field sourceNamespace is immutable once set + rule: '!has(oldSelf.sourceNamespace) || has(self.sourceNamespace)' + - message: field bubbleNamespace is immutable once set + rule: '!has(oldSelf.bubbleNamespace) || has(self.bubbleNamespace)' + - message: sourceNamespace is required for scope=Volume + rule: self.scope != 'Volume' || has(self.sourceNamespace) + status: + description: TestFailoverStatus is the observed state of one drill. + properties: + clones: + description: Clones is one entry per recovered volume. + items: + description: |- + TestFailoverClone is one recovered volume: the source it came from, the + snapshot and clone the drill built, and the PVC placed on the bubble cluster. + properties: + cloneID: + description: CloneID is the backend id of the writable clone. + type: string + pvcName: + description: PVCName is the bound PVC in the bubble namespace + on the bubble cluster. + type: string + sizeBytes: + description: SizeBytes is the recovered volume's size. + format: int64 + type: integer + snapshotID: + description: |- + SnapshotID is the recovery-point snapshot: the replicated snapshot already on + the bubble cluster's backend that the clone is built from. + type: string + sourceFSType: + description: |- + SourceFSType is the source PV's CSI fsType, carried onto the bubble PV so the + node plugin stages the clone with the filesystem it actually carries. The + clone is a block copy of the source, so its filesystem is the source's; an + empty fsType makes the node plugin default to ext4 and refuse to mount an XFS + volume. + type: string + sourceHandle: + description: SourceHandle is the source volume's backend handle, + read from its PV. + type: string + sourceRef: + description: |- + SourceRef is the source volume, or group member, the recovered volume maps + to. + type: string + sourceVolumeContext: + additionalProperties: + type: string + description: |- + SourceVolumeContext is the source PV's CSI volumeAttributes, minus the + identity and provisioner keys, carried onto the bubble PV so the node plugin + receives a non-nil VolumeContext when it stages the clone. The clone's own + identity (NQN, connections, nsId, and so on) is re-resolved from the clone + handle at stage time, so only the class-level parameters are carried; the + identity keys are dropped so a failed clone lookup can never point the mount + back at the source. + type: object + sourceVolumeMode: + description: |- + SourceVolumeMode is the source PV's volumeMode (Filesystem or Block), + carried onto the bubble PV and PVC. A VM's disk is a Block claim; a bubble + claim that omitted the mode defaulted to Filesystem and the kubelet asked + the node plugin to mount a raw guest disk (2026-10-03). + type: string + required: + - sourceRef + type: object + type: array + x-kubernetes-list-map-keys: + - sourceRef + x-kubernetes-list-type: map + completedAt: + description: CompletedAt is when the drill reached a terminal phase. + format: date-time + type: string + message: + description: |- + Message is the reason the phase is what it is: one sentence, replaced as the + drill moves, and never a log. + type: string + observedGeneration: + description: |- + ObservedGeneration is the generation the rest of this status was computed + from, so a stale status can be told from a current one. + format: int64 + type: integer + phase: + description: Phase is the drill's own progress. + enum: + - Pending + - Provisioning + - Ready + - Failed + - TearingDown + type: string + readyAt: + description: ReadyAt is when every recovered PVC became bound. + format: date-time + type: string + report: + description: Report is the drill's evidence, populated as it reaches + Ready. + properties: + bubbleCluster: + description: BubbleCluster is the cluster the drill recovered + onto. + type: string + invariantsHeld: + description: |- + InvariantsHeld is true only when the source fingerprint taken before the + drill matches the one taken at Ready. A Ready drill with this false is a + defect. + type: boolean + recoveryPoint: + description: RecoveryPoint is the snapshot or group generation + the drill recovered. + type: string + recoveryPointAgeSeconds: + description: RecoveryPointAgeSeconds is the drill time minus the + recovery-point time. + format: int64 + type: integer + recoveryPointTime: + description: RecoveryPointTime is when that point was taken. + format: date-time + type: string + type: object + startedAt: + description: StartedAt is when the drill started. + format: date-time + type: string + step: + description: Step is the position of the running drill's state machine. + properties: + claim: + description: |- + Claim records that this state's side effect was started. Absent means + no pass has started it since the state was entered. + properties: + attempt: + description: Attempt counts the claims taken on this state, + starting at 1. + format: int32 + type: integer + leaseUntil: + description: |- + LeaseUntil is when the claim expires and the side effect may be fired + again. + format: date-time + type: string + state: + description: State is the state the claim was taken in. + type: string + required: + - attempt + - leaseUntil + - state + type: object + deadline: + description: |- + Deadline is when that state expires, absent when it has none. It is an + absolute instant, so a state whose deadline passed while the controller + was down restores as already expired. + format: date-time + type: string + state: + description: |- + State is the state the machine was in. Empty means the resource has not + been reconciled yet, and restores to the graph's initial state. + type: string + type: object + x-kubernetes-validations: + - message: unknown step + rule: '!has(self.state) || self.state in [''ResolvingSource'',''ResolvingPoint'',''Shipping'',''Cloning'',''Placing'',''Releasing'']' + triggered: + description: |- + Triggered records that the current step's side effect was issued, so a + restart does not repeat it. + type: boolean + type: object + type: object + served: true + storage: true + subresources: + status: {} diff --git a/operator/internal/utils/objects.go b/operator/internal/utils/objects.go index 196bf3ca6..296cde72c 100644 --- a/operator/internal/utils/objects.go +++ b/operator/internal/utils/objects.go @@ -99,6 +99,16 @@ func ResolveClusterUUID( clusterName string, ) (string, error) { + // A cluster on a different physical Kubernetes cluster has no local + // StorageCluster object to match by name -- the object only exists on + // that other cluster's own API server. The backend control plane is + // shared across clusters and addresses cluster pairs by this same UUID, + // so a raw UUID is passed through unresolved rather than requiring a + // local name match that can never succeed for a genuinely remote target. + if IsUUID(clusterName) { + return clusterName, nil + } + var clusters simplyblockv1alpha2.StorageClusterList if err := c.List(ctx, &clusters, client.InNamespace(namespace)); err != nil { return "", err diff --git a/operator/internal/utils/objects_test.go b/operator/internal/utils/objects_test.go index 4061159b6..923af7eaa 100644 --- a/operator/internal/utils/objects_test.go +++ b/operator/internal/utils/objects_test.go @@ -138,6 +138,21 @@ func TestResolveClusterAndPoolUUID(t *testing.T) { t.Fatalf("ResolveClusterCRByUUID should fail for an unknown UUID") } + // A ReplicationPair naming a cluster that lives on a different physical + // Kubernetes cluster carries only that cluster's backend UUID -- there is + // no local StorageCluster object to match by name, because the object + // itself only exists on the other cluster's own API server. Passing a raw + // UUID through unresolved (rather than requiring a local name match) is + // what makes cross-cluster ReplicationPair authoring possible at all. + const remoteUUID = "e7afccef-1d5a-4b77-88aa-bfbe90d8b3a3" + remoteResolved, err := ResolveClusterUUID(ctx, c, "ns1", remoteUUID) + if err != nil { + t.Fatalf("ResolveClusterUUID should pass a raw UUID through even with no local match: %v", err) + } + if remoteResolved != remoteUUID { + t.Fatalf("ResolveClusterUUID got %q want pass-through of %q", remoteResolved, remoteUUID) + } + if _, err := ResolveClusterCRByUUID(ctx, c, "ns2", "uuid-a"); err == nil { t.Fatalf("ResolveClusterCRByUUID should not find a cluster from a different namespace") } diff --git a/operator/internal/webapi/consistency_group.go b/operator/internal/webapi/consistency_group.go index cfbd574a8..31cdba847 100644 --- a/operator/internal/webapi/consistency_group.go +++ b/operator/internal/webapi/consistency_group.go @@ -9,11 +9,18 @@ import ( "net/http" ) -// ConsistencyGroupInfo is a group summary (design §10). +// ConsistencyGroupInfo is a group summary (design §10). PolicyID is the group's +// replication policy: a group attached with attach_group_policy stores its policy +// on the group record, so the group drill reads the policy off the group rather +// than inferring it from the policy list's placement (which a group-first attach +// leaves empty). LvsName and NodeID carry the group's pinned placement. type ConsistencyGroupInfo struct { UUID string `json:"id"` Name string `json:"name"` MemberCount int `json:"member_count"` + LvsName string `json:"lvs_name"` + NodeID string `json:"node_id"` + PolicyID string `json:"policy_id"` } // ConsistencyGroupMember is one current member of a group (design §10 /members). diff --git a/operator/internal/webapi/context.go b/operator/internal/webapi/context.go new file mode 100644 index 000000000..1c9f0082e --- /dev/null +++ b/operator/internal/webapi/context.go @@ -0,0 +1,22 @@ +package webapi + +import "context" + +type bearerTokenKey struct{} + +// WithBearerToken attaches a bearer credential to ctx that Do and +// DoWithHeaders send instead of the client's own service-account token. It is +// how a call scoped to one cluster authenticates as that cluster rather than +// as this process's own Kubernetes identity -- the only way to reach a +// control plane a different Kubernetes cluster runs (ControlPlane.spec.source.managed), +// since a Kubernetes TokenReview can never cross a cluster boundary. +func WithBearerToken(ctx context.Context, token string) context.Context { + return context.WithValue(ctx, bearerTokenKey{}, token) +} + +// BearerTokenFromContext returns the token WithBearerToken attached, and +// whether one was. +func BearerTokenFromContext(ctx context.Context) (string, bool) { + token, ok := ctx.Value(bearerTokenKey{}).(string) + return token, ok +} diff --git a/operator/internal/webapi/group_failover.go b/operator/internal/webapi/group_failover.go new file mode 100644 index 000000000..029b7fa6d --- /dev/null +++ b/operator/internal/webapi/group_failover.go @@ -0,0 +1,175 @@ +// Group test-failover reads: the control-plane calls the TestFailover controller +// makes to recover a whole consistency group. A group drill needs three things +// the per-volume path does not: the group's replication policy (the group form +// of the recovery point is keyed on the policy, not the group), the one +// group-consistent generation of replicated snapshots on the target, and each +// member volume's K8s identity (PVC name and namespace) so the recovered PVCs +// can be named and their source PVs read for staging metadata. +package webapi + +import ( + "context" + "encoding/json" + "fmt" + "net/http" + "strings" +) + +// storagePoolsListPathFmt is the cluster-scoped storage-pools list endpoint. +const storagePoolsListPathFmt = "/api/v2/clusters/%s/storage-pools/" + +// ReplicatedGroupSnapshot is one member's replicated snapshot on the target +// cluster, at one group-consistent generation. It is the cloneable point for +// that member: cluster and pool address the target backend, snapshot is the +// snapshot to clone, and size sizes the recovered PVC. LvolID is the replica +// volume on the TARGET the snapshot belongs to, not the source member; +// SourceLvolID is the source member whose data it holds, empty on a control +// plane that does not report it yet. +type ReplicatedGroupSnapshot struct { + SnapshotID string `json:"snapshot_id"` + ClusterID string `json:"cluster_id"` + PoolID string `json:"pool_id"` + LvolID string `json:"lvol_id"` + SourceLvolID string `json:"source_lvol_id"` + Size int64 `json:"size"` + GroupSeq int `json:"group_seq"` +} + +// latestGenerationResponse is the group form of latest-snapshot: one generation +// number and one replicated snapshot per current member, all on the target. +type latestGenerationResponse struct { + GroupSeq int `json:"group_seq"` + Members []ReplicatedGroupSnapshot `json:"members"` +} + +// LatestReplicatedGeneration resolves the latest group-consistent generation on +// the target for a policy, returning the generation and one replicated snapshot +// per member. The control plane refuses (400) when no generation is complete for +// every member or when members straddle generations, which is what makes the +// recovered set crash-consistent; found is false only when there is no +// generation yet (nothing has replicated). +func (c *Client) LatestReplicatedGeneration( + ctx context.Context, + clusterUUID, policyID string, +) (groupSeq int, members []ReplicatedGroupSnapshot, found bool, err error) { + endpoint := fmt.Sprintf("/api/v2/clusters/%s/replication/policies/%s/latest-generation", clusterUUID, policyID) + body, statusCode, doErr := c.Do(ctx, http.MethodGet, endpoint, nil) + if statusCode == http.StatusNotFound { + return 0, nil, false, nil + } + if doErr != nil { + return 0, nil, false, fmt.Errorf("resolve latest replicated generation: %w", doErr) + } + if statusCode >= 300 { + return 0, nil, false, fmt.Errorf("resolve latest replicated generation: status %d: %s", statusCode, string(body)) + } + var dto latestGenerationResponse + if err := json.Unmarshal(body, &dto); err != nil { + return 0, nil, false, fmt.Errorf("decode latest-generation: %w", err) + } + return dto.GroupSeq, dto.Members, len(dto.Members) > 0, nil +} + +// MemberVolume is a group member's source volume, resolved to the K8s identity +// the drill needs: the PVC name and namespace it was provisioned for, so the +// recovered PVC can be named and the source PV read for staging metadata. +// +// The control plane stores the PVC identity in one field as "namespace/name" and +// uses the separate "namespace" field for the NVMe namespace, not the K8s one, so +// the K8s namespace and name are split out of PVCRef rather than read from the +// volume's namespace field. +type MemberVolume struct { + LvolID string + PVCName string + PVCNamespace string + PoolID string + Size int64 +} + +// memberVolumeDTO is the wire shape ResolveMemberVolumes decodes before splitting +// the namespaced PVC reference into a namespace and a name. +type memberVolumeDTO struct { + LvolID string `json:"id"` + PVCRef string `json:"pvc_name"` + PoolID string `json:"pool_id"` + Size int64 `json:"size"` +} + +// ResolveMemberVolumes maps each of the given member lvol ids to its source +// volume's K8s identity on the source cluster. A group member carries only an +// lvol id, so the pool is not known up front; this enumerates the cluster's +// pools and their volumes once and matches. Members it cannot find are omitted +// from the result, so the caller can tell an incomplete resolution from a +// complete one by the map size. +func (c *Client) ResolveMemberVolumes( + ctx context.Context, + clusterUUID string, + lvolIDs []string, +) (map[string]MemberVolume, error) { + want := make(map[string]struct{}, len(lvolIDs)) + for _, id := range lvolIDs { + want[id] = struct{}{} + } + + poolsEndpoint := fmt.Sprintf(storagePoolsListPathFmt, clusterUUID) + body, statusCode, err := c.Do(ctx, http.MethodGet, poolsEndpoint, nil) + if err != nil { + return nil, fmt.Errorf("list storage pools: %w", err) + } + if statusCode >= 300 { + return nil, fmt.Errorf("list storage pools: status %d: %s", statusCode, string(body)) + } + var pools []struct { + ID string `json:"id"` + } + if err := json.Unmarshal(body, &pools); err != nil { + return nil, fmt.Errorf("unmarshal storage pools: %w", err) + } + + found := make(map[string]MemberVolume, len(lvolIDs)) + for _, pool := range pools { + if len(found) == len(want) { + break + } + volsEndpoint := fmt.Sprintf("/api/v2/clusters/%s/storage-pools/%s/volumes", clusterUUID, pool.ID) + vbody, vstatus, verr := c.Do(ctx, http.MethodGet, volsEndpoint, nil) + if verr != nil { + return nil, fmt.Errorf("list volumes in pool %s: %w", pool.ID, verr) + } + if vstatus >= 300 { + return nil, fmt.Errorf("list volumes in pool %s: status %d: %s", pool.ID, vstatus, string(vbody)) + } + var vols []memberVolumeDTO + if err := json.Unmarshal(vbody, &vols); err != nil { + return nil, fmt.Errorf("unmarshal volumes in pool %s: %w", pool.ID, err) + } + for i := range vols { + v := vols[i] + if _, ok := want[v.LvolID]; !ok { + continue + } + ns, name := splitPVCRef(v.PVCRef) + poolID := v.PoolID + if poolID == "" { + poolID = pool.ID + } + found[v.LvolID] = MemberVolume{ + LvolID: v.LvolID, + PVCName: name, + PVCNamespace: ns, + PoolID: poolID, + Size: v.Size, + } + } + } + return found, nil +} + +// splitPVCRef splits a "namespace/name" PVC reference into its namespace and +// name. A reference with no slash is taken as a bare name in no namespace. +func splitPVCRef(ref string) (namespace, name string) { + if i := strings.IndexByte(ref, '/'); i >= 0 { + return ref[:i], ref[i+1:] + } + return "", ref +} diff --git a/operator/internal/webapi/group_failover_test.go b/operator/internal/webapi/group_failover_test.go new file mode 100644 index 000000000..df6aee72c --- /dev/null +++ b/operator/internal/webapi/group_failover_test.go @@ -0,0 +1,99 @@ +package webapi + +import ( + "context" + "net/http" + "net/http/httptest" + "testing" +) + +// routingServer serves fixed JSON bodies keyed by request path, so a test can +// stand in for the control plane without the openapi spec mock. +func routingServer(t *testing.T, routes map[string]string) *Client { + t.Helper() + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + body, ok := routes[r.URL.Path] + if !ok { + http.NotFound(w, r) + return + } + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(body)) + })) + t.Cleanup(srv.Close) + return NewClient(srv.URL) +} + +func TestLatestReplicatedGenerationReturnsPerMemberSnapshots(t *testing.T) { + c := routingServer(t, map[string]string{ + "/api/v2/clusters/C/replication/policies/P/latest-generation": `{ + "group_seq": 7, + "members": [ + {"snapshot_id":"s1","cluster_id":"B","pool_id":"pb","lvol_id":"t1","size":1073741824,"group_seq":7}, + {"snapshot_id":"s2","cluster_id":"B","pool_id":"pb","lvol_id":"t2","size":1073741824,"group_seq":7} + ] + }`, + }) + seq, members, found, err := c.LatestReplicatedGeneration(context.Background(), "C", "P") + if err != nil { + t.Fatalf("LatestReplicatedGeneration: %v", err) + } + if !found { + t.Fatalf("found = false, want true when a generation exists") + } + if seq != 7 { + t.Errorf("group_seq = %d, want 7", seq) + } + if len(members) != 2 || members[0].SnapshotID != "s1" || members[1].SnapshotID != "s2" { + t.Errorf("members = %+v, want two with s1,s2", members) + } + if members[0].ClusterID != "B" || members[0].PoolID != "pb" || members[0].Size != 1073741824 { + t.Errorf("member[0] = %+v, want the target cluster/pool/size", members[0]) + } +} + +func TestLatestReplicatedGenerationNotFoundWhenNothingReplicated(t *testing.T) { + // No route registered -> 404 -> found=false, no error (nothing has replicated). + c := routingServer(t, map[string]string{}) + _, _, found, err := c.LatestReplicatedGeneration(context.Background(), "C", "P") + if err != nil { + t.Fatalf("LatestReplicatedGeneration on 404: %v", err) + } + if found { + t.Errorf("found = true, want false on 404") + } +} + +func TestResolveMemberVolumesMapsLvolIDsToPVCNames(t *testing.T) { + // Two pools; the two members live in different pools. Resolution must find + // both and carry their pvc name, namespace, pool, and size. + // pvc_name is the namespaced "namespace/name" form, and pool_id is null in the + // list (the iterating pool supplies it); "namespace" is the NVMe namespace and + // must be ignored. + c := routingServer(t, map[string]string{ + "/api/v2/clusters/C/storage-pools/": `[{"id":"pool-1"},{"id":"pool-2"}]`, + "/api/v2/clusters/C/storage-pools/pool-1/volumes": `[ + {"id":"lvol-a","pvc_name":"app/data-1","namespace":"nvme-ns-uuid","pool_id":null,"size":1073741824}, + {"id":"lvol-x","pvc_name":"app/other","namespace":"nvme-ns-uuid","pool_id":null,"size":1073741824} + ]`, + "/api/v2/clusters/C/storage-pools/pool-2/volumes": `[ + {"id":"lvol-b","pvc_name":"app/data-2","namespace":"nvme-ns-uuid","pool_id":null,"size":1073741824} + ]`, + }) + got, err := c.ResolveMemberVolumes(context.Background(), "C", []string{"lvol-a", "lvol-b"}) + if err != nil { + t.Fatalf("ResolveMemberVolumes: %v", err) + } + if len(got) != 2 { + t.Fatalf("resolved %d members, want 2: %+v", len(got), got) + } + if got["lvol-a"].PVCName != "data-1" || got["lvol-a"].PVCNamespace != "app" || got["lvol-a"].PoolID != "pool-1" { + t.Errorf("lvol-a = %+v, want name=data-1 ns=app pool=pool-1", got["lvol-a"]) + } + if got["lvol-b"].PVCName != "data-2" || got["lvol-b"].PVCNamespace != "app" || got["lvol-b"].PoolID != "pool-2" { + t.Errorf("lvol-b = %+v, want name=data-2 ns=app pool=pool-2", got["lvol-b"]) + } + if _, ok := got["lvol-x"]; ok { + t.Errorf("resolved an unrequested member lvol-x") + } +} diff --git a/operator/internal/webapi/request.go b/operator/internal/webapi/request.go index 7734cebf2..59f2e3729 100644 --- a/operator/internal/webapi/request.go +++ b/operator/internal/webapi/request.go @@ -51,8 +51,14 @@ func (c *Client) DoWithHeaders( return nil, nil, 0, fmt.Errorf("create request: %w", err) } - // Attach auth header - req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", c.saToken)) + // Attach auth header. A token WithBearerToken attached to ctx wins over + // this process's own service-account token, for a call scoped to a + // cluster whose control plane lives on a different Kubernetes cluster. + token := c.saToken + if override, ok := BearerTokenFromContext(ctx); ok { + token = override + } + req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", token)) req.Header.Set("Content-Type", "application/json") // Execute the request diff --git a/operator/internal/webapi/request_test.go b/operator/internal/webapi/request_test.go index 01406ea25..5a3da827c 100644 --- a/operator/internal/webapi/request_test.go +++ b/operator/internal/webapi/request_test.go @@ -73,6 +73,37 @@ func TestDoAgainstSpecMockSendsHeadersBodyAndReturnsResponse(t *testing.T) { } } +// A call scoped to one cluster authenticates as that cluster when +// WithBearerToken names one, instead of as this process's own service +// account -- the only way to reach a control plane a different Kubernetes +// cluster runs, since a TokenReview can never cross a cluster boundary. +func TestDoSendsTheBearerTokenAttachedToContextInsteadOfTheServiceAccountToken(t *testing.T) { + mock := webapimock.NewSpecServerFromFile(t, "../../../shared/openapi.json", false) + defer mock.Close() + + mock.Register( + http.MethodGet, + "/api/v2/clusters/cluster-uuid/", + webapimock.RouteResponse{Status: http.StatusOK, Body: `{}`}, + ) + + c := NewClient(mock.URL()) + c.saToken = "operators-own-service-account-token" + + ctx := WithBearerToken(context.Background(), "cluster-uuids-own-secret") + if _, _, err := c.Do(ctx, http.MethodGet, "/api/v2/clusters/cluster-uuid/", nil); err != nil { + t.Fatalf("Do returned error: %v", err) + } + + reqs := mock.Requests() + if len(reqs) != 1 { + t.Fatalf("expected one request, got %d", len(reqs)) + } + if got := reqs[0].Headers["Authorization"]; got != "Bearer cluster-uuids-own-secret" { + t.Fatalf("authorization header = %q, want the context's bearer token", got) + } +} + func TestDoAgainstStrictSpecMockReturns400ForUnknownPath(t *testing.T) { mock := webapimock.NewSpecServerFromFile(t, "../../../shared/openapi.json", false) defer mock.Close() diff --git a/operator/internal/webhook/volumemigration_validator.go b/operator/internal/webhook/volumemigration_validator.go index e34f9d7b0..681ddb1b6 100644 --- a/operator/internal/webhook/volumemigration_validator.go +++ b/operator/internal/webhook/volumemigration_validator.go @@ -61,9 +61,14 @@ func (v *VolumeMigrationValidator) Handle(ctx context.Context, req admission.Req return admission.Allowed("consistency-group membership undeterminable; deferring to the backend") } if member { + // A group moves only as a whole: every member, and every subsystem + // holding one, in one group migration (sbcli co-location design §3). + // A VolumeMigration moves one subsystem, so it would split the group. return admission.Denied(fmt.Sprintf( - "volume %s is a member of consistency group %s and cannot be migrated; "+ - "a group's members are pinned to one logical volume store (§8.4)", volumeUUID, groupID)) + "volume %s is a member of consistency group %s and cannot be migrated alone: "+ + "a group's members live on one logical volume store, so the group moves as a whole "+ + "through the control plane's group migration "+ + "(POST /api/v2/clusters//consistency-groups//migration)", volumeUUID, groupID)) } return admission.Allowed("target volume is not a consistency-group member") } diff --git a/operator/test/e2e/rbac_test.go b/operator/test/e2e/rbac_test.go index d27c4ad60..35c1ac80b 100644 --- a/operator/test/e2e/rbac_test.go +++ b/operator/test/e2e/rbac_test.go @@ -50,14 +50,14 @@ import ( // after the Manager Describe regardless of Ginkgo's randomized container order. const ( - rbacFooNS = "rbac-cluster-foo" - rbacBarNS = "rbac-cluster-bar" - rbacViewerSA = "viewer-sa" - rbacEditorSA = "editor-sa" - rbacOutsiderSA = "outsider-sa" - rbacScopedSA = "scoped-sa" - rbacScopedRoleName = "rbac-foo-admin" - rbacScopedClusterAllowed = "rbac-allowed" + rbacFooNS = "rbac-cluster-foo" + rbacBarNS = "rbac-cluster-bar" + rbacViewerSA = "viewer-sa" + rbacEditorSA = "editor-sa" + rbacOutsiderSA = "outsider-sa" + rbacScopedSA = "scoped-sa" + rbacScopedRoleName = "rbac-foo-admin" + rbacScopedClusterAllowed = "rbac-allowed" rbacScopedClusterForbidden = "rbac-forbidden" ) diff --git a/shared/openapi.json b/shared/openapi.json index 1aabaa88e..3eff3aff3 100644 --- a/shared/openapi.json +++ b/shared/openapi.json @@ -295,52 +295,6 @@ } } }, - "/api/v2/clusters/{cluster_id}/backup-config": { - "get": { - "summary": "Clusters:Backup-Config:Get", - "description": "The cluster's backup configuration, with credentials masked.\n\nThe credentials are ``SecretStr``, which FastAPI's JSON serialization\nrenders as ``**********``.", - "operationId": "clusters_backup_config_get_api_v2_clusters__cluster_id__backup_config_get", - "security": [ - { - "HTTPBearer": [] - } - ], - "parameters": [ - { - "name": "cluster_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - } - } - ], - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/BackupConfig-Output" - } - } - } - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - } - } - }, "/api/v2/clusters/{cluster_id}/capacity": { "get": { "summary": "Clusters:Capacity", @@ -2293,67 +2247,6 @@ } } }, - "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/add": { - "post": { - "summary": "Clusters:Storage Nodes:Devices:Add", - "operationId": "clusters_storage_nodes_devices_add_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__add_post", - "security": [ - { - "HTTPBearer": [] - } - ], - "parameters": [ - { - "name": "cluster_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - } - }, - { - "name": "storage_node_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Storage Node Id" - } - }, - { - "name": "device_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Device Id" - } - } - ], - "responses": { - "204": { - "description": "Successful Response" - }, - "409": { - "description": "The device is not new" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - } - } - }, "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/remove": { "post": { "summary": "Clusters:Storage Nodes:Devices:Remove", @@ -2422,141 +2315,6 @@ } } }, - "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/fail": { - "post": { - "summary": "Clusters:Storage Nodes:Devices:Fail", - "operationId": "clusters_storage_nodes_devices_fail_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__fail_post", - "security": [ - { - "HTTPBearer": [] - } - ], - "parameters": [ - { - "name": "cluster_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - } - }, - { - "name": "storage_node_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Storage Node Id" - } - }, - { - "name": "device_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Device Id" - } - } - ], - "responses": { - "204": { - "description": "Successful Response" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - } - } - }, - "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/replace": { - "post": { - "summary": "Clusters:Storage Nodes:Devices:Replace", - "description": "Add a failed and migrated device back as a new, empty device.\n\nThe replacement carries the failed device's information but none of its\ndata, and arrives in `new` state \u2014 `POST ../devices/{id}/add` puts it into\nservice.", - "operationId": "clusters_storage_nodes_devices_replace_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__replace_post", - "security": [ - { - "HTTPBearer": [] - } - ], - "parameters": [ - { - "name": "cluster_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - } - }, - { - "name": "storage_node_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Storage Node Id" - } - }, - { - "name": "device_id", - "in": "path", - "required": true, - "schema": { - "type": "string", - "format": "uuid", - "title": "Device Id" - } - }, - { - "name": "response-format", - "in": "query", - "required": false, - "schema": { - "enum": [ - "empty", - "full", - "identifier" - ], - "type": "string", - "default": "identifier", - "title": "Response-Format" - } - } - ], - "responses": { - "201": { - "description": "Successful Response" - }, - "409": { - "description": "The device cannot be replaced" - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - } - } - }, "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/restart": { "post": { "summary": "Clusters:Storage Nodes:Devices:Restart", @@ -6043,77 +5801,10 @@ } } }, - "/api/v2/clusters/{cluster_id}/backups/discover": { - "post": { - "summary": "Clusters:Backups:Discover", - "description": "List the backups a bucket contains, without importing anything.\n\nA POST because it carries credentials, which have no business in a query\nstring. Takes no cluster state at all: this is what an operator runs when\nthe cluster that wrote the backups no longer exists.", - "operationId": "clusters_backups_discover_api_v2_clusters__cluster_id__backups_discover_post", - "security": [ - { - "HTTPBearer": [] - } - ], - "parameters": [ - { - "name": "cluster_id", - "in": "path", - "required": true, - "schema": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" - } - ], - "title": "Cluster Id" - } - } - ], - "requestBody": { - "required": true, - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/BackupConfig-Input" - } - } - } - }, - "responses": { - "200": { - "description": "Successful Response", - "content": { - "application/json": { - "schema": { - "type": "array", - "items": { - "$ref": "#/components/schemas/BackupManifest" - }, - "title": "Response Clusters Backups Discover Api V2 Clusters Cluster Id Backups Discover Post" - } - } - } - }, - "422": { - "description": "Validation Error", - "content": { - "application/json": { - "schema": { - "$ref": "#/components/schemas/HTTPValidationError" - } - } - } - } - } - } - }, - "/api/v2/clusters/{cluster_id}/backups/export": { - "get": { - "summary": "Clusters:Backups:Export", - "operationId": "clusters_backups_export_api_v2_clusters__cluster_id__backups_export_get", + "/api/v2/clusters/{cluster_id}/backups/export": { + "get": { + "summary": "Clusters:Backups:Export", + "operationId": "clusters_backups_export_api_v2_clusters__cluster_id__backups_export_get", "security": [ { "HTTPBearer": [] @@ -9189,330 +8880,426 @@ } } } - } - }, - "components": { - "schemas": { - "AlertDTO": { - "properties": { - "id": { - "type": "string", - "title": "Id" - }, - "kind": { - "type": "string", - "title": "Kind" - }, - "severity": { - "type": "string", - "enum": [ - "critical", - "warning" - ], - "title": "Severity" - }, - "status": { - "type": "string", - "enum": [ - "firing", - "resolved" - ], - "title": "Status" - }, - "message": { - "type": "string", - "title": "Message" - }, - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - }, - "node_id": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" - } - ], - "title": "Node Id" + }, + "/api/v2/clusters/{cluster_id}/consistency-groups/{group_id}/replication/resolution": { + "get": { + "tags": [ + "consistency-groups" + ], + "summary": "Clusters:Consistency-Groups:Replication:Resolution", + "description": "Where the group's data lives now, keyed by the handles its PVs keep.\n\nAfter a relocate the group a VGR names is empty -- its demoted members were\ndeleted so the way back stays open -- and its data lives in the peer group of\nthe same name. The CSI driver resolves the VGR's original group handle here:\nthe group holding live members, and each protected volume's original handle\nwith the volume serving it now (2026-10-04: WordPress's VRG waited for\ndestination info for ever against the emptied source group). Never a 404\nfor an existing group: ``active_group_id`` is empty when nothing serves it.", + "operationId": "clusters_consistency_groups_replication_resolution_api_v2_clusters__cluster_id__consistency_groups__group_id__replication_resolution_get", + "security": [ + { + "HTTPBearer": [] + } + ], + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + } }, - "device_id": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" + { + "name": "group_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Group Id" + } + } + ], + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ConsistencyGroupResolutionDTO" + } } - ], - "title": "Device Id" + } }, - "since": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } - ], - "title": "Since" - }, - "first_seen": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" + } + } + } + } + }, + "/api/v2/clusters/{cluster_id}/backup-config": { + "get": { + "summary": "Clusters:Backup-Config:Get", + "description": "The cluster's backup configuration, with credentials masked.\n\nThe credentials are ``SecretStr``, which FastAPI's JSON serialization\nrenders as ``**********``.", + "operationId": "clusters_backup_config_get_api_v2_clusters__cluster_id__backup_config_get", + "security": [ + { + "HTTPBearer": [] + } + ], + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + } + } + ], + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/BackupConfig-Output" + } } - ], - "title": "First Seen" + } }, - "resolved_at": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } - ], - "title": "Resolved At" - }, - "details": { - "additionalProperties": true, - "type": "object", - "title": "Details" + } + } + } + } + }, + "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/add": { + "post": { + "summary": "Clusters:Storage Nodes:Devices:Add", + "operationId": "clusters_storage_nodes_devices_add_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__add_post", + "security": [ + { + "HTTPBearer": [] } - }, - "type": "object", - "required": [ - "id", - "kind", - "severity", - "status", - "message", - "cluster_id", - "node_id", - "device_id", - "since", - "first_seen", - "resolved_at", - "details" ], - "title": "AlertDTO", - "description": "One condition that currently needs an operator.\n\nDeliberately NOT an EventObj. An event is a journal entry -- it happened,\nit is kept forever, and nothing ever retracts it. An alert is a claim\nabout the present that goes away by itself when it stops being true, so\nit carries the object it is about and the time the condition started\nrather than the time something was logged. ``id`` is derived from the\nkind and the object, so it is stable across polls and a consumer can\ndedupe on it without keeping state." - }, - "BackupConfig-Input": { - "properties": { - "bucket_name": { - "type": "string", - "minLength": 1, - "title": "Bucket Name" - }, - "region": { - "anyOf": [ - { - "type": "string", - "minLength": 1 - }, - { - "type": "null" - } - ], - "title": "Region" - }, - "endpoint": { - "anyOf": [ - { - "type": "string", - "maxLength": 2083, - "minLength": 1, - "format": "uri" - }, - { - "type": "null" - } - ], - "title": "Endpoint" - }, - "secondary_target": { - "$ref": "#/components/schemas/SecondaryTarget", - "default": 0 - }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false - }, - "snapshot_backups": { - "type": "boolean", - "title": "Snapshot Backups", - "default": true + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + } }, - "verify_tls": { - "type": "boolean", - "title": "Verify Tls", - "default": true + { + "name": "storage_node_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Storage Node Id" + } }, - "use_path_style": { - "type": "boolean", - "title": "Use Path Style", - "default": true + { + "name": "device_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Device Id" + } + } + ], + "responses": { + "204": { + "description": "Successful Response" }, - "credentials": { - "anyOf": [ - { - "$ref": "#/components/schemas/S3Credentials" - }, - { - "type": "null" - } - ] + "409": { + "description": "The device is not new" }, - "s3_thread_pool_size": { - "anyOf": [ - { - "type": "integer", - "minimum": 1.0 - }, - { - "type": "null" + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } - ], - "title": "S3 Thread Pool Size" + } + } + } + } + }, + "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/fail": { + "post": { + "summary": "Clusters:Storage Nodes:Devices:Fail", + "operationId": "clusters_storage_nodes_devices_fail_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__fail_post", + "security": [ + { + "HTTPBearer": [] } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "bucket_name" ], - "title": "BackupConfig", - "description": "A cluster's backup configuration: a location plus how to authenticate to it." - }, - "BackupConfig-Output": { - "properties": { - "bucket_name": { - "type": "string", - "minLength": 1, - "title": "Bucket Name" + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + } }, - "region": { - "anyOf": [ - { - "type": "string", - "minLength": 1 - }, - { - "type": "null" - } - ], - "title": "Region" + { + "name": "storage_node_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Storage Node Id" + } }, - "endpoint": { - "type": "string", - "title": "Endpoint" + { + "name": "device_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Device Id" + } + } + ], + "responses": { + "204": { + "description": "Successful Response" }, - "secondary_target": { - "type": "integer" + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + } + } + }, + "/api/v2/clusters/{cluster_id}/storage-nodes/{storage_node_id}/devices/{device_id}/replace": { + "post": { + "summary": "Clusters:Storage Nodes:Devices:Replace", + "description": "Add a failed and migrated device back as a new, empty device.\n\nThe replacement carries the failed device's information but none of its\ndata, and arrives in `new` state \u2014 `POST ../devices/{id}/add` puts it into\nservice.", + "operationId": "clusters_storage_nodes_devices_replace_api_v2_clusters__cluster_id__storage_nodes__storage_node_id__devices__device_id__replace_post", + "security": [ + { + "HTTPBearer": [] + } + ], + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + } }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false + { + "name": "storage_node_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Storage Node Id" + } }, - "snapshot_backups": { - "type": "boolean", - "title": "Snapshot Backups", - "default": true + { + "name": "device_id", + "in": "path", + "required": true, + "schema": { + "type": "string", + "format": "uuid", + "title": "Device Id" + } }, - "verify_tls": { - "type": "boolean", - "title": "Verify Tls", - "default": true + { + "name": "response-format", + "in": "query", + "required": false, + "schema": { + "enum": [ + "empty", + "full", + "identifier" + ], + "type": "string", + "default": "identifier", + "title": "Response-Format" + } + } + ], + "responses": { + "201": { + "description": "Successful Response" }, - "use_path_style": { - "type": "boolean", - "title": "Use Path Style", - "default": true + "409": { + "description": "The device cannot be replaced" }, - "credentials": { - "anyOf": [ - { - "$ref": "#/components/schemas/S3Credentials" - }, - { - "type": "null" + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } } - ] - }, - "s3_thread_pool_size": { - "anyOf": [ - { - "type": "integer", - "minimum": 1.0 - }, - { - "type": "null" + } + } + } + } + }, + "/api/v2/clusters/{cluster_id}/backups/discover": { + "post": { + "summary": "Clusters:Backups:Discover", + "description": "List the backups a bucket contains, without importing anything.\n\nA POST because it carries credentials, which have no business in a query\nstring. Takes no cluster state at all: this is what an operator runs when\nthe cluster that wrote the backups no longer exists.", + "operationId": "clusters_backups_discover_api_v2_clusters__cluster_id__backups_discover_post", + "security": [ + { + "HTTPBearer": [] + } + ], + "parameters": [ + { + "name": "cluster_id", + "in": "path", + "required": true, + "schema": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Cluster Id" + } + } + ], + "requestBody": { + "required": true, + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/BackupConfig-Input" } - ], - "title": "S3 Thread Pool Size" + } } }, - "additionalProperties": false, - "type": "object", - "required": [ - "bucket_name" - ], - "title": "BackupConfig", - "description": "A cluster's backup configuration: a location plus how to authenticate to it." - }, - "BackupDTO": { + "responses": { + "200": { + "description": "Successful Response", + "content": { + "application/json": { + "schema": { + "type": "array", + "items": { + "$ref": "#/components/schemas/BackupManifest" + }, + "title": "Response Clusters Backups Discover Api V2 Clusters Cluster Id Backups Discover Post" + } + } + } + }, + "422": { + "description": "Validation Error", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/HTTPValidationError" + } + } + } + } + } + } + } + }, + "components": { + "schemas": { + "AlertDTO": { "properties": { "id": { "type": "string", - "format": "uuid", "title": "Id" }, - "s3_id": { - "type": "integer", - "title": "S3 Id" - }, - "lvol_id": { + "kind": { "type": "string", - "format": "uuid", - "title": "Lvol Id" + "title": "Kind" }, - "lvol_name": { + "severity": { "type": "string", - "title": "Lvol Name" + "enum": [ + "critical", + "warning" + ], + "title": "Severity" }, - "snapshot_id": { + "status": { "type": "string", - "format": "uuid", - "title": "Snapshot Id" + "enum": [ + "firing", + "resolved" + ], + "title": "Status" }, - "snapshot_name": { + "message": { "type": "string", - "title": "Snapshot Name" + "title": "Message" }, - "node_id": { + "cluster_id": { "type": "string", "format": "uuid", - "title": "Node Id" + "title": "Cluster Id" }, - "status": { - "type": "string", - "title": "Status" + "node_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Node Id" }, - "prev_backup_id": { + "device_id": { "anyOf": [ { "type": "string", @@ -9522,297 +9309,148 @@ "type": "null" } ], - "title": "Prev Backup Id" - }, - "size": { - "type": "integer", - "title": "Size" - }, - "created_at": { - "type": "integer", - "title": "Created At" - }, - "completed_at": { - "type": "integer", - "title": "Completed At" - }, - "encrypted": { - "type": "boolean", - "title": "Encrypted" - } - }, - "type": "object", - "required": [ - "id", - "s3_id", - "lvol_id", - "lvol_name", - "snapshot_id", - "snapshot_name", - "node_id", - "status", - "size", - "created_at", - "completed_at", - "encrypted" - ], - "title": "BackupDTO" - }, - "BackupExport-Input": { - "properties": { - "schema_version": { - "type": "integer", - "title": "Schema Version", - "default": 1 - }, - "groups": { - "items": { - "$ref": "#/components/schemas/LocatedManifests-Input" - }, - "type": "array", - "title": "Groups" - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "groups" - ], - "title": "BackupExport", - "description": "Backups carried out of a cluster in a file, grouped by where they live.\n\nGrouped rather than one location for the whole document because a cluster\ncan hold backups in several buckets at once -- its own, plus any it has\nimported -- and stamping all of them with a single bucket leaves the ones it\ndoes not describe unrestorable, which is discovered during the recovery they\nwere meant to serve.\n\nA group's manifests are all in one bucket by construction: a chain cannot\nspan buckets, so the only way to collect backups from several is to walk\nmore than one chain." - }, - "BackupExport-Output": { - "properties": { - "schema_version": { - "type": "integer", - "title": "Schema Version", - "default": 1 - }, - "groups": { - "items": { - "$ref": "#/components/schemas/LocatedManifests-Output" - }, - "type": "array", - "title": "Groups" - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "groups" - ], - "title": "BackupExport", - "description": "Backups carried out of a cluster in a file, grouped by where they live.\n\nGrouped rather than one location for the whole document because a cluster\ncan hold backups in several buckets at once -- its own, plus any it has\nimported -- and stamping all of them with a single bucket leaves the ones it\ndoes not describe unrestorable, which is discovered during the recovery they\nwere meant to serve.\n\nA group's manifests are all in one bucket by construction: a chain cannot\nspan buckets, so the only way to collect backups from several is to walk\nmore than one chain." - }, - "BackupLocation-Input": { - "properties": { - "bucket_name": { - "type": "string", - "minLength": 1, - "title": "Bucket Name" + "title": "Device Id" }, - "region": { + "since": { "anyOf": [ { - "type": "string", - "minLength": 1 + "type": "string" }, { "type": "null" } ], - "title": "Region" + "title": "Since" }, - "endpoint": { + "first_seen": { "anyOf": [ { - "type": "string", - "maxLength": 2083, - "minLength": 1, - "format": "uri" + "type": "string" }, { "type": "null" } ], - "title": "Endpoint" - }, - "secondary_target": { - "$ref": "#/components/schemas/SecondaryTarget", - "default": 0 - }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false - }, - "snapshot_backups": { - "type": "boolean", - "title": "Snapshot Backups", - "default": true - }, - "verify_tls": { - "type": "boolean", - "title": "Verify Tls", - "default": true - }, - "use_path_style": { - "type": "boolean", - "title": "Use Path Style", - "default": true - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "bucket_name" - ], - "title": "BackupLocation", - "description": "Where a backup's objects are, and how to interpret them. Never secret.\n\nEvery field here affects whether the objects can be read back at all, which\nis why the whole model is embedded in each backup rather than looked up from\nthe cluster that happened to create it." - }, - "BackupLocation-Output": { - "properties": { - "bucket_name": { - "type": "string", - "minLength": 1, - "title": "Bucket Name" + "title": "First Seen" }, - "region": { + "resolved_at": { "anyOf": [ { - "type": "string", - "minLength": 1 + "type": "string" }, { "type": "null" } ], - "title": "Region" - }, - "endpoint": { - "type": "string", - "title": "Endpoint" - }, - "secondary_target": { - "type": "integer" - }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false - }, - "snapshot_backups": { - "type": "boolean", - "title": "Snapshot Backups", - "default": true - }, - "verify_tls": { - "type": "boolean", - "title": "Verify Tls", - "default": true + "title": "Resolved At" }, - "use_path_style": { - "type": "boolean", - "title": "Use Path Style", - "default": true + "details": { + "additionalProperties": true, + "type": "object", + "title": "Details" } }, - "additionalProperties": false, "type": "object", "required": [ - "bucket_name" + "id", + "kind", + "severity", + "status", + "message", + "cluster_id", + "node_id", + "device_id", + "since", + "first_seen", + "resolved_at", + "details" ], - "title": "BackupLocation", - "description": "Where a backup's objects are, and how to interpret them. Never secret.\n\nEvery field here affects whether the objects can be read back at all, which\nis why the whole model is embedded in each backup rather than looked up from\nthe cluster that happened to create it." + "title": "AlertDTO", + "description": "One condition that currently needs an operator.\n\nDeliberately NOT an EventObj. An event is a journal entry -- it happened,\nit is kept forever, and nothing ever retracts it. An alert is a claim\nabout the present that goes away by itself when it stops being true, so\nit carries the object it is about and the time the condition started\nrather than the time something was logged. ``id`` is derived from the\nkind and the object, so it is stable across polls and a consumer can\ndedupe on it without keeping state." }, - "BackupManifest": { + "BackupDTO": { "properties": { - "schema_version": { - "type": "integer", - "title": "Schema Version", - "default": 1 - }, - "backup_id": { + "id": { "type": "string", "format": "uuid", - "title": "Backup Id" + "title": "Id" }, "s3_id": { "type": "integer", "title": "S3 Id" }, - "created_at": { - "type": "integer", - "title": "Created At" + "lvol_id": { + "type": "string", + "format": "uuid", + "title": "Lvol Id" }, - "completed_at": { - "type": "integer", - "title": "Completed At" + "lvol_name": { + "type": "string", + "title": "Lvol Name" }, - "size": { - "type": "integer", - "title": "Size" + "snapshot_id": { + "type": "string", + "format": "uuid", + "title": "Snapshot Id" }, - "prev_backup_id": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" - } - ], - "title": "Prev Backup Id" + "snapshot_name": { + "type": "string", + "title": "Snapshot Name" + }, + "node_id": { + "type": "string", + "format": "uuid", + "title": "Node Id" + }, + "status": { + "type": "string", + "title": "Status" }, - "encryption": { + "prev_backup_id": { "anyOf": [ { - "oneOf": [ - { - "$ref": "#/components/schemas/FDBKeyDescriptor" - }, - { - "$ref": "#/components/schemas/HCPKeyDescriptor" - } - ], - "discriminator": { - "propertyName": "type", - "mapping": { - "fdb": "#/components/schemas/FDBKeyDescriptor", - "hcp": "#/components/schemas/HCPKeyDescriptor" - } - } + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Encryption" + "title": "Prev Backup Id" }, - "source": { - "$ref": "#/components/schemas/ManifestSource" + "size": { + "type": "integer", + "title": "Size" }, - "volume": { - "$ref": "#/components/schemas/ManifestVolume" + "created_at": { + "type": "integer", + "title": "Created At" }, - "dataplane": { - "$ref": "#/components/schemas/ManifestDataPlane" + "completed_at": { + "type": "integer", + "title": "Completed At" + }, + "encrypted": { + "type": "boolean", + "title": "Encrypted" } }, - "additionalProperties": false, "type": "object", "required": [ - "backup_id", + "id", "s3_id", + "lvol_id", + "lvol_name", + "snapshot_id", + "snapshot_name", + "node_id", + "status", + "size", "created_at", "completed_at", - "size", - "source", - "volume", - "dataplane" + "encrypted" ], - "title": "BackupManifest" + "title": "BackupDTO" }, "BackupPolicyDTO": { "properties": { @@ -11343,65 +10981,21 @@ }, "HCPKeyDescriptor": { "properties": { - "dek_path": { - "type": "string", - "title": "Dek Path" - }, - "type": { - "type": "string", - "const": "hcp", - "title": "Type", - "default": "hcp" - }, - "kek_name": { - "type": "string", - "title": "Kek Name" - }, - "vault_base_url": { + "source_cluster_id": { "anyOf": [ { "type": "string", - "maxLength": 2083, - "minLength": 1, - "format": "uri" - }, - { - "type": "null" - } - ], - "title": "Vault Base Url" - }, - "transit_mount": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Transit Mount" - }, - "kv_mount": { - "anyOf": [ - { - "type": "string" + "format": "uuid" }, { "type": "null" } ], - "title": "Kv Mount" + "title": "Source Cluster Id" } }, - "additionalProperties": false, "type": "object", - "required": [ - "dek_path", - "kek_name" - ], - "title": "HCPKeyDescriptor", - "description": "Keys held in HashiCorp Vault, wrapped under a named transit key." + "title": "GroupFailbackParams" }, "HTTPValidationError": { "properties": { @@ -11450,88 +11044,186 @@ "type": "object", "title": "HashicorpVaultSettings" }, - "LocatedManifests-Input": { + "ManagementNodeDTO": { "properties": { - "location": { - "$ref": "#/components/schemas/BackupLocation-Input" + "id": { + "type": "string", + "format": "uuid", + "title": "Id" }, - "manifests": { - "items": { - "$ref": "#/components/schemas/BackupManifest" - }, - "type": "array", - "title": "Manifests" + "status": { + "type": "string", + "title": "Status" + }, + "hostname": { + "type": "string", + "title": "Hostname" + }, + "ip": { + "type": "string", + "format": "ipv4", + "title": "Ip" } }, - "additionalProperties": false, "type": "object", "required": [ - "location", - "manifests" + "id", + "status", + "hostname", + "ip" ], - "title": "LocatedManifests", - "description": "Manifests that were read from one location, and that location." + "title": "ManagementNodeDTO" }, - "LocatedManifests-Output": { + "MigrationDTO": { "properties": { - "location": { - "$ref": "#/components/schemas/BackupLocation-Output" + "id": { + "type": "string", + "format": "uuid", + "title": "Id" }, - "manifests": { + "lvol_id": { + "type": "string", + "title": "Lvol Id" + }, + "source_node_id": { + "type": "string", + "title": "Source Node Id" + }, + "active_source_node_id": { + "type": "string", + "title": "Active Source Node Id" + }, + "target_node_id": { + "type": "string", + "title": "Target Node Id" + }, + "phase": { + "type": "string", + "title": "Phase" + }, + "status": { + "type": "string", + "title": "Status" + }, + "snaps_total": { + "type": "integer", + "title": "Snaps Total" + }, + "snaps_migrated": { + "type": "integer", + "title": "Snaps Migrated" + }, + "intermediate_snap_rounds": { + "type": "integer", + "title": "Intermediate Snap Rounds" + }, + "max_intermediate_snap_rounds": { + "type": "integer", + "title": "Max Intermediate Snap Rounds" + }, + "retry_count": { + "type": "integer", + "title": "Retry Count" + }, + "max_retries": { + "type": "integer", + "title": "Max Retries" + }, + "error_message": { + "type": "string", + "title": "Error Message" + }, + "started_at": { + "type": "integer", + "title": "Started At" + }, + "completed_at": { + "type": "integer", + "title": "Completed At" + }, + "connect_strings": { "items": { - "$ref": "#/components/schemas/BackupManifest" + "$ref": "#/components/schemas/NvmeConnectEntry" }, "type": "array", - "title": "Manifests" + "title": "Connect Strings", + "default": [] } }, - "additionalProperties": false, "type": "object", "required": [ - "location", - "manifests" + "id", + "lvol_id", + "source_node_id", + "active_source_node_id", + "target_node_id", + "phase", + "status", + "snaps_total", + "snaps_migrated", + "intermediate_snap_rounds", + "max_intermediate_snap_rounds", + "retry_count", + "max_retries", + "error_message", + "started_at", + "completed_at" ], - "title": "LocatedManifests", - "description": "Manifests that were read from one location, and that location." + "title": "MigrationDTO" }, - "ManagementNodeDTO": { + "NvmeConnectEntry": { "properties": { - "id": { + "transport": { "type": "string", - "format": "uuid", - "title": "Id" + "title": "Transport" + }, + "ip": { + "type": "string", + "title": "Ip" + }, + "port": { + "type": "integer", + "title": "Port" }, - "status": { + "nqn": { "type": "string", - "title": "Status" + "title": "Nqn" }, - "hostname": { - "type": "string", - "title": "Hostname" + "reconnect-delay": { + "type": "integer", + "title": "Reconnect-Delay" }, - "ip": { + "ctrl-loss-tmo": { + "type": "integer", + "title": "Ctrl-Loss-Tmo" + }, + "fast-io-fail-tmo": { + "type": "integer", + "title": "Fast-Io-Fail-Tmo" + }, + "nr-io-queues": { + "type": "integer", + "title": "Nr-Io-Queues" + }, + "keep-alive-tmo": { + "type": "integer", + "title": "Keep-Alive-Tmo" + }, + "host-iface": { "type": "string", - "format": "ipv4", - "title": "Ip" - } - }, - "type": "object", - "required": [ - "id", - "status", - "hostname", - "ip" - ], - "title": "ManagementNodeDTO" - }, - "ManifestDataPlane": { - "properties": { - "key_format": { + "title": "Host-Iface", + "default": "" + }, + "tls": { + "type": "boolean", + "title": "Tls", + "default": false + }, + "connect": { "type": "string", - "title": "Key Format", - "default": "{s3_id}/{mid}/{extent}" + "title": "Connect" }, - "cluster_size": { + "ns-id": { "anyOf": [ { "type": "integer" @@ -11540,32 +11232,17 @@ "type": "null" } ], - "title": "Cluster Size" - }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false - } - }, - "additionalProperties": false, - "type": "object", - "title": "ManifestDataPlane", - "description": "How the objects are encoded, so a later format change is detectable.\n\nEverything here has to be recorded because reading the bucket cannot\nrecover it -- unlike the bucket's name, region and endpoint, which the\nreader necessarily supplied to get this far." - }, - "ManifestSource": { - "properties": { - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" + "title": "Ns-Id" }, - "node_id": { - "type": "string", - "format": "uuid", - "title": "Node Id" + "allowed-hosts": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Allowed-Hosts", + "default": [] }, - "cluster_name": { + "target-lvol-id": { "anyOf": [ { "type": "string" @@ -11574,374 +11251,475 @@ "type": "null" } ], - "title": "Cluster Name" + "title": "Target-Lvol-Id" } }, - "additionalProperties": false, "type": "object", "required": [ - "cluster_id", - "node_id" + "transport", + "ip", + "port", + "nqn", + "reconnect-delay", + "ctrl-loss-tmo", + "fast-io-fail-tmo", + "nr-io-queues", + "keep-alive-tmo", + "connect" ], - "title": "ManifestSource", - "description": "Where this backup came from. Provenance for an operator reading a bucket.\n\nNothing may resolve configuration or keys through these -- that dependency\non the originating cluster is the whole problem being removed." + "title": "NvmeConnectEntry" }, - "ManifestVolume": { + "PolicyParams": { "properties": { - "lvol_id": { - "type": "string", - "format": "uuid", - "title": "Lvol Id" - }, - "lvol_name": { + "policy_name": { "type": "string", - "title": "Lvol Name" + "title": "Policy Name" }, - "snapshot_id": { + "target_id": { "type": "string", "format": "uuid", - "title": "Snapshot Id" - }, - "snapshot_name": { - "type": "string", - "title": "Snapshot Name" + "title": "Target Id" }, - "size": { + "interval_min": { "type": "integer", - "title": "Size" - }, - "pool_name": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Pool Name" + "minimum": 0.0, + "title": "Interval Min", + "default": 1 }, - "ha_type": { + "mode": { "anyOf": [ { "type": "string", "enum": [ - "single", - "ha" + "failover", + "migration" ] }, { "type": "null" } ], - "title": "Ha Type" + "title": "Mode" }, - "fabric": { + "keep_replicated": { "anyOf": [ { - "type": "string", - "enum": [ - "tcp", - "rdma", - "tcp,rdma" - ] + "type": "integer", + "minimum": 2.0 }, { "type": "null" } ], - "title": "Fabric" + "title": "Keep Replicated" }, - "lvol_priority_class": { + "rpo_target_seconds": { "anyOf": [ { - "type": "integer" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Lvol Priority Class" + "title": "Rpo Target Seconds" }, - "max_size": { - "anyOf": [ - { - "type": "integer" - }, - { - "type": "null" - } - ], - "title": "Max Size" + "consistency_group": { + "type": "boolean", + "title": "Consistency Group", + "default": false + } + }, + "type": "object", + "required": [ + "policy_name", + "target_id" + ], + "title": "PolicyParams" + }, + "PoolHostParams": { + "properties": { + "host_nqn": { + "type": "string", + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?", + "title": "Host Nqn" + } + }, + "type": "object", + "required": [ + "host_nqn" + ], + "title": "PoolHostParams" + }, + "ReplicateLVolParams": { + "properties": { + "lvol_id": { + "type": "string", + "format": "uuid", + "title": "Lvol Id" + } + }, + "type": "object", + "required": [ + "lvol_id" + ], + "title": "ReplicateLVolParams" + }, + "ReplicatedGenerationDTO": { + "properties": { + "group_seq": { + "type": "integer", + "minimum": 0.0, + "title": "Group Seq" + }, + "members": { + "items": { + "$ref": "#/components/schemas/ReplicatedSnapshotDTO" + }, + "type": "array", + "title": "Members" + } + }, + "type": "object", + "required": [ + "group_seq", + "members" + ], + "title": "ReplicatedGenerationDTO", + "description": "One complete, fully replicated consistency-group generation, every\nmember addressed as a cloneable object on the secondary." + }, + "ReplicatedSnapshotDTO": { + "properties": { + "snapshot_id": { + "type": "string", + "format": "uuid", + "title": "Snapshot Id" + }, + "cluster_id": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" }, - "rw_ios_per_sec": { + "pool_id": { "anyOf": [ { - "type": "integer" + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Rw Ios Per Sec" + "title": "Pool Id" }, - "rw_mbytes_per_sec": { + "lvol_id": { "anyOf": [ { - "type": "integer" + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Rw Mbytes Per Sec" + "title": "Lvol Id" }, - "r_mbytes_per_sec": { - "anyOf": [ - { - "type": "integer" - }, - { - "type": "null" - } - ], - "title": "R Mbytes Per Sec" + "size": { + "type": "integer", + "minimum": 0.0, + "title": "Size" }, - "w_mbytes_per_sec": { - "anyOf": [ - { - "type": "integer" - }, - { - "type": "null" - } - ], - "title": "W Mbytes Per Sec" + "used_size": { + "type": "integer", + "minimum": 0.0, + "title": "Used Size" + }, + "created_at": { + "type": "string", + "format": "date-time", + "title": "Created At" + }, + "group_id": { + "type": "string", + "title": "Group Id", + "default": "" + }, + "group_seq": { + "type": "integer", + "minimum": 0.0, + "title": "Group Seq", + "default": 0 } }, - "additionalProperties": false, "type": "object", "required": [ - "lvol_id", - "lvol_name", "snapshot_id", - "snapshot_name", - "size" + "cluster_id", + "size", + "used_size", + "created_at" ], - "title": "ManifestVolume", - "description": "The shape of the volume this backup was taken from.\n\nSplit in two by what is knowable. The identity and size come off the backup\nrecord and are always present. The settings below them come off the live\nvolume, so they are absent together once that volume is deleted -- and\nabsent is not the same answer as ``0``, which for a QoS cap means\n\"unlimited\" and for a priority class is a real class.\n\nNothing reads the settings yet; restore still creates its volume with\nhardcoded defaults. They are recorded anyway because a manifest is read\nyears after it is written, and a backup taken today cannot be given a shape\nretroactively once its volume is gone.\n\nThe volume's allow-list is deliberately not among them. Who may attach is a\nproperty of the pool a volume lives in, not of the bytes a backup holds, and\na restore lands in whichever pool it is given -- possibly in another cluster,\nwhere the source volume's NQNs mean nothing. So a restored volume takes the\ntarget pool's host configuration, and a stale allow-list from the source\nnever overrides it." + "title": "ReplicatedSnapshotDTO", + "description": "A fully replicated snapshot on the secondary, addressed as a cloneable\nobject. ``lvol_id`` is the volume the snapshot belongs to on the\nSECONDARY cluster, not the source volume the caller asked about, because\nthat is the identity the ordinary CSI clone path resolves a\n``dataSource`` against." }, - "MigrationDTO": { + "ReplicationPolicyDTO": { "properties": { "id": { "type": "string", "format": "uuid", "title": "Id" }, - "lvol_id": { + "cluster_id": { "type": "string", - "title": "Lvol Id" + "format": "uuid", + "title": "Cluster Id" }, - "source_node_id": { + "policy_name": { "type": "string", - "title": "Source Node Id" + "title": "Policy Name" }, - "active_source_node_id": { + "target_id": { "type": "string", - "title": "Active Source Node Id" + "format": "uuid", + "title": "Target Id" }, - "target_node_id": { - "type": "string", - "title": "Target Node Id" + "interval_min": { + "type": "integer", + "minimum": 0.0, + "title": "Interval Min" }, - "phase": { + "mode": { "type": "string", - "title": "Phase" + "enum": [ + "failover", + "migration" + ], + "title": "Mode" + }, + "keep_replicated": { + "type": "integer", + "title": "Keep Replicated" }, "status": { "type": "string", + "enum": [ + "active", + "inactive" + ], "title": "Status" }, - "snaps_total": { - "type": "integer", - "title": "Snaps Total" - }, - "snaps_migrated": { - "type": "integer", - "title": "Snaps Migrated" - }, - "intermediate_snap_rounds": { - "type": "integer", - "title": "Intermediate Snap Rounds" - }, - "max_intermediate_snap_rounds": { - "type": "integer", - "title": "Max Intermediate Snap Rounds" + "rpo_target_seconds": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Rpo Target Seconds" }, - "retry_count": { - "type": "integer", - "title": "Retry Count" + "consistency_group": { + "type": "boolean", + "title": "Consistency Group", + "default": false }, - "max_retries": { - "type": "integer", - "title": "Max Retries" + "group_node_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Group Node Id" }, - "error_message": { + "group_lvs_name": { "type": "string", - "title": "Error Message" - }, - "started_at": { - "type": "integer", - "title": "Started At" + "title": "Group Lvs Name", + "default": "" }, - "completed_at": { + "group_last_seq": { "type": "integer", - "title": "Completed At" - }, - "connect_strings": { - "items": { - "$ref": "#/components/schemas/NvmeConnectEntry" - }, - "type": "array", - "title": "Connect Strings", - "default": [] + "title": "Group Last Seq", + "default": 0 } }, "type": "object", "required": [ "id", - "lvol_id", - "source_node_id", - "active_source_node_id", - "target_node_id", - "phase", - "status", - "snaps_total", - "snaps_migrated", - "intermediate_snap_rounds", - "max_intermediate_snap_rounds", - "retry_count", - "max_retries", - "error_message", - "started_at", - "completed_at" + "cluster_id", + "policy_name", + "target_id", + "interval_min", + "mode", + "keep_replicated", + "status" ], - "title": "MigrationDTO" + "title": "ReplicationPolicyDTO" }, - "NvmeConnectEntry": { + "ReplicationRelationshipDTO": { "properties": { - "transport": { + "replication_id": { "type": "string", - "title": "Transport" + "format": "uuid", + "title": "Replication Id" }, - "ip": { - "type": "string", - "title": "Ip" + "source_lvol_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Source Lvol Id" + }, + "target_lvol_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Target Lvol Id" + }, + "source_cluster_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Source Cluster Id" + }, + "target_cluster_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Target Cluster Id" }, - "port": { - "type": "integer", - "title": "Port" + "target_pool_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Target Pool Id" }, - "nqn": { + "mode": { "type": "string", - "title": "Nqn" - }, - "reconnect-delay": { - "type": "integer", - "title": "Reconnect-Delay" + "enum": [ + "failover", + "migration" + ], + "title": "Mode" }, - "ctrl-loss-tmo": { - "type": "integer", - "title": "Ctrl-Loss-Tmo" + "state": { + "type": "string", + "enum": [ + "replicating", + "cutover_pending", + "cutover_done", + "failed_over" + ], + "title": "State" }, - "fast-io-fail-tmo": { - "type": "integer", - "title": "Fast-Io-Fail-Tmo" + "direction": { + "type": "string", + "enum": [ + "to_target", + "to_source" + ], + "title": "Direction" }, - "nr-io-queues": { - "type": "integer", - "title": "Nr-Io-Queues" + "target_nqn": { + "type": "string", + "title": "Target Nqn" }, - "keep-alive-tmo": { + "target_ns_id": { "type": "integer", - "title": "Keep-Alive-Tmo" - }, - "host-iface": { - "type": "string", - "title": "Host-Iface", - "default": "" + "title": "Target Ns Id" }, - "tls": { + "is_source": { "type": "boolean", - "title": "Tls", - "default": false - }, - "connect": { - "type": "string", - "title": "Connect" + "title": "Is Source" }, - "ns-id": { + "active": { "anyOf": [ { - "type": "integer" + "type": "string" }, { "type": "null" } ], - "title": "Ns-Id" - }, - "allowed-hosts": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Allowed-Hosts", - "default": [] + "title": "Active" }, - "target-lvol-id": { + "active_lvol_id": { "anyOf": [ { - "type": "string" + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Target-Lvol-Id" + "title": "Active Lvol Id" } }, "type": "object", "required": [ - "transport", - "ip", - "port", - "nqn", - "reconnect-delay", - "ctrl-loss-tmo", - "fast-io-fail-tmo", - "nr-io-queues", - "keep-alive-tmo", - "connect" + "replication_id", + "source_lvol_id", + "target_lvol_id", + "source_cluster_id", + "target_cluster_id", + "mode", + "state", + "direction", + "target_nqn", + "target_ns_id", + "is_source" ], - "title": "NvmeConnectEntry" + "title": "ReplicationRelationshipDTO" }, - "PolicyParams": { + "ReplicationStartParams": { "properties": { - "policy_name": { - "type": "string", - "title": "Policy Name" - }, - "target_id": { - "type": "string", - "format": "uuid", - "title": "Target Id" - }, - "interval_min": { - "type": "integer", - "minimum": 0.0, - "title": "Interval Min", - "default": 1 + "replication_cluster_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Replication Cluster Id" }, "mode": { "anyOf": [ @@ -11958,19 +11736,118 @@ ], "title": "Mode" }, - "keep_replicated": { + "interval_min": { "anyOf": [ { "type": "integer", - "minimum": 2.0 + "minimum": 0.0 }, { "type": "null" } ], - "title": "Keep Replicated" + "title": "Interval Min" + } + }, + "type": "object", + "title": "ReplicationStartParams" + }, + "ReplicationStatusDTO": { + "properties": { + "role": { + "type": "string", + "enum": [ + "source", + "secondary", + "failed_over", + "none" + ], + "title": "Role" }, - "rpo_target_seconds": { + "state": { + "type": "string", + "enum": [ + "in_sync", + "replicating", + "lagging", + "degraded", + "error", + "not_replicating" + ], + "title": "State" + }, + "last_replicated_at": { + "anyOf": [ + { + "type": "string", + "format": "date-time" + }, + { + "type": "null" + } + ], + "title": "Last Replicated At" + }, + "lag_seconds": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Lag Seconds" + }, + "lag_budget_seconds": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Lag Budget Seconds" + }, + "outstanding_count": { + "type": "integer", + "minimum": 0.0, + "title": "Outstanding Count", + "default": 0 + }, + "outstanding_bytes": { + "type": "integer", + "minimum": 0.0, + "title": "Outstanding Bytes", + "default": 0 + }, + "failing_count": { + "type": "integer", + "minimum": 0.0, + "title": "Failing Count", + "default": 0 + }, + "max_retry_reached": { + "type": "boolean", + "title": "Max Retry Reached", + "default": false + }, + "last_cycle_bytes": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Last Cycle Bytes" + }, + "last_cycle_seconds": { "anyOf": [ { "type": "integer", @@ -11980,85 +11857,179 @@ "type": "null" } ], - "title": "Rpo Target Seconds" + "title": "Last Cycle Seconds" }, - "consistency_group": { + "resyncing": { "type": "boolean", - "title": "Consistency Group", + "title": "Resyncing", "default": false } }, "type": "object", "required": [ - "policy_name", - "target_id" + "role", + "state" ], - "title": "PolicyParams" + "title": "ReplicationStatusDTO", + "description": "The typed steady-state replication status of one volume.\n\nServes what ``lvol_controller.get_replication_info`` computes, for the\nvolume's WHOLE replicated life \u2014 unlike ``ReplicationRelationshipDTO``,\nwhich only exists once a cutover or fail-over has created a relationship\nrecord. ``state: not_replicating, role: none`` is a valid answer, never a\n404, because the csi-addons adapter polls this on every reconcile." }, - "PoolHostParams": { + "ReplicationTargetDTO": { "properties": { - "host_nqn": { + "id": { "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?", - "title": "Host Nqn" + "format": "uuid", + "title": "Id" + }, + "cluster_id": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + }, + "target_name": { + "type": "string", + "title": "Target Name" + }, + "target_cluster_id": { + "type": "string", + "format": "uuid", + "title": "Target Cluster Id" + }, + "target_pool_uuid": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Target Pool Uuid" + }, + "timeout_sec": { + "type": "integer", + "minimum": 0.0, + "title": "Timeout Sec" + }, + "status": { + "type": "string", + "enum": [ + "active", + "inactive" + ], + "title": "Status" } }, "type": "object", "required": [ - "host_nqn" + "id", + "cluster_id", + "target_name", + "target_cluster_id", + "target_pool_uuid", + "timeout_sec", + "status" ], - "title": "PoolHostParams" + "title": "ReplicationTargetDTO" }, - "ReplicateLVolParams": { - "properties": { - "lvol_id": { - "type": "string", - "format": "uuid", - "title": "Lvol Id" + "RootModel_Union__CreateParams___CloneParams__": { + "anyOf": [ + { + "$ref": "#/components/schemas/_CreateParams" + }, + { + "$ref": "#/components/schemas/_CloneParams" } - }, - "type": "object", - "required": [ - "lvol_id" ], - "title": "ReplicateLVolParams" + "title": "RootModel[Union[_CreateParams, _CloneParams]]" }, - "ReplicatedGenerationDTO": { + "SnapshotDTO": { "properties": { - "group_seq": { + "id": { + "type": "string", + "format": "uuid", + "title": "Id" + }, + "name": { + "type": "string", + "title": "Name" + }, + "status": { + "type": "string", + "title": "Status" + }, + "health_check": { + "type": "boolean", + "title": "Health Check" + }, + "size": { "type": "integer", "minimum": 0.0, - "title": "Group Seq" + "title": "Size" }, - "members": { - "items": { - "$ref": "#/components/schemas/ReplicatedSnapshotDTO" - }, - "type": "array", - "title": "Members" + "used_size": { + "type": "integer", + "minimum": 0.0, + "title": "Used Size" + }, + "migrating": { + "type": "boolean", + "title": "Migrating" + }, + "lvol": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Lvol" + }, + "created_at": { + "type": "string", + "format": "date-time", + "title": "Created At" + }, + "group_id": { + "type": "string", + "title": "Group Id" + }, + "group_seq": { + "type": "integer", + "title": "Group Seq" } }, "type": "object", "required": [ - "group_seq", - "members" + "id", + "name", + "status", + "health_check", + "size", + "used_size", + "migrating", + "lvol", + "created_at", + "group_id", + "group_seq" ], - "title": "ReplicatedGenerationDTO", - "description": "One complete, fully replicated consistency-group generation, every\nmember addressed as a cloneable object on the secondary." + "title": "SnapshotDTO" }, - "ReplicatedSnapshotDTO": { + "StorageNodeDTO": { "properties": { - "snapshot_id": { + "id": { "type": "string", "format": "uuid", - "title": "Snapshot Id" + "title": "Id" }, "cluster_id": { "type": "string", "format": "uuid", "title": "Cluster Id" }, - "pool_id": { + "secondary_node_id": { "anyOf": [ { "type": "string", @@ -12068,9 +12039,9 @@ "type": "null" } ], - "title": "Pool Id" + "title": "Secondary Node Id" }, - "lvol_id": { + "tertiary_node_id": { "anyOf": [ { "type": "string", @@ -12080,468 +12051,537 @@ "type": "null" } ], - "title": "Lvol Id" + "title": "Tertiary Node Id" }, - "size": { + "status": { + "type": "string", + "enum": [ + "online", + "offline", + "suspended", + "in_shutdown", + "removed", + "in_restart", + "in_creation", + "unreachable", + "schedulable", + "down", + "in_removal", + "pending_removal", + "migrating_devices", + "migrating_lvols", + "removed_failed" + ], + "title": "Status" + }, + "uptime": { + "anyOf": [ + { + "type": "string", + "format": "duration" + }, + { + "type": "null" + } + ], + "title": "Uptime" + }, + "hostname": { + "type": "string", + "title": "Hostname" + }, + "host_nqn": { + "type": "string", + "title": "Host Nqn" + }, + "cpu_total_count": { "type": "integer", "minimum": 0.0, - "title": "Size" + "title": "Cpu Total Count" }, - "used_size": { + "cpu_spdk_count": { "type": "integer", "minimum": 0.0, - "title": "Used Size" + "title": "Cpu Spdk Count" }, - "created_at": { - "type": "string", - "format": "date-time", - "title": "Created At" + "cpu_poller_count": { + "type": "integer", + "minimum": 0.0, + "title": "Cpu Poller Count" }, - "group_id": { - "type": "string", - "title": "Group Id", - "default": "" + "memory": { + "type": "integer", + "minimum": 0.0, + "title": "Memory" + }, + "hugepage_memory": { + "type": "integer", + "minimum": 0.0, + "title": "Hugepage Memory" + }, + "spdk_mem": { + "type": "integer", + "title": "Spdk Mem" + }, + "lvols": { + "type": "integer", + "title": "Lvols" }, - "group_seq": { + "lvols_max": { "type": "integer", "minimum": 0.0, - "title": "Group Seq", - "default": 0 - } - }, - "type": "object", - "required": [ - "snapshot_id", - "cluster_id", - "size", - "used_size", - "created_at" - ], - "title": "ReplicatedSnapshotDTO", - "description": "A fully replicated snapshot on the secondary, addressed as a cloneable\nobject. ``lvol_id`` is the volume the snapshot belongs to on the\nSECONDARY cluster, not the source volume the caller asked about, because\nthat is the identity the ordinary CSI clone path resolves a\n``dataSource`` against." - }, - "ReplicationPolicyDTO": { - "properties": { - "id": { - "type": "string", - "format": "uuid", - "title": "Id" - }, - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" + "title": "Lvols Max" }, - "policy_name": { - "type": "string", - "title": "Policy Name" + "snapshots_max": { + "type": "integer", + "minimum": 0.0, + "title": "Snapshots Max" }, - "target_id": { - "type": "string", - "format": "uuid", - "title": "Target Id" + "rpc_port": { + "type": "integer", + "exclusiveMaximum": 65536.0, + "minimum": 0.0, + "title": "Rpc Port" }, - "interval_min": { + "lvol_subsys_port": { "type": "integer", + "exclusiveMaximum": 65536.0, "minimum": 0.0, - "title": "Interval Min" + "title": "Lvol Subsys Port" }, - "mode": { - "type": "string", - "enum": [ - "failover", - "migration" - ], - "title": "Mode" + "hublvol_port": { + "type": "integer", + "exclusiveMaximum": 65536.0, + "minimum": 0.0, + "title": "Hublvol Port" }, - "keep_replicated": { + "nvmf_port": { "type": "integer", - "title": "Keep Replicated" + "exclusiveMaximum": 65536.0, + "minimum": 0.0, + "title": "Nvmf Port" }, - "status": { + "mgmt_ip": { "type": "string", - "enum": [ - "active", - "inactive" - ], - "title": "Status" + "format": "ipv4", + "title": "Mgmt Ip" }, - "rpo_target_seconds": { + "health_check": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "boolean" }, { "type": "null" } ], - "title": "Rpo Target Seconds" - }, - "consistency_group": { - "type": "boolean", - "title": "Consistency Group", - "default": false + "title": "Health Check" }, - "group_node_id": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" - } - ], - "title": "Group Node Id" + "device_count": { + "type": "integer", + "title": "Device Count" }, - "group_lvs_name": { - "type": "string", - "title": "Group Lvs Name", - "default": "" + "online_device_count": { + "type": "integer", + "title": "Online Device Count" }, - "group_last_seq": { + "failure_domain": { "type": "integer", - "title": "Group Last Seq", - "default": 0 + "title": "Failure Domain" + }, + "capacity": { + "$ref": "#/components/schemas/CapacityStatDTO" } }, "type": "object", "required": [ "id", "cluster_id", - "policy_name", - "target_id", - "interval_min", - "mode", - "keep_replicated", - "status" + "secondary_node_id", + "status", + "uptime", + "hostname", + "host_nqn", + "cpu_total_count", + "cpu_spdk_count", + "cpu_poller_count", + "memory", + "hugepage_memory", + "spdk_mem", + "lvols", + "lvols_max", + "snapshots_max", + "rpc_port", + "lvol_subsys_port", + "hublvol_port", + "nvmf_port", + "mgmt_ip", + "health_check", + "device_count", + "online_device_count", + "failure_domain", + "capacity" ], - "title": "ReplicationPolicyDTO" + "title": "StorageNodeDTO" }, - "ReplicationRelationshipDTO": { + "StorageNodeParams": { "properties": { - "replication_id": { + "node_address": { "type": "string", - "format": "uuid", - "title": "Replication Id" + "title": "Node Address" }, - "source_lvol_id": { - "anyOf": [ - { - "type": "string", - "format": "uuid" - }, - { - "type": "null" - } - ], - "title": "Source Lvol Id" + "interface_name": { + "type": "string", + "title": "Interface Name" }, - "target_lvol_id": { + "max_snapshots": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "integer" }, { "type": "null" } ], - "title": "Target Lvol Id" + "title": "Max Snapshots", + "default": 500 }, - "source_cluster_id": { + "ha_jm": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "boolean" }, { "type": "null" } ], - "title": "Source Cluster Id" + "title": "Ha Jm", + "default": true }, - "target_cluster_id": { + "test_device": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "boolean" }, { "type": "null" } ], - "title": "Target Cluster Id" + "title": "Test Device", + "default": false }, - "target_pool_id": { + "spdk_image": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "string" }, { "type": "null" } ], - "title": "Target Pool Id" - }, - "mode": { - "type": "string", - "enum": [ - "failover", - "migration" - ], - "title": "Mode" + "title": "Spdk Image", + "default": "" }, - "state": { - "type": "string", - "enum": [ - "replicating", - "cutover_pending", - "cutover_done", - "failed_over" - ], - "title": "State" + "spdk_debug": { + "type": "boolean", + "title": "Spdk Debug", + "default": false }, - "direction": { - "type": "string", - "enum": [ - "to_target", - "to_source" - ], - "title": "Direction" + "data_nics": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Data Nics", + "default": [] }, - "target_nqn": { + "namespace": { "type": "string", - "title": "Target Nqn" - }, - "target_ns_id": { - "type": "integer", - "title": "Target Ns Id" - }, - "is_source": { - "type": "boolean", - "title": "Is Source" + "title": "Namespace", + "default": "default" }, - "active": { + "id_device_by_nqn": { "anyOf": [ { - "type": "string" + "type": "boolean" }, { "type": "null" } ], - "title": "Active" + "title": "Id Device By Nqn", + "default": false }, - "active_lvol_id": { + "jm_percent": { + "type": "integer", + "maximum": 100.0, + "minimum": 0.0, + "title": "Jm Percent", + "default": 3 + }, + "partitions": { + "type": "integer", + "title": "Partitions", + "default": 1 + }, + "iobuf_small_pool_count": { + "type": "integer", + "title": "Iobuf Small Pool Count", + "default": 0 + }, + "iobuf_large_pool_count": { + "type": "integer", + "title": "Iobuf Large Pool Count", + "default": 0 + }, + "cr_name": { + "type": "string", + "title": "Cr Name", + "default": "" + }, + "cr_namespace": { + "type": "string", + "title": "Cr Namespace", + "default": "" + }, + "cr_plural": { + "type": "string", + "title": "Cr Plural", + "default": "" + }, + "ha_jm_count": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "integer" }, { "type": "null" } ], - "title": "Active Lvol Id" - } - }, - "type": "object", - "required": [ - "replication_id", - "source_lvol_id", - "target_lvol_id", - "source_cluster_id", - "target_cluster_id", - "mode", - "state", - "direction", - "target_nqn", - "target_ns_id", - "is_source" - ], - "title": "ReplicationRelationshipDTO" - }, - "ReplicationStartParams": { - "properties": { - "replication_cluster_id": { + "title": "Ha Jm Count" + }, + "format_4k": { + "type": "boolean", + "title": "Format 4K", + "default": false + }, + "spdk_proxy_image": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "string" }, { "type": "null" } ], - "title": "Replication Cluster Id" + "title": "Spdk Proxy Image" }, - "mode": { + "spdk_sys_mem": { "anyOf": [ { - "type": "string", - "enum": [ - "failover", - "migration" - ] + "type": "string" }, { "type": "null" } ], - "title": "Mode" + "title": "Spdk Sys Mem" }, - "interval_min": { + "failure_domain": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "integer" }, { "type": "null" } ], - "title": "Interval Min" + "title": "Failure Domain" + }, + "expand": { + "type": "boolean", + "title": "Expand", + "default": false + }, + "force_format": { + "type": "boolean", + "title": "Force Format", + "default": false } }, "type": "object", - "title": "ReplicationStartParams" + "required": [ + "node_address", + "interface_name" + ], + "title": "StorageNodeParams" }, - "ReplicationStatusDTO": { + "StoragePoolDTO": { "properties": { - "role": { + "id": { "type": "string", - "enum": [ - "source", - "secondary", - "failed_over", - "none" - ], - "title": "Role" + "format": "uuid", + "title": "Id" }, - "state": { + "cluster_id": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + }, + "name": { + "type": "string", + "title": "Name" + }, + "status": { "type": "string", "enum": [ - "in_sync", - "replicating", - "lagging", - "degraded", - "error", - "not_replicating" + "active", + "inactive" ], - "title": "State" + "title": "Status" }, - "last_replicated_at": { - "anyOf": [ - { - "type": "string", - "format": "date-time" - }, - { - "type": "null" - } - ], - "title": "Last Replicated At" + "max_size": { + "type": "integer", + "minimum": 0.0, + "title": "Max Size" }, - "lag_seconds": { - "anyOf": [ - { - "type": "integer", - "minimum": 0.0 - }, - { - "type": "null" - } - ], - "title": "Lag Seconds" + "volume_max_size": { + "type": "integer", + "minimum": 0.0, + "title": "Volume Max Size" }, - "lag_budget_seconds": { + "max_rw_iops": { + "type": "integer", + "minimum": 0.0, + "title": "Max Rw Iops" + }, + "max_rw_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max Rw Mbytes" + }, + "max_r_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max R Mbytes" + }, + "max_w_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max W Mbytes" + }, + "capacity": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "$ref": "#/components/schemas/CapacityStatDTO" }, { "type": "null" } - ], - "title": "Lag Budget Seconds" + ] + }, + "dhchap": { + "type": "boolean", + "title": "Dhchap", + "default": false + }, + "allowed_hosts": { + "items": { + "type": "string", + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" + }, + "type": "array", + "title": "Allowed Hosts", + "default": [] + } + }, + "type": "object", + "required": [ + "id", + "cluster_id", + "name", + "status", + "max_size", + "volume_max_size", + "max_rw_iops", + "max_rw_mbytes", + "max_r_mbytes", + "max_w_mbytes", + "capacity" + ], + "title": "StoragePoolDTO" + }, + "StoragePoolParams": { + "properties": { + "name": { + "type": "string", + "title": "Name" + }, + "pool_max": { + "type": "integer", + "minimum": 0.0, + "title": "Pool Max", + "default": 0 + }, + "volume_max_size": { + "type": "integer", + "minimum": 0.0, + "title": "Volume Max Size", + "default": 0 }, - "outstanding_count": { + "max_rw_iops": { "type": "integer", "minimum": 0.0, - "title": "Outstanding Count", + "title": "Max Rw Iops", "default": 0 }, - "outstanding_bytes": { + "max_rw_mbytes": { "type": "integer", "minimum": 0.0, - "title": "Outstanding Bytes", + "title": "Max Rw Mbytes", "default": 0 }, - "failing_count": { + "max_r_mbytes": { "type": "integer", "minimum": 0.0, - "title": "Failing Count", + "title": "Max R Mbytes", "default": 0 }, - "max_retry_reached": { + "max_w_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max W Mbytes", + "default": 0 + }, + "dhchap": { "type": "boolean", - "title": "Max Retry Reached", + "title": "Dhchap", "default": false }, - "last_cycle_bytes": { - "anyOf": [ - { - "type": "integer", - "minimum": 0.0 - }, - { - "type": "null" - } - ], - "title": "Last Cycle Bytes" + "cr_name": { + "type": "string", + "title": "Cr Name", + "default": "" }, - "last_cycle_seconds": { - "anyOf": [ - { - "type": "integer", - "minimum": 0.0 - }, - { - "type": "null" - } - ], - "title": "Last Cycle Seconds" + "cr_namespace": { + "type": "string", + "title": "Cr Namespace", + "default": "" }, - "resyncing": { - "type": "boolean", - "title": "Resyncing", - "default": false + "cr_plural": { + "type": "string", + "title": "Cr Plural", + "default": "" } }, "type": "object", "required": [ - "role", - "state" + "name" ], - "title": "ReplicationStatusDTO", - "description": "The typed steady-state replication status of one volume.\n\nServes what ``lvol_controller.get_replication_info`` computes, for the\nvolume's WHOLE replicated life \u2014 unlike ``ReplicationRelationshipDTO``,\nwhich only exists once a cutover or fail-over has created a relationship\nrecord. ``state: not_replicating, role: none`` is a valid answer, never a\n404, because the csi-addons adapter polls this on every reconcile." + "title": "StoragePoolParams" }, - "ReplicationTargetDTO": { + "TargetParams": { "properties": { - "id": { - "type": "string", - "format": "uuid", - "title": "Id" - }, - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - }, "target_name": { "type": "string", "title": "Target Name" @@ -12551,7 +12591,7 @@ "format": "uuid", "title": "Target Cluster Id" }, - "target_pool_uuid": { + "target_pool_id": { "anyOf": [ { "type": "string", @@ -12561,512 +12601,505 @@ "type": "null" } ], - "title": "Target Pool Uuid" + "title": "Target Pool Id" }, "timeout_sec": { - "type": "integer", - "minimum": 0.0, - "title": "Timeout Sec" - }, - "status": { - "type": "string", - "enum": [ - "active", - "inactive" + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } ], - "title": "Status" + "title": "Timeout Sec" } }, "type": "object", "required": [ - "id", - "cluster_id", "target_name", - "target_cluster_id", - "target_pool_uuid", - "timeout_sec", - "status" - ], - "title": "ReplicationTargetDTO" - }, - "RootModel_Union__CreateParams___CloneParams__": { - "anyOf": [ - { - "$ref": "#/components/schemas/_CreateParams" - }, - { - "$ref": "#/components/schemas/_CloneParams" - } - ], - "title": "RootModel[Union[_CreateParams, _CloneParams]]" - }, - "S3Credentials": { - "properties": { - "access_key_id": { - "type": "string", - "format": "password", - "title": "Access Key Id", - "writeOnly": true - }, - "secret_access_key": { - "type": "string", - "format": "password", - "title": "Secret Access Key", - "writeOnly": true - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "access_key_id", - "secret_access_key" - ], - "title": "S3Credentials", - "description": "A static key pair.\n\nA pair rather than two independent fields, so \"access key set, secret\nmissing\" is unrepresentable instead of something a validator has to catch." - }, - "SecondaryTarget": { - "type": "integer", - "enum": [ - 0, - 1 + "target_cluster_id" ], - "title": "SecondaryTarget", - "description": "The kind of secondary store, numbered as the data plane's RPC expects." + "title": "TargetParams" }, - "SnapshotDTO": { + "TaskDTO": { "properties": { "id": { "type": "string", "format": "uuid", "title": "Id" }, - "name": { - "type": "string", - "title": "Name" - }, - "status": { + "cluster_id": { "type": "string", - "title": "Status" - }, - "health_check": { - "type": "boolean", - "title": "Health Check" - }, - "size": { - "type": "integer", - "minimum": 0.0, - "title": "Size" - }, - "used_size": { - "type": "integer", - "minimum": 0.0, - "title": "Used Size" + "format": "uuid", + "title": "Cluster Id" }, - "migrating": { - "type": "boolean", - "title": "Migrating" + "device_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Device Id" }, - "lvol": { + "storage_node_id": { "anyOf": [ { - "type": "string" + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Lvol" + "title": "Storage Node Id" }, - "created_at": { + "status": { "type": "string", - "format": "date-time", - "title": "Created At" + "enum": [ + "new", + "running", + "suspended", + "done" + ], + "title": "Status" }, - "group_id": { - "type": "string", - "title": "Group Id" + "canceled": { + "type": "boolean", + "title": "Canceled" }, - "group_seq": { - "type": "integer", - "title": "Group Seq" - } - }, - "type": "object", - "required": [ - "id", - "name", - "status", - "health_check", - "size", - "used_size", - "migrating", - "lvol", - "created_at", - "group_id", - "group_seq" - ], - "title": "SnapshotDTO" - }, - "StorageNodeDTO": { - "properties": { - "id": { + "function_name": { "type": "string", - "format": "uuid", - "title": "Id" + "enum": [ + "device_restart", + "node_restart", + "device_migration", + "failed_device_migration", + "new_device_migration", + "node_add", + "node_removal", + "port_allow", + "balancing_on_restart", + "balancing_on_dev_rem", + "balancing_on_dev_add", + "jc_comp_resume", + "snapshot_replication", + "lvol_sync_del", + "lvol_sync_op", + "lvol_migration", + "lvol_batch_migration", + "s3_backup", + "s3_backup_restore", + "s3_backup_merge", + "cluster_expand", + "replication_final", + "fdb_backup" + ], + "title": "Function Name" }, - "cluster_id": { + "function_params": { + "additionalProperties": true, + "type": "object", + "title": "Function Params" + }, + "function_result": { "type": "string", - "format": "uuid", - "title": "Cluster Id" + "title": "Function Result" }, - "secondary_node_id": { + "max_retry": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Secondary Node Id" + "title": "Max Retry" }, - "tertiary_node_id": { + "retry": { + "type": "integer", + "minimum": 0.0, + "title": "Retry" + } + }, + "type": "object", + "required": [ + "id", + "cluster_id", + "device_id", + "storage_node_id", + "status", + "canceled", + "function_name", + "function_params", + "function_result", + "max_retry", + "retry" + ], + "title": "TaskDTO" + }, + "UpdatableClusterParameters": { + "properties": { + "name": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "string" }, { "type": "null" } ], - "title": "Tertiary Node Id" - }, - "status": { - "type": "string", - "enum": [ - "online", - "offline", - "suspended", - "in_shutdown", - "removed", - "in_restart", - "in_creation", - "unreachable", - "schedulable", - "down", - "in_removal", - "pending_removal", - "migrating_devices", - "migrating_lvols", - "removed_failed" - ], - "title": "Status" - }, - "uptime": { + "title": "Name" + } + }, + "type": "object", + "title": "UpdatableClusterParameters" + }, + "UpdatableLVolParams": { + "properties": { + "name": { "anyOf": [ { - "type": "string", - "format": "duration" + "type": "string" }, { "type": "null" } ], - "title": "Uptime" - }, - "hostname": { - "type": "string", - "title": "Hostname" - }, - "host_nqn": { - "type": "string", - "title": "Host Nqn" - }, - "cpu_total_count": { - "type": "integer", - "minimum": 0.0, - "title": "Cpu Total Count" - }, - "cpu_spdk_count": { - "type": "integer", - "minimum": 0.0, - "title": "Cpu Spdk Count" - }, - "cpu_poller_count": { - "type": "integer", - "minimum": 0.0, - "title": "Cpu Poller Count" - }, - "memory": { - "type": "integer", - "minimum": 0.0, - "title": "Memory" - }, - "hugepage_memory": { - "type": "integer", - "minimum": 0.0, - "title": "Hugepage Memory" - }, - "spdk_mem": { - "type": "integer", - "title": "Spdk Mem" - }, - "lvols": { - "type": "integer", - "title": "Lvols" - }, - "lvols_max": { - "type": "integer", - "minimum": 0.0, - "title": "Lvols Max" - }, - "snapshots_max": { - "type": "integer", - "minimum": 0.0, - "title": "Snapshots Max" + "title": "Name" }, - "rpc_port": { + "max_rw_iops": { "type": "integer", - "exclusiveMaximum": 65536.0, "minimum": 0.0, - "title": "Rpc Port" + "title": "Max Rw Iops", + "default": 0 }, - "lvol_subsys_port": { + "max_rw_mbytes": { "type": "integer", - "exclusiveMaximum": 65536.0, "minimum": 0.0, - "title": "Lvol Subsys Port" + "title": "Max Rw Mbytes", + "default": 0 }, - "hublvol_port": { + "max_r_mbytes": { "type": "integer", - "exclusiveMaximum": 65536.0, "minimum": 0.0, - "title": "Hublvol Port" + "title": "Max R Mbytes", + "default": 0 }, - "nvmf_port": { + "max_w_mbytes": { "type": "integer", - "exclusiveMaximum": 65536.0, "minimum": 0.0, - "title": "Nvmf Port" - }, - "mgmt_ip": { - "type": "string", - "format": "ipv4", - "title": "Mgmt Ip" + "title": "Max W Mbytes", + "default": 0 }, - "health_check": { + "size": { "anyOf": [ { - "type": "boolean" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Health Check" - }, - "device_count": { - "type": "integer", - "title": "Device Count" - }, - "online_device_count": { - "type": "integer", - "title": "Online Device Count" - }, - "failure_domain": { - "type": "integer", - "title": "Failure Domain" + "title": "Size" }, - "capacity": { - "$ref": "#/components/schemas/CapacityStatDTO" + "replication_policy_id": { + "anyOf": [ + { + "type": "string", + "format": "uuid" + }, + { + "type": "null" + } + ], + "title": "Replication Policy Id" } }, "type": "object", - "required": [ - "id", - "cluster_id", - "secondary_node_id", - "status", - "uptime", - "hostname", - "host_nqn", - "cpu_total_count", - "cpu_spdk_count", - "cpu_poller_count", - "memory", - "hugepage_memory", - "spdk_mem", - "lvols", - "lvols_max", - "snapshots_max", - "rpc_port", - "lvol_subsys_port", - "hublvol_port", - "nvmf_port", - "mgmt_ip", - "health_check", - "device_count", - "online_device_count", - "failure_domain", - "capacity" - ], - "title": "StorageNodeDTO" + "title": "UpdatableLVolParams" }, - "StorageNodeParams": { + "UpdatableStoragePoolParams": { "properties": { - "node_address": { - "type": "string", - "title": "Node Address" + "name": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Name" }, - "interface_name": { - "type": "string", - "title": "Interface Name" + "max_size": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Max Size" }, - "max_snapshots": { + "volume_max_size": { "anyOf": [ { - "type": "integer" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Max Snapshots", - "default": 500 + "title": "Volume Max Size" }, - "ha_jm": { + "max_rw_iops": { "anyOf": [ { - "type": "boolean" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Ha Jm", - "default": true + "title": "Max Rw Iops" }, - "test_device": { + "max_rw_mbytes": { "anyOf": [ { - "type": "boolean" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Test Device", - "default": false + "title": "Max Rw Mbytes" }, - "spdk_image": { + "max_r_mbytes": { "anyOf": [ { - "type": "string" + "type": "integer", + "minimum": 0.0 }, { "type": "null" } ], - "title": "Spdk Image", - "default": "" + "title": "Max R Mbytes" }, - "spdk_debug": { - "type": "boolean", - "title": "Spdk Debug", - "default": false + "max_w_mbytes": { + "anyOf": [ + { + "type": "integer", + "minimum": 0.0 + }, + { + "type": "null" + } + ], + "title": "Max W Mbytes" }, - "data_nics": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Data Nics", - "default": [] + "lvols_cr_name": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Lvols Cr Name" }, - "namespace": { - "type": "string", - "title": "Namespace", - "default": "default" + "lvols_cr_namespace": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "title": "Lvols Cr Namespace" }, - "id_device_by_nqn": { + "lvols_cr_plural": { "anyOf": [ { - "type": "boolean" + "type": "string" }, { "type": "null" } ], - "title": "Id Device By Nqn", - "default": false + "title": "Lvols Cr Plural" + } + }, + "type": "object", + "title": "UpdatableStoragePoolParams" + }, + "ValidationError": { + "properties": { + "loc": { + "items": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "integer" + } + ] + }, + "type": "array", + "title": "Location" }, - "jm_percent": { + "msg": { + "type": "string", + "title": "Message" + }, + "type": { + "type": "string", + "title": "Error Type" + }, + "input": { + "title": "Input" + }, + "ctx": { + "type": "object", + "title": "Context" + } + }, + "type": "object", + "required": [ + "loc", + "msg", + "type" + ], + "title": "ValidationError" + }, + "VolumeDTO": { + "properties": { + "id": { + "type": "string", + "format": "uuid", + "title": "Id" + }, + "cluster_id": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" + }, + "storage_node_id": { + "type": "string", + "format": "uuid", + "title": "Storage Node Id" + }, + "name": { + "type": "string", + "title": "Name" + }, + "status": { + "type": "string", + "title": "Status" + }, + "health_check": { + "type": "boolean", + "title": "Health Check" + }, + "io_error": { + "type": "boolean", + "title": "Io Error" + }, + "migrating": { + "type": "boolean", + "title": "Migrating" + }, + "nqn": { + "type": "string", + "title": "Nqn" + }, + "hostname": { + "type": "string", + "title": "Hostname" + }, + "priority_class": { "type": "integer", - "maximum": 100.0, "minimum": 0.0, - "title": "Jm Percent", - "default": 3 + "title": "Priority Class" }, - "partitions": { + "namespace": { + "type": "string", + "title": "Namespace" + }, + "fabric": { + "type": "string", + "title": "Fabric" + }, + "nodes": { + "items": { + "type": "string" + }, + "type": "array", + "title": "Nodes" + }, + "port": { "type": "integer", - "title": "Partitions", - "default": 1 + "exclusiveMaximum": 65536.0, + "minimum": 0.0, + "title": "Port" }, - "iobuf_small_pool_count": { + "size": { "type": "integer", - "title": "Iobuf Small Pool Count", - "default": 0 + "minimum": 0.0, + "title": "Size" }, - "iobuf_large_pool_count": { + "ndcs": { "type": "integer", - "title": "Iobuf Large Pool Count", - "default": 0 + "title": "Ndcs" }, - "cr_name": { + "npcs": { + "type": "integer", + "title": "Npcs" + }, + "pool_uuid": { "type": "string", - "title": "Cr Name", - "default": "" + "title": "Pool Uuid" }, - "cr_namespace": { + "pool_name": { "type": "string", - "title": "Cr Namespace", - "default": "" + "title": "Pool Name" }, - "cr_plural": { + "pvc_name": { "type": "string", - "title": "Cr Plural", + "title": "Pvc Name", "default": "" }, - "ha_jm_count": { - "anyOf": [ - { - "type": "integer" - }, - { - "type": "null" - } - ], - "title": "Ha Jm Count" + "snapshot_name": { + "type": "string", + "title": "Snapshot Name", + "default": "" }, - "format_4k": { - "type": "boolean", - "title": "Format 4K", - "default": false + "blobid": { + "type": "integer", + "title": "Blobid" }, - "spdk_proxy_image": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Spdk Proxy Image" + "ns_id": { + "type": "integer", + "title": "Ns Id" }, - "spdk_sys_mem": { + "cloned_from": { "anyOf": [ { "type": "string" @@ -13075,70 +13108,20 @@ "type": "null" } ], - "title": "Spdk Sys Mem" - }, - "failure_domain": { - "anyOf": [ - { - "type": "integer" - }, - { - "type": "null" - } - ], - "title": "Failure Domain" + "title": "Cloned From" }, - "expand": { + "high_availability": { "type": "boolean", - "title": "Expand", - "default": false + "title": "High Availability" }, - "force_format": { + "do_replicate": { "type": "boolean", - "title": "Force Format", + "title": "Do Replicate", "default": false - } - }, - "type": "object", - "required": [ - "node_address", - "interface_name" - ], - "title": "StorageNodeParams" - }, - "StoragePoolDTO": { - "properties": { - "id": { - "type": "string", - "format": "uuid", - "title": "Id" - }, - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" - }, - "name": { - "type": "string", - "title": "Name" - }, - "status": { - "type": "string", - "enum": [ - "active", - "inactive" - ], - "title": "Status" - }, - "max_size": { - "type": "integer", - "minimum": 0.0, - "title": "Max Size" }, - "volume_max_size": { + "max_namespace_per_subsys": { "type": "integer", - "minimum": 0.0, - "title": "Volume Max Size" + "title": "Max Namespace Per Subsys" }, "max_rw_iops": { "type": "integer", @@ -13160,376 +13143,359 @@ "minimum": 0.0, "title": "Max W Mbytes" }, + "allowed_hosts": { + "items": { + "type": "string", + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" + }, + "type": "array", + "title": "Allowed Hosts" + }, + "policy": { + "type": "string", + "title": "Policy" + }, "capacity": { + "$ref": "#/components/schemas/CapacityStatDTO" + }, + "rep_info": { "anyOf": [ { - "$ref": "#/components/schemas/CapacityStatDTO" + "additionalProperties": true, + "type": "object" }, { "type": "null" } - ] + ], + "title": "Rep Info" }, - "dhchap": { + "from_source": { "type": "boolean", - "title": "Dhchap", - "default": false + "title": "From Source", + "default": true }, - "allowed_hosts": { - "items": { - "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" - }, - "type": "array", - "title": "Allowed Hosts", - "default": [] + "group_id": { + "type": "string", + "title": "Group Id", + "default": "" + }, + "group_seq": { + "type": "integer", + "title": "Group Seq", + "default": 0 } }, "type": "object", "required": [ "id", "cluster_id", + "storage_node_id", "name", "status", - "max_size", - "volume_max_size", + "health_check", + "io_error", + "migrating", + "nqn", + "hostname", + "priority_class", + "namespace", + "fabric", + "nodes", + "port", + "size", + "ndcs", + "npcs", + "pool_uuid", + "pool_name", + "blobid", + "ns_id", + "cloned_from", + "high_availability", + "max_namespace_per_subsys", "max_rw_iops", "max_rw_mbytes", "max_r_mbytes", "max_w_mbytes", + "allowed_hosts", + "policy", "capacity" ], - "title": "StoragePoolDTO" + "title": "VolumeDTO" }, - "StoragePoolParams": { + "_AddHostParams": { "properties": { - "name": { - "type": "string", - "title": "Name" - }, - "pool_max": { - "type": "integer", - "minimum": 0.0, - "title": "Pool Max", - "default": 0 - }, - "volume_max_size": { - "type": "integer", - "minimum": 0.0, - "title": "Volume Max Size", - "default": 0 - }, - "max_rw_iops": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Iops", - "default": 0 - }, - "max_rw_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Mbytes", - "default": 0 - }, - "max_r_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max R Mbytes", - "default": 0 - }, - "max_w_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max W Mbytes", - "default": 0 - }, - "dhchap": { - "type": "boolean", - "title": "Dhchap", - "default": false - }, - "cr_name": { + "host_nqn": { "type": "string", - "title": "Cr Name", - "default": "" - }, - "cr_namespace": { + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?", + "title": "Host Nqn" + } + }, + "type": "object", + "required": [ + "host_nqn" + ], + "title": "_AddHostParams" + }, + "_AttachParams": { + "properties": { + "target_type": { "type": "string", - "title": "Cr Namespace", - "default": "" + "title": "Target Type" }, - "cr_plural": { + "target_id": { "type": "string", - "title": "Cr Plural", - "default": "" + "title": "Target Id" } }, "type": "object", "required": [ - "name" + "target_type", + "target_id" ], - "title": "StoragePoolParams" + "title": "_AttachParams" }, - "TargetParams": { + "_BackupSnapshotParams": { "properties": { - "target_name": { + "snapshot_id": { "type": "string", - "title": "Target Name" - }, - "target_cluster_id": { + "title": "Snapshot Id" + } + }, + "type": "object", + "required": [ + "snapshot_id" + ], + "title": "_BackupSnapshotParams" + }, + "_CloneParams": { + "properties": { + "name": { "type": "string", - "format": "uuid", - "title": "Target Cluster Id" + "title": "Name" }, - "target_pool_id": { + "snapshot_id": { "anyOf": [ { "type": "string", - "format": "uuid" + "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$" }, { "type": "null" } ], - "title": "Target Pool Id" + "title": "Snapshot Id" }, - "timeout_sec": { + "size": { + "type": "integer", + "minimum": 0.0, + "title": "Size", + "default": 0 + }, + "pvc_name": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Timeout Sec" - } - }, - "type": "object", - "required": [ - "target_name", - "target_cluster_id" - ], - "title": "TargetParams" - }, - "TaskDTO": { - "properties": { - "id": { - "type": "string", - "format": "uuid", - "title": "Id" - }, - "cluster_id": { - "type": "string", - "format": "uuid", - "title": "Cluster Id" + "title": "Pvc Name" }, - "device_id": { + "pvc_namespace": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "string" }, { "type": "null" } ], - "title": "Device Id" + "title": "Pvc Namespace" }, - "storage_node_id": { + "delete_snap_on_lvol_delete": { + "type": "boolean", + "title": "Delete Snap On Lvol Delete", + "default": false + }, + "consistency_group": { "anyOf": [ { - "type": "string", - "format": "uuid" + "type": "string" }, { "type": "null" } ], - "title": "Storage Node Id" + "title": "Consistency Group" + } + }, + "type": "object", + "required": [ + "name", + "snapshot_id" + ], + "title": "_CloneParams" + }, + "_ContinueParams": { + "properties": { + "max_retries": { + "type": "integer", + "title": "Max Retries", + "default": 10 }, - "status": { + "deadline_seconds": { + "type": "integer", + "title": "Deadline Seconds", + "default": 14400 + } + }, + "type": "object", + "title": "_ContinueParams" + }, + "_CreateParams": { + "properties": { + "name": { "type": "string", - "enum": [ - "new", - "running", - "suspended", - "done" - ], - "title": "Status" + "title": "Name" }, - "canceled": { - "type": "boolean", - "title": "Canceled" + "size": { + "type": "integer", + "minimum": 0.0, + "title": "Size" }, - "function_name": { - "type": "string", - "enum": [ - "device_restart", - "node_restart", - "device_migration", - "failed_device_migration", - "new_device_migration", - "node_add", - "node_removal", - "port_allow", - "balancing_on_restart", - "balancing_on_dev_rem", - "balancing_on_dev_add", - "jc_comp_resume", - "snapshot_replication", - "lvol_sync_del", - "lvol_sync_op", - "lvol_migration", - "lvol_batch_migration", - "s3_backup", - "s3_backup_restore", - "s3_backup_merge", - "cluster_expand", - "replication_final", - "fdb_backup" - ], - "title": "Function Name" + "max_rw_iops": { + "type": "integer", + "minimum": 0.0, + "title": "Max Rw Iops", + "default": 0 }, - "function_params": { - "additionalProperties": true, - "type": "object", - "title": "Function Params" + "max_rw_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max Rw Mbytes", + "default": 0 }, - "function_result": { - "type": "string", - "title": "Function Result" + "max_r_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max R Mbytes", + "default": 0 }, - "max_retry": { + "max_w_mbytes": { + "type": "integer", + "minimum": 0.0, + "title": "Max W Mbytes", + "default": 0 + }, + "ha_type": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string", + "enum": [ + "single", + "ha" + ] }, { "type": "null" } ], - "title": "Max Retry" + "title": "Ha Type" }, - "retry": { - "type": "integer", - "minimum": 0.0, - "title": "Retry" - } - }, - "type": "object", - "required": [ - "id", - "cluster_id", - "device_id", - "storage_node_id", - "status", - "canceled", - "function_name", - "function_params", - "function_result", - "max_retry", - "retry" - ], - "title": "TaskDTO" - }, - "UnresolvedBackupConfig": { - "properties": { - "bucket_name": { + "host_id": { "anyOf": [ { - "type": "string", - "minLength": 1 + "type": "string" }, { "type": "null" } ], - "title": "Bucket Name" + "title": "Host Id" }, - "region": { + "priority_class": { + "type": "integer", + "enum": [ + 0, + 1 + ], + "title": "Priority Class", + "default": 0 + }, + "namespaced": { "anyOf": [ { - "type": "string", - "minLength": 1 + "type": "boolean" }, { "type": "null" } ], - "title": "Region" + "title": "Namespaced", + "default": false }, - "endpoint": { + "pvc_name": { "anyOf": [ { - "type": "string", - "maxLength": 2083, - "minLength": 1, - "format": "uri" + "type": "string" }, { "type": "null" } ], - "title": "Endpoint" + "title": "Pvc Name" }, - "secondary_target": { - "$ref": "#/components/schemas/SecondaryTarget", + "ndcs": { + "type": "integer", + "minimum": 0.0, + "title": "Ndcs", "default": 0 }, - "with_compression": { - "type": "boolean", - "title": "With Compression", - "default": false - }, - "snapshot_backups": { - "type": "boolean", - "title": "Snapshot Backups", - "default": true - }, - "verify_tls": { - "type": "boolean", - "title": "Verify Tls", - "default": true - }, - "use_path_style": { - "type": "boolean", - "title": "Use Path Style", - "default": true - }, - "credentials": { + "npcs": { + "type": "integer", + "minimum": 0.0, + "title": "Npcs", + "default": 0 + }, + "allowed_hosts": { "anyOf": [ { - "$ref": "#/components/schemas/S3Credentials" + "items": { + "type": "string", + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" + }, + "type": "array" }, { "type": "null" } - ] + ], + "title": "Allowed Hosts" }, - "s3_thread_pool_size": { + "fabric": { + "type": "string", + "title": "Fabric", + "default": "tcp" + }, + "max_namespace_per_subsys": { "anyOf": [ { - "type": "integer", - "minimum": 1.0 + "type": "integer" }, { "type": "null" } ], - "title": "S3 Thread Pool Size" - } - }, - "additionalProperties": false, - "type": "object", - "title": "UnresolvedBackupConfig", - "description": "A backup configuration as a caller can state it, before a cluster resolves it.\n\nIdentical to :class:`BackupConfig` except that ``bucket_name`` may be absent,\nbecause the default is derived from a cluster id that does not exist yet at\ncluster-create time (``Cluster.default_backup_bucket_name``). Hand one to\n``Cluster.set_backup_config``, which resolves it; nothing further down ever\nsees a configuration without a bucket.\n\nWhich is also why an instance must be turned back into a plain dict at the\nboundary it arrived on rather than passed along as a ``BackupConfig``: the\ninherited :meth:`location` cannot produce a location for a bucket nobody has\nnamed yet." - }, - "UpdatableClusterParameters": { - "properties": { - "name": { + "title": "Max Namespace Per Subsys" + }, + "do_replicate": { + "type": "boolean", + "title": "Do Replicate", + "default": false + }, + "replication_cluster_id": { "anyOf": [ { "type": "string" @@ -13538,15 +13504,9 @@ "type": "null" } ], - "title": "Name" - } - }, - "type": "object", - "title": "UpdatableClusterParameters" - }, - "UpdatableLVolParams": { - "properties": { - "name": { + "title": "Replication Cluster Id" + }, + "replication_policy": { "anyOf": [ { "type": "string" @@ -13555,146 +13515,241 @@ "type": "null" } ], - "title": "Name" - }, - "max_rw_iops": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Iops", - "default": 0 - }, - "max_rw_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Mbytes", - "default": 0 - }, - "max_r_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max R Mbytes", - "default": 0 - }, - "max_w_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max W Mbytes", - "default": 0 + "title": "Replication Policy" }, - "size": { + "consistency_group": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Size" + "title": "Consistency Group" }, - "replication_policy_id": { + "encrypt": { + "type": "boolean", + "title": "Encrypt", + "default": false + } + }, + "type": "object", + "required": [ + "name", + "size" + ], + "title": "_CreateParams" + }, + "_MigrationParams": { + "properties": { + "target_node_id": { + "type": "string", + "format": "uuid", + "title": "Target Node Id" + }, + "ctrl_loss_tmo": { + "type": "integer", + "title": "Ctrl Loss Tmo", + "default": 3600 + }, + "host_nqn": { "anyOf": [ { "type": "string", - "format": "uuid" + "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" }, { "type": "null" } ], - "title": "Replication Policy Id" + "title": "Host Nqn" } }, "type": "object", - "title": "UpdatableLVolParams" + "required": [ + "target_node_id" + ], + "title": "_MigrationParams" }, - "UpdatableStoragePoolParams": { + "_PolicyCreateParams": { "properties": { "name": { + "type": "string", + "title": "Name" + }, + "versions": { "anyOf": [ { - "type": "string" + "type": "integer" }, { "type": "null" } ], - "title": "Name" + "title": "Versions", + "default": 0 }, - "max_size": { + "age": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Max Size" + "title": "Age", + "default": "" }, - "volume_max_size": { + "schedule": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Volume Max Size" + "title": "Schedule", + "default": "" + } + }, + "type": "object", + "required": [ + "name" + ], + "title": "_PolicyCreateParams" + }, + "_ReplicationParams": { + "properties": { + "snapshot_replication_target_cluster": { + "type": "string", + "title": "Snapshot Replication Target Cluster" }, - "max_rw_iops": { + "snapshot_replication_timeout": { + "type": "integer", + "title": "Snapshot Replication Timeout", + "default": 0 + }, + "target_pool": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Max Rw Iops" + "title": "Target Pool" + } + }, + "type": "object", + "required": [ + "snapshot_replication_target_cluster" + ], + "title": "_ReplicationParams" + }, + "_RestartParams": { + "properties": { + "force": { + "type": "boolean", + "title": "Force", + "default": false }, - "max_rw_mbytes": { + "reattach_volume": { + "type": "boolean", + "title": "Reattach Volume", + "default": false + }, + "node_address": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Max Rw Mbytes" + "title": "Node Address" }, - "max_r_mbytes": { + "new_ssd_pcie": { + "items": { + "type": "string" + }, + "type": "array", + "title": "New Ssd Pcie", + "default": [] + } + }, + "type": "object", + "title": "_RestartParams" + }, + "_RestoreParams": { + "properties": { + "backup_id": { + "type": "string", + "title": "Backup Id" + }, + "lvol_name": { + "type": "string", + "title": "Lvol Name" + }, + "pool": { + "type": "string", + "title": "Pool" + }, + "target_node_id": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "type": "string" }, { "type": "null" } ], - "title": "Max R Mbytes" + "title": "Target Node Id" }, - "max_w_mbytes": { + "s3_credentials": { "anyOf": [ { - "type": "integer", - "minimum": 0.0 + "$ref": "#/components/schemas/S3Credentials" }, { "type": "null" } - ], - "title": "Max W Mbytes" + ] + } + }, + "type": "object", + "required": [ + "backup_id", + "lvol_name", + "pool" + ], + "title": "_RestoreParams" + }, + "_SnapshotParams": { + "properties": { + "name": { + "type": "string", + "title": "Name" }, - "lvols_cr_name": { + "backup": { + "type": "boolean", + "title": "Backup", + "default": false + } + }, + "type": "object", + "required": [ + "name" + ], + "title": "_SnapshotParams" + }, + "_UpdateParams": { + "properties": { + "management_image": { "anyOf": [ { "type": "string" @@ -13703,9 +13758,9 @@ "type": "null" } ], - "title": "Lvols Cr Name" + "title": "Management Image" }, - "lvols_cr_namespace": { + "spdk_image": { "anyOf": [ { "type": "string" @@ -13714,479 +13769,561 @@ "type": "null" } ], - "title": "Lvols Cr Namespace" + "title": "Spdk Image" }, - "lvols_cr_plural": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Lvols Cr Plural" + "restart": { + "type": "boolean", + "title": "Restart", + "default": false } }, "type": "object", - "title": "UpdatableStoragePoolParams" + "required": [ + "management_image", + "spdk_image" + ], + "title": "_UpdateParams" }, - "ValidationError": { + "ConsistencyGroupResolutionDTO": { "properties": { - "loc": { - "items": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "integer" - } - ] - }, - "type": "array", - "title": "Location" + "cluster_id": { + "type": "string", + "title": "Cluster Id" }, - "msg": { + "group_id": { "type": "string", - "title": "Message" + "title": "Group Id" }, - "type": { + "active_cluster_id": { "type": "string", - "title": "Error Type" + "title": "Active Cluster Id", + "default": "" }, - "input": { - "title": "Input" + "active_group_id": { + "type": "string", + "title": "Active Group Id", + "default": "" }, - "ctx": { - "type": "object", - "title": "Context" + "members": { + "items": { + "$ref": "#/components/schemas/ConsistencyGroupLineageMemberDTO" + }, + "type": "array", + "title": "Members", + "default": [] } }, "type": "object", "required": [ - "loc", - "msg", - "type" + "cluster_id", + "group_id" ], - "title": "ValidationError" + "title": "ConsistencyGroupResolutionDTO", + "description": "Where a consistency group's data lives now (replication_policy_controller.\nresolve_group). ``active_*`` are empty when no group holds a live member." }, - "VolumeDTO": { + "ConsistencyGroupLineageMemberDTO": { "properties": { - "id": { + "origin_handle": { "type": "string", - "format": "uuid", - "title": "Id" + "title": "Origin Handle" }, - "cluster_id": { + "active_handle": { "type": "string", - "format": "uuid", - "title": "Cluster Id" - }, - "storage_node_id": { + "title": "Active Handle" + } + }, + "type": "object", + "required": [ + "origin_handle", + "active_handle" + ], + "title": "ConsistencyGroupLineageMemberDTO", + "description": "One protected volume of a consistency group: the handle its\nPersistentVolume carries and the volume serving its data now." + }, + "BackupConfig-Input": { + "properties": { + "bucket_name": { "type": "string", - "format": "uuid", - "title": "Storage Node Id" + "minLength": 1, + "title": "Bucket Name" }, - "name": { - "type": "string", - "title": "Name" + "region": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "null" + } + ], + "title": "Region" }, - "status": { - "type": "string", - "title": "Status" + "endpoint": { + "anyOf": [ + { + "type": "string", + "maxLength": 2083, + "minLength": 1, + "format": "uri" + }, + { + "type": "null" + } + ], + "title": "Endpoint" }, - "health_check": { - "type": "boolean", - "title": "Health Check" + "secondary_target": { + "$ref": "#/components/schemas/SecondaryTarget", + "default": 0 }, - "io_error": { + "with_compression": { "type": "boolean", - "title": "Io Error" + "title": "With Compression", + "default": false }, - "migrating": { + "snapshot_backups": { "type": "boolean", - "title": "Migrating" - }, - "nqn": { - "type": "string", - "title": "Nqn" - }, - "hostname": { - "type": "string", - "title": "Hostname" - }, - "priority_class": { - "type": "integer", - "minimum": 0.0, - "title": "Priority Class" - }, - "namespace": { - "type": "string", - "title": "Namespace" - }, - "fabric": { - "type": "string", - "title": "Fabric" - }, - "nodes": { - "items": { - "type": "string" - }, - "type": "array", - "title": "Nodes" - }, - "port": { - "type": "integer", - "exclusiveMaximum": 65536.0, - "minimum": 0.0, - "title": "Port" - }, - "size": { - "type": "integer", - "minimum": 0.0, - "title": "Size" - }, - "ndcs": { - "type": "integer", - "title": "Ndcs" - }, - "npcs": { - "type": "integer", - "title": "Npcs" - }, - "pool_uuid": { - "type": "string", - "title": "Pool Uuid" - }, - "pool_name": { - "type": "string", - "title": "Pool Name" + "title": "Snapshot Backups", + "default": true }, - "pvc_name": { - "type": "string", - "title": "Pvc Name", - "default": "" + "verify_tls": { + "type": "boolean", + "title": "Verify Tls", + "default": true }, - "snapshot_name": { - "type": "string", - "title": "Snapshot Name", - "default": "" + "use_path_style": { + "type": "boolean", + "title": "Use Path Style", + "default": true }, - "blobid": { - "type": "integer", - "title": "Blobid" + "credentials": { + "anyOf": [ + { + "$ref": "#/components/schemas/S3Credentials" + }, + { + "type": "null" + } + ] }, - "ns_id": { - "type": "integer", - "title": "Ns Id" + "s3_thread_pool_size": { + "anyOf": [ + { + "type": "integer", + "minimum": 1.0 + }, + { + "type": "null" + } + ], + "title": "S3 Thread Pool Size" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "bucket_name" + ], + "title": "BackupConfig", + "description": "A cluster's backup configuration: a location plus how to authenticate to it." + }, + "BackupConfig-Output": { + "properties": { + "bucket_name": { + "type": "string", + "minLength": 1, + "title": "Bucket Name" }, - "cloned_from": { + "region": { "anyOf": [ { - "type": "string" + "type": "string", + "minLength": 1 }, { "type": "null" } ], - "title": "Cloned From" + "title": "Region" }, - "high_availability": { - "type": "boolean", - "title": "High Availability" + "endpoint": { + "type": "string", + "title": "Endpoint" }, - "do_replicate": { + "secondary_target": { + "type": "integer" + }, + "with_compression": { "type": "boolean", - "title": "Do Replicate", + "title": "With Compression", "default": false }, - "max_namespace_per_subsys": { - "type": "integer", - "title": "Max Namespace Per Subsys" + "snapshot_backups": { + "type": "boolean", + "title": "Snapshot Backups", + "default": true }, - "max_rw_iops": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Iops" + "verify_tls": { + "type": "boolean", + "title": "Verify Tls", + "default": true }, - "max_rw_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Mbytes" + "use_path_style": { + "type": "boolean", + "title": "Use Path Style", + "default": true }, - "max_r_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max R Mbytes" + "credentials": { + "anyOf": [ + { + "$ref": "#/components/schemas/S3Credentials" + }, + { + "type": "null" + } + ] }, - "max_w_mbytes": { + "s3_thread_pool_size": { + "anyOf": [ + { + "type": "integer", + "minimum": 1.0 + }, + { + "type": "null" + } + ], + "title": "S3 Thread Pool Size" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "bucket_name" + ], + "title": "BackupConfig", + "description": "A cluster's backup configuration: a location plus how to authenticate to it." + }, + "BackupExport-Input": { + "properties": { + "schema_version": { "type": "integer", - "minimum": 0.0, - "title": "Max W Mbytes" + "title": "Schema Version", + "default": 1 }, - "allowed_hosts": { + "groups": { "items": { - "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" + "$ref": "#/components/schemas/LocatedManifests-Input" }, "type": "array", - "title": "Allowed Hosts" + "title": "Groups" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "groups" + ], + "title": "BackupExport", + "description": "Backups carried out of a cluster in a file, grouped by where they live.\n\nGrouped rather than one location for the whole document because a cluster\ncan hold backups in several buckets at once -- its own, plus any it has\nimported -- and stamping all of them with a single bucket leaves the ones it\ndoes not describe unrestorable, which is discovered during the recovery they\nwere meant to serve.\n\nA group's manifests are all in one bucket by construction: a chain cannot\nspan buckets, so the only way to collect backups from several is to walk\nmore than one chain." + }, + "BackupExport-Output": { + "properties": { + "schema_version": { + "type": "integer", + "title": "Schema Version", + "default": 1 }, - "policy": { + "groups": { + "items": { + "$ref": "#/components/schemas/LocatedManifests-Output" + }, + "type": "array", + "title": "Groups" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "groups" + ], + "title": "BackupExport", + "description": "Backups carried out of a cluster in a file, grouped by where they live.\n\nGrouped rather than one location for the whole document because a cluster\ncan hold backups in several buckets at once -- its own, plus any it has\nimported -- and stamping all of them with a single bucket leaves the ones it\ndoes not describe unrestorable, which is discovered during the recovery they\nwere meant to serve.\n\nA group's manifests are all in one bucket by construction: a chain cannot\nspan buckets, so the only way to collect backups from several is to walk\nmore than one chain." + }, + "BackupLocation-Input": { + "properties": { + "bucket_name": { "type": "string", - "title": "Policy" + "minLength": 1, + "title": "Bucket Name" }, - "capacity": { - "$ref": "#/components/schemas/CapacityStatDTO" + "region": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "null" + } + ], + "title": "Region" }, - "rep_info": { + "endpoint": { "anyOf": [ { - "additionalProperties": true, - "type": "object" + "type": "string", + "maxLength": 2083, + "minLength": 1, + "format": "uri" }, { "type": "null" } ], - "title": "Rep Info" + "title": "Endpoint" }, - "from_source": { + "secondary_target": { + "$ref": "#/components/schemas/SecondaryTarget", + "default": 0 + }, + "with_compression": { "type": "boolean", - "title": "From Source", + "title": "With Compression", + "default": false + }, + "snapshot_backups": { + "type": "boolean", + "title": "Snapshot Backups", "default": true }, - "group_id": { - "type": "string", - "title": "Group Id", - "default": "" + "verify_tls": { + "type": "boolean", + "title": "Verify Tls", + "default": true }, - "group_seq": { - "type": "integer", - "title": "Group Seq", - "default": 0 - } - }, - "type": "object", - "required": [ - "id", - "cluster_id", - "storage_node_id", - "name", - "status", - "health_check", - "io_error", - "migrating", - "nqn", - "hostname", - "priority_class", - "namespace", - "fabric", - "nodes", - "port", - "size", - "ndcs", - "npcs", - "pool_uuid", - "pool_name", - "blobid", - "ns_id", - "cloned_from", - "high_availability", - "max_namespace_per_subsys", - "max_rw_iops", - "max_rw_mbytes", - "max_r_mbytes", - "max_w_mbytes", - "allowed_hosts", - "policy", - "capacity" - ], - "title": "VolumeDTO" - }, - "_AddHostParams": { - "properties": { - "host_nqn": { - "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?", - "title": "Host Nqn" + "use_path_style": { + "type": "boolean", + "title": "Use Path Style", + "default": true } }, + "additionalProperties": false, "type": "object", "required": [ - "host_nqn" + "bucket_name" ], - "title": "_AddHostParams" + "title": "BackupLocation", + "description": "Where a backup's objects are, and how to interpret them. Never secret.\n\nEvery field here affects whether the objects can be read back at all, which\nis why the whole model is embedded in each backup rather than looked up from\nthe cluster that happened to create it." }, - "_AttachParams": { + "BackupLocation-Output": { "properties": { - "target_type": { + "bucket_name": { "type": "string", - "title": "Target Type" + "minLength": 1, + "title": "Bucket Name" }, - "target_id": { - "type": "string", - "title": "Target Id" - } - }, - "type": "object", - "required": [ - "target_type", - "target_id" - ], - "title": "_AttachParams" - }, - "_BackupSnapshotParams": { - "properties": { - "snapshot_id": { + "region": { + "anyOf": [ + { + "type": "string", + "minLength": 1 + }, + { + "type": "null" + } + ], + "title": "Region" + }, + "endpoint": { "type": "string", - "title": "Snapshot Id" + "title": "Endpoint" + }, + "secondary_target": { + "type": "integer" + }, + "with_compression": { + "type": "boolean", + "title": "With Compression", + "default": false + }, + "snapshot_backups": { + "type": "boolean", + "title": "Snapshot Backups", + "default": true + }, + "verify_tls": { + "type": "boolean", + "title": "Verify Tls", + "default": true + }, + "use_path_style": { + "type": "boolean", + "title": "Use Path Style", + "default": true } }, + "additionalProperties": false, "type": "object", "required": [ - "snapshot_id" + "bucket_name" ], - "title": "_BackupSnapshotParams" + "title": "BackupLocation", + "description": "Where a backup's objects are, and how to interpret them. Never secret.\n\nEvery field here affects whether the objects can be read back at all, which\nis why the whole model is embedded in each backup rather than looked up from\nthe cluster that happened to create it." }, - "_CloneParams": { + "BackupManifest": { "properties": { - "name": { + "schema_version": { + "type": "integer", + "title": "Schema Version", + "default": 1 + }, + "backup_id": { "type": "string", - "title": "Name" + "format": "uuid", + "title": "Backup Id" }, - "snapshot_id": { - "anyOf": [ - { - "type": "string", - "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$" - }, - { - "type": "null" - } - ], - "title": "Snapshot Id" + "s3_id": { + "type": "integer", + "title": "S3 Id" + }, + "created_at": { + "type": "integer", + "title": "Created At" + }, + "completed_at": { + "type": "integer", + "title": "Completed At" }, "size": { "type": "integer", - "minimum": 0.0, - "title": "Size", - "default": 0 + "title": "Size" }, - "pvc_name": { + "prev_backup_id": { "anyOf": [ { - "type": "string" + "type": "string", + "format": "uuid" }, { "type": "null" } ], - "title": "Pvc Name" + "title": "Prev Backup Id" }, - "pvc_namespace": { + "encryption": { "anyOf": [ { - "type": "string" + "oneOf": [ + { + "$ref": "#/components/schemas/FDBKeyDescriptor" + }, + { + "$ref": "#/components/schemas/HCPKeyDescriptor" + } + ], + "discriminator": { + "propertyName": "type", + "mapping": { + "fdb": "#/components/schemas/FDBKeyDescriptor", + "hcp": "#/components/schemas/HCPKeyDescriptor" + } + } }, { "type": "null" } ], - "title": "Pvc Namespace" + "title": "Encryption" }, - "delete_snap_on_lvol_delete": { - "type": "boolean", - "title": "Delete Snap On Lvol Delete", - "default": false + "source": { + "$ref": "#/components/schemas/ManifestSource" }, - "consistency_group": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Consistency Group" + "volume": { + "$ref": "#/components/schemas/ManifestVolume" + }, + "dataplane": { + "$ref": "#/components/schemas/ManifestDataPlane" } }, + "additionalProperties": false, "type": "object", "required": [ - "name", - "snapshot_id" + "backup_id", + "s3_id", + "created_at", + "completed_at", + "size", + "source", + "volume", + "dataplane" ], - "title": "_CloneParams" + "title": "BackupManifest" }, - "_ContinueParams": { + "FDBKeyDescriptor": { "properties": { - "max_retries": { - "type": "integer", - "title": "Max Retries", - "default": 10 + "dek_path": { + "type": "string", + "title": "Dek Path" }, - "deadline_seconds": { - "type": "integer", - "title": "Deadline Seconds", - "default": 14400 + "type": { + "type": "string", + "const": "fdb", + "title": "Type", + "default": "fdb" } }, + "additionalProperties": false, "type": "object", - "title": "_ContinueParams" + "required": [ + "dek_path" + ], + "title": "FDBKeyDescriptor", + "description": "Keys held in the cluster's own FoundationDB, by ``LocalKMS``." }, - "_CreateParams": { + "HCPKeyDescriptor": { "properties": { - "name": { + "dek_path": { "type": "string", - "title": "Name" - }, - "size": { - "type": "integer", - "minimum": 0.0, - "title": "Size" - }, - "max_rw_iops": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Iops", - "default": 0 - }, - "max_rw_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max Rw Mbytes", - "default": 0 + "title": "Dek Path" }, - "max_r_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max R Mbytes", - "default": 0 + "type": { + "type": "string", + "const": "hcp", + "title": "Type", + "default": "hcp" }, - "max_w_mbytes": { - "type": "integer", - "minimum": 0.0, - "title": "Max W Mbytes", - "default": 0 + "kek_name": { + "type": "string", + "title": "Kek Name" }, - "ha_type": { + "vault_base_url": { "anyOf": [ { "type": "string", - "enum": [ - "single", - "ha" - ] + "maxLength": 2083, + "minLength": 1, + "format": "uri" + }, + { + "type": "null" + } + ], + "title": "Vault Base Url" + }, + "transit_mount": { + "anyOf": [ + { + "type": "string" }, { "type": "null" } ], - "title": "Ha Type" + "title": "Transit Mount" }, - "host_id": { + "kv_mount": { "anyOf": [ { "type": "string" @@ -14195,30 +14332,104 @@ "type": "null" } ], - "title": "Host Id" + "title": "Kv Mount" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "dek_path", + "kek_name" + ], + "title": "HCPKeyDescriptor", + "description": "Keys held in HashiCorp Vault, wrapped under a named transit key." + }, + "LocatedManifests-Input": { + "properties": { + "location": { + "$ref": "#/components/schemas/BackupLocation-Input" }, - "priority_class": { - "type": "integer", - "enum": [ - 0, - 1 - ], - "title": "Priority Class", - "default": 0 + "manifests": { + "items": { + "$ref": "#/components/schemas/BackupManifest" + }, + "type": "array", + "title": "Manifests" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "location", + "manifests" + ], + "title": "LocatedManifests", + "description": "Manifests that were read from one location, and that location." + }, + "LocatedManifests-Output": { + "properties": { + "location": { + "$ref": "#/components/schemas/BackupLocation-Output" }, - "namespaced": { + "manifests": { + "items": { + "$ref": "#/components/schemas/BackupManifest" + }, + "type": "array", + "title": "Manifests" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "location", + "manifests" + ], + "title": "LocatedManifests", + "description": "Manifests that were read from one location, and that location." + }, + "ManifestDataPlane": { + "properties": { + "key_format": { + "type": "string", + "title": "Key Format", + "default": "{s3_id}/{mid}/{extent}" + }, + "cluster_size": { "anyOf": [ { - "type": "boolean" + "type": "integer" }, { "type": "null" } ], - "title": "Namespaced", + "title": "Cluster Size" + }, + "with_compression": { + "type": "boolean", + "title": "With Compression", "default": false + } + }, + "additionalProperties": false, + "type": "object", + "title": "ManifestDataPlane", + "description": "How the objects are encoded, so a later format change is detectable.\n\nEverything here has to be recorded because reading the bucket cannot\nrecover it -- unlike the bucket's name, region and endpoint, which the\nreader necessarily supplied to get this far." + }, + "ManifestSource": { + "properties": { + "cluster_id": { + "type": "string", + "format": "uuid", + "title": "Cluster Id" }, - "pvc_name": { + "node_id": { + "type": "string", + "format": "uuid", + "title": "Node Id" + }, + "cluster_name": { "anyOf": [ { "type": "string" @@ -14227,168 +14438,118 @@ "type": "null" } ], - "title": "Pvc Name" + "title": "Cluster Name" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "cluster_id", + "node_id" + ], + "title": "ManifestSource", + "description": "Where this backup came from. Provenance for an operator reading a bucket.\n\nNothing may resolve configuration or keys through these -- that dependency\non the originating cluster is the whole problem being removed." + }, + "ManifestVolume": { + "properties": { + "lvol_id": { + "type": "string", + "format": "uuid", + "title": "Lvol Id" }, - "ndcs": { - "type": "integer", - "minimum": 0.0, - "title": "Ndcs", - "default": 0 + "lvol_name": { + "type": "string", + "title": "Lvol Name" }, - "npcs": { + "snapshot_id": { + "type": "string", + "format": "uuid", + "title": "Snapshot Id" + }, + "snapshot_name": { + "type": "string", + "title": "Snapshot Name" + }, + "size": { "type": "integer", - "minimum": 0.0, - "title": "Npcs", - "default": 0 + "title": "Size" }, - "allowed_hosts": { + "pool_name": { "anyOf": [ { - "items": { - "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" - }, - "type": "array" + "type": "string" }, { "type": "null" } ], - "title": "Allowed Hosts" - }, - "fabric": { - "type": "string", - "title": "Fabric", - "default": "tcp" + "title": "Pool Name" }, - "max_namespace_per_subsys": { + "ha_type": { "anyOf": [ { - "type": "integer" + "type": "string", + "enum": [ + "single", + "ha" + ] }, { "type": "null" } ], - "title": "Max Namespace Per Subsys" - }, - "do_replicate": { - "type": "boolean", - "title": "Do Replicate", - "default": false + "title": "Ha Type" }, - "replication_cluster_id": { + "fabric": { "anyOf": [ { - "type": "string" + "type": "string", + "enum": [ + "tcp", + "rdma", + "tcp,rdma" + ] }, { "type": "null" } ], - "title": "Replication Cluster Id" + "title": "Fabric" }, - "replication_policy": { + "lvol_priority_class": { "anyOf": [ { - "type": "string" + "type": "integer" }, { "type": "null" } ], - "title": "Replication Policy" + "title": "Lvol Priority Class" }, - "consistency_group": { + "max_size": { "anyOf": [ { - "type": "string" + "type": "integer" }, { "type": "null" } ], - "title": "Consistency Group" - }, - "encrypt": { - "type": "boolean", - "title": "Encrypt", - "default": false - } - }, - "type": "object", - "required": [ - "name", - "size" - ], - "title": "_CreateParams" - }, - "_ImportFromBucket": { - "properties": { - "bucket": { - "$ref": "#/components/schemas/BackupConfig-Input" - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "bucket" - ], - "title": "_ImportFromBucket", - "description": "Import whatever a bucket turns out to contain.\n\nThe disaster-recovery path: it needs a bucket and credentials for it, and\nnothing from the cluster that wrote the backups." - }, - "_ImportManifests": { - "properties": { - "metadata": { - "$ref": "#/components/schemas/BackupExport-Input" - } - }, - "additionalProperties": false, - "type": "object", - "required": [ - "metadata" - ], - "title": "_ImportManifests", - "description": "An export carried in the request itself, e.g. read from a file.\n\nNothing beside it names a bucket: an export groups its manifests by the\nlocation each was read from, so the caller states nothing the document has\nnot already recorded, and backups from several buckets import in one go." - }, - "_MigrationParams": { - "properties": { - "target_node_id": { - "type": "string", - "format": "uuid", - "title": "Target Node Id" - }, - "ctrl_loss_tmo": { - "type": "integer", - "title": "Ctrl Loss Tmo", - "default": 3600 + "title": "Max Size" }, - "host_nqn": { + "rw_ios_per_sec": { "anyOf": [ { - "type": "string", - "pattern": "nqn\\.\\d{4}-\\d{2}\\.(?:[a-zA-Z0-9](?:[a-zA-Z0-9\\-]{0,61}[a-zA-Z0-9])?\\.)*[a-zA-Z]{2,}(?::[a-zA-Z0-9.\\-:_]+)?" + "type": "integer" }, { "type": "null" } ], - "title": "Host Nqn" - } - }, - "type": "object", - "required": [ - "target_node_id" - ], - "title": "_MigrationParams" - }, - "_PolicyCreateParams": { - "properties": { - "name": { - "type": "string", - "title": "Name" + "title": "Rw Ios Per Sec" }, - "versions": { + "rw_mbytes_per_sec": { "anyOf": [ { "type": "integer" @@ -14397,202 +14558,195 @@ "type": "null" } ], - "title": "Versions", - "default": 0 + "title": "Rw Mbytes Per Sec" }, - "age": { + "r_mbytes_per_sec": { "anyOf": [ { - "type": "string" + "type": "integer" }, { "type": "null" } ], - "title": "Age", - "default": "" + "title": "R Mbytes Per Sec" }, - "schedule": { + "w_mbytes_per_sec": { "anyOf": [ { - "type": "string" + "type": "integer" }, { "type": "null" } ], - "title": "Schedule", - "default": "" + "title": "W Mbytes Per Sec" } }, + "additionalProperties": false, "type": "object", "required": [ - "name" + "lvol_id", + "lvol_name", + "snapshot_id", + "snapshot_name", + "size" ], - "title": "_PolicyCreateParams" + "title": "ManifestVolume", + "description": "The shape of the volume this backup was taken from.\n\nSplit in two by what is knowable. The identity and size come off the backup\nrecord and are always present. The settings below them come off the live\nvolume, so they are absent together once that volume is deleted -- and\nabsent is not the same answer as ``0``, which for a QoS cap means\n\"unlimited\" and for a priority class is a real class.\n\nNothing reads the settings yet; restore still creates its volume with\nhardcoded defaults. They are recorded anyway because a manifest is read\nyears after it is written, and a backup taken today cannot be given a shape\nretroactively once its volume is gone.\n\nThe volume's allow-list is deliberately not among them. Who may attach is a\nproperty of the pool a volume lives in, not of the bytes a backup holds, and\na restore lands in whichever pool it is given -- possibly in another cluster,\nwhere the source volume's NQNs mean nothing. So a restored volume takes the\ntarget pool's host configuration, and a stale allow-list from the source\nnever overrides it." }, - "_ReplicationParams": { + "S3Credentials": { "properties": { - "snapshot_replication_target_cluster": { + "access_key_id": { "type": "string", - "title": "Snapshot Replication Target Cluster" - }, - "snapshot_replication_timeout": { - "type": "integer", - "title": "Snapshot Replication Timeout", - "default": 0 + "format": "password", + "title": "Access Key Id", + "writeOnly": true }, - "target_pool": { - "anyOf": [ - { - "type": "string" - }, - { - "type": "null" - } - ], - "title": "Target Pool" + "secret_access_key": { + "type": "string", + "format": "password", + "title": "Secret Access Key", + "writeOnly": true } }, + "additionalProperties": false, "type": "object", "required": [ - "snapshot_replication_target_cluster" + "access_key_id", + "secret_access_key" ], - "title": "_ReplicationParams" + "title": "S3Credentials", + "description": "A static key pair.\n\nA pair rather than two independent fields, so \"access key set, secret\nmissing\" is unrepresentable instead of something a validator has to catch." }, - "_RestartParams": { + "SecondaryTarget": { + "type": "integer", + "enum": [ + 0, + 1 + ], + "title": "SecondaryTarget", + "description": "The kind of secondary store, numbered as the data plane's RPC expects." + }, + "UnresolvedBackupConfig": { "properties": { - "force": { - "type": "boolean", - "title": "Force", - "default": false - }, - "reattach_volume": { - "type": "boolean", - "title": "Reattach Volume", - "default": false - }, - "node_address": { + "bucket_name": { "anyOf": [ { - "type": "string" + "type": "string", + "minLength": 1 }, { "type": "null" } ], - "title": "Node Address" - }, - "new_ssd_pcie": { - "items": { - "type": "string" - }, - "type": "array", - "title": "New Ssd Pcie", - "default": [] - } - }, - "type": "object", - "title": "_RestartParams" - }, - "_RestoreParams": { - "properties": { - "backup_id": { - "type": "string", - "title": "Backup Id" - }, - "lvol_name": { - "type": "string", - "title": "Lvol Name" - }, - "pool": { - "type": "string", - "title": "Pool" + "title": "Bucket Name" }, - "target_node_id": { + "region": { "anyOf": [ { - "type": "string" + "type": "string", + "minLength": 1 }, { "type": "null" } ], - "title": "Target Node Id" + "title": "Region" }, - "s3_credentials": { + "endpoint": { "anyOf": [ { - "$ref": "#/components/schemas/S3Credentials" + "type": "string", + "maxLength": 2083, + "minLength": 1, + "format": "uri" }, { "type": "null" } - ] - } - }, - "type": "object", - "required": [ - "backup_id", - "lvol_name", - "pool" - ], - "title": "_RestoreParams" - }, - "_SnapshotParams": { - "properties": { - "name": { - "type": "string", - "title": "Name" + ], + "title": "Endpoint" }, - "backup": { + "secondary_target": { + "$ref": "#/components/schemas/SecondaryTarget", + "default": 0 + }, + "with_compression": { "type": "boolean", - "title": "Backup", + "title": "With Compression", "default": false - } - }, - "type": "object", - "required": [ - "name" - ], - "title": "_SnapshotParams" - }, - "_UpdateParams": { - "properties": { - "management_image": { + }, + "snapshot_backups": { + "type": "boolean", + "title": "Snapshot Backups", + "default": true + }, + "verify_tls": { + "type": "boolean", + "title": "Verify Tls", + "default": true + }, + "use_path_style": { + "type": "boolean", + "title": "Use Path Style", + "default": true + }, + "credentials": { "anyOf": [ { - "type": "string" + "$ref": "#/components/schemas/S3Credentials" }, { "type": "null" } - ], - "title": "Management Image" + ] }, - "spdk_image": { + "s3_thread_pool_size": { "anyOf": [ { - "type": "string" + "type": "integer", + "minimum": 1.0 }, { "type": "null" } ], - "title": "Spdk Image" - }, - "restart": { - "type": "boolean", - "title": "Restart", - "default": false + "title": "S3 Thread Pool Size" } }, + "additionalProperties": false, + "type": "object", + "title": "UnresolvedBackupConfig", + "description": "A backup configuration as a caller can state it, before a cluster resolves it.\n\nIdentical to :class:`BackupConfig` except that ``bucket_name`` may be absent,\nbecause the default is derived from a cluster id that does not exist yet at\ncluster-create time (``Cluster.default_backup_bucket_name``). Hand one to\n``Cluster.set_backup_config``, which resolves it; nothing further down ever\nsees a configuration without a bucket.\n\nWhich is also why an instance must be turned back into a plain dict at the\nboundary it arrived on rather than passed along as a ``BackupConfig``: the\ninherited :meth:`location` cannot produce a location for a bucket nobody has\nnamed yet." + }, + "_ImportFromBucket": { + "properties": { + "bucket": { + "$ref": "#/components/schemas/BackupConfig-Input" + } + }, + "additionalProperties": false, "type": "object", "required": [ - "management_image", - "spdk_image" + "bucket" ], - "title": "_UpdateParams" + "title": "_ImportFromBucket", + "description": "Import whatever a bucket turns out to contain.\n\nThe disaster-recovery path: it needs a bucket and credentials for it, and\nnothing from the cluster that wrote the backups." + }, + "_ImportManifests": { + "properties": { + "metadata": { + "$ref": "#/components/schemas/BackupExport-Input" + } + }, + "additionalProperties": false, + "type": "object", + "required": [ + "metadata" + ], + "title": "_ImportManifests", + "description": "An export carried in the request itself, e.g. read from a file.\n\nNothing beside it names a bucket: an export groups its manifests by the\nlocation each was read from, so the caller states nothing the document has\nnot already recorded, and backups from several buckets import in one go." } }, "securitySchemes": {