From 57f657146caa237258c446d9a8ab7737f4e2ac45 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=D0=9A=D0=B8=D1=80=D0=B8=D0=BB=D0=BB=20=D0=A0=D0=B0=D0=B7?= =?UTF-8?q?=D1=83=D0=BC=D0=BE=D0=B2?= Date: Mon, 5 Oct 2026 11:08:12 +0300 Subject: [PATCH] chart: run as a StatefulSet with a ReadWriteOnce claim per replica The chart rendered a Deployment on a single plugin-cache claim. Scaling out meant ReadWriteMany, and replicas sharing a cache corrupt it: unpacking is serialised by an in-process lock only, so two pods missing the same plugin race on its remove-and-rename, and each applies cacheMaxBytes to the same bytes. On ReadWriteOnce the Deployment had to use strategy: Recreate instead, so every upgrade was a gap in service. The workload is now a StatefulSet, there for volumeClaimTemplates rather than identity: each replica gets a claim of its own, plugins--N, or an emptyDir with persistence.enabled=false. Upgrades roll, since a StatefulSet deletes a pod before starting its successor, and pods start in parallel because migrations already take an advisory lock. The headless governing Service publishes gRPC only, so the ServiceMonitor does not scrape each pod twice. - persistence.accessMode accepts ReadWriteOnce or ReadWriteOncePod; a shared mode is refused. - More than one replica, or an HPA allowed more than one, needs config.registry.s3.bucket: every claim starts empty and S3 is what fills it. - autoscaling.* renders an HPA on CPU, memory optional; replicaCount is ignored while it is on. An HPA with no metric, a target without the matching resource request, or minReplicas above maxReplicas is refused. - persistence.retentionPolicy sets the claims' whenDeleted and whenScaled, Retain by default. - persistence.existingClaim and templates/pvc.yaml are gone. Setting existingClaim fails the render rather than mounting an empty volume, and NOTES.txt names the old -plugins claim when an upgrade leaves it unmounted. Every new value tolerates being absent, because --reuse-values from an older release carries none of them. tests/values-1.0.4.yaml holds the 1.0.4 values verbatim and render.sh renders the templates against them; a partial autoscaling map is refused by name. The chart README covers scaling, upgrading from the Deployment (including rebinding the old volume when there is no S3) and resizing the immutable claim templates. The service README says why replicas must not share a cache and that every limit is per pod. CI's token-rotation check compares pod UIDs, as a StatefulSet's replacement pod keeps its name. deploy/values-easyp-service.yaml adds values for a two-replica install against S3. --- .github/workflows/go.yml | 9 +- README.md | 31 +- deploy/charts/easyp-service/README.md | 183 ++++++- .../charts/easyp-service/templates/NOTES.txt | 41 +- .../easyp-service/templates/_helpers.tpl | 98 +++- .../charts/easyp-service/templates/hpa.yaml | 46 ++ .../charts/easyp-service/templates/pvc.yaml | 21 - .../templates/service-headless.yaml | 23 + .../{deployment.yaml => statefulset.yaml} | 76 ++- deploy/charts/easyp-service/tests/render.sh | 186 ++++++- .../easyp-service/tests/values-1.0.4.yaml | 455 ++++++++++++++++++ deploy/charts/easyp-service/values.yaml | 50 +- 12 files changed, 1115 insertions(+), 104 deletions(-) create mode 100644 deploy/charts/easyp-service/templates/hpa.yaml delete mode 100644 deploy/charts/easyp-service/templates/pvc.yaml create mode 100644 deploy/charts/easyp-service/templates/service-headless.yaml rename deploy/charts/easyp-service/templates/{deployment.yaml => statefulset.yaml} (70%) create mode 100644 deploy/charts/easyp-service/tests/values-1.0.4.yaml diff --git a/.github/workflows/go.yml b/.github/workflows/go.yml index 0673e1b..341cbaa 100644 --- a/.github/workflows/go.yml +++ b/.github/workflows/go.yml @@ -300,7 +300,7 @@ jobs: if: failure() run: | kubectl get pods -o wide - kubectl describe deploy/easyp-easyp-service | tail -40 + kubectl describe statefulset/easyp-easyp-service | tail -40 kubectl describe pod -l app.kubernetes.io/name=easyp-service | tail -60 kubectl logs -l app.kubernetes.io/name=easyp-service --all-containers --tail=100 || true kubectl get events --sort-by=.lastTimestamp | tail -30 @@ -315,14 +315,17 @@ jobs: # Chart 0.3.1 fixed exactly this and nothing tested it: the secret was not # hashed into the pod template, so rotating a write token left the old pod # running and the new token rejected until someone restarted by hand. + # By UID, not name: a StatefulSet's replacement pod is named exactly like + # the one it replaced, so comparing names would report a pod that was + # rolled as one that was not. - name: Rotating a write token rolls the pod run: | - before="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.name}')" + before="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.uid}')" helm upgrade easyp deploy/charts/easyp-service --reuse-values \ --set secrets.data.AUTH_WRITE_TOKENS="ci=$(printf 'rotated' | sha256sum | cut -d' ' -f1)" \ --wait --timeout 5m kubectl wait --for=condition=Ready pod -l app.kubernetes.io/name=easyp-service --timeout=5m - after="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.name}')" + after="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.uid}')" if [[ "$before" == "$after" ]]; then echo "::error::the pod was not replaced; the secret is not hashed into the pod template" exit 1 diff --git a/README.md b/README.md index 68fd1dc..7e73cea 100644 --- a/README.md +++ b/README.md @@ -914,24 +914,29 @@ chain and is wrapped separately. That is a single shared credential, not identity: it decides *whether* a caller may read, never *which* caller is reading. -### It does not run in more than one replica +### Replicas must not share a plugin cache The database side is safe: migrations take a Postgres session lock and audit partition maintenance takes an advisory lock, so several processes cannot collide there. -The plugin cache cannot. Unpacking finishes with a remove followed by a rename — -two steps, not atomic — and the only thing serialising it is an in-process -lock. Two pods on one `ReadWriteMany` volume race: one can delete a directory -the other is reading, and a rename onto a directory recreated in between fails. -Each pod also keeps its own in-memory accounting of the same shared bytes, so -`registry.cache_max_bytes` is applied once per pod to one volume. - -The chart's defaults are honest about this — one replica, `ReadWriteOnce`, -`Recreate`. It *permits* `replicaCount > 1` with a `ReadWriteMany` volume, and -that combination is **not supported**: it will appear to work and corrupt the -cache under concurrent misses for the same plugin. Scale by making the single -pod bigger — `maxConcurrentGenerations` and CPU — not by adding pods. +The plugin cache cannot be shared. Unpacking finishes with a remove followed by +a rename — two steps, not atomic — and the only thing serialising it is an +in-process lock. Two pods on one `ReadWriteMany` volume race: one can delete a +directory the other is reading, and a rename onto a directory recreated in +between fails. Each pod also keeps its own in-memory accounting of the same +shared bytes, so `registry.cache_max_bytes` is applied once per pod to one +volume. + +So the service scales out with a cache per pod, never one between them. The +Helm chart runs a StatefulSet that gives each replica a ReadWriteOnce claim of +its own, or an emptyDir with `persistence.enabled=false`, and refuses a shared +access mode outright. More than one replica needs `registry.s3`, since each pod +fills its own cache from object storage. + +Everything else a replica enforces is its own: the worker pool, the concurrent +generation cap and the per-client rate limit apply per pod, so a client spread +across N pods gets up to N times its limit. ### There is no down-migration path diff --git a/deploy/charts/easyp-service/README.md b/deploy/charts/easyp-service/README.md index b89d2ab..6a75a19 100644 --- a/deploy/charts/easyp-service/README.md +++ b/deploy/charts/easyp-service/README.md @@ -173,24 +173,175 @@ fail the request in a way that looks like a corrupt artifact. Watch `easyp_plugin_cache_bytes` against the limit, and `easyp_plugin_cache_evictions_total` for churn. -The PVC carries `helm.sh/resource-policy: keep`, because refilling the cache -costs more than the disk. +The claims outlive the release: they are created by the StatefulSet rather than +by Helm, so `helm uninstall` leaves them, because refilling the cache costs more +than the disk. `persistence.retentionPolicy.whenDeleted=Delete` changes that. -`replicaCount > 1` needs `persistence.accessMode=ReadWriteMany`; the chart fails -the install otherwise rather than leaving pods stuck in Pending. +## Scaling -### Upgrades take the service down briefly +The workload is a StatefulSet. The pods keep no state of their own — plugin +metadata and audit live in Postgres, archives in object storage — and nothing +addresses one by name; the StatefulSet is there for `volumeClaimTemplates`, the +one way to give each replica a ReadWriteOnce volume that outlives the pod. -With a `ReadWriteOnce` volume the deployment uses `strategy: Recreate`, so an -upgrade stops the running pod before starting its replacement. That gap is -deliberate. A rolling update would start the new pod first, and because a -ReadWriteOnce volume attaches to one node at a time, a replacement scheduled -anywhere else waits on a Multi-Attach error indefinitely — `helm upgrade` neither -completes nor fails. +| | Replicas | Cache after a restart | +|---|---|---| +| `persistence.enabled=true` (default) | any | warm: a claim per replica, `plugins--easyp-service-N` | +| `persistence.enabled=false` | any | cold: an emptyDir per pod | -If the gap is unacceptable, the answer is a `ReadWriteMany` volume, not a -different strategy: with a shared volume the chart rolls, and `replicaCount` can -exceed one. +Replicas never share a volume, and the chart has no ReadWriteMany option. +Unpacking a plugin is serialised by an in-process lock only, so two pods on one +volume race on a concurrent miss for the same plugin and corrupt it; each would +also apply `cacheMaxBytes` to the same bytes on its own. `persistence.accessMode` +accepts `ReadWriteOnce` and `ReadWriteOncePod` and refuses anything else. That +also means scaling needs nothing but ordinary block storage — no NFS, no RWX +class. + +More than one replica needs object storage: + +```bash +helm install easyp oci://ghcr.io/easyp-tech/charts/easyp-service \ + --set secrets.existingSecret=easyp-env \ + --set config.registry.s3.bucket=easyp-plugins \ + --set config.registry.s3.endpoint=https://s3.example.com \ + --set autoscaling.enabled=true \ + ... +``` + +Every replica's volume starts empty and S3 is the only thing that fills it; +without S3 the plugins directory is the one copy of every plugin, and only one +replica can hold it. The chart refuses more than one replica — or an autoscaler +allowed more than one — without `config.registry.s3.bucket`, rather than +starting pods that have no plugins. + +Storage is `replicas × persistence.size`, and each new replica downloads what it +serves. `persistence.retentionPolicy.whenScaled` decides what happens to a claim +when the StatefulSet scales down: `Retain` (the default) keeps the warm cache for +the next scale-up, `Delete` stops an autoscaler from leaving idle disks behind. +It needs Kubernetes 1.27; older API servers ignore it and retain. + +### Upgrades roll + +A StatefulSet deletes a pod before starting its replacement, so a ReadWriteOnce +claim is released before the next pod asks for it. With one replica that is a +short gap; with several, the others serve through it. Pods start in parallel +(`podManagementPolicy: Parallel`) — migrations are serialised by an advisory +lock, so there is nothing to order. + +The chart used to render a Deployment, which needed `strategy: Recreate` on a +ReadWriteOnce claim: a rolling update started the replacement first, and one +scheduled on another node waited on a Multi-Attach error forever. + +### Upgrading from a release that rendered a Deployment + +Earlier releases rendered a Deployment and one standalone claim, +`-easyp-service-plugins`, or mounted the claim named by +`persistence.existingClaim`. Both are gone: the StatefulSet's pods mount claims +from the template, `plugins--easyp-service-N`, and `existingClaim` is +refused rather than ignored, so an upgrade that still sets it stops before it +starts pods on an empty volume. The chart's own old claim carries +`helm.sh/resource-policy: keep`, so Helm leaves it in place, unmounted, and the +install notes say so when they find it. + +What to do with the old claim depends on whether the release uses object +storage: + +- **With S3**, nothing is lost. The new claims fill on demand, a cold start per + replica. Upgrade, then delete the old claim once the pods are serving. +- **Without S3**, the old claim holds the only copy of every plugin. Before + upgrading, rebind its PersistentVolume to the name the StatefulSet will look + for; the StatefulSet then adopts it as replica 0's claim instead of creating + an empty one. + +```bash +release=easyp +old=easyp-easyp-service-plugins # or the claim existingClaim named +new=plugins-easyp-easyp-service-0 + +pv="$(kubectl get pvc "$old" -o jsonpath='{.spec.volumeName}')" +class="$(kubectl get pvc "$old" -o jsonpath='{.spec.storageClassName}')" +size="$(kubectl get pvc "$old" -o jsonpath='{.spec.resources.requests.storage}')" + +# Keep the volume when its claim goes away, and stop the pod that mounts it. +kubectl patch pv "$pv" -p '{"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' +kubectl scale deploy/easyp-easyp-service --replicas=0 +kubectl delete pvc "$old" + +# Free the volume, then claim it under the StatefulSet's name. +kubectl patch pv "$pv" --type=json -p '[{"op":"remove","path":"/spec/claimRef"}]' +kubectl create -f - < --reuse-values \ + --set persistence.existingClaim=null --set persistence.size="$size" +``` + +`--reuse-values` works for this upgrade, with one thing to know. It does not +merge the new chart's defaults: it replaces them with the old release's values, +so every key added since — `autoscaling`, `persistence.retentionPolicy` — is +absent rather than defaulted. The templates read an absent key as off (or, for +retention, as Retain), so the release comes up as it was. Turning autoscaling on +in the same upgrade is the exception: `--set autoscaling.enabled=true` brings +that one key and none of its bounds, and the chart refuses it by name. Use +`--reset-then-reuse-values` (Helm 3.14+) for that, which starts from the new +defaults and applies the old release's values over them. + +If the release was already upgraded and its pod started on an empty claim, the +same steps apply with `kubectl scale statefulset/easyp-easyp-service +--replicas=0` in place of the Deployment, deleting the empty +`plugins-easyp-easyp-service-0` alongside the old claim, and scaling back to one +afterwards. + +Tooling that addressed `deploy/-easyp-service` — `kubectl rollout`, +`kubectl logs`, dashboards, alert selectors on `kube_deployment_*` — needs +`statefulset/` instead. + +### Autoscaling + +`autoscaling.enabled` renders a HorizontalPodAutoscaler on CPU — every +generation is a plugin process charged to this container, so CPU is where load +shows. It needs metrics-server; without it the HPA reports `` and never +scales. `replicaCount` is ignored while it is on, so `helm upgrade` does not +reset what the HPA chose. + +Every limit in the service is per pod: the worker pool, the concurrent +generation cap and the per-client rate limit. Capacity therefore scales with +replicas, and so does what one client can use — a client whose requests land on +N pods gets up to N times `rateLimit`. The Community plugin cap counts rows in +the database and holds across replicas; the per-pod ceilings apply to each one. + +### Resizing the plugin volumes + +A StatefulSet's claim templates are immutable, so raising `persistence.size` +fails `helm upgrade`. With a storage class that allows expansion: + +```bash +# 1. Grow the existing claims in place. +for pvc in $(kubectl get pvc -l app.kubernetes.io/instance=easyp -o name); do + kubectl patch "$pvc" -p '{"spec":{"resources":{"requests":{"storage":"50Gi"}}}}' +done + +# 2. Drop the StatefulSet object, leaving its pods and claims running. +kubectl delete statefulset easyp-easyp-service --cascade=orphan + +# 3. Create it again with the new template; it adopts the pods and claims. +helm upgrade easyp ... --reuse-values --set persistence.size=50Gi +``` + +Raise `config.registry.cacheMaxBytes` in the same upgrade if the extra room is +meant for the cache. ## Behind an ingress: `config.server.trustedProxies` @@ -315,6 +466,6 @@ overrides the file. See `values.yaml`; every key is commented. The install-time checks in `_helpers.tpl` reject combinations that would otherwise fail confusingly at -runtime — missing DSN, plaintext router against a TLS listener, multi-replica -`ReadWriteOnce`, grace period shorter than the generation timeout, an ingress +runtime — missing DSN, plaintext router against a TLS listener, a shared +volume, more than one replica without object storage, grace period shorter than the generation timeout, an ingress with no trusted proxies, peak generation buffers larger than the memory limit. diff --git a/deploy/charts/easyp-service/templates/NOTES.txt b/deploy/charts/easyp-service/templates/NOTES.txt index 21c86c1..fc689ca 100644 --- a/deploy/charts/easyp-service/templates/NOTES.txt +++ b/deploy/charts/easyp-service/templates/NOTES.txt @@ -2,12 +2,12 @@ Watch it come up: - kubectl -n {{ .Release.Namespace }} rollout status deploy/{{ include "easyp-service.fullname" . }} + kubectl -n {{ .Release.Namespace }} rollout status statefulset/{{ include "easyp-service.fullname" . }} If the pod restarts immediately, the cause is almost always configuration and it is printed on the first line of the log, not in the pod events: - kubectl -n {{ .Release.Namespace }} logs deploy/{{ include "easyp-service.fullname" . }} + kubectl -n {{ .Release.Namespace }} logs statefulset/{{ include "easyp-service.fullname" . }} Three things worth checking before you call it done: @@ -23,7 +23,7 @@ Three things worth checking before you call it done: {{- else }} 1. LICENSE — a key was supplied. Confirm it was accepted: - kubectl -n {{ .Release.Namespace }} logs deploy/{{ include "easyp-service.fullname" . }} | grep -i licen + kubectl -n {{ .Release.Namespace }} logs statefulset/{{ include "easyp-service.fullname" . }} | grep -i licen {{- end }} {{ if not (hasKey .Values.secrets.data "AUTH_WRITE_TOKENS") }} @@ -65,8 +65,7 @@ Three things worth checking before you call it done: PLUGIN CACHE IS ON NODE DISK. persistence is disabled, so each pod keeps its own copy of the plugin cache in - an emptyDir. That is a supported way to run several replicas without - ReadWriteMany storage, but the space is the node's, and it is charged per + an emptyDir, lost with the pod. The space is the node's, and it is charged per replica rather than once: {{ .Values.replicaCount }} × {{ .Values.persistence.ephemeralSizeLimit }} across the nodes these pods land on @@ -77,6 +76,38 @@ PLUGIN CACHE IS ON NODE DISK. resources.requests.ephemeral-storage is what the scheduler actually reads. {{- end }} +{{- if .Values.persistence.enabled }} + +PLUGIN CACHE IS ONE VOLUME PER REPLICA. + + Each pod has its own {{ .Values.persistence.size }} claim (plugins-{{ include "easyp-service.fullname" . }}-N) and fills it + from object storage on first use, so a new replica starts cold. Resizing + cannot go through `helm upgrade` — the claim template is immutable; see the + chart README, "Resizing the plugin volumes". +{{- /* +Releases before the StatefulSet created one claim, -plugins, and kept +it on removal. Nothing mounts it any more; without S3 it holds the only copy of +every plugin, which is worth saying at the moment it stops being used. +*/}} +{{- $legacy := printf "%s-plugins" (include "easyp-service.fullname" .) }} +{{- if and .Release.IsUpgrade (lookup "v1" "PersistentVolumeClaim" .Release.Namespace $legacy) }} + + THE PREVIOUS CLAIM, {{ $legacy }}, IS NO LONGER MOUNTED. It was kept, not + deleted.{{ if not .Values.config.registry.s3.bucket }} Without S3 it holds the only copy of every plugin, and + the new claim started empty: rebind its volume to plugins-{{ include "easyp-service.fullname" . }}-0 as the + chart README describes, "Upgrading from a release that rendered a Deployment".{{ else }} + The new claims fill from S3, so delete it once the pods are serving.{{ end }} +{{- end }} +{{- end }} + +{{- if eq (include "easyp-service.autoscalingEnabled" .) "true" }} + +AUTOSCALING between {{ .Values.autoscaling.minReplicas }} and {{ .Values.autoscaling.maxReplicas }} replicas. It needs metrics-server; without it the +HPA shows targets and never scales. Check: + + kubectl -n {{ .Release.Namespace }} get hpa {{ include "easyp-service.fullname" . }} +{{- end }} + {{ if not .Values.ingress.enabled }} The API is not exposed outside the cluster (ingress.enabled=false). To try it: diff --git a/deploy/charts/easyp-service/templates/_helpers.tpl b/deploy/charts/easyp-service/templates/_helpers.tpl index e107bc0..694d47a 100644 --- a/deploy/charts/easyp-service/templates/_helpers.tpl +++ b/deploy/charts/easyp-service/templates/_helpers.tpl @@ -89,10 +89,27 @@ what a cert-manager CA issuer produces. {{- end }} {{- end }} +{{/* +Whether autoscaling is on, read so that an absent autoscaling map means off. + +`helm upgrade --reuse-values` replaces this chart's defaults with the previous +release's values, so a release installed before a key existed arrives without +it — and `.Values.autoscaling.enabled` on a missing map is a nil-pointer +failure that stops the upgrade. Absent is what an older release was: off. + +Every value added after 1.0.4 is read this way or through `with`, which skips a +missing one. tests/values-1.0.4.yaml pins it: the templates must render against +that release's values alone. +*/}} +{{- define "easyp-service.autoscalingEnabled" -}} +{{- dig "autoscaling" "enabled" false .Values.AsMap -}} +{{- end }} + {{- define "easyp-service.mutualTLS" -}} {{- and .Values.tls.enabled (ne .Values.tls.clientCASecret "-") -}} {{- end }} + {{/* Converts a Kubernetes quantity like "25Gi" to plain bytes, so that the volume size and the cache limit can be compared. Only the suffixes a volume size @@ -168,9 +185,84 @@ nothing but a log line to say so. {{- fail "easyp-service: secrets.create is true but secrets.data.DB_POSTGRES_DSN is empty." }} {{- end }} -{{- if gt (int .Values.replicaCount) 1 }} -{{- if and .Values.persistence.enabled (eq .Values.persistence.accessMode "ReadWriteOnce") }} -{{- fail "easyp-service: replicaCount > 1 needs persistence.accessMode=ReadWriteMany, otherwise only one pod can mount the plugin cache and the rest stay Pending." }} +{{- /* +Replicas never share a volume. Unpacking a plugin is serialised by an in-process +lock only, so two pods on one ReadWriteMany claim race on a concurrent miss for +the same plugin and corrupt it — and each applies cacheMaxBytes to the same +bytes on its own. Refused rather than documented, because it appears to work. +*/}} +{{- /* +persistence.existingClaim was a value until the chart moved to a StatefulSet, +and Helm ignores values nothing reads. Ignored, an upgrade carrying it would +mount a fresh, empty claim while the one named here — without S3, the only copy +of every plugin — sat unmounted, and nothing would say so. +*/}} +{{- if .Values.persistence.existingClaim }} +{{- fail (printf "easyp-service: persistence.existingClaim was removed; every replica now gets a claim of its own from the StatefulSet's volumeClaimTemplates, named plugins-%s-N. To keep the data in %q, rebind its PersistentVolume to plugins-%s-0 (see the chart README, \"Upgrading from a release that rendered a Deployment\"), then unset persistence.existingClaim." (include "easyp-service.fullname" .) .Values.persistence.existingClaim (include "easyp-service.fullname" .)) }} +{{- end }} + +{{- if not (has .Values.persistence.accessMode (list "ReadWriteOnce" "ReadWriteOncePod")) }} +{{- fail (printf "easyp-service: persistence.accessMode must be ReadWriteOnce or ReadWriteOncePod, got %q. Replicas must not share a plugin cache: pods unpacking the same plugin at once corrupt it. Each replica gets a claim of its own instead." .Values.persistence.accessMode) }} +{{- end }} + +{{- /* +With autoscaling the ceiling is what counts: a pod the HPA adds under load has +to be able to start. +*/}} +{{- $autoscaling := eq (include "easyp-service.autoscalingEnabled" .) "true" }} +{{- /* Not ternary: it evaluates both branches, and the HPA's may not exist. */}} +{{- $maxReplicas := int .Values.replicaCount }} +{{- if $autoscaling }} +{{- $maxReplicas = int .Values.autoscaling.maxReplicas }} +{{- end }} +{{- if gt $maxReplicas 1 }} +{{- /* +Every replica's volume starts empty. Without S3 the plugins directory is the one +copy of every plugin, so a second replica would have none and fail every +generation with NotFound. +*/}} +{{- if not .Values.config.registry.s3.bucket }} +{{- fail "easyp-service: more than one replica needs config.registry.s3.bucket. Each replica's cache starts empty and object storage is the only thing that fills it; without S3 the plugins directory is the sole copy of every plugin, and only one replica can hold it." }} +{{- end }} +{{- end }} + +{{- if $autoscaling }} +{{- /* +Present but partial: `--reuse-values --set autoscaling.enabled=true` on a release +from before autoscaling existed carries that one key and none of the defaults. +Refused rather than guessed, so the numbers live in values.yaml alone. +*/}} +{{- if not (and (hasKey .Values.autoscaling "minReplicas") (hasKey .Values.autoscaling "maxReplicas")) }} +{{- fail "easyp-service: autoscaling.enabled needs autoscaling.minReplicas and autoscaling.maxReplicas. A --reuse-values upgrade from a release that predates autoscaling carries none of its defaults; upgrade with --reset-then-reuse-values, or set both." }} +{{- end }} +{{- if gt (int .Values.autoscaling.minReplicas) (int .Values.autoscaling.maxReplicas) }} +{{- fail (printf "easyp-service: autoscaling.minReplicas (%d) exceeds autoscaling.maxReplicas (%d)." (int .Values.autoscaling.minReplicas) (int .Values.autoscaling.maxReplicas)) }} +{{- end }} +{{- if not (or .Values.autoscaling.targetCPUUtilizationPercentage .Values.autoscaling.targetMemoryUtilizationPercentage) }} +{{- fail "easyp-service: autoscaling.enabled needs targetCPUUtilizationPercentage or targetMemoryUtilizationPercentage; an HPA with no metric never scales." }} +{{- end }} +{{- /* +Utilisation is a percentage of the request. Without one the HPA reports + and holds at whatever it last had, which looks like a quiet day. +*/}} +{{- if .Values.autoscaling.targetCPUUtilizationPercentage }} +{{- if not (and .Values.resources.requests .Values.resources.requests.cpu) }} +{{- fail "easyp-service: autoscaling.targetCPUUtilizationPercentage needs resources.requests.cpu; utilisation is measured against the request." }} +{{- end }} +{{- end }} +{{- if .Values.autoscaling.targetMemoryUtilizationPercentage }} +{{- if not (and .Values.resources.requests .Values.resources.requests.memory) }} +{{- fail "easyp-service: autoscaling.targetMemoryUtilizationPercentage needs resources.requests.memory; utilisation is measured against the request." }} +{{- end }} +{{- end }} +{{- end }} + +{{- with .Values.persistence.retentionPolicy }} +{{- range $when := list "whenDeleted" "whenScaled" }} +{{- /* A key left out means Retain, as the StatefulSet renders it. */}} +{{- if and (hasKey $.Values.persistence.retentionPolicy $when) (not (has (index $.Values.persistence.retentionPolicy $when) (list "Retain" "Delete"))) }} +{{- fail (printf "easyp-service: persistence.retentionPolicy.%s must be Retain or Delete." $when) }} +{{- end }} {{- end }} {{- end }} diff --git a/deploy/charts/easyp-service/templates/hpa.yaml b/deploy/charts/easyp-service/templates/hpa.yaml new file mode 100644 index 0000000..8b40182 --- /dev/null +++ b/deploy/charts/easyp-service/templates/hpa.yaml @@ -0,0 +1,46 @@ +{{- if eq (include "easyp-service.autoscalingEnabled" .) "true" }} +{{/* +Scales on CPU because the work is CPU: every generation is a plugin process +running inside this container, so its usage is charged here. Memory is offered +but off by default — the buffers it would react to are bounded by +maxConcurrentGenerations per pod, and freed as soon as a response is sent, so it +tracks CPU without adding anything. + +Needs metrics-server. Without it the HPA exists, reports and scales +nothing, which is why NOTES.txt names it. +*/}} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "easyp-service.fullname" . }} + labels: + {{- include "easyp-service.labels" . | nindent 4 }} +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: StatefulSet + name: {{ include "easyp-service.fullname" . }} + minReplicas: {{ .Values.autoscaling.minReplicas }} + maxReplicas: {{ .Values.autoscaling.maxReplicas }} + metrics: + {{- with .Values.autoscaling.targetCPUUtilizationPercentage }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ . }} + {{- end }} + {{- with .Values.autoscaling.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ . }} + {{- end }} + {{- with .Values.autoscaling.behavior }} + behavior: + {{- toYaml . | nindent 4 }} + {{- end }} +{{- end }} diff --git a/deploy/charts/easyp-service/templates/pvc.yaml b/deploy/charts/easyp-service/templates/pvc.yaml deleted file mode 100644 index 509384d..0000000 --- a/deploy/charts/easyp-service/templates/pvc.yaml +++ /dev/null @@ -1,21 +0,0 @@ -{{- if and .Values.persistence.enabled (not .Values.persistence.existingClaim) }} -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: {{ printf "%s-plugins" (include "easyp-service.fullname" .) }} - labels: - {{- include "easyp-service.labels" . | nindent 4 }} - annotations: - # Deliberately kept when the release is removed: re-downloading the whole - # plugin set costs far more than the disk does. - helm.sh/resource-policy: keep -spec: - accessModes: - - {{ .Values.persistence.accessMode }} - resources: - requests: - storage: {{ .Values.persistence.size }} - {{- if .Values.persistence.storageClass }} - storageClassName: {{ .Values.persistence.storageClass }} - {{- end }} -{{- end }} diff --git a/deploy/charts/easyp-service/templates/service-headless.yaml b/deploy/charts/easyp-service/templates/service-headless.yaml new file mode 100644 index 0000000..910d46c --- /dev/null +++ b/deploy/charts/easyp-service/templates/service-headless.yaml @@ -0,0 +1,23 @@ +{{/* +The governing Service a StatefulSet has to name. Nothing is meant to call it — +clients use the regular Service — so it publishes the gRPC port alone. +Specifically not metrics: the ServiceMonitor selects Services by the same labels, +and a second Service carrying the metrics port would have every pod scraped +twice, doubling every counter summed across targets. +*/}} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "easyp-service.fullname" . }}-headless + labels: + {{- include "easyp-service.labels" . | nindent 4 }} +spec: + clusterIP: None + selector: + {{- include "easyp-service.selectorLabels" . | nindent 4 }} + ports: + - name: grpc + port: {{ .Values.ports.grpc }} + targetPort: grpc + protocol: TCP + appProtocol: h2c diff --git a/deploy/charts/easyp-service/templates/deployment.yaml b/deploy/charts/easyp-service/templates/statefulset.yaml similarity index 70% rename from deploy/charts/easyp-service/templates/deployment.yaml rename to deploy/charts/easyp-service/templates/statefulset.yaml index 1dac88c..2e78481 100644 --- a/deploy/charts/easyp-service/templates/deployment.yaml +++ b/deploy/charts/easyp-service/templates/statefulset.yaml @@ -1,6 +1,18 @@ {{- include "easyp-service.validate" . }} +{{/* +A StatefulSet for its volumeClaimTemplates rather than for identity: the pods +are interchangeable and nothing addresses one by name. A claim template is the +one way to give each replica a ReadWriteOnce volume of its own that outlives the +pod, which is what lets the service scale out on plain block storage. Replicas +never share one volume — see the accessMode check in _helpers.tpl. + +Upgrades roll. A StatefulSet deletes a pod before starting its replacement, so a +ReadWriteOnce claim is released before the next pod asks for it; the Deployment +this replaced needed Recreate for that, or waited on a Multi-Attach error +forever. +*/}} apiVersion: apps/v1 -kind: Deployment +kind: StatefulSet metadata: name: {{ include "easyp-service.fullname" . }} labels: @@ -13,19 +25,26 @@ metadata: reloader.stakater.com/auto: "true" {{- end }} spec: + {{- if ne (include "easyp-service.autoscalingEnabled" .) "true" }} replicas: {{ .Values.replicaCount }} - {{- if and .Values.persistence.enabled (ne .Values.persistence.accessMode "ReadWriteMany") }} - # Recreate, because a ReadWriteOnce volume can be attached to one node at a - # time. A rolling update starts the replacement pod before retiring the old - # one, and if the scheduler picks a different node that pod waits on a - # Multi-Attach error forever — the rollout never completes and never fails. - # The choice is not between downtime and none; it is between a short gap and - # a stuck upgrade. Give the pod a ReadWriteMany volume to roll instead. - strategy: - type: Recreate - {{- else }} - strategy: + {{- end }} + serviceName: {{ include "easyp-service.fullname" . }}-headless + # Replicas share nothing but the database, and migrations are serialised by + # an advisory lock, so there is no reason to start them one at a time. Ordered + # startup would make a scale-out from 2 to 8 take six startup probes in a row. + podManagementPolicy: Parallel + updateStrategy: type: RollingUpdate + {{- if .Values.persistence.enabled }} + {{- with .Values.persistence.retentionPolicy }} + # Honoured from Kubernetes 1.27; older API servers drop the field, which + # leaves the claims in place — the same as Retain. Absent altogether (a + # --reuse-values upgrade from 1.0.4), the field is left out, and Kubernetes + # defaults both to Retain as well. + persistentVolumeClaimRetentionPolicy: + whenDeleted: {{ .whenDeleted | default "Retain" }} + whenScaled: {{ .whenScaled | default "Retain" }} + {{- end }} {{- end }} selector: matchLabels: @@ -40,7 +59,7 @@ spec: annotations: # The configuration is read once at startup, so a changed ConfigMap does # nothing until the pods are replaced. Kubernetes does not roll a - # Deployment when a mounted ConfigMap changes; hashing it into the pod + # StatefulSet when a mounted ConfigMap changes; hashing it into the pod # template is what turns `helm upgrade --set config.…` into a rollout # rather than a silent no-op. checksum/config: {{ include (print $.Template.BasePath "/configmap.yaml") . | sha256sum }} @@ -152,18 +171,16 @@ spec: - name: config configMap: name: {{ include "easyp-service.fullname" . }}-config + {{- /* With persistence on, volumeClaimTemplates below supply it. */}} + {{- if not .Values.persistence.enabled }} - name: plugins - {{- if .Values.persistence.enabled }} - persistentVolumeClaim: - claimName: {{ .Values.persistence.existingClaim | default (printf "%s-plugins" (include "easyp-service.fullname" .)) }} - {{- else }} # Ephemeral: every restart re-downloads plugin archives from S3. # sizeLimit is what keeps an overrun local to this pod — without it the # cache grows into the node's ephemeral storage, and the kubelet # resolves disk pressure by evicting pods that need not be this one. emptyDir: sizeLimit: {{ .Values.persistence.ephemeralSizeLimit }} - {{- end }} + {{- end }} {{- if .Values.tls.enabled }} - name: server-tls secret: @@ -190,3 +207,26 @@ spec: topologySpreadConstraints: {{- toYaml . | nindent 8 }} {{- end }} + {{- if .Values.persistence.enabled }} + # A claim per pod, plugins--N. Each starts empty and fills from + # object storage, which is why more than one replica needs S3. Not Helm-owned: + # `helm uninstall` leaves them, and retentionPolicy decides on scale-down. + # + # Immutable once created: a StatefulSet refuses any change here, so raising + # persistence.size fails `helm upgrade`. The README's "Resizing the plugin + # volumes" section has the procedure. + volumeClaimTemplates: + - metadata: + name: plugins + labels: + {{- include "easyp-service.selectorLabels" . | nindent 10 }} + spec: + accessModes: + - {{ .Values.persistence.accessMode }} + resources: + requests: + storage: {{ .Values.persistence.size }} + {{- if .Values.persistence.storageClass }} + storageClassName: {{ .Values.persistence.storageClass }} + {{- end }} + {{- end }} diff --git a/deploy/charts/easyp-service/tests/render.sh b/deploy/charts/easyp-service/tests/render.sh index ac9c4be..89b5a6a 100755 --- a/deploy/charts/easyp-service/tests/render.sh +++ b/deploy/charts/easyp-service/tests/render.sh @@ -386,38 +386,122 @@ fi echo echo "== rollout strategy ==" -# A ReadWriteOnce volume attaches to one node at a time, so a rolling update can -# strand the replacement pod on a Multi-Attach error and hang the upgrade -# indefinitely. Nothing in `helm upgrade` reports that as a failure, which is -# why it is checked here rather than discovered in a cluster. -if expect_render "a ReadWriteOnce volume forces Recreate"; then - if grep -q "type: Recreate" <<<"$out"; then - pass "a ReadWriteOnce volume forces Recreate" +# One workload kind in every storage mode. It used to be a Deployment, which had +# to Recreate on a ReadWriteOnce claim: a rolling update started the replacement +# first, and a replacement on another node waited on Multi-Attach forever. A +# StatefulSet deletes the pod before starting its successor, so every mode rolls. +# A Deployment reappearing here would bring that hang back with it. +for mode in "claim per replica|" \ + "emptyDir|--set persistence.enabled=false --set resources.limits.ephemeral-storage=30Gi"; do + what="${mode%%|*}" + read -r -a args <<<"${mode#*|}" + + if expect_render "$what: a StatefulSet that rolls" ${args[@]+"${args[@]}"}; then + if grep -q "^kind: Deployment$" <<<"$out"; then + fail "$what: a StatefulSet that rolls: a Deployment rendered" + elif ! grep -q "^kind: StatefulSet$" <<<"$out"; then + fail "$what: a StatefulSet that rolls: no StatefulSet rendered" + elif ! grep -A1 "updateStrategy:" <<<"$out" | grep -q "type: RollingUpdate"; then + fail "$what: a StatefulSet that rolls: updateStrategy is not RollingUpdate" + else + pass "$what: a StatefulSet that rolls" + fi + fi +done + +# A StatefulSet must name a governing Service that exists, or the API server +# accepts it and the pods never get their DNS records. +if expect_render "the StatefulSet names the headless Service the chart renders"; then + governing="$(grep -E '^ serviceName: ' <<<"$out" | awk '{print $2}')" + if [[ -n "$governing" ]] && grep -qE "^ name: ${governing}$" <<<"$out" && grep -q "clusterIP: None" <<<"$out"; then + pass "the StatefulSet names the headless Service the chart renders" else - fail "a ReadWriteOnce volume forces Recreate: strategy is not Recreate"$'\n'"$(grep -A2 'strategy:' <<<"$out")" + fail "the StatefulSet names the headless Service the chart renders: serviceName '$governing' has no headless Service" fi fi -# The converse: a shared volume has no attach conflict, so the upgrade should -# roll rather than take the service down for it. -if expect_render "a ReadWriteMany volume rolls" \ - --set persistence.accessMode=ReadWriteMany; then - if grep -q "type: RollingUpdate" <<<"$out"; then - pass "a ReadWriteMany volume rolls" +# The headless Service carries the same selector labels as the real one, and the +# ServiceMonitor selects Services by those labels. With a metrics port on both, +# every pod would be scraped twice and every summed counter doubled. +if expect_render "the headless Service does not publish metrics" \ + --show-only templates/service-headless.yaml; then + if grep -q "name: metrics" <<<"$out"; then + fail "the headless Service does not publish metrics: it does, so the ServiceMonitor would scrape each pod twice" else - fail "a ReadWriteMany volume rolls: strategy is not RollingUpdate" + pass "the headless Service does not publish metrics" fi fi -# Without persistence the volume is an emptyDir, which every pod gets its own -# copy of. Nothing to conflict over, so nothing to take downtime for. -if expect_render "no persistence rolls" \ - --set persistence.enabled=false \ - --set resources.limits.ephemeral-storage=30Gi; then - if grep -q "type: RollingUpdate" <<<"$out"; then - pass "no persistence rolls" +echo +echo "== one volume per replica ==" + +# Plenty of clusters have no ReadWriteMany class, or one too slow to execute +# plugins from — and replicas must not share a cache anyway. Every replica gets +# a ReadWriteOnce claim of its own from volumeClaimTemplates. +S3=(--set config.registry.s3.bucket=plugins) + +if expect_render "every replica gets a claim of its own" \ + "${S3[@]}" --set replicaCount=3; then + problems="" + grep -q "^kind: PersistentVolumeClaim$" <<<"$out" && problems="$problems a-standalone-claim-rendered" + grep -q "volumeClaimTemplates:" <<<"$out" || problems="$problems no-volumeClaimTemplates" + grep -q "persistentVolumeClaim:" <<<"$out" && problems="$problems pod-names-a-claim" + grep -A1 "accessModes:" <<<"$out" | grep -q "ReadWriteOnce" || problems="$problems not-ReadWriteOnce" + grep -qE "^ replicas: 3$" <<<"$out" || problems="$problems replicas-not-3" + + if [[ -n "$problems" ]]; then + fail "every replica gets a claim of its own:$problems" + else + pass "every replica gets a claim of its own" + fi +fi + +# Two pods on one volume corrupt the cache on a concurrent miss for the same +# plugin, and it appears to work until then. Refused, not documented. +expect_failure "a shared access mode is refused" "must be ReadWriteOnce or ReadWriteOncePod" \ + --set persistence.accessMode=ReadWriteMany + +if expect_render "ReadWriteOncePod is accepted" \ + --set persistence.accessMode=ReadWriteOncePod; then + pass "ReadWriteOncePod is accepted" +fi + +# existingClaim was a value until the StatefulSet, and Helm ignores values +# nothing reads. Ignored, an upgrade carrying it would mount an empty claim +# while the named one — without S3, the only copy of every plugin — sat unused. +expect_failure "the removed existingClaim is refused, not ignored" "persistence.existingClaim was removed" \ + --set persistence.existingClaim=mine + +# An empty volume with nothing to fill it from fails every generation. +expect_failure "several replicas without object storage are refused" "needs config.registry.s3.bucket" \ + --set replicaCount=2 + +expect_failure "several emptyDir replicas without object storage are refused" "needs config.registry.s3.bucket" \ + --set replicaCount=2 --set persistence.enabled=false \ + --set resources.limits.ephemeral-storage=30Gi + +# The HPA's ceiling is what matters: a pod it adds under load has to be able +# to start. +expect_failure "autoscaling without object storage is refused" "needs config.registry.s3.bucket" \ + --set autoscaling.enabled=true + +expect_failure "autoscaling with no metric is refused" "an HPA with no metric never scales" \ + "${S3[@]}" --set autoscaling.enabled=true \ + --set autoscaling.targetCPUUtilizationPercentage=null + +expect_failure "a retention policy other than Retain or Delete is refused" "must be Retain or Delete" \ + --set persistence.retentionPolicy.whenScaled=Keep + +# A replicas field alongside an HPA is reset by every `helm upgrade`, scaling the +# workload back to replicaCount until the HPA notices. +if expect_render "autoscaling leaves the replica count to the HPA" \ + "${S3[@]}" --set autoscaling.enabled=true; then + if grep -qE "^ replicas:" <<<"$out"; then + fail "autoscaling leaves the replica count to the HPA: the workload still sets replicas" + elif ! grep -A3 "scaleTargetRef:" <<<"$out" | grep -q "kind: StatefulSet"; then + fail "autoscaling leaves the replica count to the HPA: the HPA does not target the StatefulSet" else - fail "no persistence rolls: strategy is not RollingUpdate" + pass "autoscaling leaves the replica count to the HPA" fi fi @@ -474,6 +558,62 @@ if expect_render "the defaults ship a network policy"; then fi fi +echo +echo "== upgrading with --reuse-values ==" + +# `helm upgrade --reuse-values` replaces the new chart's defaults with the old +# release's values, so every key added since that release is absent rather than +# defaulted. A template reading one as `.Values.new.key` fails with a nil pointer +# and the upgrade stops. It did, for autoscaling, before anything checked it. +# +# Rendering against the old release's values.yaml alone is what the templates +# see on that upgrade. The copy is needed because --reuse-values is a property +# of an installed release; helm template has no way to ask for it. +reuse_chart="$(mktemp -d -t easyp-reuse.XXXXXX)" +cp -R "$CHART/." "$reuse_chart/" +cp "$CHART/tests/values-1.0.4.yaml" "$reuse_chart/values.yaml" + +reuse() { helm template test "$reuse_chart" "${BASE[@]}" "$@"; } + +if out="$(reuse 2>&1)"; then + problems="" + grep -q "^kind: StatefulSet$" <<<"$out" || problems="$problems no-StatefulSet" + grep -q "^kind: HorizontalPodAutoscaler$" <<<"$out" && problems="$problems an-HPA-nobody-asked-for" + grep -qE "^ replicas: 1$" <<<"$out" || problems="$problems replicas-not-1" + + if [[ -n "$problems" ]]; then + fail "an upgrade from 1.0.4 with --reuse-values renders:$problems" + else + pass "an upgrade from 1.0.4 with --reuse-values renders" + fi +else + fail "an upgrade from 1.0.4 with --reuse-values renders"$'\n'"$out" +fi + +# Turning autoscaling on in the same upgrade brings one key of the map and none +# of its defaults. Refused by name, rather than an HPA with no bounds. +if out="$(reuse --set autoscaling.enabled=true --set config.registry.s3.bucket=plugins 2>&1)"; then + fail "a partial autoscaling map is refused by name: it rendered" +elif grep -qF "needs autoscaling.minReplicas and autoscaling.maxReplicas" <<<"$out"; then + pass "a partial autoscaling map is refused by name" +else + fail "a partial autoscaling map is refused by name: failed some other way"$'\n'"$out" +fi + +# One retention key set, the other absent: the absent one is Retain, which is +# also what Kubernetes would pick, not an empty field the API server rejects. +if out="$(reuse --set persistence.retentionPolicy.whenScaled=Delete 2>&1)"; then + if grep -q "whenDeleted: Retain" <<<"$out" && grep -q "whenScaled: Delete" <<<"$out"; then + pass "a partial retention policy fills the rest with Retain" + else + fail "a partial retention policy fills the rest with Retain: $(grep -A2 'persistentVolumeClaimRetentionPolicy' <<<"$out" | tr '\n' ' ')" + fi +else + fail "a partial retention policy fills the rest with Retain"$'\n'"$out" +fi + +rm -rf "$reuse_chart" + echo echo "== install paths a customer actually takes ==" diff --git a/deploy/charts/easyp-service/tests/values-1.0.4.yaml b/deploy/charts/easyp-service/tests/values-1.0.4.yaml new file mode 100644 index 0000000..63dedb2 --- /dev/null +++ b/deploy/charts/easyp-service/tests/values-1.0.4.yaml @@ -0,0 +1,455 @@ +# The chart's values.yaml as released in 1.0.4, unchanged below this header. +# +# `helm upgrade --reuse-values` does not merge a new chart's defaults: it +# replaces them with the previous release's values. So on an upgrade from 1.0.4 +# the templates see exactly this file plus whatever the release overrode, and +# every key added since is simply absent. tests/render.sh renders the current +# templates against this file alone; a template that reads a newer key without +# tolerating its absence fails there instead of in a customer's upgrade. +# +# Do not edit it to make a check pass. When the oldest release worth upgrading +# from moves, replace it with that release's values.yaml, verbatim. +# +# Default values for easyp-service. +# +# Three things have no working default and must be decided before the chart will +# install. Each is refused at install time rather than at runtime, so a missing +# one is a message naming it, not a pod that comes up wrong. +# +# 1. A database DSN — see `secrets`. There is nothing sensible to default to. +# +# 2. TLS material, because `tls.enabled` defaults to true. Supply one of: +# tls.serverSecret= you bring the certificate +# certManager.enabled=true the chart issues it +# tls.enabled=false plaintext, trusted network +# The default stays on: a listener that speaks plaintext is the wrong thing +# to arrive at by saying nothing. But it does mean there is no install that +# consists only of a DSN, and this list said otherwise until v0.14.0. +# +# 3. `ingress.host`, and with it `config.server.trustedProxies`, if you want +# the gRPC API reachable from outside the cluster. Both are required +# together — see the comment on trustedProxies for why the second one is +# not optional. + +replicaCount: 1 + +image: + repository: ghcr.io/easyp-tech/service + # Empty means Chart.appVersion. Pin an explicit tag in production: a floating + # tag makes a rollout non-reproducible. + tag: "" + pullPolicy: IfNotPresent + +imagePullSecrets: [] +nameOverride: "" +fullnameOverride: "" + +serviceAccount: + create: true + annotations: {} + name: "" + +podAnnotations: {} +podLabels: {} +nodeSelector: {} +tolerations: [] +affinity: {} +topologySpreadConstraints: [] + +# The service reads its certificate and key once, at startup, so a rotated +# certificate is not picked up until the pod restarts. With stakater/Reloader +# installed this annotation restarts the deployment when the secret changes. +# Without it, rotation is a manual `kubectl rollout restart`. +reloader: + enabled: true + +# Sized against what the service is configured to do, not guessed. The plugin's +# output is read into memory whole, so peak usage tracks +# config.workerPool.maxConcurrentGenerations × config.registry.maxOutputSize — +# 16 × 64 MiB = 1 GiB of buffers with the defaults below, and the same bytes +# again once marshalled into the gRPC response. _helpers.tpl refuses an install +# where that product no longer fits in the limit. +# +# The earlier default was 1Gi, which the buffers alone filled. The pod was +# OOMKilled under load, and an OOMKill reads as a crash rather than as overload: +# no log line, no rejected-generation counter, no saturation alert. +resources: + requests: + cpu: 500m + memory: 1Gi + # Writable layer and logs only. Both the plugin cache and the archives + # staged mid-download live on the plugins volume, not here — the download + # used to land in /tmp on the writable layer, where a few hundred megabytes + # times the concurrent generations was an eviction rather than an overload. + # With persistence disabled see persistence.ephemeralSizeLimit, which is + # what actually bounds that volume. + ephemeral-storage: 1Gi + limits: + # Four cores for sixteen concurrent plugin processes. On two they each ran + # at an eighth speed and approached generationTimeoutSeconds, which would + # have surfaced as DeadlineExceeded — a broken plugin, to anyone reading — + # rather than as the honest ResourceExhausted the limiter returns. + cpu: "4" + memory: 4Gi + ephemeral-storage: 2Gi + +# Ports the container listens on. The service defaults to 23410-23413 when no +# configuration is given; the chart always sets them explicitly. +ports: + grpc: 8080 + # Singular, matching the service's own server.port.metric. It was `metrics` + # until v0.13.0 — one word with two spellings across three files. + metric: 8081 + health: 8082 + mcp: 8083 + +# The MCP endpoint: a read-only HTTP surface AI tooling uses to read the plugin +# catalog and the easyp.yaml schema. It exposes nothing the anonymous gRPC +# reads do not, but it sits outside the gRPC interceptor chain — no TLS, no +# rate limit, no audit — so serving it is a deployment's explicit decision. +# Off: the container does not listen on ports.mcp and the Service does not +# publish it. Turn it on inside a trusted network or behind an ingress that +# terminates TLS. +mcp: + enabled: false + +service: + type: ClusterIP + annotations: {} + +# Seconds Kubernetes waits after SIGTERM before SIGKILL. +# +# Three numbers have to line up, and _helpers.tpl enforces the order: +# +# generationTimeoutSeconds < forceShutdownAfterSeconds < terminationGracePeriodSeconds +# +# A generation the service accepted must be able to finish; the process must be +# able to exit on its own after that; and Kubernetes must wait for both rather +# than reaching for SIGKILL first. +terminationGracePeriodSeconds: 180 + +# --------------------------------------------------------------------------- +# Configuration that is not secret. Rendered into a ConfigMap and mounted at +# /etc/easyp/config.yml — the same config file every other deployment of this +# service reads. Secrets are not here; they arrive as environment variables from +# the secret above and override the file. +# +# The numbers below repeat the defaults compiled into the service, on purpose: +# this is the chart's documented interface and `helm show values` is where an +# operator looks them up. tests/render.sh checks that they still agree with the +# binary, so a default changed on one side and not the other fails there rather +# than in a cluster. +# --------------------------------------------------------------------------- +config: + logLevel: info + + server: + # CIDRs whose X-Forwarded-For and X-Real-IP headers may be believed. + # + # Required once ingress.enabled is true, and the chart refuses the install + # without it. Behind a proxy every request arrives from the proxy's address, + # so with this empty the rate limit and the per-caller concurrency limit + # become one bucket shared by every client at once, and the audit log + # records the ingress instead of who acted. None of that fails visibly. + # + # Set it to the range your ingress controller's pods run in — the pod CIDR + # of that node pool, not the whole cluster: anything listed here can claim + # to be any client. + trustedProxies: [] + + # Largest single gRPC message accepted and sent. gRPC's own defaults are + # 4 MiB in — which a request carrying a large proto tree exceeds — and + # effectively unlimited out. maxSendMsgSize must be at least + # config.registry.maxOutputSize or the service refuses to start: a plugin + # allowed to produce more than can be sent does its work for nothing. + maxRecvMsgSize: 67108864 + maxSendMsgSize: 67108864 + + # Concurrent streams one connection may hold. gRPC defaults to unlimited, + # which lets a single caller allocate goroutines and buffers until the pod + # is out of memory, before any limiter in the chain sees the requests. + maxConcurrentStreams: 256 + + # Hard exit this long after SIGTERM, if graceful shutdown has not finished. + # Expressed in seconds so the chart can compare it with the two numbers on + # either side of it. + forceShutdownAfterSeconds: 150 + + workerPool: + # Concurrent plugin lookups: a database read, plus a download and unpack + # from object storage on a cache miss. + workers: 4 + # Waiting slots, applied to lookups and to generations alike. Beyond this + # the answer is ErrServerOverloaded rather than a longer queue. + queueSize: 16 + # Concurrent plugin processes. Separate from workers on purpose: a worker is + # released once the plugin is located, and execution runs after that. + maxConcurrentGenerations: 16 + # Expressed in seconds (not "120s") so the chart can compare it against + # terminationGracePeriodSeconds. Rendered as a Go duration for the service. + generationTimeoutSeconds: 120 + maxRetries: 2 + shutdownTimeout: 30s + + rateLimit: + requestsPerSecond: 10.0 + burst: 20 + cleanupInterval: 10m + # Requests one client may have in flight at once. Rate alone does not bound + # this: a caller staying under the rate can still hold every generation slot + # with long requests. 0 disables the check. + maxConcurrentPerIP: 2 + + audit: + bufferSize: 1000 + batchSize: 100 + flushInterval: 1s + maxSaveRetries: 3 + # How long an operation waits for room in the audit queue. This is the only + # place audit can slow a request down, and it takes a backed-up queue to get + # there: a healthy writer drains batchSize/flushInterval — a hundred entries + # a second — so this expires only when the database has stopped keeping up. + # An entry that never finds room is dropped and counted under + # easyp_audit_events_lost_total{reason="enqueue_timeout"}; the operation + # itself still succeeds. + enqueueTimeout: 1s + # Bound on one write to storage, retries included. A batch that misses it is + # lost rather than holding the writer up behind a database that has stopped + # answering. + flushTimeout: 5s + # Partitions older than this are dropped on a schedule — a real delete, so + # afterwards the only copy of that month is in a backup taken while it still + # existed. Backup retention has to *exceed* this rather than match it: + # matching means the month leaves the database and the archive at about the + # same time, leaving it nowhere. + retentionMonths: 12 + preCreateMonths: 3 + partitionCheckInterval: 6h + partitionOpTimeout: 30s + + registry: + pluginsDir: /plugins + maxOutputSize: 67108864 + # Bound on the unpacked plugins on disk. Least recently used ones are + # dropped once this is exceeded; the archive in object storage is left + # alone, so an evicted plugin is one download away rather than lost. + # Keep it below persistence.size — eviction starts at the limit, and the + # volume needs room to reach it. 0 disables eviction. + cacheMaxBytes: 21474836480 # 20 GiB + # Leaving bucket empty disables S3 and makes pluginsDir the only source of + # plugin binaries. + s3: + endpoint: "" + bucket: "" + region: us-east-1 + prefix: "" + forcePathStyle: false + + telemetry: + # Empty means no collector, and the service then builds no exporter at all. + # These carry no fallback on purpose: the OTLP connection is lazy, so an + # endpoint nobody is listening on does not fail — it retries forever, and a + # pod with no observability stack would fill its log with connection errors + # while otherwise working. + otlpEndpoint: "" + pyroscopeEndpoint: "" + # Tags traces and profiles with the licence tier this release serves. Worth + # setting only where community and enterprise run side by side and their + # signals land in the same backends; on a cluster running one tier it would + # distinguish nothing. Metrics carry the same distinction already, so use + # the same word here that the metric label uses. + serviceTier: "" + + license: + # Public halves of the keys licence tokens are signed with, keyed by the key + # id that appears in the token footer. The token names one; the rest are + # here so that a signing key can be rotated without every deployment having + # to change key on the same day. + # + # These are the trust anchor: whoever sets them decides which authority may + # issue licences for this installation. The default below is easyp.tech's own + # published key, so a customer holding a LICENSE_KEY needs nothing else. + # Replace it only if you issue your own licences. + # + # Source of truth is keys/ in the licence registry (easyp-tech/licenses); + # these must be kept in step with it. + # + # Without a key, LICENSE_KEY is ignored and the service runs in community mode. + publicKeys: + "2026-08": "81322461987167d5cfd529e9cb8b96f4797f12fce6be4399a0866e250c9b6bb5" + +# Extra environment variables, appended last. Use for settings the chart does +# not model yet. +extraEnv: [] + +# --------------------------------------------------------------------------- +# Secrets +# +# The pod loads them with envFrom, so the KEYS OF THE SECRET ARE THE ENVIRONMENT +# VARIABLE NAMES. A secret you bring yourself must use exactly these keys: +# +# DB_POSTGRES_DSN required +# REGISTRY_S3_ACCESS_KEY_ID required when config.registry.s3.bucket is set +# REGISTRY_S3_SECRET_ACCESS_KEY required when config.registry.s3.bucket is set +# LICENSE_KEY optional; absent means community mode +# +# The single-key LICENSE_PUBLIC_KEY was removed in v0.13.0. The trust anchor is +# config.license.publicKeys alone: an entry per key id, or one under "*" to +# verify any key id. Without a public key, LICENSE_KEY counts for nothing. +# AUTH_WRITE_TOKENS optional; absent means no writes are allowed +# +# AUTH_WRITE_TOKENS holds sha256 digests, never the tokens: "ci=<64 hex>,me=<64 hex>". +# Generate a token with `easyp-svc auth new-token --name ci`. +# --------------------------------------------------------------------------- +secrets: + # Bring your own secret (recommended). Values put in `data` below end up in + # Helm release history and in `helm get values`, which is why create defaults + # to false. + existingSecret: "" + create: false + data: {} + +# --------------------------------------------------------------------------- +# Transport security +# +# When enabled, the gRPC listener serves TLS. Adding clientCASecret turns it +# into mutual TLS: only callers holding a certificate signed by that CA get in, +# which is what keeps the listener private when the ingress is the only party +# meant to reach it. +# --------------------------------------------------------------------------- +tls: + enabled: true + # Secret of type kubernetes.io/tls holding tls.crt and tls.key for the server. + serverSecret: "" + # Secret holding ca.crt used to verify client certificates. Empty falls back + # to ca.crt inside serverSecret, which is what cert-manager writes for a CA + # issuer. Set to "-" to serve plain server-side TLS with no client check. + clientCASecret: "" + mountPath: /certs + +# Optional: let cert-manager issue the certificates instead of supplying them. +certManager: + enabled: false + issuerRef: + name: "" + kind: ClusterIssuer + group: cert-manager.io + duration: 2160h + renewBefore: 360h + # Extra SANs for the server certificate. The in-cluster service name is + # always included. + extraDnsNames: [] + # Issues the client certificate the ingress presents to the service. + clientCertificate: + enabled: true + commonName: easyp-ingress + +# --------------------------------------------------------------------------- +# Ingress. Defaults target Traefik because it is the only controller that can +# express a mutual-TLS backend leg declaratively, via ServersTransport. +# --------------------------------------------------------------------------- +ingress: + enabled: false + className: traefik + host: "" + annotations: {} + # Secret holding the certificate served to external clients. + tlsSecret: "" + # Traefik ServersTransport describing how the router reaches the pod. Required + # whenever tls.enabled is true, otherwise the backend leg would be plaintext + # against a listener that demands a client certificate. + serversTransport: + enabled: true + # Secret of type kubernetes.io/tls with the client certificate the router + # presents to the service. + clientSecret: "" + # Name the router expects on the service certificate. + serverName: "" + insecureSkipVerify: false + +# --------------------------------------------------------------------------- +# Storage for the plugin cache. Plugin archives are downloaded lazily and +# unpacked here, and the directory is pruned: once it passes +# config.registry.cacheMaxBytes the least recently used plugin versions are +# removed. So size this for the working set you want resident, not for every +# plugin that could ever be requested. +# +# Eviction only ever removes local files. The archive stays in object storage, +# so an evicted plugin costs one download on its next request rather than being +# lost — which is why the cache can be sized to taste. +# +# Whichever of the two sizes below applies has to exceed cacheMaxBytes, and the +# chart refuses to install when it does not: eviction starts at the limit, so +# storage sized exactly to it is already full by the time the cache first needs +# headroom. +# --------------------------------------------------------------------------- +persistence: + enabled: true + existingClaim: "" + storageClass: "" + accessMode: ReadWriteOnce + size: 25Gi + # Used when enabled=false, where the cache lives in an emptyDir on the node + # instead of a volume of its own. Without a limit here the cache grows into + # the node's ephemeral storage, and disk pressure there is resolved by the + # kubelet evicting pods — not necessarily this one. With it, an overrun stops + # at the pod that caused it. + # + # Remember this is per replica: with persistence off, every pod keeps its own + # copy, so plan for replicaCount × this much on the nodes. + ephemeralSizeLimit: 25Gi + +serviceMonitor: + enabled: false + interval: 30s + scrapeTimeout: 10s + labels: {} + +# Alerting rules. Needs the Prometheus Operator CRDs, same as serviceMonitor, +# which is why this defaults to off rather than failing the install where they +# are absent. Turn it on wherever you actually run this. +# +# The rules worth having on day one are the licence ones: a licence that lapses +# breaks nothing loudly. The tier drops to community, audit stops, and the +# plugin limit starts rejecting registrations — all silently. +prometheusRule: + enabled: false + # Extra labels, usually what your Prometheus selects rules by. + labels: {} + # Where each alert's runbook_url points. Every alert gets one, anchored on its + # own name. Point this at your own copy if you keep runbooks internally — + # an alert that arrives at 3am with no procedure attached is where most of the + # time goes. + runbookBaseUrl: https://easyp.tech/docs/api-service/runbooks + # How long before expiry to start warning. The token also carries its own + # grace period, which runs after this; both exist so a renewal that slips + # does not take a customer's pipeline down. + licenceExpiryWarningDays: 14 + # Fraction of config.registry.cacheMaxBytes that counts as full. + cacheUsageWarningRatio: 0.95 + # Fraction of generations allowed to fail before alerting. + generationErrorRatio: 0.05 + # Rejected write credentials per second before alerting. + authFailureRate: 0.5 + +podDisruptionBudget: + enabled: false + minAvailable: 1 + +# On by default. This pod's whole job is running third-party binaries, so the +# set of places it can reach should be stated rather than inherited. +# +# If your database, object storage or collector is not on one of the ports below, +# add it here or the pod goes quiet in a way that looks like a hang. That is the +# one thing to check first after enabling this. +networkPolicy: + enabled: true + # Selector matching the namespace your ingress controller runs in. Empty means + # the gRPC port accepts connections from anywhere in the cluster — set it once + # you know which namespace your controller runs in. + ingressNamespaceSelector: {} + # Ports the pod is allowed to reach outbound, beyond DNS. + egressPorts: + - 5432 # PostgreSQL + - 443 # object storage, and anything a plugin fetches over HTTPS + - 4317 # OTLP diff --git a/deploy/charts/easyp-service/values.yaml b/deploy/charts/easyp-service/values.yaml index fa54d19..e250c5d 100644 --- a/deploy/charts/easyp-service/values.yaml +++ b/deploy/charts/easyp-service/values.yaml @@ -19,8 +19,29 @@ # together — see the comment on trustedProxies for why the second one is # not optional. +# Ignored while autoscaling.enabled: the HPA owns the count then, and a +# `replicas` field in the template would reset it on every `helm upgrade`. +# More than one needs config.registry.s3 — see persistence. replicaCount: 1 +# CPU is the signal because every generation is a plugin process charged to this +# container. Needs metrics-server. Each replica runs its own worker pool and +# limits, so capacity scales with it: maxReplicas × maxConcurrentGenerations +# generations at once, cluster-wide. The per-client limits (rateLimit) are per +# pod too, so a client spread across N pods gets up to N times its limit. +autoscaling: + enabled: false + minReplicas: 2 + maxReplicas: 6 + targetCPUUtilizationPercentage: 70 + targetMemoryUtilizationPercentage: null + # Passed through to the HPA as-is. Scale-down is worth slowing: a replica + # removed stops serving from its warm cache. Its claim is kept under + # retentionPolicy.whenScaled=Retain, an emptyDir is not. + behavior: + scaleDown: + stabilizationWindowSeconds: 600 + image: repository: ghcr.io/easyp-tech/service # Empty means Chart.appVersion. Pin an explicit tag in production: a floating @@ -46,7 +67,7 @@ topologySpreadConstraints: [] # The service reads its certificate and key once, at startup, so a rotated # certificate is not picked up until the pod restarts. With stakater/Reloader -# installed this annotation restarts the deployment when the secret changes. +# installed this annotation restarts the pods when the secret changes. # Without it, rotation is a manual `kubectl rollout restart`. reloader: enabled: true @@ -370,13 +391,38 @@ ingress: # chart refuses to install when it does not: eviction starts at the limit, so # storage sized exactly to it is already full by the time the cache first needs # headroom. +# +# The workload is a StatefulSet, and where the cache goes is one of two: +# +# enabled=true (default) a ReadWriteOnce claim per replica from +# volumeClaimTemplates, plugins--N. Kept across +# restarts, upgrades and — by retentionPolicy — scale-down. +# enabled=false an emptyDir per pod on node disk, lost with the pod. +# +# Replicas never share a volume, so there is no ReadWriteMany option: unpacking +# a plugin is serialised by an in-process lock only, and two pods on one volume +# corrupt the cache on a concurrent miss for the same plugin. +# +# More than one replica needs config.registry.s3. Each replica's volume starts +# empty and object storage is the only thing that fills it. # --------------------------------------------------------------------------- persistence: enabled: true - existingClaim: "" storageClass: "" + # ReadWriteOnce or ReadWriteOncePod; anything shared is refused, see above. accessMode: ReadWriteOnce + # Per replica, so the total is replicas × this. It cannot be changed by + # `helm upgrade`: a StatefulSet's claim templates are immutable. See the + # README's "Resizing the plugin volumes". size: 25Gi + # What happens to a replica's claim when the StatefulSet is scaled down or + # deleted. Retain keeps the warm cache for the next scale-up and survives + # `helm uninstall`, as the single claim of earlier releases did; Delete stops + # an autoscaler from leaving a trail of idle disks. Honoured from Kubernetes + # 1.27; older API servers ignore it and retain. + retentionPolicy: + whenDeleted: Retain + whenScaled: Retain # Used when enabled=false, where the cache lives in an emptyDir on the node # instead of a volume of its own. Without a limit here the cache grows into # the node's ephemeral storage, and disk pressure there is resolved by the