diff --git a/.github/workflows/go.yml b/.github/workflows/go.yml index 0673e1b..341cbaa 100644 --- a/.github/workflows/go.yml +++ b/.github/workflows/go.yml @@ -300,7 +300,7 @@ jobs: if: failure() run: | kubectl get pods -o wide - kubectl describe deploy/easyp-easyp-service | tail -40 + kubectl describe statefulset/easyp-easyp-service | tail -40 kubectl describe pod -l app.kubernetes.io/name=easyp-service | tail -60 kubectl logs -l app.kubernetes.io/name=easyp-service --all-containers --tail=100 || true kubectl get events --sort-by=.lastTimestamp | tail -30 @@ -315,14 +315,17 @@ jobs: # Chart 0.3.1 fixed exactly this and nothing tested it: the secret was not # hashed into the pod template, so rotating a write token left the old pod # running and the new token rejected until someone restarted by hand. + # By UID, not name: a StatefulSet's replacement pod is named exactly like + # the one it replaced, so comparing names would report a pod that was + # rolled as one that was not. - name: Rotating a write token rolls the pod run: | - before="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.name}')" + before="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.uid}')" helm upgrade easyp deploy/charts/easyp-service --reuse-values \ --set secrets.data.AUTH_WRITE_TOKENS="ci=$(printf 'rotated' | sha256sum | cut -d' ' -f1)" \ --wait --timeout 5m kubectl wait --for=condition=Ready pod -l app.kubernetes.io/name=easyp-service --timeout=5m - after="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.name}')" + after="$(kubectl get pod -l app.kubernetes.io/name=easyp-service -o jsonpath='{.items[0].metadata.uid}')" if [[ "$before" == "$after" ]]; then echo "::error::the pod was not replaced; the secret is not hashed into the pod template" exit 1 diff --git a/README.md b/README.md index 68fd1dc..7e73cea 100644 --- a/README.md +++ b/README.md @@ -914,24 +914,29 @@ chain and is wrapped separately. That is a single shared credential, not identity: it decides *whether* a caller may read, never *which* caller is reading. -### It does not run in more than one replica +### Replicas must not share a plugin cache The database side is safe: migrations take a Postgres session lock and audit partition maintenance takes an advisory lock, so several processes cannot collide there. -The plugin cache cannot. Unpacking finishes with a remove followed by a rename — -two steps, not atomic — and the only thing serialising it is an in-process -lock. Two pods on one `ReadWriteMany` volume race: one can delete a directory -the other is reading, and a rename onto a directory recreated in between fails. -Each pod also keeps its own in-memory accounting of the same shared bytes, so -`registry.cache_max_bytes` is applied once per pod to one volume. - -The chart's defaults are honest about this — one replica, `ReadWriteOnce`, -`Recreate`. It *permits* `replicaCount > 1` with a `ReadWriteMany` volume, and -that combination is **not supported**: it will appear to work and corrupt the -cache under concurrent misses for the same plugin. Scale by making the single -pod bigger — `maxConcurrentGenerations` and CPU — not by adding pods. +The plugin cache cannot be shared. Unpacking finishes with a remove followed by +a rename — two steps, not atomic — and the only thing serialising it is an +in-process lock. Two pods on one `ReadWriteMany` volume race: one can delete a +directory the other is reading, and a rename onto a directory recreated in +between fails. Each pod also keeps its own in-memory accounting of the same +shared bytes, so `registry.cache_max_bytes` is applied once per pod to one +volume. + +So the service scales out with a cache per pod, never one between them. The +Helm chart runs a StatefulSet that gives each replica a ReadWriteOnce claim of +its own, or an emptyDir with `persistence.enabled=false`, and refuses a shared +access mode outright. More than one replica needs `registry.s3`, since each pod +fills its own cache from object storage. + +Everything else a replica enforces is its own: the worker pool, the concurrent +generation cap and the per-client rate limit apply per pod, so a client spread +across N pods gets up to N times its limit. ### There is no down-migration path diff --git a/deploy/charts/easyp-service/README.md b/deploy/charts/easyp-service/README.md index b89d2ab..6a75a19 100644 --- a/deploy/charts/easyp-service/README.md +++ b/deploy/charts/easyp-service/README.md @@ -173,24 +173,175 @@ fail the request in a way that looks like a corrupt artifact. Watch `easyp_plugin_cache_bytes` against the limit, and `easyp_plugin_cache_evictions_total` for churn. -The PVC carries `helm.sh/resource-policy: keep`, because refilling the cache -costs more than the disk. +The claims outlive the release: they are created by the StatefulSet rather than +by Helm, so `helm uninstall` leaves them, because refilling the cache costs more +than the disk. `persistence.retentionPolicy.whenDeleted=Delete` changes that. -`replicaCount > 1` needs `persistence.accessMode=ReadWriteMany`; the chart fails -the install otherwise rather than leaving pods stuck in Pending. +## Scaling -### Upgrades take the service down briefly +The workload is a StatefulSet. The pods keep no state of their own — plugin +metadata and audit live in Postgres, archives in object storage — and nothing +addresses one by name; the StatefulSet is there for `volumeClaimTemplates`, the +one way to give each replica a ReadWriteOnce volume that outlives the pod. -With a `ReadWriteOnce` volume the deployment uses `strategy: Recreate`, so an -upgrade stops the running pod before starting its replacement. That gap is -deliberate. A rolling update would start the new pod first, and because a -ReadWriteOnce volume attaches to one node at a time, a replacement scheduled -anywhere else waits on a Multi-Attach error indefinitely — `helm upgrade` neither -completes nor fails. +| | Replicas | Cache after a restart | +|---|---|---| +| `persistence.enabled=true` (default) | any | warm: a claim per replica, `plugins--easyp-service-N` | +| `persistence.enabled=false` | any | cold: an emptyDir per pod | -If the gap is unacceptable, the answer is a `ReadWriteMany` volume, not a -different strategy: with a shared volume the chart rolls, and `replicaCount` can -exceed one. +Replicas never share a volume, and the chart has no ReadWriteMany option. +Unpacking a plugin is serialised by an in-process lock only, so two pods on one +volume race on a concurrent miss for the same plugin and corrupt it; each would +also apply `cacheMaxBytes` to the same bytes on its own. `persistence.accessMode` +accepts `ReadWriteOnce` and `ReadWriteOncePod` and refuses anything else. That +also means scaling needs nothing but ordinary block storage — no NFS, no RWX +class. + +More than one replica needs object storage: + +```bash +helm install easyp oci://ghcr.io/easyp-tech/charts/easyp-service \ + --set secrets.existingSecret=easyp-env \ + --set config.registry.s3.bucket=easyp-plugins \ + --set config.registry.s3.endpoint=https://s3.example.com \ + --set autoscaling.enabled=true \ + ... +``` + +Every replica's volume starts empty and S3 is the only thing that fills it; +without S3 the plugins directory is the one copy of every plugin, and only one +replica can hold it. The chart refuses more than one replica — or an autoscaler +allowed more than one — without `config.registry.s3.bucket`, rather than +starting pods that have no plugins. + +Storage is `replicas × persistence.size`, and each new replica downloads what it +serves. `persistence.retentionPolicy.whenScaled` decides what happens to a claim +when the StatefulSet scales down: `Retain` (the default) keeps the warm cache for +the next scale-up, `Delete` stops an autoscaler from leaving idle disks behind. +It needs Kubernetes 1.27; older API servers ignore it and retain. + +### Upgrades roll + +A StatefulSet deletes a pod before starting its replacement, so a ReadWriteOnce +claim is released before the next pod asks for it. With one replica that is a +short gap; with several, the others serve through it. Pods start in parallel +(`podManagementPolicy: Parallel`) — migrations are serialised by an advisory +lock, so there is nothing to order. + +The chart used to render a Deployment, which needed `strategy: Recreate` on a +ReadWriteOnce claim: a rolling update started the replacement first, and one +scheduled on another node waited on a Multi-Attach error forever. + +### Upgrading from a release that rendered a Deployment + +Earlier releases rendered a Deployment and one standalone claim, +`-easyp-service-plugins`, or mounted the claim named by +`persistence.existingClaim`. Both are gone: the StatefulSet's pods mount claims +from the template, `plugins--easyp-service-N`, and `existingClaim` is +refused rather than ignored, so an upgrade that still sets it stops before it +starts pods on an empty volume. The chart's own old claim carries +`helm.sh/resource-policy: keep`, so Helm leaves it in place, unmounted, and the +install notes say so when they find it. + +What to do with the old claim depends on whether the release uses object +storage: + +- **With S3**, nothing is lost. The new claims fill on demand, a cold start per + replica. Upgrade, then delete the old claim once the pods are serving. +- **Without S3**, the old claim holds the only copy of every plugin. Before + upgrading, rebind its PersistentVolume to the name the StatefulSet will look + for; the StatefulSet then adopts it as replica 0's claim instead of creating + an empty one. + +```bash +release=easyp +old=easyp-easyp-service-plugins # or the claim existingClaim named +new=plugins-easyp-easyp-service-0 + +pv="$(kubectl get pvc "$old" -o jsonpath='{.spec.volumeName}')" +class="$(kubectl get pvc "$old" -o jsonpath='{.spec.storageClassName}')" +size="$(kubectl get pvc "$old" -o jsonpath='{.spec.resources.requests.storage}')" + +# Keep the volume when its claim goes away, and stop the pod that mounts it. +kubectl patch pv "$pv" -p '{"spec":{"persistentVolumeReclaimPolicy":"Retain"}}' +kubectl scale deploy/easyp-easyp-service --replicas=0 +kubectl delete pvc "$old" + +# Free the volume, then claim it under the StatefulSet's name. +kubectl patch pv "$pv" --type=json -p '[{"op":"remove","path":"/spec/claimRef"}]' +kubectl create -f - < --reuse-values \ + --set persistence.existingClaim=null --set persistence.size="$size" +``` + +`--reuse-values` works for this upgrade, with one thing to know. It does not +merge the new chart's defaults: it replaces them with the old release's values, +so every key added since — `autoscaling`, `persistence.retentionPolicy` — is +absent rather than defaulted. The templates read an absent key as off (or, for +retention, as Retain), so the release comes up as it was. Turning autoscaling on +in the same upgrade is the exception: `--set autoscaling.enabled=true` brings +that one key and none of its bounds, and the chart refuses it by name. Use +`--reset-then-reuse-values` (Helm 3.14+) for that, which starts from the new +defaults and applies the old release's values over them. + +If the release was already upgraded and its pod started on an empty claim, the +same steps apply with `kubectl scale statefulset/easyp-easyp-service +--replicas=0` in place of the Deployment, deleting the empty +`plugins-easyp-easyp-service-0` alongside the old claim, and scaling back to one +afterwards. + +Tooling that addressed `deploy/-easyp-service` — `kubectl rollout`, +`kubectl logs`, dashboards, alert selectors on `kube_deployment_*` — needs +`statefulset/` instead. + +### Autoscaling + +`autoscaling.enabled` renders a HorizontalPodAutoscaler on CPU — every +generation is a plugin process charged to this container, so CPU is where load +shows. It needs metrics-server; without it the HPA reports `` and never +scales. `replicaCount` is ignored while it is on, so `helm upgrade` does not +reset what the HPA chose. + +Every limit in the service is per pod: the worker pool, the concurrent +generation cap and the per-client rate limit. Capacity therefore scales with +replicas, and so does what one client can use — a client whose requests land on +N pods gets up to N times `rateLimit`. The Community plugin cap counts rows in +the database and holds across replicas; the per-pod ceilings apply to each one. + +### Resizing the plugin volumes + +A StatefulSet's claim templates are immutable, so raising `persistence.size` +fails `helm upgrade`. With a storage class that allows expansion: + +```bash +# 1. Grow the existing claims in place. +for pvc in $(kubectl get pvc -l app.kubernetes.io/instance=easyp -o name); do + kubectl patch "$pvc" -p '{"spec":{"resources":{"requests":{"storage":"50Gi"}}}}' +done + +# 2. Drop the StatefulSet object, leaving its pods and claims running. +kubectl delete statefulset easyp-easyp-service --cascade=orphan + +# 3. Create it again with the new template; it adopts the pods and claims. +helm upgrade easyp ... --reuse-values --set persistence.size=50Gi +``` + +Raise `config.registry.cacheMaxBytes` in the same upgrade if the extra room is +meant for the cache. ## Behind an ingress: `config.server.trustedProxies` @@ -315,6 +466,6 @@ overrides the file. See `values.yaml`; every key is commented. The install-time checks in `_helpers.tpl` reject combinations that would otherwise fail confusingly at -runtime — missing DSN, plaintext router against a TLS listener, multi-replica -`ReadWriteOnce`, grace period shorter than the generation timeout, an ingress +runtime — missing DSN, plaintext router against a TLS listener, a shared +volume, more than one replica without object storage, grace period shorter than the generation timeout, an ingress with no trusted proxies, peak generation buffers larger than the memory limit. diff --git a/deploy/charts/easyp-service/templates/NOTES.txt b/deploy/charts/easyp-service/templates/NOTES.txt index 21c86c1..fc689ca 100644 --- a/deploy/charts/easyp-service/templates/NOTES.txt +++ b/deploy/charts/easyp-service/templates/NOTES.txt @@ -2,12 +2,12 @@ Watch it come up: - kubectl -n {{ .Release.Namespace }} rollout status deploy/{{ include "easyp-service.fullname" . }} + kubectl -n {{ .Release.Namespace }} rollout status statefulset/{{ include "easyp-service.fullname" . }} If the pod restarts immediately, the cause is almost always configuration and it is printed on the first line of the log, not in the pod events: - kubectl -n {{ .Release.Namespace }} logs deploy/{{ include "easyp-service.fullname" . }} + kubectl -n {{ .Release.Namespace }} logs statefulset/{{ include "easyp-service.fullname" . }} Three things worth checking before you call it done: @@ -23,7 +23,7 @@ Three things worth checking before you call it done: {{- else }} 1. LICENSE — a key was supplied. Confirm it was accepted: - kubectl -n {{ .Release.Namespace }} logs deploy/{{ include "easyp-service.fullname" . }} | grep -i licen + kubectl -n {{ .Release.Namespace }} logs statefulset/{{ include "easyp-service.fullname" . }} | grep -i licen {{- end }} {{ if not (hasKey .Values.secrets.data "AUTH_WRITE_TOKENS") }} @@ -65,8 +65,7 @@ Three things worth checking before you call it done: PLUGIN CACHE IS ON NODE DISK. persistence is disabled, so each pod keeps its own copy of the plugin cache in - an emptyDir. That is a supported way to run several replicas without - ReadWriteMany storage, but the space is the node's, and it is charged per + an emptyDir, lost with the pod. The space is the node's, and it is charged per replica rather than once: {{ .Values.replicaCount }} × {{ .Values.persistence.ephemeralSizeLimit }} across the nodes these pods land on @@ -77,6 +76,38 @@ PLUGIN CACHE IS ON NODE DISK. resources.requests.ephemeral-storage is what the scheduler actually reads. {{- end }} +{{- if .Values.persistence.enabled }} + +PLUGIN CACHE IS ONE VOLUME PER REPLICA. + + Each pod has its own {{ .Values.persistence.size }} claim (plugins-{{ include "easyp-service.fullname" . }}-N) and fills it + from object storage on first use, so a new replica starts cold. Resizing + cannot go through `helm upgrade` — the claim template is immutable; see the + chart README, "Resizing the plugin volumes". +{{- /* +Releases before the StatefulSet created one claim, -plugins, and kept +it on removal. Nothing mounts it any more; without S3 it holds the only copy of +every plugin, which is worth saying at the moment it stops being used. +*/}} +{{- $legacy := printf "%s-plugins" (include "easyp-service.fullname" .) }} +{{- if and .Release.IsUpgrade (lookup "v1" "PersistentVolumeClaim" .Release.Namespace $legacy) }} + + THE PREVIOUS CLAIM, {{ $legacy }}, IS NO LONGER MOUNTED. It was kept, not + deleted.{{ if not .Values.config.registry.s3.bucket }} Without S3 it holds the only copy of every plugin, and + the new claim started empty: rebind its volume to plugins-{{ include "easyp-service.fullname" . }}-0 as the + chart README describes, "Upgrading from a release that rendered a Deployment".{{ else }} + The new claims fill from S3, so delete it once the pods are serving.{{ end }} +{{- end }} +{{- end }} + +{{- if eq (include "easyp-service.autoscalingEnabled" .) "true" }} + +AUTOSCALING between {{ .Values.autoscaling.minReplicas }} and {{ .Values.autoscaling.maxReplicas }} replicas. It needs metrics-server; without it the +HPA shows targets and never scales. Check: + + kubectl -n {{ .Release.Namespace }} get hpa {{ include "easyp-service.fullname" . }} +{{- end }} + {{ if not .Values.ingress.enabled }} The API is not exposed outside the cluster (ingress.enabled=false). To try it: diff --git a/deploy/charts/easyp-service/templates/_helpers.tpl b/deploy/charts/easyp-service/templates/_helpers.tpl index e107bc0..694d47a 100644 --- a/deploy/charts/easyp-service/templates/_helpers.tpl +++ b/deploy/charts/easyp-service/templates/_helpers.tpl @@ -89,10 +89,27 @@ what a cert-manager CA issuer produces. {{- end }} {{- end }} +{{/* +Whether autoscaling is on, read so that an absent autoscaling map means off. + +`helm upgrade --reuse-values` replaces this chart's defaults with the previous +release's values, so a release installed before a key existed arrives without +it — and `.Values.autoscaling.enabled` on a missing map is a nil-pointer +failure that stops the upgrade. Absent is what an older release was: off. + +Every value added after 1.0.4 is read this way or through `with`, which skips a +missing one. tests/values-1.0.4.yaml pins it: the templates must render against +that release's values alone. +*/}} +{{- define "easyp-service.autoscalingEnabled" -}} +{{- dig "autoscaling" "enabled" false .Values.AsMap -}} +{{- end }} + {{- define "easyp-service.mutualTLS" -}} {{- and .Values.tls.enabled (ne .Values.tls.clientCASecret "-") -}} {{- end }} + {{/* Converts a Kubernetes quantity like "25Gi" to plain bytes, so that the volume size and the cache limit can be compared. Only the suffixes a volume size @@ -168,9 +185,84 @@ nothing but a log line to say so. {{- fail "easyp-service: secrets.create is true but secrets.data.DB_POSTGRES_DSN is empty." }} {{- end }} -{{- if gt (int .Values.replicaCount) 1 }} -{{- if and .Values.persistence.enabled (eq .Values.persistence.accessMode "ReadWriteOnce") }} -{{- fail "easyp-service: replicaCount > 1 needs persistence.accessMode=ReadWriteMany, otherwise only one pod can mount the plugin cache and the rest stay Pending." }} +{{- /* +Replicas never share a volume. Unpacking a plugin is serialised by an in-process +lock only, so two pods on one ReadWriteMany claim race on a concurrent miss for +the same plugin and corrupt it — and each applies cacheMaxBytes to the same +bytes on its own. Refused rather than documented, because it appears to work. +*/}} +{{- /* +persistence.existingClaim was a value until the chart moved to a StatefulSet, +and Helm ignores values nothing reads. Ignored, an upgrade carrying it would +mount a fresh, empty claim while the one named here — without S3, the only copy +of every plugin — sat unmounted, and nothing would say so. +*/}} +{{- if .Values.persistence.existingClaim }} +{{- fail (printf "easyp-service: persistence.existingClaim was removed; every replica now gets a claim of its own from the StatefulSet's volumeClaimTemplates, named plugins-%s-N. To keep the data in %q, rebind its PersistentVolume to plugins-%s-0 (see the chart README, \"Upgrading from a release that rendered a Deployment\"), then unset persistence.existingClaim." (include "easyp-service.fullname" .) .Values.persistence.existingClaim (include "easyp-service.fullname" .)) }} +{{- end }} + +{{- if not (has .Values.persistence.accessMode (list "ReadWriteOnce" "ReadWriteOncePod")) }} +{{- fail (printf "easyp-service: persistence.accessMode must be ReadWriteOnce or ReadWriteOncePod, got %q. Replicas must not share a plugin cache: pods unpacking the same plugin at once corrupt it. Each replica gets a claim of its own instead." .Values.persistence.accessMode) }} +{{- end }} + +{{- /* +With autoscaling the ceiling is what counts: a pod the HPA adds under load has +to be able to start. +*/}} +{{- $autoscaling := eq (include "easyp-service.autoscalingEnabled" .) "true" }} +{{- /* Not ternary: it evaluates both branches, and the HPA's may not exist. */}} +{{- $maxReplicas := int .Values.replicaCount }} +{{- if $autoscaling }} +{{- $maxReplicas = int .Values.autoscaling.maxReplicas }} +{{- end }} +{{- if gt $maxReplicas 1 }} +{{- /* +Every replica's volume starts empty. Without S3 the plugins directory is the one +copy of every plugin, so a second replica would have none and fail every +generation with NotFound. +*/}} +{{- if not .Values.config.registry.s3.bucket }} +{{- fail "easyp-service: more than one replica needs config.registry.s3.bucket. Each replica's cache starts empty and object storage is the only thing that fills it; without S3 the plugins directory is the sole copy of every plugin, and only one replica can hold it." }} +{{- end }} +{{- end }} + +{{- if $autoscaling }} +{{- /* +Present but partial: `--reuse-values --set autoscaling.enabled=true` on a release +from before autoscaling existed carries that one key and none of the defaults. +Refused rather than guessed, so the numbers live in values.yaml alone. +*/}} +{{- if not (and (hasKey .Values.autoscaling "minReplicas") (hasKey .Values.autoscaling "maxReplicas")) }} +{{- fail "easyp-service: autoscaling.enabled needs autoscaling.minReplicas and autoscaling.maxReplicas. A --reuse-values upgrade from a release that predates autoscaling carries none of its defaults; upgrade with --reset-then-reuse-values, or set both." }} +{{- end }} +{{- if gt (int .Values.autoscaling.minReplicas) (int .Values.autoscaling.maxReplicas) }} +{{- fail (printf "easyp-service: autoscaling.minReplicas (%d) exceeds autoscaling.maxReplicas (%d)." (int .Values.autoscaling.minReplicas) (int .Values.autoscaling.maxReplicas)) }} +{{- end }} +{{- if not (or .Values.autoscaling.targetCPUUtilizationPercentage .Values.autoscaling.targetMemoryUtilizationPercentage) }} +{{- fail "easyp-service: autoscaling.enabled needs targetCPUUtilizationPercentage or targetMemoryUtilizationPercentage; an HPA with no metric never scales." }} +{{- end }} +{{- /* +Utilisation is a percentage of the request. Without one the HPA reports + and holds at whatever it last had, which looks like a quiet day. +*/}} +{{- if .Values.autoscaling.targetCPUUtilizationPercentage }} +{{- if not (and .Values.resources.requests .Values.resources.requests.cpu) }} +{{- fail "easyp-service: autoscaling.targetCPUUtilizationPercentage needs resources.requests.cpu; utilisation is measured against the request." }} +{{- end }} +{{- end }} +{{- if .Values.autoscaling.targetMemoryUtilizationPercentage }} +{{- if not (and .Values.resources.requests .Values.resources.requests.memory) }} +{{- fail "easyp-service: autoscaling.targetMemoryUtilizationPercentage needs resources.requests.memory; utilisation is measured against the request." }} +{{- end }} +{{- end }} +{{- end }} + +{{- with .Values.persistence.retentionPolicy }} +{{- range $when := list "whenDeleted" "whenScaled" }} +{{- /* A key left out means Retain, as the StatefulSet renders it. */}} +{{- if and (hasKey $.Values.persistence.retentionPolicy $when) (not (has (index $.Values.persistence.retentionPolicy $when) (list "Retain" "Delete"))) }} +{{- fail (printf "easyp-service: persistence.retentionPolicy.%s must be Retain or Delete." $when) }} +{{- end }} {{- end }} {{- end }} diff --git a/deploy/charts/easyp-service/templates/hpa.yaml b/deploy/charts/easyp-service/templates/hpa.yaml new file mode 100644 index 0000000..8b40182 --- /dev/null +++ b/deploy/charts/easyp-service/templates/hpa.yaml @@ -0,0 +1,46 @@ +{{- if eq (include "easyp-service.autoscalingEnabled" .) "true" }} +{{/* +Scales on CPU because the work is CPU: every generation is a plugin process +running inside this container, so its usage is charged here. Memory is offered +but off by default — the buffers it would react to are bounded by +maxConcurrentGenerations per pod, and freed as soon as a response is sent, so it +tracks CPU without adding anything. + +Needs metrics-server. Without it the HPA exists, reports and scales +nothing, which is why NOTES.txt names it. +*/}} +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: {{ include "easyp-service.fullname" . }} + labels: + {{- include "easyp-service.labels" . | nindent 4 }} +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: StatefulSet + name: {{ include "easyp-service.fullname" . }} + minReplicas: {{ .Values.autoscaling.minReplicas }} + maxReplicas: {{ .Values.autoscaling.maxReplicas }} + metrics: + {{- with .Values.autoscaling.targetCPUUtilizationPercentage }} + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: {{ . }} + {{- end }} + {{- with .Values.autoscaling.targetMemoryUtilizationPercentage }} + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: {{ . }} + {{- end }} + {{- with .Values.autoscaling.behavior }} + behavior: + {{- toYaml . | nindent 4 }} + {{- end }} +{{- end }} diff --git a/deploy/charts/easyp-service/templates/pvc.yaml b/deploy/charts/easyp-service/templates/pvc.yaml deleted file mode 100644 index 509384d..0000000 --- a/deploy/charts/easyp-service/templates/pvc.yaml +++ /dev/null @@ -1,21 +0,0 @@ -{{- if and .Values.persistence.enabled (not .Values.persistence.existingClaim) }} -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: {{ printf "%s-plugins" (include "easyp-service.fullname" .) }} - labels: - {{- include "easyp-service.labels" . | nindent 4 }} - annotations: - # Deliberately kept when the release is removed: re-downloading the whole - # plugin set costs far more than the disk does. - helm.sh/resource-policy: keep -spec: - accessModes: - - {{ .Values.persistence.accessMode }} - resources: - requests: - storage: {{ .Values.persistence.size }} - {{- if .Values.persistence.storageClass }} - storageClassName: {{ .Values.persistence.storageClass }} - {{- end }} -{{- end }} diff --git a/deploy/charts/easyp-service/templates/service-headless.yaml b/deploy/charts/easyp-service/templates/service-headless.yaml new file mode 100644 index 0000000..910d46c --- /dev/null +++ b/deploy/charts/easyp-service/templates/service-headless.yaml @@ -0,0 +1,23 @@ +{{/* +The governing Service a StatefulSet has to name. Nothing is meant to call it — +clients use the regular Service — so it publishes the gRPC port alone. +Specifically not metrics: the ServiceMonitor selects Services by the same labels, +and a second Service carrying the metrics port would have every pod scraped +twice, doubling every counter summed across targets. +*/}} +apiVersion: v1 +kind: Service +metadata: + name: {{ include "easyp-service.fullname" . }}-headless + labels: + {{- include "easyp-service.labels" . | nindent 4 }} +spec: + clusterIP: None + selector: + {{- include "easyp-service.selectorLabels" . | nindent 4 }} + ports: + - name: grpc + port: {{ .Values.ports.grpc }} + targetPort: grpc + protocol: TCP + appProtocol: h2c diff --git a/deploy/charts/easyp-service/templates/deployment.yaml b/deploy/charts/easyp-service/templates/statefulset.yaml similarity index 70% rename from deploy/charts/easyp-service/templates/deployment.yaml rename to deploy/charts/easyp-service/templates/statefulset.yaml index 1dac88c..2e78481 100644 --- a/deploy/charts/easyp-service/templates/deployment.yaml +++ b/deploy/charts/easyp-service/templates/statefulset.yaml @@ -1,6 +1,18 @@ {{- include "easyp-service.validate" . }} +{{/* +A StatefulSet for its volumeClaimTemplates rather than for identity: the pods +are interchangeable and nothing addresses one by name. A claim template is the +one way to give each replica a ReadWriteOnce volume of its own that outlives the +pod, which is what lets the service scale out on plain block storage. Replicas +never share one volume — see the accessMode check in _helpers.tpl. + +Upgrades roll. A StatefulSet deletes a pod before starting its replacement, so a +ReadWriteOnce claim is released before the next pod asks for it; the Deployment +this replaced needed Recreate for that, or waited on a Multi-Attach error +forever. +*/}} apiVersion: apps/v1 -kind: Deployment +kind: StatefulSet metadata: name: {{ include "easyp-service.fullname" . }} labels: @@ -13,19 +25,26 @@ metadata: reloader.stakater.com/auto: "true" {{- end }} spec: + {{- if ne (include "easyp-service.autoscalingEnabled" .) "true" }} replicas: {{ .Values.replicaCount }} - {{- if and .Values.persistence.enabled (ne .Values.persistence.accessMode "ReadWriteMany") }} - # Recreate, because a ReadWriteOnce volume can be attached to one node at a - # time. A rolling update starts the replacement pod before retiring the old - # one, and if the scheduler picks a different node that pod waits on a - # Multi-Attach error forever — the rollout never completes and never fails. - # The choice is not between downtime and none; it is between a short gap and - # a stuck upgrade. Give the pod a ReadWriteMany volume to roll instead. - strategy: - type: Recreate - {{- else }} - strategy: + {{- end }} + serviceName: {{ include "easyp-service.fullname" . }}-headless + # Replicas share nothing but the database, and migrations are serialised by + # an advisory lock, so there is no reason to start them one at a time. Ordered + # startup would make a scale-out from 2 to 8 take six startup probes in a row. + podManagementPolicy: Parallel + updateStrategy: type: RollingUpdate + {{- if .Values.persistence.enabled }} + {{- with .Values.persistence.retentionPolicy }} + # Honoured from Kubernetes 1.27; older API servers drop the field, which + # leaves the claims in place — the same as Retain. Absent altogether (a + # --reuse-values upgrade from 1.0.4), the field is left out, and Kubernetes + # defaults both to Retain as well. + persistentVolumeClaimRetentionPolicy: + whenDeleted: {{ .whenDeleted | default "Retain" }} + whenScaled: {{ .whenScaled | default "Retain" }} + {{- end }} {{- end }} selector: matchLabels: @@ -40,7 +59,7 @@ spec: annotations: # The configuration is read once at startup, so a changed ConfigMap does # nothing until the pods are replaced. Kubernetes does not roll a - # Deployment when a mounted ConfigMap changes; hashing it into the pod + # StatefulSet when a mounted ConfigMap changes; hashing it into the pod # template is what turns `helm upgrade --set config.…` into a rollout # rather than a silent no-op. checksum/config: {{ include (print $.Template.BasePath "/configmap.yaml") . | sha256sum }} @@ -152,18 +171,16 @@ spec: - name: config configMap: name: {{ include "easyp-service.fullname" . }}-config + {{- /* With persistence on, volumeClaimTemplates below supply it. */}} + {{- if not .Values.persistence.enabled }} - name: plugins - {{- if .Values.persistence.enabled }} - persistentVolumeClaim: - claimName: {{ .Values.persistence.existingClaim | default (printf "%s-plugins" (include "easyp-service.fullname" .)) }} - {{- else }} # Ephemeral: every restart re-downloads plugin archives from S3. # sizeLimit is what keeps an overrun local to this pod — without it the # cache grows into the node's ephemeral storage, and the kubelet # resolves disk pressure by evicting pods that need not be this one. emptyDir: sizeLimit: {{ .Values.persistence.ephemeralSizeLimit }} - {{- end }} + {{- end }} {{- if .Values.tls.enabled }} - name: server-tls secret: @@ -190,3 +207,26 @@ spec: topologySpreadConstraints: {{- toYaml . | nindent 8 }} {{- end }} + {{- if .Values.persistence.enabled }} + # A claim per pod, plugins--N. Each starts empty and fills from + # object storage, which is why more than one replica needs S3. Not Helm-owned: + # `helm uninstall` leaves them, and retentionPolicy decides on scale-down. + # + # Immutable once created: a StatefulSet refuses any change here, so raising + # persistence.size fails `helm upgrade`. The README's "Resizing the plugin + # volumes" section has the procedure. + volumeClaimTemplates: + - metadata: + name: plugins + labels: + {{- include "easyp-service.selectorLabels" . | nindent 10 }} + spec: + accessModes: + - {{ .Values.persistence.accessMode }} + resources: + requests: + storage: {{ .Values.persistence.size }} + {{- if .Values.persistence.storageClass }} + storageClassName: {{ .Values.persistence.storageClass }} + {{- end }} + {{- end }} diff --git a/deploy/charts/easyp-service/tests/render.sh b/deploy/charts/easyp-service/tests/render.sh index ac9c4be..89b5a6a 100755 --- a/deploy/charts/easyp-service/tests/render.sh +++ b/deploy/charts/easyp-service/tests/render.sh @@ -386,38 +386,122 @@ fi echo echo "== rollout strategy ==" -# A ReadWriteOnce volume attaches to one node at a time, so a rolling update can -# strand the replacement pod on a Multi-Attach error and hang the upgrade -# indefinitely. Nothing in `helm upgrade` reports that as a failure, which is -# why it is checked here rather than discovered in a cluster. -if expect_render "a ReadWriteOnce volume forces Recreate"; then - if grep -q "type: Recreate" <<<"$out"; then - pass "a ReadWriteOnce volume forces Recreate" +# One workload kind in every storage mode. It used to be a Deployment, which had +# to Recreate on a ReadWriteOnce claim: a rolling update started the replacement +# first, and a replacement on another node waited on Multi-Attach forever. A +# StatefulSet deletes the pod before starting its successor, so every mode rolls. +# A Deployment reappearing here would bring that hang back with it. +for mode in "claim per replica|" \ + "emptyDir|--set persistence.enabled=false --set resources.limits.ephemeral-storage=30Gi"; do + what="${mode%%|*}" + read -r -a args <<<"${mode#*|}" + + if expect_render "$what: a StatefulSet that rolls" ${args[@]+"${args[@]}"}; then + if grep -q "^kind: Deployment$" <<<"$out"; then + fail "$what: a StatefulSet that rolls: a Deployment rendered" + elif ! grep -q "^kind: StatefulSet$" <<<"$out"; then + fail "$what: a StatefulSet that rolls: no StatefulSet rendered" + elif ! grep -A1 "updateStrategy:" <<<"$out" | grep -q "type: RollingUpdate"; then + fail "$what: a StatefulSet that rolls: updateStrategy is not RollingUpdate" + else + pass "$what: a StatefulSet that rolls" + fi + fi +done + +# A StatefulSet must name a governing Service that exists, or the API server +# accepts it and the pods never get their DNS records. +if expect_render "the StatefulSet names the headless Service the chart renders"; then + governing="$(grep -E '^ serviceName: ' <<<"$out" | awk '{print $2}')" + if [[ -n "$governing" ]] && grep -qE "^ name: ${governing}$" <<<"$out" && grep -q "clusterIP: None" <<<"$out"; then + pass "the StatefulSet names the headless Service the chart renders" else - fail "a ReadWriteOnce volume forces Recreate: strategy is not Recreate"$'\n'"$(grep -A2 'strategy:' <<<"$out")" + fail "the StatefulSet names the headless Service the chart renders: serviceName '$governing' has no headless Service" fi fi -# The converse: a shared volume has no attach conflict, so the upgrade should -# roll rather than take the service down for it. -if expect_render "a ReadWriteMany volume rolls" \ - --set persistence.accessMode=ReadWriteMany; then - if grep -q "type: RollingUpdate" <<<"$out"; then - pass "a ReadWriteMany volume rolls" +# The headless Service carries the same selector labels as the real one, and the +# ServiceMonitor selects Services by those labels. With a metrics port on both, +# every pod would be scraped twice and every summed counter doubled. +if expect_render "the headless Service does not publish metrics" \ + --show-only templates/service-headless.yaml; then + if grep -q "name: metrics" <<<"$out"; then + fail "the headless Service does not publish metrics: it does, so the ServiceMonitor would scrape each pod twice" else - fail "a ReadWriteMany volume rolls: strategy is not RollingUpdate" + pass "the headless Service does not publish metrics" fi fi -# Without persistence the volume is an emptyDir, which every pod gets its own -# copy of. Nothing to conflict over, so nothing to take downtime for. -if expect_render "no persistence rolls" \ - --set persistence.enabled=false \ - --set resources.limits.ephemeral-storage=30Gi; then - if grep -q "type: RollingUpdate" <<<"$out"; then - pass "no persistence rolls" +echo +echo "== one volume per replica ==" + +# Plenty of clusters have no ReadWriteMany class, or one too slow to execute +# plugins from — and replicas must not share a cache anyway. Every replica gets +# a ReadWriteOnce claim of its own from volumeClaimTemplates. +S3=(--set config.registry.s3.bucket=plugins) + +if expect_render "every replica gets a claim of its own" \ + "${S3[@]}" --set replicaCount=3; then + problems="" + grep -q "^kind: PersistentVolumeClaim$" <<<"$out" && problems="$problems a-standalone-claim-rendered" + grep -q "volumeClaimTemplates:" <<<"$out" || problems="$problems no-volumeClaimTemplates" + grep -q "persistentVolumeClaim:" <<<"$out" && problems="$problems pod-names-a-claim" + grep -A1 "accessModes:" <<<"$out" | grep -q "ReadWriteOnce" || problems="$problems not-ReadWriteOnce" + grep -qE "^ replicas: 3$" <<<"$out" || problems="$problems replicas-not-3" + + if [[ -n "$problems" ]]; then + fail "every replica gets a claim of its own:$problems" + else + pass "every replica gets a claim of its own" + fi +fi + +# Two pods on one volume corrupt the cache on a concurrent miss for the same +# plugin, and it appears to work until then. Refused, not documented. +expect_failure "a shared access mode is refused" "must be ReadWriteOnce or ReadWriteOncePod" \ + --set persistence.accessMode=ReadWriteMany + +if expect_render "ReadWriteOncePod is accepted" \ + --set persistence.accessMode=ReadWriteOncePod; then + pass "ReadWriteOncePod is accepted" +fi + +# existingClaim was a value until the StatefulSet, and Helm ignores values +# nothing reads. Ignored, an upgrade carrying it would mount an empty claim +# while the named one — without S3, the only copy of every plugin — sat unused. +expect_failure "the removed existingClaim is refused, not ignored" "persistence.existingClaim was removed" \ + --set persistence.existingClaim=mine + +# An empty volume with nothing to fill it from fails every generation. +expect_failure "several replicas without object storage are refused" "needs config.registry.s3.bucket" \ + --set replicaCount=2 + +expect_failure "several emptyDir replicas without object storage are refused" "needs config.registry.s3.bucket" \ + --set replicaCount=2 --set persistence.enabled=false \ + --set resources.limits.ephemeral-storage=30Gi + +# The HPA's ceiling is what matters: a pod it adds under load has to be able +# to start. +expect_failure "autoscaling without object storage is refused" "needs config.registry.s3.bucket" \ + --set autoscaling.enabled=true + +expect_failure "autoscaling with no metric is refused" "an HPA with no metric never scales" \ + "${S3[@]}" --set autoscaling.enabled=true \ + --set autoscaling.targetCPUUtilizationPercentage=null + +expect_failure "a retention policy other than Retain or Delete is refused" "must be Retain or Delete" \ + --set persistence.retentionPolicy.whenScaled=Keep + +# A replicas field alongside an HPA is reset by every `helm upgrade`, scaling the +# workload back to replicaCount until the HPA notices. +if expect_render "autoscaling leaves the replica count to the HPA" \ + "${S3[@]}" --set autoscaling.enabled=true; then + if grep -qE "^ replicas:" <<<"$out"; then + fail "autoscaling leaves the replica count to the HPA: the workload still sets replicas" + elif ! grep -A3 "scaleTargetRef:" <<<"$out" | grep -q "kind: StatefulSet"; then + fail "autoscaling leaves the replica count to the HPA: the HPA does not target the StatefulSet" else - fail "no persistence rolls: strategy is not RollingUpdate" + pass "autoscaling leaves the replica count to the HPA" fi fi @@ -474,6 +558,62 @@ if expect_render "the defaults ship a network policy"; then fi fi +echo +echo "== upgrading with --reuse-values ==" + +# `helm upgrade --reuse-values` replaces the new chart's defaults with the old +# release's values, so every key added since that release is absent rather than +# defaulted. A template reading one as `.Values.new.key` fails with a nil pointer +# and the upgrade stops. It did, for autoscaling, before anything checked it. +# +# Rendering against the old release's values.yaml alone is what the templates +# see on that upgrade. The copy is needed because --reuse-values is a property +# of an installed release; helm template has no way to ask for it. +reuse_chart="$(mktemp -d -t easyp-reuse.XXXXXX)" +cp -R "$CHART/." "$reuse_chart/" +cp "$CHART/tests/values-1.0.4.yaml" "$reuse_chart/values.yaml" + +reuse() { helm template test "$reuse_chart" "${BASE[@]}" "$@"; } + +if out="$(reuse 2>&1)"; then + problems="" + grep -q "^kind: StatefulSet$" <<<"$out" || problems="$problems no-StatefulSet" + grep -q "^kind: HorizontalPodAutoscaler$" <<<"$out" && problems="$problems an-HPA-nobody-asked-for" + grep -qE "^ replicas: 1$" <<<"$out" || problems="$problems replicas-not-1" + + if [[ -n "$problems" ]]; then + fail "an upgrade from 1.0.4 with --reuse-values renders:$problems" + else + pass "an upgrade from 1.0.4 with --reuse-values renders" + fi +else + fail "an upgrade from 1.0.4 with --reuse-values renders"$'\n'"$out" +fi + +# Turning autoscaling on in the same upgrade brings one key of the map and none +# of its defaults. Refused by name, rather than an HPA with no bounds. +if out="$(reuse --set autoscaling.enabled=true --set config.registry.s3.bucket=plugins 2>&1)"; then + fail "a partial autoscaling map is refused by name: it rendered" +elif grep -qF "needs autoscaling.minReplicas and autoscaling.maxReplicas" <<<"$out"; then + pass "a partial autoscaling map is refused by name" +else + fail "a partial autoscaling map is refused by name: failed some other way"$'\n'"$out" +fi + +# One retention key set, the other absent: the absent one is Retain, which is +# also what Kubernetes would pick, not an empty field the API server rejects. +if out="$(reuse --set persistence.retentionPolicy.whenScaled=Delete 2>&1)"; then + if grep -q "whenDeleted: Retain" <<<"$out" && grep -q "whenScaled: Delete" <<<"$out"; then + pass "a partial retention policy fills the rest with Retain" + else + fail "a partial retention policy fills the rest with Retain: $(grep -A2 'persistentVolumeClaimRetentionPolicy' <<<"$out" | tr '\n' ' ')" + fi +else + fail "a partial retention policy fills the rest with Retain"$'\n'"$out" +fi + +rm -rf "$reuse_chart" + echo echo "== install paths a customer actually takes ==" diff --git a/deploy/charts/easyp-service/tests/values-1.0.4.yaml b/deploy/charts/easyp-service/tests/values-1.0.4.yaml new file mode 100644 index 0000000..63dedb2 --- /dev/null +++ b/deploy/charts/easyp-service/tests/values-1.0.4.yaml @@ -0,0 +1,455 @@ +# The chart's values.yaml as released in 1.0.4, unchanged below this header. +# +# `helm upgrade --reuse-values` does not merge a new chart's defaults: it +# replaces them with the previous release's values. So on an upgrade from 1.0.4 +# the templates see exactly this file plus whatever the release overrode, and +# every key added since is simply absent. tests/render.sh renders the current +# templates against this file alone; a template that reads a newer key without +# tolerating its absence fails there instead of in a customer's upgrade. +# +# Do not edit it to make a check pass. When the oldest release worth upgrading +# from moves, replace it with that release's values.yaml, verbatim. +# +# Default values for easyp-service. +# +# Three things have no working default and must be decided before the chart will +# install. Each is refused at install time rather than at runtime, so a missing +# one is a message naming it, not a pod that comes up wrong. +# +# 1. A database DSN — see `secrets`. There is nothing sensible to default to. +# +# 2. TLS material, because `tls.enabled` defaults to true. Supply one of: +# tls.serverSecret= you bring the certificate +# certManager.enabled=true the chart issues it +# tls.enabled=false plaintext, trusted network +# The default stays on: a listener that speaks plaintext is the wrong thing +# to arrive at by saying nothing. But it does mean there is no install that +# consists only of a DSN, and this list said otherwise until v0.14.0. +# +# 3. `ingress.host`, and with it `config.server.trustedProxies`, if you want +# the gRPC API reachable from outside the cluster. Both are required +# together — see the comment on trustedProxies for why the second one is +# not optional. + +replicaCount: 1 + +image: + repository: ghcr.io/easyp-tech/service + # Empty means Chart.appVersion. Pin an explicit tag in production: a floating + # tag makes a rollout non-reproducible. + tag: "" + pullPolicy: IfNotPresent + +imagePullSecrets: [] +nameOverride: "" +fullnameOverride: "" + +serviceAccount: + create: true + annotations: {} + name: "" + +podAnnotations: {} +podLabels: {} +nodeSelector: {} +tolerations: [] +affinity: {} +topologySpreadConstraints: [] + +# The service reads its certificate and key once, at startup, so a rotated +# certificate is not picked up until the pod restarts. With stakater/Reloader +# installed this annotation restarts the deployment when the secret changes. +# Without it, rotation is a manual `kubectl rollout restart`. +reloader: + enabled: true + +# Sized against what the service is configured to do, not guessed. The plugin's +# output is read into memory whole, so peak usage tracks +# config.workerPool.maxConcurrentGenerations × config.registry.maxOutputSize — +# 16 × 64 MiB = 1 GiB of buffers with the defaults below, and the same bytes +# again once marshalled into the gRPC response. _helpers.tpl refuses an install +# where that product no longer fits in the limit. +# +# The earlier default was 1Gi, which the buffers alone filled. The pod was +# OOMKilled under load, and an OOMKill reads as a crash rather than as overload: +# no log line, no rejected-generation counter, no saturation alert. +resources: + requests: + cpu: 500m + memory: 1Gi + # Writable layer and logs only. Both the plugin cache and the archives + # staged mid-download live on the plugins volume, not here — the download + # used to land in /tmp on the writable layer, where a few hundred megabytes + # times the concurrent generations was an eviction rather than an overload. + # With persistence disabled see persistence.ephemeralSizeLimit, which is + # what actually bounds that volume. + ephemeral-storage: 1Gi + limits: + # Four cores for sixteen concurrent plugin processes. On two they each ran + # at an eighth speed and approached generationTimeoutSeconds, which would + # have surfaced as DeadlineExceeded — a broken plugin, to anyone reading — + # rather than as the honest ResourceExhausted the limiter returns. + cpu: "4" + memory: 4Gi + ephemeral-storage: 2Gi + +# Ports the container listens on. The service defaults to 23410-23413 when no +# configuration is given; the chart always sets them explicitly. +ports: + grpc: 8080 + # Singular, matching the service's own server.port.metric. It was `metrics` + # until v0.13.0 — one word with two spellings across three files. + metric: 8081 + health: 8082 + mcp: 8083 + +# The MCP endpoint: a read-only HTTP surface AI tooling uses to read the plugin +# catalog and the easyp.yaml schema. It exposes nothing the anonymous gRPC +# reads do not, but it sits outside the gRPC interceptor chain — no TLS, no +# rate limit, no audit — so serving it is a deployment's explicit decision. +# Off: the container does not listen on ports.mcp and the Service does not +# publish it. Turn it on inside a trusted network or behind an ingress that +# terminates TLS. +mcp: + enabled: false + +service: + type: ClusterIP + annotations: {} + +# Seconds Kubernetes waits after SIGTERM before SIGKILL. +# +# Three numbers have to line up, and _helpers.tpl enforces the order: +# +# generationTimeoutSeconds < forceShutdownAfterSeconds < terminationGracePeriodSeconds +# +# A generation the service accepted must be able to finish; the process must be +# able to exit on its own after that; and Kubernetes must wait for both rather +# than reaching for SIGKILL first. +terminationGracePeriodSeconds: 180 + +# --------------------------------------------------------------------------- +# Configuration that is not secret. Rendered into a ConfigMap and mounted at +# /etc/easyp/config.yml — the same config file every other deployment of this +# service reads. Secrets are not here; they arrive as environment variables from +# the secret above and override the file. +# +# The numbers below repeat the defaults compiled into the service, on purpose: +# this is the chart's documented interface and `helm show values` is where an +# operator looks them up. tests/render.sh checks that they still agree with the +# binary, so a default changed on one side and not the other fails there rather +# than in a cluster. +# --------------------------------------------------------------------------- +config: + logLevel: info + + server: + # CIDRs whose X-Forwarded-For and X-Real-IP headers may be believed. + # + # Required once ingress.enabled is true, and the chart refuses the install + # without it. Behind a proxy every request arrives from the proxy's address, + # so with this empty the rate limit and the per-caller concurrency limit + # become one bucket shared by every client at once, and the audit log + # records the ingress instead of who acted. None of that fails visibly. + # + # Set it to the range your ingress controller's pods run in — the pod CIDR + # of that node pool, not the whole cluster: anything listed here can claim + # to be any client. + trustedProxies: [] + + # Largest single gRPC message accepted and sent. gRPC's own defaults are + # 4 MiB in — which a request carrying a large proto tree exceeds — and + # effectively unlimited out. maxSendMsgSize must be at least + # config.registry.maxOutputSize or the service refuses to start: a plugin + # allowed to produce more than can be sent does its work for nothing. + maxRecvMsgSize: 67108864 + maxSendMsgSize: 67108864 + + # Concurrent streams one connection may hold. gRPC defaults to unlimited, + # which lets a single caller allocate goroutines and buffers until the pod + # is out of memory, before any limiter in the chain sees the requests. + maxConcurrentStreams: 256 + + # Hard exit this long after SIGTERM, if graceful shutdown has not finished. + # Expressed in seconds so the chart can compare it with the two numbers on + # either side of it. + forceShutdownAfterSeconds: 150 + + workerPool: + # Concurrent plugin lookups: a database read, plus a download and unpack + # from object storage on a cache miss. + workers: 4 + # Waiting slots, applied to lookups and to generations alike. Beyond this + # the answer is ErrServerOverloaded rather than a longer queue. + queueSize: 16 + # Concurrent plugin processes. Separate from workers on purpose: a worker is + # released once the plugin is located, and execution runs after that. + maxConcurrentGenerations: 16 + # Expressed in seconds (not "120s") so the chart can compare it against + # terminationGracePeriodSeconds. Rendered as a Go duration for the service. + generationTimeoutSeconds: 120 + maxRetries: 2 + shutdownTimeout: 30s + + rateLimit: + requestsPerSecond: 10.0 + burst: 20 + cleanupInterval: 10m + # Requests one client may have in flight at once. Rate alone does not bound + # this: a caller staying under the rate can still hold every generation slot + # with long requests. 0 disables the check. + maxConcurrentPerIP: 2 + + audit: + bufferSize: 1000 + batchSize: 100 + flushInterval: 1s + maxSaveRetries: 3 + # How long an operation waits for room in the audit queue. This is the only + # place audit can slow a request down, and it takes a backed-up queue to get + # there: a healthy writer drains batchSize/flushInterval — a hundred entries + # a second — so this expires only when the database has stopped keeping up. + # An entry that never finds room is dropped and counted under + # easyp_audit_events_lost_total{reason="enqueue_timeout"}; the operation + # itself still succeeds. + enqueueTimeout: 1s + # Bound on one write to storage, retries included. A batch that misses it is + # lost rather than holding the writer up behind a database that has stopped + # answering. + flushTimeout: 5s + # Partitions older than this are dropped on a schedule — a real delete, so + # afterwards the only copy of that month is in a backup taken while it still + # existed. Backup retention has to *exceed* this rather than match it: + # matching means the month leaves the database and the archive at about the + # same time, leaving it nowhere. + retentionMonths: 12 + preCreateMonths: 3 + partitionCheckInterval: 6h + partitionOpTimeout: 30s + + registry: + pluginsDir: /plugins + maxOutputSize: 67108864 + # Bound on the unpacked plugins on disk. Least recently used ones are + # dropped once this is exceeded; the archive in object storage is left + # alone, so an evicted plugin is one download away rather than lost. + # Keep it below persistence.size — eviction starts at the limit, and the + # volume needs room to reach it. 0 disables eviction. + cacheMaxBytes: 21474836480 # 20 GiB + # Leaving bucket empty disables S3 and makes pluginsDir the only source of + # plugin binaries. + s3: + endpoint: "" + bucket: "" + region: us-east-1 + prefix: "" + forcePathStyle: false + + telemetry: + # Empty means no collector, and the service then builds no exporter at all. + # These carry no fallback on purpose: the OTLP connection is lazy, so an + # endpoint nobody is listening on does not fail — it retries forever, and a + # pod with no observability stack would fill its log with connection errors + # while otherwise working. + otlpEndpoint: "" + pyroscopeEndpoint: "" + # Tags traces and profiles with the licence tier this release serves. Worth + # setting only where community and enterprise run side by side and their + # signals land in the same backends; on a cluster running one tier it would + # distinguish nothing. Metrics carry the same distinction already, so use + # the same word here that the metric label uses. + serviceTier: "" + + license: + # Public halves of the keys licence tokens are signed with, keyed by the key + # id that appears in the token footer. The token names one; the rest are + # here so that a signing key can be rotated without every deployment having + # to change key on the same day. + # + # These are the trust anchor: whoever sets them decides which authority may + # issue licences for this installation. The default below is easyp.tech's own + # published key, so a customer holding a LICENSE_KEY needs nothing else. + # Replace it only if you issue your own licences. + # + # Source of truth is keys/ in the licence registry (easyp-tech/licenses); + # these must be kept in step with it. + # + # Without a key, LICENSE_KEY is ignored and the service runs in community mode. + publicKeys: + "2026-08": "81322461987167d5cfd529e9cb8b96f4797f12fce6be4399a0866e250c9b6bb5" + +# Extra environment variables, appended last. Use for settings the chart does +# not model yet. +extraEnv: [] + +# --------------------------------------------------------------------------- +# Secrets +# +# The pod loads them with envFrom, so the KEYS OF THE SECRET ARE THE ENVIRONMENT +# VARIABLE NAMES. A secret you bring yourself must use exactly these keys: +# +# DB_POSTGRES_DSN required +# REGISTRY_S3_ACCESS_KEY_ID required when config.registry.s3.bucket is set +# REGISTRY_S3_SECRET_ACCESS_KEY required when config.registry.s3.bucket is set +# LICENSE_KEY optional; absent means community mode +# +# The single-key LICENSE_PUBLIC_KEY was removed in v0.13.0. The trust anchor is +# config.license.publicKeys alone: an entry per key id, or one under "*" to +# verify any key id. Without a public key, LICENSE_KEY counts for nothing. +# AUTH_WRITE_TOKENS optional; absent means no writes are allowed +# +# AUTH_WRITE_TOKENS holds sha256 digests, never the tokens: "ci=<64 hex>,me=<64 hex>". +# Generate a token with `easyp-svc auth new-token --name ci`. +# --------------------------------------------------------------------------- +secrets: + # Bring your own secret (recommended). Values put in `data` below end up in + # Helm release history and in `helm get values`, which is why create defaults + # to false. + existingSecret: "" + create: false + data: {} + +# --------------------------------------------------------------------------- +# Transport security +# +# When enabled, the gRPC listener serves TLS. Adding clientCASecret turns it +# into mutual TLS: only callers holding a certificate signed by that CA get in, +# which is what keeps the listener private when the ingress is the only party +# meant to reach it. +# --------------------------------------------------------------------------- +tls: + enabled: true + # Secret of type kubernetes.io/tls holding tls.crt and tls.key for the server. + serverSecret: "" + # Secret holding ca.crt used to verify client certificates. Empty falls back + # to ca.crt inside serverSecret, which is what cert-manager writes for a CA + # issuer. Set to "-" to serve plain server-side TLS with no client check. + clientCASecret: "" + mountPath: /certs + +# Optional: let cert-manager issue the certificates instead of supplying them. +certManager: + enabled: false + issuerRef: + name: "" + kind: ClusterIssuer + group: cert-manager.io + duration: 2160h + renewBefore: 360h + # Extra SANs for the server certificate. The in-cluster service name is + # always included. + extraDnsNames: [] + # Issues the client certificate the ingress presents to the service. + clientCertificate: + enabled: true + commonName: easyp-ingress + +# --------------------------------------------------------------------------- +# Ingress. Defaults target Traefik because it is the only controller that can +# express a mutual-TLS backend leg declaratively, via ServersTransport. +# --------------------------------------------------------------------------- +ingress: + enabled: false + className: traefik + host: "" + annotations: {} + # Secret holding the certificate served to external clients. + tlsSecret: "" + # Traefik ServersTransport describing how the router reaches the pod. Required + # whenever tls.enabled is true, otherwise the backend leg would be plaintext + # against a listener that demands a client certificate. + serversTransport: + enabled: true + # Secret of type kubernetes.io/tls with the client certificate the router + # presents to the service. + clientSecret: "" + # Name the router expects on the service certificate. + serverName: "" + insecureSkipVerify: false + +# --------------------------------------------------------------------------- +# Storage for the plugin cache. Plugin archives are downloaded lazily and +# unpacked here, and the directory is pruned: once it passes +# config.registry.cacheMaxBytes the least recently used plugin versions are +# removed. So size this for the working set you want resident, not for every +# plugin that could ever be requested. +# +# Eviction only ever removes local files. The archive stays in object storage, +# so an evicted plugin costs one download on its next request rather than being +# lost — which is why the cache can be sized to taste. +# +# Whichever of the two sizes below applies has to exceed cacheMaxBytes, and the +# chart refuses to install when it does not: eviction starts at the limit, so +# storage sized exactly to it is already full by the time the cache first needs +# headroom. +# --------------------------------------------------------------------------- +persistence: + enabled: true + existingClaim: "" + storageClass: "" + accessMode: ReadWriteOnce + size: 25Gi + # Used when enabled=false, where the cache lives in an emptyDir on the node + # instead of a volume of its own. Without a limit here the cache grows into + # the node's ephemeral storage, and disk pressure there is resolved by the + # kubelet evicting pods — not necessarily this one. With it, an overrun stops + # at the pod that caused it. + # + # Remember this is per replica: with persistence off, every pod keeps its own + # copy, so plan for replicaCount × this much on the nodes. + ephemeralSizeLimit: 25Gi + +serviceMonitor: + enabled: false + interval: 30s + scrapeTimeout: 10s + labels: {} + +# Alerting rules. Needs the Prometheus Operator CRDs, same as serviceMonitor, +# which is why this defaults to off rather than failing the install where they +# are absent. Turn it on wherever you actually run this. +# +# The rules worth having on day one are the licence ones: a licence that lapses +# breaks nothing loudly. The tier drops to community, audit stops, and the +# plugin limit starts rejecting registrations — all silently. +prometheusRule: + enabled: false + # Extra labels, usually what your Prometheus selects rules by. + labels: {} + # Where each alert's runbook_url points. Every alert gets one, anchored on its + # own name. Point this at your own copy if you keep runbooks internally — + # an alert that arrives at 3am with no procedure attached is where most of the + # time goes. + runbookBaseUrl: https://easyp.tech/docs/api-service/runbooks + # How long before expiry to start warning. The token also carries its own + # grace period, which runs after this; both exist so a renewal that slips + # does not take a customer's pipeline down. + licenceExpiryWarningDays: 14 + # Fraction of config.registry.cacheMaxBytes that counts as full. + cacheUsageWarningRatio: 0.95 + # Fraction of generations allowed to fail before alerting. + generationErrorRatio: 0.05 + # Rejected write credentials per second before alerting. + authFailureRate: 0.5 + +podDisruptionBudget: + enabled: false + minAvailable: 1 + +# On by default. This pod's whole job is running third-party binaries, so the +# set of places it can reach should be stated rather than inherited. +# +# If your database, object storage or collector is not on one of the ports below, +# add it here or the pod goes quiet in a way that looks like a hang. That is the +# one thing to check first after enabling this. +networkPolicy: + enabled: true + # Selector matching the namespace your ingress controller runs in. Empty means + # the gRPC port accepts connections from anywhere in the cluster — set it once + # you know which namespace your controller runs in. + ingressNamespaceSelector: {} + # Ports the pod is allowed to reach outbound, beyond DNS. + egressPorts: + - 5432 # PostgreSQL + - 443 # object storage, and anything a plugin fetches over HTTPS + - 4317 # OTLP diff --git a/deploy/charts/easyp-service/values.yaml b/deploy/charts/easyp-service/values.yaml index fa54d19..e250c5d 100644 --- a/deploy/charts/easyp-service/values.yaml +++ b/deploy/charts/easyp-service/values.yaml @@ -19,8 +19,29 @@ # together — see the comment on trustedProxies for why the second one is # not optional. +# Ignored while autoscaling.enabled: the HPA owns the count then, and a +# `replicas` field in the template would reset it on every `helm upgrade`. +# More than one needs config.registry.s3 — see persistence. replicaCount: 1 +# CPU is the signal because every generation is a plugin process charged to this +# container. Needs metrics-server. Each replica runs its own worker pool and +# limits, so capacity scales with it: maxReplicas × maxConcurrentGenerations +# generations at once, cluster-wide. The per-client limits (rateLimit) are per +# pod too, so a client spread across N pods gets up to N times its limit. +autoscaling: + enabled: false + minReplicas: 2 + maxReplicas: 6 + targetCPUUtilizationPercentage: 70 + targetMemoryUtilizationPercentage: null + # Passed through to the HPA as-is. Scale-down is worth slowing: a replica + # removed stops serving from its warm cache. Its claim is kept under + # retentionPolicy.whenScaled=Retain, an emptyDir is not. + behavior: + scaleDown: + stabilizationWindowSeconds: 600 + image: repository: ghcr.io/easyp-tech/service # Empty means Chart.appVersion. Pin an explicit tag in production: a floating @@ -46,7 +67,7 @@ topologySpreadConstraints: [] # The service reads its certificate and key once, at startup, so a rotated # certificate is not picked up until the pod restarts. With stakater/Reloader -# installed this annotation restarts the deployment when the secret changes. +# installed this annotation restarts the pods when the secret changes. # Without it, rotation is a manual `kubectl rollout restart`. reloader: enabled: true @@ -370,13 +391,38 @@ ingress: # chart refuses to install when it does not: eviction starts at the limit, so # storage sized exactly to it is already full by the time the cache first needs # headroom. +# +# The workload is a StatefulSet, and where the cache goes is one of two: +# +# enabled=true (default) a ReadWriteOnce claim per replica from +# volumeClaimTemplates, plugins--N. Kept across +# restarts, upgrades and — by retentionPolicy — scale-down. +# enabled=false an emptyDir per pod on node disk, lost with the pod. +# +# Replicas never share a volume, so there is no ReadWriteMany option: unpacking +# a plugin is serialised by an in-process lock only, and two pods on one volume +# corrupt the cache on a concurrent miss for the same plugin. +# +# More than one replica needs config.registry.s3. Each replica's volume starts +# empty and object storage is the only thing that fills it. # --------------------------------------------------------------------------- persistence: enabled: true - existingClaim: "" storageClass: "" + # ReadWriteOnce or ReadWriteOncePod; anything shared is refused, see above. accessMode: ReadWriteOnce + # Per replica, so the total is replicas × this. It cannot be changed by + # `helm upgrade`: a StatefulSet's claim templates are immutable. See the + # README's "Resizing the plugin volumes". size: 25Gi + # What happens to a replica's claim when the StatefulSet is scaled down or + # deleted. Retain keeps the warm cache for the next scale-up and survives + # `helm uninstall`, as the single claim of earlier releases did; Delete stops + # an autoscaler from leaving a trail of idle disks. Honoured from Kubernetes + # 1.27; older API servers ignore it and retain. + retentionPolicy: + whenDeleted: Retain + whenScaled: Retain # Used when enabled=false, where the cache lives in an emptyDir on the node # instead of a volume of its own. Without a limit here the cache grows into # the node's ephemeral storage, and disk pressure there is resolved by the