Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions charts/grid-operator/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,7 @@ RELEASE=grid-operator; NAMESPACE=grid-system; for crd in agenttoolproviders grid
|-----|------|---------|-------------|
| `crds.enabled` | bool | `true` | Install and upgrade the Grid CRDs. `false` when a platform owns them. |
| `crds.keep` | bool | `true` | Keep the CRDs on `helm uninstall` and an Argo CD delete or prune. |
| `grid.providers` | object | `{}` | InferenceProviders this site serves, keyed by name. `model` defaults to the name, `providerKind` to `vllm`, `backendKind` to `local_model`. `maxRunning` is the most requests one endpoint runs at once (vLLM max-num-seqs); without it the provider's capacity, and so its saturation, is unknown. |
| `rbac.enrollmentNamespace` | string | `""` | The grid-enrollment namespace. The render fails if the operator would get Secret access there. |
| `rbac.metricsScraper` | bool | `true` | Create the metrics scraper ServiceAccount, allowed only GET on the nonResourceURL /metrics, and let the operator mint short-lived tokens for it. An llm-d EPP serving bearer-authenticated metrics (the default) admits a scrape with that token (metricsConfig.auth type serviceAccountToken). The operator never sends its own token. |
| `replicaCount` | int | `1` | Operator replicas. Must be 1 (schema-enforced). |
Expand Down
81 changes: 81 additions & 0 deletions charts/grid-operator/templates/crds/inferenceprovider.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -24,8 +24,19 @@ spec:
- jsonPath: .spec.providerKind
name: Provider
type: string
- jsonPath: .status.state
name: Status
type: string
- jsonPath: .metadata.creationTimestamp
name: Age
type: date
- jsonPath: .status.conditions[?(@.type=="Ready")].reason
name: Reason
priority: 1
type: string
- jsonPath: .status.phase
name: Phase
priority: 1
type: string
name: v1alpha1
schema:
Expand Down Expand Up @@ -289,6 +300,17 @@ spec:
- message: set exactly one of caSecretRef and caConfigMapRef
rule: has(self.caSecretRef) != has(self.caConfigMapRef)
type: object
maxRunning:
description: |-
Most requests one endpoint runs at once: vLLM max-num-seqs.

The site's operator multiplies it by the EPP's fresh ready endpoints and publishes
the product as `grid_provider_capacity_requests`, the capacity gateways weigh the
provider's load against. When absent, or no endpoint is fresh, capacity is unknown.
format: uint32
minimum: 1.0
nullable: true
type: integer
metricsConfig:
description: |-
Prometheus metrics scraping configuration.
Expand Down Expand Up @@ -407,6 +429,14 @@ spec:
description: Metric name for normalised queue depth (0.0–1.0).
nullable: true
type: string
readyEndpoints:
description: |-
Metric name counting the pool's ready endpoints, read for the `Ready` condition.

Defaults to `llm_d_epp_ready_endpoints`, then `inference_pool_ready_pods`.
Filtered by `poolName` when set. Zero marks the provider not ready.
nullable: true
type: string
type: object
staleMetricsSeconds:
description: |-
Expand Down Expand Up @@ -764,6 +794,46 @@ spec:
description: Observed status of an [`InferenceProvider`].
nullable: true
properties:
conditions:
description: |-
Observed conditions. `Ready` says whether the provider can currently serve a request.

Written by the operator's signals loop, never by provider reconciliation.
items:
description: One observed condition, shaped like `metav1.Condition`.
properties:
lastTransitionTime:
description: When `status` last changed, RFC 3339.
format: date-time
type: string
message:
default: ""
description: Human-readable detail.
type: string
observedGeneration:
description: The `metadata.generation` this was computed against.
format: int64
nullable: true
type: integer
reason:
description: CamelCase reason for the status.
type: string
status:
description: '`True`, `False`, or `Unknown`.'
type: string
type:
description: Condition type, such as `Ready`.
type: string
required:
- lastTransitionTime
- reason
- status
- type
type: object
type: array
x-kubernetes-list-map-keys:
- type
x-kubernetes-list-type: map
matchingSites:
default: []
description: Sites matched by the site selector.
Expand Down Expand Up @@ -811,6 +881,17 @@ spec:
`HealthCheckTlsIdentityMismatch`
nullable: true
type: string
state:
description: |-
`Ready`, `NotReady`, or `Unknown`: the `Ready` condition's status, for display.

Written with the condition, never by provider reconciliation.
enum:
- Ready
- NotReady
- Unknown
nullable: true
type: string
type: object
required:
- spec
Expand Down
9 changes: 0 additions & 9 deletions charts/grid-operator/templates/deployment.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -108,15 +108,6 @@ spec:
value: {{ printf "%s-metrics-scraper" (include "grid-operator.fullname" .) | quote }}
- name: GRID_METRICS_SCRAPER_NAMESPACE
value: {{ .Release.Namespace | quote }}
# Binds each scrape token to this pod, so it stops working when the pod goes.
- name: GRID_POD_NAME
valueFrom:
fieldRef:
fieldPath: metadata.name
- name: GRID_POD_UID
valueFrom:
fieldRef:
fieldPath: metadata.uid
{{- end }}
{{- if (.Values.signals).enabled }}
- name: GRID_SIGNALS_LOCAL_ADDR
Expand Down
3 changes: 3 additions & 0 deletions charts/grid-operator/templates/inferenceprovider.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,9 @@ spec:
{{- with .capacityWeight }}
capacityWeight: {{ . }}
{{- end }}
{{- with .maxRunning }}
maxRunning: {{ . }}
{{- end }}
{{- with .routingClusterRef }}
routingClusterRef: {{ . | quote }}
{{- end }}
Expand Down
26 changes: 26 additions & 0 deletions charts/grid-operator/tests/grid_test.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -135,6 +135,32 @@ tests:
- name: Qwen3-Coder-30B-A3B
capabilities: [text_generation]

- it: passes maxRunning through to the provider spec
template: templates/inferenceprovider.yaml
set:
site.name: east
grid.id: lab
grid.providers:
qwen3:
endpoint: http://qwen3.llm-d.svc:80
maxRunning: 64
asserts:
- equal:
path: spec.maxRunning
value: 64

- it: leaves maxRunning unset when a provider omits it
template: templates/inferenceprovider.yaml
set:
site.name: east
grid.id: lab
grid.providers:
qwen3:
endpoint: http://qwen3.llm-d.svc:80
asserts:
- notExists:
path: spec.maxRunning

- it: keeps YAML-ambiguous names strings
templates:
- templates/gridnetwork.yaml
Expand Down
3 changes: 3 additions & 0 deletions charts/grid-operator/values.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -28,11 +28,14 @@ grid:
swimKeySecretName: grid-swim-key
# -- InferenceProviders this site serves, keyed by name, as in the grid-site chart.
# model defaults to the name, providerKind to vllm, and backendKind to local_model.
# maxRunning is the most requests one endpoint runs at once (vLLM max-num-seqs); without
# it the provider's capacity, and so its saturation, is unknown.
providers: {}
# qwen3:
# model: Qwen3-Coder-30B-A3B
# endpoint: http://qwen3-epp-service.llm-d.svc:80
# routingClusterRef: site-d
# maxRunning: 64

# -- Grid CRDs.
crds:
Expand Down
23 changes: 23 additions & 0 deletions charts/praxis-gateway/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -228,6 +228,10 @@ Praxis AI image; these values may advance independently.
| `gatewayConfig.auth.validateCA` | object | empty | CA for the validate call (`configMap` or `secret`, `key`). Set as `SSL_CERT_FILE`, which replaces the platform trust store for the validate call and https backends without a per-backend CA or `upstreamCA`. mutual_tls backends and `upstreamCA` are unaffected. See the recipe below. |
| `networkPolicy.enabled` | bool | `false` | Render a NetworkPolicy that limits which pods can reach the listener port, where the CNI enforces NetworkPolicy. It is not authentication. Node and host-network traffic handling is CNI-specific (OVN-Kubernetes: the `policy-group.network.openshift.io/host-network` label), and a LoadBalancer with `externalTrafficPolicy: Cluster` can SNAT clients to node IPs. |
| `networkPolicy.from` | list | `[]` | NetworkPolicyPeer entries allowed in. Required when enabled. With `auth.mode: none`, list only the authenticating front. `{podSelector: {}}` admits every pod in this namespace. An empty `namespaceSelector` and an `ipBlock` of `0.0.0.0/0` or `::/0` admit everyone and fail the render. An all-address `ipBlock` with `except` entries is allowed. The check reads selector emptiness and the cidr only, so `matchExpressions` that happen to select every pod pass. A provider gateway behind a LoadBalancer that SNATs clients to node IPs needs `ipBlock` peers for those node addresses. |
| `metricsListener.enabled` | bool | `false` | Serve `GET /metrics` over TLS on its own port and ClusterIP Service, for an in-cluster Prometheus. The admin listener refuses a non-loopback Host, so Prometheus cannot scrape it. Needs the grid-gateway image, `existingSecret`, `fromNamespaces`, and `networkPolicy.enabled`. The port answers only `/metrics` but has no authentication, so the NetworkPolicy is its access control, and that holds only where the CNI enforces NetworkPolicy. |
| `metricsListener.existingSecret` | string | `""` | Secret with `tls.crt` and `tls.key`. On OpenShift, request it with `metricsListener.service.annotations` `service.beta.openshift.io/serving-cert-secret-name`. The listener reloads the cert when the Secret changes. |
| `metricsListener.fromNamespaces` | list | `[]` | Namespace names allowed to reach the metrics port, for example `openshift-user-workload-monitoring`. |
| `metricsListener.serviceMonitor.enabled` | bool | `false` | Render a ServiceMonitor that verifies the cert against `caConfigMap` (for example `openshift-service-ca.crt`, key `service-ca.crt`) and renames Praxis's `cluster` label to `backend`, since ACM uses `cluster` for the managed cluster. |
| `gatewayConfig.upstreamCA.secretName` | string | `""` | CA bundle for backend TLS without a per-cluster CA (`upstream_ca_file`). |
| `gatewayConfig.listenerTls.enabled` | bool | `false` | Terminate TLS at the listener from `existingSecret`, in render or BYO mode. Names the port `https`. The cert mounts at `listenerTls.mountPath` (`/etc/praxis/listener-tls`), so a BYO config moving off `tls.enabled` must point its listener `cert_path`/`key_path` there. On OpenShift, annotate the Service with `service.beta.openshift.io/serving-cert-secret-name`. |
| `port.containerPort` | int | `8080` | Container port. |
Expand Down Expand Up @@ -274,6 +278,25 @@ Praxis AI image; these values may advance independently.
| `topologySpreadConstraints` | list | `[]` | Topology spread constraints. |
| `priorityClassName` | string | `""` | Pod priority class. |

## Metrics

The metrics listener serves the Praxis registry, including the grid gateway's own series. With the ServiceMonitor, Praxis's `cluster` label arrives as `backend`.

| Metric | Labels | Meaning |
|--------|--------|---------|
| `grid_route_decisions_total` | `site`, `reason` | Requests `grid_site_route` decided. `site` is a site name from the serving config, or empty for a refusal. No label comes from the request. |
| `grid_route_site_score` | `site`, `cluster` | The queue depth the last route order used for each candidate, lower first. `inf` when unmeasured, `NaN` when excluded or demoted, or when the pair left the topology. |

`reason` is one of five values, and adding one is a deliberate change:

| `reason` | `site` | Response |
|----------|--------|----------|
| `routed` | chosen site | Sent to a healthy site. How it ranked is in `grid_route_site_score`. |
| `fallback` | chosen site | Sent to a demoted site because no healthy one was left. |
| `not_ready` | empty | 503: every candidate was excluded. |
| `no_route` | empty | 503: an admitted candidate had no route from this gateway. |
| `bad_request` | empty | 400 or 404: no model, or a model no candidate serves. |

## Security

The chart enforces Kubernetes restricted security defaults:
Expand Down
33 changes: 33 additions & 0 deletions charts/praxis-gateway/templates/_helpers.tpl
Original file line number Diff line number Diff line change
Expand Up @@ -555,3 +555,36 @@ RUST_LOG for the gateway and overlay-sync: log.filter when set, else log.level,
{{- $log := .Values.log | default dict -}}
{{- $log.filter | default $log.level -}}
{{- end }}

{{/*
The metrics listener needs the grid-gateway image, a cert, and the NetworkPolicy that
limits its port; without the policy any pod could scrape it.
*/}}
{{- define "praxis-gateway.validateMetricsListener" -}}
{{- $m := .Values.metricsListener }}
{{- if $m.enabled }}
{{- if ne .Values.image.flavor "grid-gateway" }}
{{- fail "metricsListener needs image.flavor grid-gateway" }}
{{- end }}
{{- if not $m.existingSecret }}
{{- fail "metricsListener.existingSecret is required: the listener serves TLS only" }}
{{- end }}
{{- if not .Values.networkPolicy.enabled }}
{{- fail "metricsListener needs networkPolicy.enabled, which limits the metrics port to metricsListener.fromNamespaces" }}
{{- end }}
{{- if not $m.fromNamespaces }}
{{- fail "metricsListener.fromNamespaces needs at least one namespace" }}
{{- end }}
{{- $taken := list (int .Values.port.containerPort) }}
{{- if and .Values.overlay.enabled .Values.overlay.sidecar.enabled }}{{ $taken = append $taken 9091 }}{{ end }}
{{- if has (int $m.port) $taken }}
{{- fail (printf "metricsListener.port %d collides with another gateway pod port" (int $m.port)) }}
{{- end }}
{{- end }}
{{- if and $m.serviceMonitor.enabled (not $m.enabled) }}
{{- fail "metricsListener.serviceMonitor needs metricsListener.enabled" }}
{{- end }}
{{- if and $m.serviceMonitor.enabled (not $m.serviceMonitor.caConfigMap.name) }}
{{- fail "metricsListener.serviceMonitor.caConfigMap.name is required to verify the metrics cert" }}
{{- end }}
{{- end }}
28 changes: 27 additions & 1 deletion charts/praxis-gateway/templates/deployment.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
{{- include "praxis-gateway.validateConfig" . }}
{{- include "praxis-gateway.validateMounts" . }}
{{- include "praxis-gateway.validateProbes" . }}
{{- include "praxis-gateway.validateMetricsListener" . }}
apiVersion: apps/v1
kind: Deployment
metadata:
Expand Down Expand Up @@ -192,7 +193,7 @@ spec:
{{- $userLog := false }}
{{- range .Values.env }}{{- if eq .name "RUST_LOG" }}{{- $userLog = true }}{{- end }}{{- end }}
{{- $rustLog := and (not $userLog) (include "praxis-gateway.rustLog" .) }}
{{- if or $rustLog .Values.env (.Values.gridServing).enabled $withCA }}
{{- if or $rustLog .Values.env (.Values.gridServing).enabled $withCA (.Values.metricsListener).enabled }}
env:
{{- if $rustLog }}
- name: RUST_LOG
Expand All @@ -209,6 +210,16 @@ spec:
- name: SSL_CERT_FILE
value: {{ printf "%s/%s" $validateCA.mountPath $validateCA.key | quote }}
{{- end }}
{{- with .Values.metricsListener }}
{{- if .enabled }}
- name: GRID_METRICS_ADDR
value: {{ printf "0.0.0.0:%d" (int .port) | quote }}
- name: GRID_METRICS_TLS_CERT
value: {{ printf "%s/tls.crt" .mountPath | quote }}
- name: GRID_METRICS_TLS_KEY
value: {{ printf "%s/tls.key" .mountPath | quote }}
{{- end }}
{{- end }}
{{- end }}
securityContext:
runAsNonRoot: true
Expand All @@ -227,6 +238,11 @@ spec:
- name: {{ include "praxis-gateway.portName" . }}
containerPort: {{ .Values.port.containerPort }}
protocol: {{ .Values.port.protocol | quote }}
{{- if .Values.metricsListener.enabled }}
- name: metrics
containerPort: {{ .Values.metricsListener.port }}
protocol: TCP
{{- end }}
{{- with .Values.health.readiness }}
readinessProbe:
{{- /* /ready fails while any backend cluster is down. */}}
Expand Down Expand Up @@ -286,6 +302,11 @@ spec:
mountPath: {{ .Values.gatewayConfig.listenerTls.mountPath | quote }}
readOnly: true
{{- end }}
{{- if .Values.metricsListener.enabled }}
- name: metrics-tls
mountPath: {{ .Values.metricsListener.mountPath | quote }}
readOnly: true
{{- end }}
{{- if $withCA }}
- name: validate-ca
mountPath: {{ $validateCA.mountPath | quote }}
Expand Down Expand Up @@ -407,6 +428,11 @@ spec:
secret:
secretName: {{ .Values.gatewayConfig.listenerTls.existingSecret | quote }}
{{- end }}
{{- if .Values.metricsListener.enabled }}
- name: metrics-tls
secret:
secretName: {{ .Values.metricsListener.existingSecret | quote }}
{{- end }}
{{- range .Values.credentials }}
- name: credential-{{ .name }}
secret:
Expand Down
11 changes: 11 additions & 0 deletions charts/praxis-gateway/templates/networkpolicy.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -33,4 +33,15 @@ spec:
ports:
- port: {{ .Values.port.containerPort }}
protocol: {{ .Values.port.protocol }}
{{- if .Values.metricsListener.enabled }}
- from:
- namespaceSelector:
matchExpressions:
- key: kubernetes.io/metadata.name
operator: In
values: {{ toJson .Values.metricsListener.fromNamespaces }}
ports:
- port: {{ .Values.metricsListener.port }}
protocol: TCP
{{- end }}
{{- end }}
24 changes: 24 additions & 0 deletions charts/praxis-gateway/templates/service-metrics.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{{- include "praxis-gateway.normalize" . }}
{{- if .Values.metricsListener.enabled }}
{{- /* ClusterIP of its own, so a LoadBalancer gateway Service never exposes metrics. */}}
apiVersion: v1
kind: Service
metadata:
name: {{ include "praxis-gateway.fullname" . }}-metrics
labels:
{{- include "praxis-gateway.labels" . | nindent 4 }}
app.kubernetes.io/component: gateway-metrics
{{- with .Values.metricsListener.service.annotations }}
annotations:
{{- toYaml . | nindent 4 }}
{{- end }}
spec:
type: ClusterIP
ports:
- name: metrics
port: {{ .Values.metricsListener.port }}
targetPort: metrics
protocol: TCP
selector:
{{- include "praxis-gateway.selectorLabels" . | nindent 4 }}
{{- end }}
Loading
Loading