diff --git a/TESTING.md b/TESTING.md index a1c2f9476e..a1f681cde9 100644 --- a/TESTING.md +++ b/TESTING.md @@ -475,6 +475,7 @@ Kubernetes e2e environment variables: | `OPENSHELL_E2E_KUBE_TEST` | Scope to a single test (e.g. `smoke`) | | `OPENSHELL_E2E_KUBE_EXTRA_VALUES` | Colon-separated additional Helm values files | | `OPENSHELL_E2E_KUBERNETES_FEATURES` | Cargo feature flags (default: `e2e,e2e-host-gateway,e2e-kubernetes`) | +| `OPENSHELL_E2E_OPENSHIFT` | Force the OpenShift code paths on (`1`) or off (`0`), skipping cluster detection | | `IMAGE_TAG` | Gateway/supervisor image tag (default: `latest` for existing clusters) | | `OPENSHELL_REGISTRY` | Image registry prefix (default: `ghcr.io/nvidia/openshell`) | | `GATEWAY_IMAGE` | Kubernetes gateway image repository or complete tagged/digest-pinned image reference; digests require `OPENSHELL_E2E_KUBE_BUILD_IMAGES=0` | diff --git a/e2e/support/gateway-common.sh b/e2e/support/gateway-common.sh index d13d258479..5b658ddc67 100644 --- a/e2e/support/gateway-common.sh +++ b/e2e/support/gateway-common.sh @@ -566,3 +566,91 @@ e2e_print_gateway_log_on_failure() { echo "=== end gateway log ===" fi } + +# Single API path used to decide whether a cluster is OpenShift. Reading one +# group version asks the API server about `route.openshift.io` alone, so an +# unrelated unavailable aggregated APIService cannot make an OpenShift cluster +# look like vanilla Kubernetes the way `kubectl api-resources` can. +E2E_OPENSHIFT_PROBE_PATH="/apis/route.openshift.io/v1" + +# True when a failed probe proves the API group is genuinely absent rather than +# momentarily unreachable. Only a NotFound answer from the API server is +# conclusive; connection, authentication, and throttling errors are not. +e2e_openshift_probe_output_is_absent() { + local output=$1 + + [[ "${output}" == *"(NotFound)"* ]] \ + || [[ "${output}" == *"could not find the requested resource"* ]] +} + +# Decide whether the cluster behind a kubectl context is OpenShift. +# +# Prints `1` or `0` on stdout and returns 0 only when the answer is conclusive. +# Returns non-zero after logging the probe and the underlying error when it is +# not, so callers fail fast instead of silently taking the vanilla-Kubernetes +# path on a transient discovery failure. The outcome is always logged to stderr. +# +# Overrides: +# OPENSHELL_E2E_OPENSHIFT 1/true/yes or 0/false/no to skip the probe +# OPENSHELL_E2E_OPENSHIFT_PROBE_ATTEMPTS probe attempts before giving up (default 5) +# OPENSHELL_E2E_OPENSHIFT_PROBE_DELAY seconds before the first retry, doubling (default 2) +e2e_detect_openshift() { + local context=$1 + local override="${OPENSHELL_E2E_OPENSHIFT:-}" + local attempts="${OPENSHELL_E2E_OPENSHIFT_PROBE_ATTEMPTS:-5}" + local delay="${OPENSHELL_E2E_OPENSHIFT_PROBE_DELAY:-2}" + local probe="kubectl --context ${context} get --raw ${E2E_OPENSHIFT_PROBE_PATH}" + local attempt=1 + local status=0 + local output="" + + case "${override}" in + "") ;; + 1 | true | TRUE | yes | YES) + echo "OpenShift detection: cluster is OpenShift (forced by OPENSHELL_E2E_OPENSHIFT=${override})." >&2 + printf '1\n' + return 0 + ;; + 0 | false | FALSE | no | NO) + echo "OpenShift detection: cluster is not OpenShift (forced by OPENSHELL_E2E_OPENSHIFT=${override})." >&2 + printf '0\n' + return 0 + ;; + *) + echo "ERROR: OPENSHELL_E2E_OPENSHIFT must be 1/true/yes or 0/false/no, got '${override}'." >&2 + return 2 + ;; + esac + + while :; do + status=0 + output="$(kubectl --context "${context}" get --raw "${E2E_OPENSHIFT_PROBE_PATH}" 2>&1)" || status=$? + + if [ "${status}" -eq 0 ]; then + echo "OpenShift detection: cluster is OpenShift (probe: ${probe})." >&2 + printf '1\n' + return 0 + fi + + if e2e_openshift_probe_output_is_absent "${output}"; then + echo "OpenShift detection: cluster is not OpenShift (probe: ${probe} reports the API group is absent)." >&2 + printf '0\n' + return 0 + fi + + if [ "${attempt}" -ge "${attempts}" ]; then + break + fi + + echo "WARNING: OpenShift detection probe failed (attempt ${attempt}/${attempts}, exit ${status}), retrying in ${delay}s: ${output}" >&2 + sleep "${delay}" + delay=$((delay * 2)) + attempt=$((attempt + 1)) + done + + echo "ERROR: could not determine whether context '${context}' is OpenShift after ${attempts} attempt(s)." >&2 + echo "ERROR: probe: ${probe}" >&2 + echo "ERROR: last failure (exit ${status}): ${output}" >&2 + echo "ERROR: set OPENSHELL_E2E_OPENSHIFT=1 or OPENSHELL_E2E_OPENSHIFT=0 to bypass detection." >&2 + return 1 +} diff --git a/e2e/with-kube-gateway.sh b/e2e/with-kube-gateway.sh index a400ec29fe..761212453c 100755 --- a/e2e/with-kube-gateway.sh +++ b/e2e/with-kube-gateway.sh @@ -26,7 +26,9 @@ # (ci/values-openshift-e2e.yaml). # # Every OpenShift-specific branch below is gated on OPENSHIFT_DETECTED, so the -# vanilla-Kubernetes path stays exactly the same. +# vanilla-Kubernetes path stays exactly the same. Detection probes the cluster +# and aborts if it cannot answer conclusively; set OPENSHELL_E2E_OPENSHIFT to +# 1 or 0 to force the answer and skip the probe. # # Set OPENSHELL_E2E_KUBE_EXTRA_VALUES to one or more colon-separated Helm values # files, relative to the repository root or absolute, to layer additional chart @@ -1153,9 +1155,14 @@ AGENT_SANDBOX_VERSION="${AGENT_SANDBOX_VERSION}" \ bash "${ROOT}/e2e/support/install-agent-sandbox.sh" --context "${KUBE_CONTEXT}" # Detect OpenShift up front so fixtures deployed below can apply SCC-compatible -# handling; the gateway setup further down reuses this flag. -if kctl api-resources --api-group=route.openshift.io --no-headers 2>/dev/null | grep -q .; then - OPENSHIFT_DETECTED=1 +# handling; the gateway setup further down reuses this flag. A probe that cannot +# reach a conclusive answer aborts the run: silently assuming vanilla Kubernetes +# would reconfigure every OpenShift-gated branch below. +if ! OPENSHIFT_DETECTED="$(e2e_detect_openshift "${KUBE_CONTEXT}")"; then + exit 2 +fi + +if [ "${OPENSHIFT_DETECTED}" = "1" ]; then if ! command -v oc >/dev/null 2>&1; then echo "ERROR: oc CLI is required for OpenShift SCC management but was not found." >&2 exit 2 diff --git a/tasks/scripts/test-e2e-openshift-detection.sh b/tasks/scripts/test-e2e-openshift-detection.sh new file mode 100755 index 0000000000..7914b80c15 --- /dev/null +++ b/tasks/scripts/test-e2e-openshift-detection.sh @@ -0,0 +1,174 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025-2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +# shellcheck source=e2e/support/gateway-common.sh +source "${ROOT}/e2e/support/gateway-common.sh" + +WORKDIR="$(mktemp -d "${TMPDIR:-/tmp}/openshell-openshift-detect.XXXXXX")" +trap 'rm -rf "${WORKDIR}"' EXIT + +# A stand-in kubectl placed earlier on PATH. Behavior is selected by +# FAKE_KUBECTL_MODE and every invocation is counted in FAKE_KUBECTL_STATE so +# tests can assert how many times the probe actually ran. +mkdir -p "${WORKDIR}/bin" +cat >"${WORKDIR}/bin/kubectl" <<'FAKE' +#!/usr/bin/env bash +set -euo pipefail + +count=0 +if [ -s "${FAKE_KUBECTL_STATE}" ]; then + count="$(cat "${FAKE_KUBECTL_STATE}")" +fi +count=$((count + 1)) +printf '%s\n' "${count}" >"${FAKE_KUBECTL_STATE}" + +api_resource_list='{"kind":"APIResourceList","groupVersion":"route.openshift.io/v1"}' + +case "${FAKE_KUBECTL_MODE}" in + openshift) + printf '%s\n' "${api_resource_list}" + ;; + vanilla) + echo 'Error from server (NotFound): the server could not find the requested resource' >&2 + exit 1 + ;; + transient) + if [ "${count}" -lt "${FAKE_KUBECTL_SUCCEED_ON:-2}" ]; then + echo 'Unable to connect to the server: dial tcp 10.0.0.1:6443: i/o timeout' >&2 + exit 1 + fi + printf '%s\n' "${api_resource_list}" + ;; + broken) + echo 'error: You must be logged in to the server (Unauthorized)' >&2 + exit 1 + ;; + *) + echo "fake kubectl: unknown FAKE_KUBECTL_MODE '${FAKE_KUBECTL_MODE}'" >&2 + exit 64 + ;; +esac +FAKE +chmod +x "${WORKDIR}/bin/kubectl" +PATH="${WORKDIR}/bin:${PATH}" +export PATH + +DETECT_STDERR="${WORKDIR}/stderr" +DETECT_RESULT="" +DETECT_STATUS=0 +PROBE_CALLS=0 + +# Run the detection helper against the fake kubectl and record its stdout, +# exit status, stderr, and how many times the probe ran. +run_detection() { + local mode=$1 + shift + + export FAKE_KUBECTL_MODE="${mode}" + export FAKE_KUBECTL_STATE="${WORKDIR}/calls" + : >"${FAKE_KUBECTL_STATE}" + # Keep retries fast; the production default sleeps between attempts. + export OPENSHELL_E2E_OPENSHIFT_PROBE_DELAY=0 + + DETECT_STATUS=0 + DETECT_RESULT="$(env "$@" e2e_detect_openshift_run)" || DETECT_STATUS=$? + PROBE_CALLS=0 + if [ -s "${FAKE_KUBECTL_STATE}" ]; then + PROBE_CALLS="$(cat "${FAKE_KUBECTL_STATE}")" + fi +} + +# `env` cannot call a shell function, so route per-test environment through a +# tiny wrapper script that sources the helpers and invokes the function. +cat >"${WORKDIR}/bin/e2e_detect_openshift_run" <&2 + echo " result='${DETECT_RESULT}' status=${DETECT_STATUS} probe_calls=${PROBE_CALLS}" >&2 + echo " stderr:" >&2 + sed 's/^/ /' "${DETECT_STDERR}" >&2 || true + exit 1 +} + +assert_detection() { + local description=$1 + local expected_result=$2 + local expected_calls=$3 + + if [ "${DETECT_STATUS}" -ne 0 ]; then + fail "${description}: expected success, got exit ${DETECT_STATUS}" + fi + if [ "${DETECT_RESULT}" != "${expected_result}" ]; then + fail "${description}: expected result '${expected_result}', got '${DETECT_RESULT}'" + fi + if [ "${PROBE_CALLS}" -ne "${expected_calls}" ]; then + fail "${description}: expected ${expected_calls} probe call(s), got ${PROBE_CALLS}" + fi +} + +assert_stderr_contains() { + local description=$1 + local needle=$2 + + if ! grep -qF -- "${needle}" "${DETECT_STDERR}"; then + fail "${description}: expected stderr to contain '${needle}'" + fi +} + +# An OpenShift cluster answers the route.openshift.io/v1 probe. +run_detection openshift 2>"${DETECT_STDERR}" +assert_detection "OpenShift cluster is detected" 1 1 +assert_stderr_contains "OpenShift cluster is detected" "cluster is OpenShift" + +# A clean NotFound is a conclusive answer, so it must not be retried and must +# not be reported as an error. +run_detection vanilla 2>"${DETECT_STDERR}" +assert_detection "conclusive absence reports not-OpenShift" 0 1 +assert_stderr_contains "conclusive absence is logged" "cluster is not OpenShift" + +# A discovery blip must not be mistaken for vanilla Kubernetes. +run_detection transient FAKE_KUBECTL_SUCCEED_ON=3 2>"${DETECT_STDERR}" +assert_detection "transient failure then success is detected" 1 3 + +# A probe that never reaches a conclusive answer must fail loudly rather than +# silently selecting the vanilla-Kubernetes path. +run_detection broken OPENSHELL_E2E_OPENSHIFT_PROBE_ATTEMPTS=3 2>"${DETECT_STDERR}" +if [ "${DETECT_STATUS}" -eq 0 ]; then + fail "persistent probe failure must exit non-zero" +fi +if [ "${DETECT_RESULT}" = "0" ]; then + fail "persistent probe failure must not report a conclusive not-OpenShift answer" +fi +if [ "${PROBE_CALLS}" -ne 3 ]; then + fail "persistent probe failure should exhaust the configured attempts" +fi +assert_stderr_contains "failure names the probe" "/apis/route.openshift.io/v1" +assert_stderr_contains "failure names the underlying error" "Unauthorized" + +# The override short-circuits the probe in both directions. +run_detection broken OPENSHELL_E2E_OPENSHIFT=1 2>"${DETECT_STDERR}" +assert_detection "override forces detection on" 1 0 +assert_stderr_contains "override on is logged" "OPENSHELL_E2E_OPENSHIFT" + +run_detection openshift OPENSHELL_E2E_OPENSHIFT=false 2>"${DETECT_STDERR}" +assert_detection "override forces detection off" 0 0 +assert_stderr_contains "override off is logged" "OPENSHELL_E2E_OPENSHIFT" + +# A malformed override is a configuration error, not a silent default. +run_detection openshift OPENSHELL_E2E_OPENSHIFT=maybe 2>"${DETECT_STDERR}" +if [ "${DETECT_STATUS}" -eq 0 ]; then + fail "malformed override must exit non-zero" +fi +assert_stderr_contains "malformed override is explained" "OPENSHELL_E2E_OPENSHIFT" + +echo "E2E OpenShift detection tests passed." diff --git a/tasks/test.toml b/tasks/test.toml index 8a90f3c084..fa35fd0e9e 100644 --- a/tasks/test.toml +++ b/tasks/test.toml @@ -14,6 +14,7 @@ depends = [ "test:build-env", "test:gateway-pull-policy", "test:e2e-image-overrides", + "test:e2e-openshift-detection", "test:gateway-config", "test:packaging-assets", "test:qualification-summary", @@ -61,6 +62,12 @@ run = "tasks/scripts/test-e2e-image-overrides.sh" run_windows = "echo Skipping test:e2e-image-overrides: Unix E2E wrappers do not apply on Windows." hide = true +["test:e2e-openshift-detection"] +description = "Test E2E OpenShift cluster detection" +run = "tasks/scripts/test-e2e-openshift-detection.sh" +run_windows = "echo Skipping test:e2e-openshift-detection: Unix E2E wrappers do not apply on Windows." +hide = true + ["test:packaging-assets"] description = "Run static packaging asset tests" run = "tasks/scripts/test-packaging-assets.sh"