diff --git a/ci/pipeline.yml b/ci/pipeline.yml index 9defee75..f26f37e4 100644 --- a/ci/pipeline.yml +++ b/ci/pipeline.yml @@ -88,6 +88,7 @@ groups: - update-alicloud-resolute-stemcell - update-azure-resolute-stemcell - update-docker-resolute-stemcell + - update-docker-resolute-rosetta-stemcell - update-gcp-resolute-stemcell - update-openstack-resolute-stemcell - update-openstack-raw-resolute-stemcell @@ -1334,6 +1335,23 @@ jobs: STEMCELL_NAME: warden-ubuntu-resolute-stemcell - <<: *commit-when-new-asset +- name: update-docker-resolute-rosetta-stemcell + plan: + - in_parallel: + - get: bosh-deployment + params: + clean_tags: true + - get: warden-ubuntu-resolute-rosetta-stemcell + trigger: true + - task: update-stemcell + file: bosh-deployment/ci/tasks/update-stemcell.yml + input_mapping: + stemcell: warden-ubuntu-resolute-rosetta-stemcell + params: + CPI_OPS_FILE: docker/use-resolute-rosetta.yml + STEMCELL_NAME: warden-ubuntu-resolute-rosetta-stemcell + - <<: *commit-when-new-asset + - name: update-gcp-resolute-stemcell plan: - in_parallel: @@ -1665,6 +1683,14 @@ resources: source: name: bosh-warden-boshlite-ubuntu-resolute +# Keeps an x86_64 userland but ships native arm64 systemd daemons, so Resolute +# bootstraps under Rosetta 2 on Apple Silicon. Only docker/create-env consumes +# it; see docs/bosh-on-docker.md. +- name: warden-ubuntu-resolute-rosetta-stemcell + type: bosh-io-stemcell + source: + name: bosh-warden-boshlite-ubuntu-resolute-rosetta + - name: vsphere-ubuntu-resolute-stemcell type: bosh-io-stemcell source: diff --git a/docker/create-env b/docker/create-env new file mode 100755 index 00000000..e0fdda4e --- /dev/null +++ b/docker/create-env @@ -0,0 +1,862 @@ +#!/usr/bin/env bash +# +# Stand up a local BOSH Director on the Docker CPI. +# +# Run it from the directory you want your deployment files in -- state.json, +# creds.yml and bosh.env are written to $PWD, the same way +# virtualbox/create-env.sh works. Run with --help for the full flag list. + +# -E matters: without errtrace the ERR trap does not fire inside function bodies, +# and since every phase here is a function, a failure would exit silently -- +# which under --silent means no error, no log path, just a non-zero status. +set -Eeu -o pipefail + +# With BOSH_TTY=true the CLI appends a decorative "Succeeded" line to *stdout*, +# which silently corrupts everything this script captures with $( ) -- a CA cert +# with a trailing "Succeeded" fails as "Parsing certificate 2: Missing PEM +# block", and a --path lookup returns a value with a second line glued on. The +# progress output that BOSH_TTY exists for is requested explicitly with --tty on +# the create-env/delete-env calls instead, which is the only place it belongs. +export BOSH_TTY=false + +bosh_deployment="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)" + +# The Director's own address, the Docker bridge it lives on, and the socket the +# CPI talks to. These match the range in docker/cloud-config.yml; change them +# together or not at all. +NETWORK_NAME=${NETWORK_NAME:-bosh-net} +NETWORK_SUBNET=${NETWORK_SUBNET:-10.245.0.0/16} +INTERNAL_IP=${INTERNAL_IP:-10.245.0.10} +INTERNAL_GW=${INTERNAL_GW:-10.245.0.1} +DIRECTOR_NAME=${DIRECTOR_NAME:-bosh-docker} +# The daemon-side path, not the client's DOCKER_HOST. docker/unix-sock.yml +# bind-mounts this into the Director's container, and the CPI strips the +# unix:// prefix (vm.Factory.cleanMounts) before handing it to the daemon, which +# resolves it against its own filesystem. On Docker Desktop and Colima alike +# that is the Linux VM's /var/run/docker.sock, whatever socket your CLI uses. +DOCKER_HOST_URI=${DOCKER_HOST_URI:-unix:///var/run/docker.sock} + +# Where the CLI on *this* machine reaches the daemon, which is not the same +# thing. bosh create-env runs a CPI locally to build the Director's container, +# and that one is configured from /cloud_provider, so it needs the client +# endpoint. They coincide on Linux and Docker Desktop and diverge under Colima, +# whose socket lives in ~/.colima. Detected in preflight; empty means "same as +# DOCKER_HOST_URI, nothing to override". +CLIENT_DOCKER_HOST= + +# Overridable so --dry-run can render both host branches from any machine; see +# the --dry-run checks in tests/run-checks.sh. +HOST_OS=${CREATE_ENV_OS:-$(uname -s)} +HOST_ARCH=${CREATE_ENV_ARCH:-$(uname -m)} + +# Generated ops files and the record of how the Director was created. Kept out of +# the way of state.json/creds.yml, and read back by --destroy so the ops list +# cannot drift between create and delete. +WORK_DIR="${PWD}/.create-env" +RUN_RECORD="${WORK_DIR}/run" +LOG_FILE="${PWD}/create-env.log" + +FLAVOR=default +SILENT=false +RECREATE=false +DRY_RUN=false +DESTROY=false +STEMCELL_OVERRIDE= +BOSH_RELEASE= +USER_OPS=() +USER_VARS=() + +# --------------------------------------------------------------------------- +# output +# --------------------------------------------------------------------------- + +if [ -t 2 ]; then red=$'\033[31m'; yellow=$'\033[33m'; reset=$'\033[0m' +else red=; yellow=; reset=; fi + +# fd 4 is the real stderr. --silent redirects everything else into $LOG_FILE, so +# errors and the final hint still need somewhere to go. +exec 4>&2 + +STEP() { echo; echo "==\\"; echo "===> $*"; echo "==/"; echo; } +say() { echo " $*"; } +warn() { echo "${yellow}warning: $*${reset}" >&4; } +die() { echo "${red}error: $*${reset}" >&4; exit 1; } + +on_err() { + local status=$1 command=$2 + # A failing $( ) fires ERR twice under errtrace: once inside the substitution's + # subshell and once for the assignment that contains it. Report only the outer + # one, so a single failure is not printed twice with the log tail each time. + [ "$BASH_SUBSHELL" -eq 0 ] || return 0 + echo "${red}error: command failed (exit ${status}): ${command}${reset}" >&4 + if $SILENT; then + echo "${red}last 20 lines of ${LOG_FILE}:${reset}" >&4 + tail -20 "$LOG_FILE" >&4 2>/dev/null || true + echo "${red}full log: ${LOG_FILE}${reset}" >&4 + fi +} + +usage() { + cat <&2; die "unknown argument '$1'" ;; + esac + done +} + +# --------------------------------------------------------------------------- +# the ops/vars builder both create and destroy render from +# --------------------------------------------------------------------------- + +# Absolutise a path without requiring it to exist yet. +abspath() { + case "$1" in + /*) echo "$1" ;; + *) echo "${PWD}/${1#./}" ;; + esac +} + +resolute_rosetta_needed() { + [ "$FLAVOR" = resolute ] && [ "$HOST_OS" = Darwin ] && [ "$HOST_ARCH" = arm64 ] +} + +# OPS_ARGS and VAR_ARGS, in the one order that works. The constraints are real: +# +# - use-resolute.yml replaces the bosh-docker-cpi release cpi.yml appends, so +# it has to come after it. +# - use-compiled-resolute-releases.yml replaces /releases/name=uaa and +# /releases/name=credhub, so it has to come after uaa.yml and credhub.yml, +# which append their own uncompiled entries. +# - use-resolute-rosetta.yml overrides only the stemcell, on top of +# use-resolute.yml. +# +# Everything the caller passes lands after all of it. +build_ops() { + OPS_ARGS=() + VAR_ARGS=() + + local ops=(docker/cpi.yml) + + if [ "$FLAVOR" = resolute ]; then + ops+=(docker/use-resolute.yml) + if resolute_rosetta_needed; then + ops+=(docker/use-resolute-rosetta.yml) + fi + fi + + ops+=(uaa.yml credhub.yml) + + if [ "$FLAVOR" = resolute ]; then + ops+=(misc/use-compiled-resolute-releases.yml) + fi + + ops+=(docker/unix-sock.yml docker/dns.yml jumpbox-user.yml) + + local f + for f in "${ops[@]}"; do + OPS_ARGS+=(-o "${bosh_deployment}/${f}") + done + + if [ -n "$BOSH_RELEASE" ]; then + local release_path + release_path=$(abspath "$BOSH_RELEASE") + case "$BOSH_RELEASE" in + *.tgz|*.tar.gz) OPS_ARGS+=(-o "${bosh_deployment}/local-bosh-release-tarball.yml") ;; + *) OPS_ARGS+=(-o "${bosh_deployment}/local-bosh-release.yml") ;; + esac + VAR_ARGS+=(-v "local_bosh_release=${release_path}") + fi + + VAR_ARGS+=( + -v "director_name=${DIRECTOR_NAME}" + -v "internal_ip=${INTERNAL_IP}" + -v "internal_gw=${INTERNAL_GW}" + -v "internal_cidr=${NETWORK_SUBNET}" + -v "docker_host=${DOCKER_HOST_URI}" + -v "network=${NETWORK_NAME}" + ) + + # After everything the script adds, so a caller can override any of it. + for f in ${USER_OPS[@]+"${USER_OPS[@]}"}; do + OPS_ARGS+=(-o "$(abspath "$f")") + done + for f in ${USER_VARS[@]+"${USER_VARS[@]}"}; do + VAR_ARGS+=(-v "$f") + done + + # An explicit --stemcell wins over the ops files, including the rosetta one. + if [ -n "$STEMCELL_OVERRIDE" ]; then + OPS_ARGS+=(-o "$(write_stemcell_opsfile "$STEMCELL_OVERRIDE" stemcell-override.yml)") + fi + + # Point only the create-env/delete-env CPI at the client's socket. The + # Director's own CPI job keeps talking to the bind-mounted + # unix:///docker/docker.sock that docker/unix-sock.yml sets up, and the bind + # source keeps the daemon-side DOCKER_HOST_URI -- a virtiofs-mounted socket + # from the Mac side is not connectable from inside the VM. + if [ -n "$CLIENT_DOCKER_HOST" ]; then + OPS_ARGS+=(-o "$(write_bootstrap_cpi_opsfile)") + fi +} + +write_bootstrap_cpi_opsfile() { + local path="${WORK_DIR}/bootstrap-cpi-socket.yml" + cat > "$path" < "$RUN_RECORD" +} + +load_run_record() { + [ -f "$RUN_RECORD" ] || die "no ${RUN_RECORD#"${PWD}"/} here -- run --destroy from the directory you created the Director in." + # shellcheck source=/dev/null + . "$RUN_RECORD" +} + +# --------------------------------------------------------------------------- +# stemcell +# --------------------------------------------------------------------------- + +# An ops file replacing the stemcell the manifest pinned. Used for --stemcell and +# for pointing create-env at the tarball we already downloaded. +write_stemcell_opsfile() { + local source=$1 name=$2 path url sha1 + path="${WORK_DIR}/${name}" + + case "$source" in + http://*|https://*|file://*) url=$source ;; + *) + [ -f "$source" ] || die "no such stemcell: ${source}" + url="file://$(abspath "$source")" + ;; + esac + + if [ "${url#file://}" != "$url" ]; then + sha1=$(sha1_of "${url#file://}") + else + # Remote and unverifiable without downloading it first; the CLI uploads it + # unverified, which is what `bosh upload-stemcell URL` does anyway. + sha1= + fi + + { + echo "# Generated by docker/create-env. Not checked in -- tests/run-checks.sh" + echo "# interpolates every .yml in the repo." + echo "- name: stemcell" + echo " path: /resource_pools/name=vms/stemcell?" + echo " type: replace" + echo " value:" + [ -n "$sha1" ] && echo " sha1: ${sha1}" + echo " url: ${url}" + } > "$path" + + echo "$path" +} + +sha1_of() { + if command -v shasum > /dev/null; then + shasum -a 1 "$1" | cut -d' ' -f1 + else + sha1sum "$1" | cut -d' ' -f1 + fi +} + +# Read the stemcell the current ops stack resolves to, into STEMCELL_URL and +# STEMCELL_SHA1. Never a table in this script -- CI keeps the ops files current +# and this reads them back. +resolve_stemcell() { + local manifest + manifest=$(bosh int "${bosh_deployment}/bosh.yml" \ + "${OPS_ARGS[@]}" ${VAR_ARGS[@]+"${VAR_ARGS[@]}"} \ + --path /resource_pools/name=vms/stemcell) + + STEMCELL_URL=$(echo "$manifest" | sed -n 's/^url: *//p') + STEMCELL_SHA1=$(echo "$manifest" | sed -n 's/^sha1: *//p') + + [ -n "$STEMCELL_URL" ] || die "could not read a stemcell url out of the interpolated manifest" +} + +# The stemcell is needed twice: create-env builds the Director from it, and the +# upload step pushes the same tarball to that Director. Fetch it once, into $PWD +# next to state.json, and point both at that copy. --destroy leaves it alone, so +# recreating in the same directory does not re-download. +cache_stemcell() { + STEMCELL_LOCAL= + + if [ "${STEMCELL_URL#file://}" != "$STEMCELL_URL" ]; then + STEMCELL_LOCAL=${STEMCELL_URL#file://} + [ -f "$STEMCELL_LOCAL" ] || die "the manifest pins a local stemcell that is not here: ${STEMCELL_LOCAL} +Nothing tracked in this repo pins a file:// stemcell, so this is either a +--stemcell you passed or an ops file of your own. Build that tarball, or point +--stemcell somewhere that exists." + say "using the local stemcell already pinned: ${STEMCELL_LOCAL}" + return + fi + + STEMCELL_LOCAL="${PWD}/$(basename "${STEMCELL_URL%%\?*}")" + case "$STEMCELL_LOCAL" in + *.tgz|*.tar.gz) ;; + *) STEMCELL_LOCAL="${STEMCELL_LOCAL}.tgz" ;; + esac + + # A non-empty sha1 is required to reuse: a remote --stemcell override has none, + # and then a same-named file left from a previous run would be served for a URL + # that may now point at something else. + if [ -f "$STEMCELL_LOCAL" ] && [ -n "$STEMCELL_SHA1" ] && stemcell_matches; then + say "reusing ${STEMCELL_LOCAL##*/} (sha1 matches the manifest)" + elif [ -f "$STEMCELL_LOCAL" ] && [ -z "$STEMCELL_SHA1" ]; then + say "re-downloading ${STEMCELL_LOCAL##*/}: the manifest pins no sha1 for it, so the cached copy cannot be verified" + download_stemcell + else + download_stemcell + stemcell_matches || { + warn "sha1 mismatch on ${STEMCELL_LOCAL##*/}; re-downloading once" + download_stemcell + stemcell_matches || die "sha1 mismatch on ${STEMCELL_LOCAL} after re-download (wanted ${STEMCELL_SHA1})" + } + fi + + OPS_ARGS+=(-o "$(write_stemcell_opsfile "$STEMCELL_LOCAL" stemcell-local.yml)") +} + +stemcell_matches() { + # No sha1 to check against means nothing to disagree with. + [ -n "$STEMCELL_SHA1" ] || return 0 + [ "$(sha1_of "$STEMCELL_LOCAL")" = "$STEMCELL_SHA1" ] +} + +download_stemcell() { + say "downloading ${STEMCELL_URL}" + curl --fail --location --retry 3 --retry-delay 5 \ + --output "${STEMCELL_LOCAL}.part" "$STEMCELL_URL" + mv "${STEMCELL_LOCAL}.part" "$STEMCELL_LOCAL" +} + +# --------------------------------------------------------------------------- +# preflight +# --------------------------------------------------------------------------- + +drop_all_proxy() { + [ -n "${BOSH_ALL_PROXY:-}" ] || return 0 + warn "ignoring BOSH_ALL_PROXY=${BOSH_ALL_PROXY}" + unset BOSH_ALL_PROXY +} + +preflight() { + STEP "Preflight" + + command -v bosh > /dev/null || die "no bosh CLI on PATH" + say "bosh: $(bosh --version | head -1)" + + drop_all_proxy + + command -v docker > /dev/null || die "no docker CLI on PATH" + docker info > /dev/null 2>&1 || die "cannot reach the Docker daemon. Start it (on macOS with Colima: colima start) and try again." + say "docker: $(docker version --format '{{.Server.Version}}')" + + # The CLI's endpoint, which the local create-env CPI needs. Left empty when it + # already agrees with the daemon-side path, so Linux and Docker Desktop render + # exactly the upstream ops stack. + local endpoint + endpoint=${DOCKER_HOST:-$(docker context inspect --format '{{.Endpoints.docker.Host}}' 2>/dev/null || true)} + if [ -n "$endpoint" ] && [ "$endpoint" != "$DOCKER_HOST_URI" ]; then + CLIENT_DOCKER_HOST=$endpoint + say "this CLI reaches the daemon at ${CLIENT_DOCKER_HOST}; the Director's CPI keeps using ${DOCKER_HOST_URI} through the bind mount" + else + say "docker socket: ${DOCKER_HOST_URI}" + fi + + if docker network inspect "$NETWORK_NAME" > /dev/null 2>&1; then + # One line per pool, then the first IPv4 one. A network with IPv6 enabled or + # with two pools has several, and concatenating them would hand bosh + # something like "10.245.0.0/16fd00::/64" as internal_cidr. + local subnet + subnet=$(docker network inspect "$NETWORK_NAME" \ + --format '{{range .IPAM.Config}}{{println .Subnet}}{{end}}' 2> /dev/null \ + | grep -m1 '\.' || true) + if [ -z "$subnet" ]; then + warn "could not read an IPv4 subnet from docker network '${NETWORK_NAME}'; leaving internal_cidr at ${NETWORK_SUBNET}" + elif [ "$subnet" != "$NETWORK_SUBNET" ]; then + # An unrelated range cannot carry the Director's address at all -- which is + # what `docker network create bosh-net` without --subnet leaves behind, + # since Docker then picks something in 172.16/12. Adopting it would defer + # the failure to a create_vm networking error that explains none of this. + if ! cidr_contains "$subnet" "$INTERNAL_IP" || ! cidr_contains "$subnet" "$INTERNAL_GW"; then + die "docker network '${NETWORK_NAME}' is ${subnet}, which does not contain ${INTERNAL_IP} and ${INTERNAL_GW}. +Recreate it on ${NETWORK_SUBNET}: + + docker network rm ${NETWORK_NAME} + +and let create-env make it, or set INTERNAL_IP and INTERNAL_GW to addresses inside ${subnet}." + fi + # Otherwise adopt it rather than carry on with a netmask that disagrees + # with the bridge the Director is actually on. + say "docker network '${NETWORK_NAME}' already exists as ${subnet}; using that as internal_cidr instead of ${NETWORK_SUBNET}" + warn "'${NETWORK_NAME}' is ${subnet}, which differs from docker/cloud-config.yml's 10.245.0.0/16 range. The Director allocates from the bottom of that range, so a narrower subnet works until it hands out an address Docker's does not contain. Recreate the network to match." + NETWORK_SUBNET=$subnet + else + say "docker network '${NETWORK_NAME}': ${subnet}" + fi + else + say "creating docker network '${NETWORK_NAME}' (${NETWORK_SUBNET}, gateway ${INTERNAL_GW})" + docker network create \ + --subnet="$NETWORK_SUBNET" --gateway="$INTERNAL_GW" "$NETWORK_NAME" > /dev/null + fi + + # A Director already sitting on our IP -- typically one created from a + # different deployment directory -- otherwise surfaces as a create_vm failure + # deep in the CPI ("failed to set up container networking: Address already in + # use"), which says nothing about what is holding it. + # Read from a process substitution rather than a pipeline: a `while` loop whose + # last iteration does not match exits non-zero, which under errexit/pipefail + # would abort preflight with nothing but "command failed" whenever some other + # container happens to be on the network. + local squatter='' name ip + while read -r name; do + ip=$(docker inspect \ + -f "{{(index .NetworkSettings.Networks \"${NETWORK_NAME}\").IPAddress}}" \ + "$name" 2> /dev/null || true) + if [ "$ip" = "$INTERNAL_IP" ]; then + squatter=$name + break + fi + done < <(docker ps --filter "network=${NETWORK_NAME}" --format '{{.Names}}') + if [ -n "$squatter" ] && [ "$squatter" != "$(current_vm_cid)" ]; then + die "${INTERNAL_IP} is already taken on the '${NETWORK_NAME}' network by container '${squatter}', which this deployment directory does not own. +That is usually a Director created from another directory. Destroy it there +first, or point this one somewhere else with INTERNAL_IP and NETWORK_NAME." + fi + + if [ "$HOST_OS" = Darwin ]; then + # Docker's bridge networks live inside a Linux VM that is not routable from + # macOS, so every step from create-env onward talks to an unreachable IP + # without a tunnel. docker-mac-net-connect adds host routes for the subnet. + # + # Ask the gateway rather than reading the routing table: macOS netstat + # abbreviates destinations (10.245.0.0/24 prints as "10.245/24"), so + # grepping for the subnet as written gives a false negative, and a VPN's + # own route can cover the subnet without carrying traffic to Docker. + if ping -c1 -t2 "$INTERNAL_GW" > /dev/null 2>&1; then + say "docker network reachable from the host (${INTERNAL_GW} answers)" + else + warn "cannot reach ${INTERNAL_GW} from this host. On macOS the Docker bridge lives inside a Linux VM and is not reachable without a tunnel, so create-env will fail waiting for the agent on ${INTERNAL_IP}. Install and start docker-mac-net-connect: + + brew install chipmk/tap/docker-mac-net-connect + sudo brew services start chipmk/tap/docker-mac-net-connect +" + fi + fi +} + +# IPv4 containment, in bash: adding a python3 dependency for this would be a +# strange trade in a script that otherwise needs only bash, docker and bosh. +ipv4_to_int() { + local IFS=. o1 o2 o3 o4 + read -r o1 o2 o3 o4 <<< "$1" + case "${o1}${o2}${o3}${o4}" in + '' | *[!0-9]*) return 1 ;; + esac + echo $(((o1 << 24) + (o2 << 16) + (o3 << 8) + o4)) +} + +cidr_contains() { + local cidr=$1 addr=$2 net bits mask net_int addr_int + net=${cidr%/*} + bits=${cidr#*/} + case "$bits" in + '' | *[!0-9]*) return 1 ;; + esac + [ "$bits" -le 32 ] || return 1 + net_int=$(ipv4_to_int "$net") || return 1 + addr_int=$(ipv4_to_int "$addr") || return 1 + [ "$bits" -eq 0 ] && return 0 + mask=$(((0xFFFFFFFF << (32 - bits)) & 0xFFFFFFFF)) + [ $((net_int & mask)) -eq $((addr_int & mask)) ] +} + +# The VM this directory's state.json already owns, if any -- so a plain re-run +# against our own Director is not mistaken for a collision. +current_vm_cid() { + [ -f "${PWD}/state.json" ] || return 0 + # sed rather than a JSON parser: this keeps the script's dependencies to bash, + # docker and bosh. A python3 one-liner here that silently returned empty when + # python3 was missing would make preflight report this directory's own + # Director as somebody else's. + sed -n 's/.*"current_vm_cid"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' \ + "${PWD}/state.json" | head -1 +} + +guard_repo() { + # Compared as a path component: a bare prefix test also matches siblings such + # as ../bosh-deployment-work, which is a perfectly good place to run this. + if [ "$PWD" = "$bosh_deployment" ] \ + || [ "${PWD##"${bosh_deployment}"/}" != "${PWD}" ] \ + || [ -e docker/create-env ]; then + die "It looks like you are running this inside ${bosh_deployment}. +Run it from the directory you want your deployment files in, so creds.yml does +not end up in the repo." + fi +} + +# --------------------------------------------------------------------------- +# phases +# --------------------------------------------------------------------------- + +announce_flavor() { + case "$FLAVOR" in + resolute) + say "flavor: resolute (docker/use-resolute.yml + misc/use-compiled-resolute-releases.yml)" + if resolute_rosetta_needed; then + say "host: ${HOST_OS}/${HOST_ARCH} -- adding docker/use-resolute-rosetta.yml, because Resolute's systemd 259 uses pidfd syscalls Rosetta 2 does not translate" + else + say "host: ${HOST_OS}/${HOST_ARCH} -- no rosetta stemcell needed" + fi + ;; + *) + say "flavor: default (the stemcell docker/cpi.yml pins)" + say "host: ${HOST_OS}/${HOST_ARCH}" + ;; + esac + [ -n "$STEMCELL_OVERRIDE" ] && say "--stemcell ${STEMCELL_OVERRIDE} overrides the stemcell the ops files pin" + [ -n "$BOSH_RELEASE" ] && say "--bosh-release ${BOSH_RELEASE}" + return 0 +} + +print_invocation() { + local verb=$1 + printf 'bosh %s %s/bosh.yml\n' "$verb" "$bosh_deployment" + printf ' --state %s/state.json\n' "$PWD" + printf ' --vars-store %s/creds.yml\n' "$PWD" + local i + for ((i = 0; i < ${#OPS_ARGS[@]}; i += 2)); do + printf ' %s %s\n' "${OPS_ARGS[i]}" "${OPS_ARGS[i + 1]}" + done + for ((i = 0; i < ${#VAR_ARGS[@]}; i += 2)); do + printf ' %s %s\n' "${VAR_ARGS[i]}" "${VAR_ARGS[i + 1]}" + done + $RECREATE && printf ' --recreate\n' + return 0 +} + +do_create_env() { + STEP "Creating the Director" + # create-env writes progress to /dev/tty, so it prints nothing at all when + # redirected -- a 15 minute run then looks exactly like a hang. Always ask for + # it: it is what you would see on a terminal anyway, and it is the only way a + # piped or --silent run reports progress. + local extra=(--tty) + $RECREATE && extra+=(--recreate) + + bosh create-env "${bosh_deployment}/bosh.yml" \ + --state "${PWD}/state.json" \ + --vars-store "${PWD}/creds.yml" \ + "${OPS_ARGS[@]}" ${VAR_ARGS[@]+"${VAR_ARGS[@]}"} \ + ${extra[@]+"${extra[@]}"} +} + +write_env_file() { + STEP "Writing bosh.env" + + # bosh.env is shell, and this path is interpolated into commands inside it, so + # a deployment directory containing a space or a shell metacharacter has to be + # escaped -- otherwise sourcing it hands `bosh interpolate` a truncated path. + local creds + printf -v creds '%q' "${PWD}/creds.yml" + + # No secrets in here: it interpolates out of creds.yml at source time. + cat > "${PWD}/bosh.env" < bosh.env" + elif [ "$(readlink .envrc 2> /dev/null || true)" = bosh.env ]; then + say "bosh.env written, .envrc -> bosh.env already" + else + warn "leaving the existing .envrc here alone -- it is not ours. Source bosh.env yourself." + fi + + # shellcheck source=/dev/null + . "${PWD}/bosh.env" + + STEP "Targeting the Director" + bosh env +} + +update_cloud_config() { + STEP "Uploading the cloud config" + # -v network is not optional: the Director stores cloud configs verbatim, so + # without it every create_vm looks for a Docker network literally named + # ((network)). + bosh -n update-cloud-config "${bosh_deployment}/docker/cloud-config.yml" \ + -v "network=${NETWORK_NAME}" + say "uploaded with network=${NETWORK_NAME}" +} + +upload_stemcell() { + STEP "Uploading the stemcell" + local extra=() + [ -n "$STEMCELL_SHA1" ] && extra+=(--sha1 "$STEMCELL_SHA1") + bosh -n upload-stemcell "$STEMCELL_LOCAL" ${extra[@]+"${extra[@]}"} + bosh stemcells +} + +update_runtime_config() { + STEP "Uploading the bosh-dns runtime config" + bosh -n update-runtime-config "${bosh_deployment}/runtime-configs/dns.yml" + say "uploaded" +} + +# --------------------------------------------------------------------------- +# entry points +# --------------------------------------------------------------------------- + +create() { + + if $SILENT; then + : > "$LOG_FILE" + exec >> "$LOG_FILE" 2>&1 + echo "non-error output for this run only; see the terminal for failures" >&4 + fi + + STEP "Configuration" + announce_flavor + + # Preflight first: it discovers the docker socket and the network's real + # subnet, both of which build_ops bakes into the ops files and vars. Skipped + # for --dry-run, which promises to touch neither Docker nor the network. + $DRY_RUN || preflight + + STEP "Rendering the deployment" + build_ops + + STEP "Resolving the stemcell" + resolve_stemcell + say "stemcell: ${STEMCELL_URL}" + [ -n "$STEMCELL_SHA1" ] && say "sha1: ${STEMCELL_SHA1}" + + # Written before the dry-run bails out, so `--destroy --dry-run` can render the + # same ops list and tests/run-checks.sh can compare the two -- but never over a + # record that describes a Director this directory already owns, because + # --destroy replays it to delete that Director. + if ! $DRY_RUN || [ ! -f "${PWD}/state.json" ]; then + save_run_record + else + say "leaving ${RUN_RECORD#"${PWD}"/} alone: it records the Director this directory already owns" + fi + + if $DRY_RUN; then + if [ "${STEMCELL_URL#file://}" != "$STEMCELL_URL" ] && [ ! -f "${STEMCELL_URL#file://}" ]; then + warn "the pinned stemcell ${STEMCELL_URL#file://} is not on this machine; a real run would fail here" + fi + STEP "Would run" + print_invocation create-env + return 0 + fi + + cache_stemcell + + do_create_env + write_env_file + update_cloud_config + upload_stemcell + update_runtime_config + + STEP "Done" + say "Credentials are in ${PWD}/creds.yml and VM state in ${PWD}/state.json. Keep both." + say "To use the Director:" + echo + echo " source bosh.env" + echo + say "That also sets BOSH_AGENT_ENDPOINT/BOSH_AGENT_CERTIFICATE, so" + say "'bosh ssh --director' can get onto the Director itself." + $SILENT && echo "done -- source bosh.env" >&4 + return 0 +} + +destroy() { + load_run_record + + STEP "Configuration" + say "replaying the ops files and vars recorded in ${RUN_RECORD#"${PWD}"/}" + announce_flavor + build_ops + + # No stemcell handling here: delete-env does not fetch one, and the cached + # tarball stays so recreating in this directory does not re-download it. + if $DRY_RUN; then + STEP "Would run" + print_invocation delete-env + return 0 + fi + + STEP "Deleting the Director" + drop_all_proxy + local extra=(--tty) + bosh delete-env "${bosh_deployment}/bosh.yml" \ + --state "${PWD}/state.json" \ + --vars-store "${PWD}/creds.yml" \ + "${OPS_ARGS[@]}" ${VAR_ARGS[@]+"${VAR_ARGS[@]}"} \ + ${extra[@]+"${extra[@]}"} + + # These now name a Director that does not exist; keeping them only invites + # confusion with the next run. state.json needs no handling here -- delete-env + # removes it once the deployment is fully deleted. + rm -f "${PWD}/creds.yml" "${PWD}/bosh.env" + # Only the symlink we made, never one that was already here pointing elsewhere. + if [ "$(readlink .envrc 2> /dev/null || true)" = bosh.env ]; then + rm -f .envrc + fi + + # And the whole working directory, rather than the files this version happens + # to generate: that is what prunes output left by an older create-env whose + # generated files had different names. Everything in here is regenerated on the + # next create, and the run record it holds describes a Director that is now + # gone. The stemcell tarball lives in $PWD, not in here, so it survives. + rm -rf "$WORK_DIR" + + STEP "Done" + say "Director deleted. creds.yml, bosh.env and ${WORK_DIR##*/}/ removed, and delete-env took state.json with it. The stemcell tarball is still here, so recreating in this directory will not re-download it." + return 0 +} + +main() { + parse_args "$@" + trap 'on_err $? "$BASH_COMMAND"' ERR + + guard_repo + mkdir -p "$WORK_DIR" + + if $DESTROY; then destroy; else create; fi +} + +main "$@" diff --git a/docs/bosh-on-docker.md b/docs/bosh-on-docker.md index a83a05ac..8d6ba347 100644 --- a/docs/bosh-on-docker.md +++ b/docs/bosh-on-docker.md @@ -1,25 +1,14 @@ # BOSH on Docker -These steps bring up a local BOSH Director on the Docker CPI and deploy the -`nats` test deployment against it. They are the interactive equivalent of what -[`ci/tasks/test-docker.sh`](../ci/tasks/test-docker.sh) runs in CI. +`docker/create-env` stands up a local BOSH Director on the Docker CPI, on macOS +or Linux, and leaves you with a file you can `source`. It is the interactive +equivalent of what [`ci/tasks/test-docker.sh`](../ci/tasks/test-docker.sh) runs +in CI. ## Prerequisites -- A running Docker daemon reachable at `unix:///var/run/docker.sock` +- A running Docker daemon - The `bosh` CLI (7.x) -- A user-defined bridge network for the Director and its VMs: - - ```bash - docker network create --subnet=10.245.0.0/16 --gateway=10.245.0.1 bosh-net - ``` - - The gateway must match the `internal_gw` you pass to `create-env` and the - `gateway` in `docker/cloud-config.yml`. Size the subnet to cover that file's - `range` (`10.245.0.0/16`) — the Director allocates from the bottom of the - range, so a `/24` works for small deployments, but once you have enough VMs - the Director will hand out an address that Docker's subnet does not contain. - - **On macOS**, [`docker-mac-net-connect`](https://github.com/chipmk/docker-mac-net-connect) running in the background: @@ -30,150 +19,154 @@ These steps bring up a local BOSH Director on the Docker CPI and deploy the Docker runs inside a Linux VM on macOS, and that VM's bridge networks are not routable from the host — so without it there is no IP-level access from the Mac - to any container. Every step from step 2 onward talks to the Director at - `10.245.0.10` directly, so all of them fail. `docker-mac-net-connect` opens a - WireGuard tunnel into the VM and adds host routes for the Docker subnets. - Verify before continuing: + to any container, and everything after `create-env` times out against + `10.245.0.10`. `docker-mac-net-connect` opens a WireGuard tunnel into the VM + and adds host routes for the Docker subnets. `create-env` checks this by + pinging the bridge's gateway and warns if it does not answer. - ```bash - docker run --rm -d --name nettest --network bosh-net busybox sleep 60 - ping -c1 "$(docker inspect -f '{{(index .NetworkSettings.Networks "bosh-net").IPAddress}}' nettest)" - docker rm -f nettest - ``` +The Docker bridge network the Director and its VMs live on (`bosh-net`, +`10.245.0.0/16`, gateway `10.245.0.1`) is created for you if it does not exist. -## A note on DNS +## Deploy the Director -The Docker CPI runs every BOSH VM as a container, and the BOSH agent writes the -network's `dns` setting into `/etc/systemd/resolved.conf.d/10-bosh.conf` inside -that container. `bosh.yml` defaults that to Google's `8.8.8.8`, which fails -closed on any network that blocks public resolvers. The Director comes up fine -but cannot download remote releases or stemcells: +Run it from the directory you want your deployment files in — `state.json`, +`creds.yml`, `bosh.env` and the stemcell tarball are written to `$PWD`, and the +script refuses to run inside this checkout so credentials cannot land in the +repo. +```bash +mkdir -p ~/bosh-docker && cd ~/bosh-docker +/path/to/bosh-deployment/docker/create-env ``` -Task 47 | Downloading remote release: Downloading remote release (00:00:20) - L Error: Downloading remote release failed. -``` - -The failure is easy to misread, because ICMP to `8.8.8.8` is often answered by -the local Docker VM's NAT even when TCP and UDP port 53 are dropped. Confirm it -from inside the Director with `resolvectl status` and `getent hosts bosh.io` -rather than with `ping`. -Use `docker/dns.yml` and the `dns` value already set in -`docker/cloud-config.yml`, which both point at `127.0.0.11` — Docker's embedded -DNS server. It exists in every container's own network namespace and forwards to -whatever resolvers the Docker host is using, so it works regardless of the -network the host is on. - -## 1. Deploy the Director +That does the whole thing: preflight, `bosh create-env`, write and source +`bosh.env`, `bosh env`, upload the cloud config, upload the stemcell the +manifest pinned, upload the bosh-dns runtime config. When it finishes: ```bash -export DEPLOYMENT_DIR=~/bosh-docker # holds state.json and creds.yml -export BOSH_DEPLOYMENT=/path/to/bosh-deployment -mkdir -p "${DEPLOYMENT_DIR}" && cd "${DEPLOYMENT_DIR}" - -bosh create-env "${BOSH_DEPLOYMENT}/bosh.yml" \ - --state=state.json \ - --vars-store=creds.yml \ - -o "${BOSH_DEPLOYMENT}/docker/cpi.yml" \ - -o "${BOSH_DEPLOYMENT}/uaa.yml" \ - -o "${BOSH_DEPLOYMENT}/credhub.yml" \ - -o "${BOSH_DEPLOYMENT}/docker/unix-sock.yml" \ - -o "${BOSH_DEPLOYMENT}/docker/dns.yml" \ - -o "${BOSH_DEPLOYMENT}/jumpbox-user.yml" \ - -v director_name=bosh-docker \ - -v internal_cidr=10.245.0.0/24 \ - -v internal_gw=10.245.0.1 \ - -v internal_ip=10.245.0.10 \ - -v docker_host="unix:///var/run/docker.sock" \ - -v network=bosh-net +source bosh.env ``` -`uaa.yml` and `credhub.yml` are not optional here: both `ci/assets/nats.yml` and -`runtime-configs/dns.yml` declare a `variables:` block, and the Director can only -generate those credentials with a config server. Without them the Director comes -up reporting `config_server: disabled` and the deploy in step 6 fails to resolve -`((nats_password))`. - -`docker/unix-sock.yml` bind-mounts the host's Docker socket into the Director so -the CPI can create sibling containers. Drop it and pass `-v docker_tls=...` -instead if you are talking to a remote, TLS-protected daemon. +`bosh.env` holds no secrets — it interpolates out of `creds.yml` at source time. +A `.envrc` symlink to it is created so direnv users get the same behaviour. -Ops-file order matters if you also use one of the `misc/use-compiled-*-releases.yml` -files: those replace `/releases/name=uaa` and `/releases/name=credhub`, so they -must come *after* `uaa.yml` and `credhub.yml`, which append their own -uncompiled entries. - -`create-env` writes its progress to `/dev/tty`, so it emits **nothing** when you -redirect it to a file or a pipe. Pass `--tty` (or set `BOSH_TTY=true`) whenever -you capture the output — otherwise a long run looks identical to a hung one. - -## 2. Target the Director +It also exports `BOSH_AGENT_ENDPOINT` and `BOSH_AGENT_CERTIFICATE`, so you can get +a shell on the Director itself: ```bash -bosh int creds.yml --path /director_ssl/ca > ca.crt +bosh ssh --director +``` -export BOSH_ENVIRONMENT=10.245.0.10 -export BOSH_CA_CERT="${PWD}/ca.crt" -export BOSH_CLIENT=admin -export BOSH_CLIENT_SECRET="$(bosh int creds.yml --path /admin_password)" +To tear it down, from the same directory: -bosh env +```bash +/path/to/bosh-deployment/docker/create-env --destroy ``` -`bosh env` should report `config_server: enabled`. If it says `disabled`, go back -to step 1 and add `uaa.yml` and `credhub.yml`. - -## 3. Upload the cloud config +`--destroy` reads the flavor and ops files back out of `.create-env/run`, so it +deletes with exactly what the create built — no flags to remember, and no list +to keep in sync. It removes `creds.yml` and `bosh.env` afterwards, because both +now describe a Director that no longer exists. `bosh delete-env` removes +`state.json` itself once the deployment is gone. The stemcell tarball stays, so +recreating in the same directory does not re-download ~1GB. -```bash -bosh -n update-cloud-config "${BOSH_DEPLOYMENT}/docker/cloud-config.yml" \ - -v network=bosh-net +### Flags -bosh cloud-config # verify +``` +docker/create-env [--resolute] [--silent] [--recreate] [--dry-run] + [--bosh-release PATH] [--stemcell URL_OR_PATH] + [-o FILE] [--var k=v] +docker/create-env --destroy +docker/create-env --help ``` -`-v network=bosh-net` is required. `docker/cloud-config.yml` uses `((network))` -for the subnet's `cloud_properties.name`, and the Director stores cloud configs -verbatim — omit the var and it uploads the literal string `((network))`, after -which every `create_vm` fails looking for a Docker network by that name. The -verification step above is worth doing: the broken config is only visible as the -unresolved `((network))` in the output. +- `--resolute` uses the Resolute stemcell and the compiled Resolute releases — + `docker/use-resolute.yml` plus `misc/use-compiled-resolute-releases.yml`, + which is the combination the `test-resolute` job in `ci/pipeline.yml` already + uses. On Apple Silicon it also applies `docker/use-resolute-rosetta.yml`, + whose stemcell keeps an x86_64 userland but ships native arm64 systemd + daemons — Resolute's systemd 259 stops services with pidfd syscalls that + Rosetta 2 does not translate. +- `--bosh-release PATH` points the bosh release at a local build. A directory + uses `local-bosh-release.yml` (`version: create`); a `.tgz` uses + `local-bosh-release-tarball.yml` (`version: latest`). +- `--stemcell URL_OR_PATH` overrides the stemcell the ops files pin, including + the Apple Silicon auto-detection. Use it for "I just built a stemcell, use + it", rather than editing a tracked ops file. +- `-o FILE` and `--var k=v` are applied after everything the script adds, so + they override any of it. This is the escape hatch for anything there is no + flag for — a different `docker_host` for a remote TLS daemon, say. +- `--recreate` passes `--recreate` through to `bosh create-env`. See + [Recovering from an interrupted create-env](#recovering-from-an-interrupted-create-env). +- `--silent` puts everything but errors into `create-env.log` and prints the + failing command plus the tail of that log if the run fails. It passes `--tty` + to `create-env`, which otherwise writes its progress to `/dev/tty` and so + logs nothing at all when redirected. +- `--dry-run` prints the `create-env` invocation and the stemcell it resolved, + and touches neither Docker nor the network. + +### Ops-file ordering + +The script owns this, and that is most of the reason for it to exist: + +- `docker/use-resolute.yml` replaces the `bosh-docker-cpi` release that + `docker/cpi.yml` appends, so it must come after it. +- `misc/use-compiled-resolute-releases.yml` replaces `/releases/name=uaa` and + `/releases/name=credhub`, so it must come after `uaa.yml` and `credhub.yml`, + which append their own uncompiled entries. +- `docker/use-resolute-rosetta.yml` overrides only the stemcell, on top of + `docker/use-resolute.yml`. + +Get any of those wrong and nothing complains — you just get the wrong release or +stemcell in the deployed Director. `tests/run-checks.sh` asserts the resulting +ops list and stemcell URL for every flavor for exactly that reason. -## 4. Upload a stemcell +`docker/unix-sock.yml` bind-mounts the host's Docker socket into the Director so +the CPI can create sibling containers. Pass `--var docker_tls=...` and +`-o` your own file instead if you are talking to a remote, TLS-protected daemon. -```bash -bosh upload-stemcell \ - https://storage.googleapis.com/bosh-core-stemcells/1.484/bosh-stemcell-1.484-warden-boshlite-ubuntu-noble.tgz +## A note on DNS -bosh stemcells -``` +The Docker CPI runs every BOSH VM as a container, and the BOSH agent writes the +network's `dns` setting into `/etc/systemd/resolved.conf.d/10-bosh.conf` inside +that container. `bosh.yml` defaults that to Google's `8.8.8.8`, which fails +closed on any network that blocks public resolvers. The Director comes up fine +but cannot download remote releases or stemcells: -Use the same version referenced by `docker/cpi.yml`, or your own locally built -warden stemcell. Note the `OS` column — step 6 needs it. +``` +Task 47 | Downloading remote release: Downloading remote release (00:00:20) + L Error: Downloading remote release failed. +``` -## 5. Upload the bosh-dns runtime config +The failure is easy to misread, because ICMP to `8.8.8.8` is often answered by +the local Docker VM's NAT even when TCP and UDP port 53 are dropped. Confirm it +from inside the Director with `resolvectl status` and `getent hosts bosh.io` +rather than with `ping`. -```bash -bosh -n update-runtime-config "${BOSH_DEPLOYMENT}/runtime-configs/dns.yml" -``` +`create-env` applies `docker/dns.yml`, and `docker/cloud-config.yml` carries the +matching value for deployed VMs. Both point at `127.0.0.11` — Docker's embedded +DNS server, which exists in every container's own network namespace and forwards +to whatever resolvers the Docker host is using, so it works regardless of the +network the host is on. -On Noble and Resolute stemcells this addon sets `configure_systemd_resolved: true` -and `disable_recursors: true`, so bosh-dns answers only for BOSH domains and -leaves everything else to systemd-resolved — which is exactly why the `127.0.0.11` -value from step 3 has to be right for deployed VMs too. +## Verify it: deploy nats -## 6. Deploy nats +`create-env` stops once the Director is configured. Deploying something is the +step that proves it works, and it is the same thing CI does: ```bash -bosh -n -d nats deploy "${BOSH_DEPLOYMENT}/ci/assets/nats.yml" \ +source bosh.env +bosh stemcells # note the OS column + +bosh -n -d nats deploy /path/to/bosh-deployment/ci/assets/nats.yml \ -v stemcell_os=ubuntu-noble bosh -d nats instances --ps bosh -n -d nats run-errand smoke-tests ``` -`stemcell_os` must match the `OS` of the stemcell uploaded in step 4. +`stemcell_os` must match the `OS` of the uploaded stemcell — `ubuntu-resolute` +if you used `--resolute`. A healthy result looks like this — two `nats` instances, each running `bosh-dns`, `bosh-dns-healthcheck`, `nats-tls-healthcheck` and `nats-tls-wrapper`: @@ -190,6 +183,22 @@ nats/50084029-f03c-4784-a0e0-84eaeb5ba815 - running and the errand exits `0` with `Detected no non-TLS hosts` on stderr, which is expected — this deployment only runs the TLS leg of the smoke tests. +## What CI covers, and what it does not + +`tests/run-checks.sh` runs on every commit in a plain container with no Docker +daemon, so it covers `create-env` statically: shellcheck, `--help`, the +`--dry-run` ops list for each flavor on both `uname` branches, and the stemcell +URL each flavor resolves to. + +The end-to-end coverage is the `test-docker` job, which is privileged and does +stand up a real Director — but it sources `start-bosh` from the +`bosh-docker-cpi` image rather than running this script, because that image +handles nested-container specifics `create-env` does not. So **CI exercises the +ops files and the manifest, not `docker/create-env` itself.** That gap is +deliberate for now: `create-env` in a privileged Concourse container is a +materially different environment from a laptop, and making it work there is its +own piece of work. + ## Troubleshooting ### Package compilation hangs forever on an emulated stemcell @@ -224,4 +233,4 @@ names the exact agent call it is polling. If `create-env` is interrupted after it has created the VM but before it applies a spec, the container is left with no jobs and the agent reports the instance as `unknown/0`. Re-running plain `create-env` then tries to update that VM in place. -Pass `--recreate` to replace the VM instead, which skips the in-place update path. +Pass `--recreate` to replace the VM instead, which skips the in-place update path. \ No newline at end of file diff --git a/tests/run-checks.sh b/tests/run-checks.sh index 3e5b80ba..5072d19a 100755 --- a/tests/run-checks.sh +++ b/tests/run-checks.sh @@ -8,9 +8,16 @@ cd "${script_dir}/.." tmp_file="/tmp/bosh-deployment-test" touch "${tmp_file}" +# create-env refuses to run inside this checkout and writes its generated ops +# files into $PWD, so every check below runs from a scratch directory. One +# mktemp root per invocation, removed whole, so cleanup never globs over paths +# this run does not own. +create_env_dir=$(mktemp -d "${TMPDIR:-/tmp}/bosh-deployment-create-env-check.XXXXXX") + function clean_tmp() { rm -f "${tmp_file}" rm -f "${tmp_file}."* + rm -rf -- "${create_env_dir}" } trap clean_tmp EXIT @@ -30,6 +37,159 @@ grep -r -i bosh-compiled-release-tarballs.s3.amazonaws.com . | grep -v grep | gr echo -e "\nUsed stemcells\n" grep -r -i d/stemcells . | grep -v grep | grep -v ./.git +echo -e "\ndocker/create-env\n" + +# The script refuses to run inside this checkout, and it writes generated ops +# files into $PWD, so every check below runs from create_env_dir (created above, +# alongside the cleanup that owns it). +create_env="${PWD}/docker/create-env" + +if command -v shellcheck > /dev/null; then + echo "- shellcheck" + shellcheck "${create_env}" +else + echo "- shellcheck (SKIPPED: not installed in this image)" +fi + +echo "- --help exits 0" +"${create_env}" --help > /dev/null + +# Render the create-env invocation for one flavor/host and print just the ops +# files, relative to the checkout, one per line. +function create_env_ops() { + local os=$1 arch=$2 + shift 2 + ( + cd "${create_env_dir}" + CREATE_ENV_OS="${os}" CREATE_ENV_ARCH="${arch}" "${create_env}" --dry-run "$@" + ) | sed -n "s|^ -o ${PWD}/||p" +} + +# Ops-file *ordering* is otherwise completely silent: whether +# misc/use-compiled-resolute-releases.yml lands after credhub.yml, and whether +# docker/use-resolute-rosetta.yml lands after docker/use-resolute.yml, only +# shows up as the wrong release or stemcell URL in the deployed Director. +function assert_ops() { + local label=$1 expected=$2 actual=$3 + if [ "${expected}" != "${actual}" ]; then + echo "ERROR: ${label} ops files are not what we expect" >&2 + diff -u <(echo "${expected}") <(echo "${actual}") >&2 || true + exit 1 + fi + echo "- ${label} ops files" +} + +default_ops="docker/cpi.yml +uaa.yml +credhub.yml +docker/unix-sock.yml +docker/dns.yml +jumpbox-user.yml" + +resolute_ops="docker/cpi.yml +docker/use-resolute.yml +uaa.yml +credhub.yml +misc/use-compiled-resolute-releases.yml +docker/unix-sock.yml +docker/dns.yml +jumpbox-user.yml" + +resolute_rosetta_ops="docker/cpi.yml +docker/use-resolute.yml +docker/use-resolute-rosetta.yml +uaa.yml +credhub.yml +misc/use-compiled-resolute-releases.yml +docker/unix-sock.yml +docker/dns.yml +jumpbox-user.yml" + +assert_ops "default (Linux/x86_64)" "${default_ops}" "$(create_env_ops Linux x86_64)" +assert_ops "default (Darwin/arm64)" "${default_ops}" "$(create_env_ops Darwin arm64)" +assert_ops "--resolute (Linux/x86_64)" "${resolute_ops}" "$(create_env_ops Linux x86_64 --resolute)" +assert_ops "--resolute (Darwin/arm64)" "${resolute_rosetta_ops}" "$(create_env_ops Darwin arm64 --resolute)" + +# --destroy has to delete with exactly what create built, and it reads the +# flavor back out of .create-env/run rather than being told again. +echo "- --destroy renders the same ops files as create" +# Re-render the create record first: each --dry-run rewrites .create-env/run, so +# without this the assertion below depends on which case ran last. +create_env_ops Darwin arm64 --resolute > /dev/null +destroy_ops=$( + cd "${create_env_dir}" + CREATE_ENV_OS=Darwin CREATE_ENV_ARCH=arm64 "${create_env}" --destroy --dry-run +) +assert_ops "--destroy after --resolute (Darwin/arm64)" \ + "${resolute_rosetta_ops}" "$(echo "${destroy_ops}" | sed -n "s|^ -o ${PWD}/||p")" +echo "${destroy_ops}" | grep -q '^bosh delete-env ' \ + || { echo "ERROR: --destroy did not render a delete-env invocation" >&2; exit 1; } + +# The run record is sourced from inside a function, so the arrays in it have to +# be plain assignments; `declare -a` would scope them to that function and +# --destroy would delete with a different ops stack than it created. +echo "- --destroy replays the caller's -o and -v arguments" +custom_ops="${tmp_file}.custom-ops.yml" +echo '[]' > "${custom_ops}" +create_env_ops Linux x86_64 -o "${custom_ops}" -v custom_check=1 > /dev/null +destroy_custom=$( + cd "${create_env_dir}" + CREATE_ENV_OS=Linux CREATE_ENV_ARCH=x86_64 "${create_env}" --destroy --dry-run +) +for expected in "-o ${custom_ops}" "-v custom_check=1"; do + if ! echo "${destroy_custom}" | grep -q -- "${expected}"; then + echo "ERROR: --destroy dropped '${expected}' from the replayed invocation" >&2 + echo "${destroy_custom}" >&2 + exit 1 + fi +done + +# The script must never carry its own stemcell table -- CI bumps the ops files, +# and create-env reads the pin back out of the interpolated manifest. +echo "- stemcell pins come from the ops files" +function assert_stemcell_url() { + local label=$1 expected=$2 actual=$3 + [ "${expected}" = "${actual}" ] \ + || { echo "ERROR: ${label} stemcell is '${actual}', expected '${expected}'" >&2; exit 1; } + echo " ${label}: ${actual}" +} + +noble_url=$(command bosh int bosh.yml -o docker/cpi.yml --path /resource_pools/name=vms/stemcell/url) +resolute_url=$(command bosh int bosh.yml -o docker/cpi.yml -o docker/use-resolute.yml \ + --path /resource_pools/name=vms/stemcell/url) + +function create_env_stemcell() { + local os=$1 arch=$2 + shift 2 + ( + cd "${create_env_dir}" + CREATE_ENV_OS="${os}" CREATE_ENV_ARCH="${arch}" "${create_env}" --dry-run "$@" + ) | sed -n 's|^ *stemcell: ||p' +} + +assert_stemcell_url "default" "${noble_url}" "$(create_env_stemcell Linux x86_64)" +assert_stemcell_url "--resolute" "${resolute_url}" "$(create_env_stemcell Linux x86_64 --resolute)" + +# On Apple Silicon the rosetta ops file has to win, and it may only change the +# stemcell -- the compiled-for-resolute docker CPI from use-resolute.yml stays. +rosetta_url=$(command bosh int bosh.yml -o docker/cpi.yml -o docker/use-resolute.yml \ + -o docker/use-resolute-rosetta.yml --path /resource_pools/name=vms/stemcell/url) +assert_stemcell_url "--resolute on Apple Silicon" \ + "${rosetta_url}" "$(create_env_stemcell Darwin arm64 --resolute)" +[ "${rosetta_url}" != "${resolute_url}" ] \ + || { echo "ERROR: docker/use-resolute-rosetta.yml did not override the stemcell" >&2; exit 1; } +cpi_url=$(command bosh int bosh.yml -o docker/cpi.yml -o docker/use-resolute.yml \ + -o docker/use-resolute-rosetta.yml --path /releases/name=bosh-docker-cpi/url) +case "${cpi_url}" in + *ubuntu-resolute*) echo " --resolute on Apple Silicon keeps the resolute docker CPI" ;; + *) echo "ERROR: rosetta ops file clobbered the resolute docker CPI: ${cpi_url}" >&2; exit 1 ;; +esac + +echo "- an explicit --stemcell overrides the pin" +override=$(create_env_stemcell Darwin arm64 --resolute --stemcell https://example.com/my.tgz) +[ "${override}" = "https://example.com/my.tgz" ] \ + || { echo "ERROR: --stemcell did not win, got '${override}'" >&2; exit 1; } + echo -e "\nExamples\n" echo "- AWS"