From 53d47d3a90d84093a14ab14b56e2d6a2474f0bfb Mon Sep 17 00:00:00 2001 From: Julien Goux Date: Fri, 2 Oct 2026 09:26:52 +0200 Subject: [PATCH 1/3] ci(stack): retry slim pickup when a merge-queued PR holds the branch --- .github/workflows/slim-release-published.yml | 72 ++++++++++++++++++-- docs/adr/0026-slim-artifact-mirrors.md | 21 ++++-- 2 files changed, 81 insertions(+), 12 deletions(-) diff --git a/.github/workflows/slim-release-published.yml b/.github/workflows/slim-release-published.yml index 2307ab2b01..4ddd3fe689 100644 --- a/.github/workflows/slim-release-published.yml +++ b/.github/workflows/slim-release-published.yml @@ -1,4 +1,5 @@ name: Slim Release Published +run-name: slim-release-published ${{ github.event.client_payload.service }} # Reconciles the stack catalog against supabase/slim-services releases. See # docs/adr/0026-slim-artifact-mirrors.md, "Hotfix and upgrade pickup". @@ -19,9 +20,10 @@ jobs: pickup: runs-on: ubuntu-latest # Setup (~10 min) + release visibility (<1 min) + the S3-mirror wait (up to ~10 min, normally only - # for the first planned item) + pin, render and push per item. - timeout-minutes: 45 + # for the first planned item) + pin, render and push per item + the merge-queue wait (up to 30 min). + timeout-minutes: 75 permissions: + actions: read contents: read steps: - name: Checkout the default branch @@ -103,12 +105,18 @@ jobs: # One branch per planned item, rebuilt from the recorded base commit, never from a previous # iteration's HEAD; `--release` pins the planned release, never "highest at apply time". # `bun`/`pnpm` run under `env -u APP_TOKEN`: only `git push` and `gh` ever see the app token. + # A branch whose pull request is in the merge queue rejects pushes; the step waits for the queue + # to release it, then stops; a fresh run re-plans every remaining item from the new default branch. + # That is the service's pending run when one exists, since a replay would cancel it, otherwise a + # re-dispatch of this event. - name: Apply planned updates if: steps.plan.outputs.count != '0' env: APP_TOKEN: ${{ steps.app-token.outputs.token }} GITHUB_TOKEN: ${{ github.token }} SERVICE: ${{ steps.validate.outputs.service }} + UPSTREAM_VERSION: ${{ steps.validate.outputs.upstream_version }} + REVISION: ${{ steps.validate.outputs.revision }} RELEASE_VERSION: ${{ steps.validate.outputs.release_version }} BASE_SHA: ${{ steps.base.outputs.sha }} DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} @@ -119,6 +127,14 @@ jobs: push_url="${PUSH_REMOTE_URL:-https://x-access-token:${APP_TOKEN}@github.com/${GITHUB_REPOSITORY}.git}" manual_prefix="bun .github/scripts/sync-artifacts-catalog.ts --service ${SERVICE} --release" + in_merge_queue() { + # shellcheck disable=SC2016 # `$owner`, `$name` and `$branch` are GraphQL variables. + GH_TOKEN="$APP_TOKEN" gh api graphql \ + -f query='query($owner: String!, $name: String!, $branch: String!) { repository(owner: $owner, name: $name) { pullRequests(headRefName: $branch, states: OPEN, first: 10) { nodes { isInMergeQueue isCrossRepository } } } }' \ + -f owner="${GITHUB_REPOSITORY%/*}" -f name="${GITHUB_REPOSITORY#*/}" -f branch="$1" \ + --jq '[.data.repository.pullRequests.nodes[] | select(.isCrossRepository | not) | .isInMergeQueue] | any' + } + # Reads records on fd 3, not stdin, so commands the loop runs (`bun`, `gh`, `git`) never # consume bytes meant for `read`. while IFS=$'\x1f' read -r -u 3 kind branch title release from; do @@ -146,10 +162,56 @@ jobs: git add -- "${generated_files[@]}" git commit -m "$title" - if ! git push --force "$push_url" "HEAD:refs/heads/${branch}"; then - echo "::error ::Failed to push ${branch}. Run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." - exit 1 + if ! push_output="$(git push --force "$push_url" "HEAD:refs/heads/${branch}" 2>&1)"; then + echo "$push_output" + if ! grep -q "merge queue" <<<"$push_output"; then + echo "::error ::Failed to push ${branch}. Run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." + exit 1 + fi + manual_steps="Once ${branch} leaves the merge queue, run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." + # The app token minted just before this step expires after one hour. + deadline=$((SECONDS + 1800 < 3000 ? SECONDS + 1800 : 3000)) + while :; do + if ! queued="$(in_merge_queue "$branch")"; then + echo "::error ::Failed to read the merge-queue state of ${branch}. ${manual_steps}" + exit 1 + fi + case "$queued" in + false) break ;; + true) ;; + *) + echo "::error ::Unexpected merge-queue state '${queued}' for ${branch}. ${manual_steps}" + exit 1 + ;; + esac + if [ "$SECONDS" -ge "$deadline" ]; then + echo "::error ::${branch} is still in the merge queue. ${manual_steps}" + exit 1 + fi + sleep 30 + done + if ! pending="$(GH_TOKEN="$GITHUB_TOKEN" gh run list --repo "$GITHUB_REPOSITORY" --workflow slim-release-published.yml --status pending --json displayTitle | + jq --arg title "slim-release-published ${SERVICE}" '[.[] | select(.displayTitle == $title)] | length')"; then + echo "::error ::Failed to list pending slim-release-published runs. ${manual_steps}" + exit 1 + fi + if [ "$pending" -gt 0 ]; then + echo "::warning ::${branch} was held by a merge-queued pull request; a pending ${SERVICE} run re-plans from ${DEFAULT_BRANCH}." + exit 0 + fi + if ! GH_TOKEN="$APP_TOKEN" gh api "repos/${GITHUB_REPOSITORY}/dispatches" \ + -f event_type=slim-release-published \ + -f "client_payload[service]=${SERVICE}" \ + -f "client_payload[upstream_version]=${UPSTREAM_VERSION}" \ + -f "client_payload[revision]=${REVISION}" \ + -f "client_payload[release_version]=${RELEASE_VERSION}"; then + echo "::error ::Failed to re-dispatch slim-release-published for ${SERVICE} ${RELEASE_VERSION}. Re-run this job." + exit 1 + fi + echo "::warning ::${branch} was held by a merge-queued pull request; re-dispatched slim-release-published so a fresh run re-plans from ${DEFAULT_BRANCH}." + exit 0 fi + echo "$push_output" if [ "$kind" = "upgrade" ]; then action="bumps" diff --git a/docs/adr/0026-slim-artifact-mirrors.md b/docs/adr/0026-slim-artifact-mirrors.md index 3ea6dd22ff..9063b01c22 100644 --- a/docs/adr/0026-slim-artifact-mirrors.md +++ b/docs/adr/0026-slim-artifact-mirrors.md @@ -148,13 +148,20 @@ the same day compare equal and never produce an upgrade; the manual `--release` build. Plan and apply run from the same checkout of the default branch in one job, so a re-run always -recomputes from the latest develop: a stale plan can never be applied, and a superseded PR's -branch is rewritten (force-pushed) in place rather than raced by a new one — a superseded PR is -never auto-closed. A backlog republish of an older upstream version naturally plans nothing. The -fallback, if the push or PR step fails, is a documented manual `bun -.github/scripts/sync-artifacts-catalog.ts --service --release -r` invocation, followed -by `apps/cli/scripts/render-service-dockerfile.ts` — the release itself is already committed by -then, so a failure here means "open the pull request by hand", not "republish". +recomputes from the latest develop: a stale plan can never be applied, and a superseded PR's branch +is rewritten (force-pushed) in place rather than raced by a new one — a superseded PR is never +auto-closed. A backlog republish of an older upstream version naturally plans nothing. A branch +whose PR is in the merge queue rejects the force-push; the run waits up to 30 minutes for the queue +to merge or drop that PR, then stops so a fresh run re-plans every remaining update from the updated +default branch: the service's pending run when one exists (a replay would cancel it and its own +release-visibility wait), otherwise a re-send of the same dispatch. The lookup and the re-send are +not atomic: a newer dispatch arriving between them is replaced by the replay, which plans it only if +its release is already listed by then. A queue that holds the branch longer, or a failed queue +lookup or dispatch, fails the run with the manual invocation below. The fallback, if the push or PR +step fails, is a documented manual `bun .github/scripts/sync-artifacts-catalog.ts --service +--release -r` invocation, followed by `apps/cli/scripts/render-service-dockerfile.ts` — the +release itself is already committed by then, so a failure here means "open the pull request by +hand", not "republish". The app token (contents and pull-requests write) never reaches third-party code or disk: `git push` takes it only inside an explicit URL (`PUSH_REMOTE_URL` overrides it, so a dry run can From fc02fadccc91d973dcc0ffedf63f209683affe24 Mon Sep 17 00:00:00 2001 From: Julien Goux Date: Fri, 2 Oct 2026 11:50:43 +0200 Subject: [PATCH 2/3] ci(stack): bound slim pickup re-dispatches after merge-queue rejections --- .github/workflows/slim-release-published.yml | 9 ++++++++- docs/adr/0026-slim-artifact-mirrors.md | 16 ++++++++-------- 2 files changed, 16 insertions(+), 9 deletions(-) diff --git a/.github/workflows/slim-release-published.yml b/.github/workflows/slim-release-published.yml index 4ddd3fe689..83da29b2e7 100644 --- a/.github/workflows/slim-release-published.yml +++ b/.github/workflows/slim-release-published.yml @@ -118,6 +118,7 @@ jobs: UPSTREAM_VERSION: ${{ steps.validate.outputs.upstream_version }} REVISION: ${{ steps.validate.outputs.revision }} RELEASE_VERSION: ${{ steps.validate.outputs.release_version }} + REPLAY: ${{ github.event.client_payload.replay }} BASE_SHA: ${{ steps.base.outputs.sha }} DEFAULT_BRANCH: ${{ github.event.repository.default_branch }} run: | @@ -199,12 +200,18 @@ jobs: echo "::warning ::${branch} was held by a merge-queued pull request; a pending ${SERVICE} run re-plans from ${DEFAULT_BRANCH}." exit 0 fi + # `replay` counts consecutive re-sends; anything but 0-2 stops the chain. + if ! [[ "${REPLAY:-0}" =~ ^[0-2]$ ]]; then + echo "::error ::Stopped re-dispatching after repeated merge-queue rejections of ${branch}. ${manual_steps}" + exit 1 + fi if ! GH_TOKEN="$APP_TOKEN" gh api "repos/${GITHUB_REPOSITORY}/dispatches" \ -f event_type=slim-release-published \ -f "client_payload[service]=${SERVICE}" \ -f "client_payload[upstream_version]=${UPSTREAM_VERSION}" \ -f "client_payload[revision]=${REVISION}" \ - -f "client_payload[release_version]=${RELEASE_VERSION}"; then + -f "client_payload[release_version]=${RELEASE_VERSION}" \ + -f "client_payload[replay]=$((${REPLAY:-0} + 1))"; then echo "::error ::Failed to re-dispatch slim-release-published for ${SERVICE} ${RELEASE_VERSION}. Re-run this job." exit 1 fi diff --git a/docs/adr/0026-slim-artifact-mirrors.md b/docs/adr/0026-slim-artifact-mirrors.md index 9063b01c22..0b74f9d99e 100644 --- a/docs/adr/0026-slim-artifact-mirrors.md +++ b/docs/adr/0026-slim-artifact-mirrors.md @@ -154,14 +154,14 @@ auto-closed. A backlog republish of an older upstream version naturally plans no whose PR is in the merge queue rejects the force-push; the run waits up to 30 minutes for the queue to merge or drop that PR, then stops so a fresh run re-plans every remaining update from the updated default branch: the service's pending run when one exists (a replay would cancel it and its own -release-visibility wait), otherwise a re-send of the same dispatch. The lookup and the re-send are -not atomic: a newer dispatch arriving between them is replaced by the replay, which plans it only if -its release is already listed by then. A queue that holds the branch longer, or a failed queue -lookup or dispatch, fails the run with the manual invocation below. The fallback, if the push or PR -step fails, is a documented manual `bun .github/scripts/sync-artifacts-catalog.ts --service ---release -r` invocation, followed by `apps/cli/scripts/render-service-dockerfile.ts` — the -release itself is already committed by then, so a failure here means "open the pull request by -hand", not "republish". +release-visibility wait), otherwise a re-send of the same dispatch, at most three times in a row. +The lookup and the re-send are not atomic: a newer dispatch arriving between them is replaced by the +replay, which plans it only if its release is already listed by then. A queue that holds the branch +longer, or a failed queue lookup or dispatch, fails the run with the manual invocation below. The +fallback, if the push or PR step fails, is a documented manual `bun +.github/scripts/sync-artifacts-catalog.ts --service --release -r` invocation, followed +by `apps/cli/scripts/render-service-dockerfile.ts` — the release itself is already committed by +then, so a failure here means "open the pull request by hand", not "republish". The app token (contents and pull-requests write) never reaches third-party code or disk: `git push` takes it only inside an explicit URL (`PUSH_REMOTE_URL` overrides it, so a dry run can From 7bb90ff15513d3ad6a1c9cc4c3a89fdb4216352d Mon Sep 17 00:00:00 2001 From: Julien Goux Date: Fri, 2 Oct 2026 19:33:59 +0200 Subject: [PATCH 3/3] ci(stack): keep every pending slim pickup run and narrow the queue match --- .github/workflows/slim-release-published.yml | 34 +++++++------------- docs/adr/0026-slim-artifact-mirrors.md | 15 ++++----- 2 files changed, 19 insertions(+), 30 deletions(-) diff --git a/.github/workflows/slim-release-published.yml b/.github/workflows/slim-release-published.yml index 83da29b2e7..449297676d 100644 --- a/.github/workflows/slim-release-published.yml +++ b/.github/workflows/slim-release-published.yml @@ -1,5 +1,4 @@ name: Slim Release Published -run-name: slim-release-published ${{ github.event.client_payload.service }} # Reconciles the stack catalog against supabase/slim-services releases. See # docs/adr/0026-slim-artifact-mirrors.md, "Hotfix and upgrade pickup". @@ -15,6 +14,8 @@ permissions: concurrency: group: slim-release-published-${{ github.event.client_payload.service }} cancel-in-progress: false + # The default `single` cancels a pending run when another is queued, losing its release-visibility wait. + queue: max jobs: pickup: @@ -23,7 +24,6 @@ jobs: # for the first planned item) + pin, render and push per item + the merge-queue wait (up to 30 min). timeout-minutes: 75 permissions: - actions: read contents: read steps: - name: Checkout the default branch @@ -105,10 +105,7 @@ jobs: # One branch per planned item, rebuilt from the recorded base commit, never from a previous # iteration's HEAD; `--release` pins the planned release, never "highest at apply time". # `bun`/`pnpm` run under `env -u APP_TOKEN`: only `git push` and `gh` ever see the app token. - # A branch whose pull request is in the merge queue rejects pushes; the step waits for the queue - # to release it, then stops; a fresh run re-plans every remaining item from the new default branch. - # That is the service's pending run when one exists, since a replay would cancel it, otherwise a - # re-dispatch of this event. + # A merge-queued branch rejects pushes; see ADR 0026 "Hotfix and upgrade pickup" for the wait and hand-off. - name: Apply planned updates if: steps.plan.outputs.count != '0' env: @@ -165,44 +162,37 @@ jobs: if ! push_output="$(git push --force "$push_url" "HEAD:refs/heads/${branch}" 2>&1)"; then echo "$push_output" - if ! grep -q "merge queue" <<<"$push_output"; then + if ! grep -q "queued for merging cannot be updated" <<<"$push_output"; then echo "::error ::Failed to push ${branch}. Run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." exit 1 fi - manual_steps="Once ${branch} leaves the merge queue, run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." - # The app token minted just before this step expires after one hour. + manual_steps="run manually: ${manual_prefix} ${release}, regenerate the Dockerfiles with \`bun apps/cli/scripts/render-service-dockerfile.ts\`, then open a pull request by hand." + queued_steps="Once ${branch} leaves the merge queue, ${manual_steps}" + # The app token minted just before this step expires after one hour; capping the wait at + # 50 minutes into the step leaves room for the lookups and the dispatch that follow it. deadline=$((SECONDS + 1800 < 3000 ? SECONDS + 1800 : 3000)) while :; do if ! queued="$(in_merge_queue "$branch")"; then - echo "::error ::Failed to read the merge-queue state of ${branch}. ${manual_steps}" + echo "::error ::Failed to read the merge-queue state of ${branch}. ${queued_steps}" exit 1 fi case "$queued" in false) break ;; true) ;; *) - echo "::error ::Unexpected merge-queue state '${queued}' for ${branch}. ${manual_steps}" + echo "::error ::Unexpected merge-queue state '${queued}' for ${branch}. ${queued_steps}" exit 1 ;; esac if [ "$SECONDS" -ge "$deadline" ]; then - echo "::error ::${branch} is still in the merge queue. ${manual_steps}" + echo "::error ::${branch} is still in the merge queue. ${queued_steps}" exit 1 fi sleep 30 done - if ! pending="$(GH_TOKEN="$GITHUB_TOKEN" gh run list --repo "$GITHUB_REPOSITORY" --workflow slim-release-published.yml --status pending --json displayTitle | - jq --arg title "slim-release-published ${SERVICE}" '[.[] | select(.displayTitle == $title)] | length')"; then - echo "::error ::Failed to list pending slim-release-published runs. ${manual_steps}" - exit 1 - fi - if [ "$pending" -gt 0 ]; then - echo "::warning ::${branch} was held by a merge-queued pull request; a pending ${SERVICE} run re-plans from ${DEFAULT_BRANCH}." - exit 0 - fi # `replay` counts consecutive re-sends; anything but 0-2 stops the chain. if ! [[ "${REPLAY:-0}" =~ ^[0-2]$ ]]; then - echo "::error ::Stopped re-dispatching after repeated merge-queue rejections of ${branch}. ${manual_steps}" + echo "::error ::Stopped re-dispatching after repeated merge-queue rejections of ${branch}; ${manual_steps}" exit 1 fi if ! GH_TOKEN="$APP_TOKEN" gh api "repos/${GITHUB_REPOSITORY}/dispatches" \ diff --git a/docs/adr/0026-slim-artifact-mirrors.md b/docs/adr/0026-slim-artifact-mirrors.md index 0b74f9d99e..d569ba7eb8 100644 --- a/docs/adr/0026-slim-artifact-mirrors.md +++ b/docs/adr/0026-slim-artifact-mirrors.md @@ -151,14 +151,13 @@ Plan and apply run from the same checkout of the default branch in one job, so a recomputes from the latest develop: a stale plan can never be applied, and a superseded PR's branch is rewritten (force-pushed) in place rather than raced by a new one — a superseded PR is never auto-closed. A backlog republish of an older upstream version naturally plans nothing. A branch -whose PR is in the merge queue rejects the force-push; the run waits up to 30 minutes for the queue -to merge or drop that PR, then stops so a fresh run re-plans every remaining update from the updated -default branch: the service's pending run when one exists (a replay would cancel it and its own -release-visibility wait), otherwise a re-send of the same dispatch, at most three times in a row. -The lookup and the re-send are not atomic: a newer dispatch arriving between them is replaced by the -replay, which plans it only if its release is already listed by then. A queue that holds the branch -longer, or a failed queue lookup or dispatch, fails the run with the manual invocation below. The -fallback, if the push or PR step fails, is a documented manual `bun +whose PR is in the merge queue rejects the force-push; the run waits up to 30 minutes (less when +earlier work leaves the app token too little lifetime) for the queue to merge or drop that PR, then +re-sends the same dispatch and stops, so a fresh run re-plans every remaining update from the +updated default branch. At most three re-sends run in a row. The concurrency group keeps every +pending run (`queue: max`), so a replay never cancels a newer dispatch and its release-visibility +wait. A queue that holds the branch longer, or a failed queue lookup or dispatch, fails the run with +the manual invocation below. The fallback, if the push or PR step fails, is a documented manual `bun .github/scripts/sync-artifacts-catalog.ts --service --release -r` invocation, followed by `apps/cli/scripts/render-service-dockerfile.ts` — the release itself is already committed by then, so a failure here means "open the pull request by hand", not "republish".