diff --git a/.github/ci-image/Dockerfile b/.github/ci-image/Dockerfile deleted file mode 100644 index 12443a1c..00000000 --- a/.github/ci-image/Dockerfile +++ /dev/null @@ -1,161 +0,0 @@ -# CI image for claude-code-native — Fedora 44. -# -# WHY THIS EXISTS -# The expensive part of this pipeline is not compute, it is downloads: a cold Gradle resolves and EXTRACTS -# the whole IntelliJ Platform before it compiles a line, and a GitHub runner starts cold every time. Baking -# that into an image turns minutes of download into a pull. -# -# WHAT IS DELIBERATELY *NOT* IN HERE: THE VERIFIER'S IDEs -# This image used to also bake what `verifyPlugin` downloads, and that made it 38.1 GB — 29.1 GB of it -# extracted IDEs. Every job in ci.yml pulls its own copy on its own runner, so a job whose actual work is an -# 8-second vitest run spent 5m37s in `Initialize containers` (measured, not estimated), and 38 GB on a -# runner with ~25-30 GB free on the root volume was flirting with `No space left on device`. -# -# The verifier is the only consumer of those IDEs, and it runs ONLY on a pull request from develop into -# main — a handful of times a month. Baking 29 GB into every job's pull, permanently, to save ten minutes on -# the rarest job in the pipeline is the wrong side of that trade by two orders of magnitude. `verifyPlugin` -# downloads what it needs, when it runs. -# -# There is a second reason, and it is the one that would have bitten silently: the IDE set MOVES. The -# verifier resolves from the EAP/RC channels, so the day JetBrains publishes a new build, the baked copies -# stop matching and Gradle downloads the new one anyway. The saving decayed on JetBrains' release schedule, -# not ours. -# -# WHERE TO PUBLISH IT -# ghcr.io, NOT Docker Hub. It sits on the same network as the runners (much faster pulls) and has no -# anonymous pull-rate limit — that limit is a classic cause of a pipeline failing for reasons nobody -# changed. -# -# BUILDING IT — from the repository ROOT, so /.dockerignore applies: -# -# docker build -f .github/ci-image/Dockerfile -t ghcr.io/OWNER/cc-ci:base . -# docker push ghcr.io/OWNER/cc-ci:base -# -# Tagged `:base`, not `:latest`, because this repository's own standard is to pin rather than float, and -# because a floating tag makes "which image was that job green on?" unanswerable. -# -# Used from a workflow as: -# jobs: -# test: -# runs-on: ubuntu-latest -# container: ghcr.io/OWNER/cc-ci:base -FROM fedora:44 AS base - -# Parallel downloads: dnf defaults to 3, and this image installs a JDK plus a Node toolchain over a link -# that is not the bottleneck. Set before the first transaction so every one of them benefits. -RUN echo "max_parallel_downloads=20" >> /etc/dnf/dnf.conf \ - && echo "fastestmirror=True" >> /etc/dnf/dnf.conf - -# Temurin, not Fedora's OpenJDK. -# -# Fedora 44 no longer packages java-21-openjdk — it has moved on to a newer LTS — and the JDK version is not -# ours to float: build.gradle.kts pins the toolchain to 21 because the IDE runs on JBR 21, which is the -# ceiling. Building on 25 would produce class files no target IDE can load. -# -# Adoptium's repository is the same source the `setup-java` action uses on the GitHub runners, so the image -# and the hosted pipeline compile against the same JDK rather than two different builds of "21". -# ONE transaction, not two. The previous first transaction existed only to install `dnf-plugins-core`, which -# was never used: the repository file below is written with `printf`, not with `dnf config-manager`, and -# `curl` is already in the fedora:44 base image (verified: curl-8.18.0). It was ~150 MB of Python stack -# pulled in to run a command nobody ran. -# -# `python3` is now EXPLICIT, and that is a correctness fix rather than a size one. `bin/fake-claude` — the -# deterministic stand-in the integration tests drive a real ClaudeSession against — is a `#!/usr/bin/env -# python3` script. It worked only because `dnf-plugins-core` happened to drag the interpreter in as a -# transitive dependency. Removing the unused package without naming python3 here would have made the -# integration suite fail on a missing interpreter, which is the kind of break that reads as a test bug. -# -# `git-core` rather than `git`: actions/checkout clones, fetches and checks out, and git-core provides -# /usr/bin/git for all of that. The `git` metapackage adds the Perl tooling (git-send-email and friends), -# git-core-doc and perl-libs — ~32 MB nothing in this pipeline invokes. -# -# `which`/`findutils`/`procps-ng` are assumed present by various actions and by Gradle's own probing, and -# Fedora's base image is minimal enough not to ship them. -# -# `install_weak_deps=False` drops recommended-but-unused packages; `tsflags=nodocs` drops the documentation -# that ships inside the ones we do want. -RUN curl -fsSL https://packages.adoptium.net/artifactory/api/gpg/key/public \ - -o /etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium \ - && rpm --import /etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium \ - && printf '%s\n' \ - '[Adoptium]' \ - 'name=Adoptium' \ - 'baseurl=https://packages.adoptium.net/artifactory/rpm/fedora/$releasever/$basearch' \ - 'enabled=1' \ - 'gpgcheck=1' \ - 'gpgkey=file:///etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium' \ - > /etc/yum.repos.d/adoptium.repo \ - && dnf -y --setopt=install_weak_deps=False --setopt=tsflags=nodocs install \ - temurin-21-jdk \ - nodejs npm \ - python3 \ - git-core unzip zip tar which findutils procps-ng ca-certificates \ - && dnf clean all \ - && rm -rf /var/cache/dnf \ - # gettext catalogues: translated CLI messages for tools this image only ever runs non-interactively - # under a C locale. NOT /usr/lib/locale, which is the locale DEFINITIONS the JVM and glibc resolve - # against — deleting that would change how the build behaves, not just how it reads. - && rm -rf /usr/share/locale - -# JAVA_HOME is resolved rather than hardcoded: the exact path carries the package's build number and would -# silently break on the next base-image bump. -RUN JH="$(dirname "$(dirname "$(readlink -f "$(command -v javac)")")")" \ - && echo "JAVA_HOME=$JH" >> /etc/environment \ - && ln -sfn "$JH" /opt/java-21 \ - && "$JH/bin/java" -version -# A stable symlink, so JAVA_HOME does not carry Temurin's build number and break on the next image rebuild. -ENV JAVA_HOME=/opt/java-21 -ENV PATH="${JAVA_HOME}/bin:${PATH}" - -# Gradle writes here, and the path must match what the job will use, or the warm caches below are invisible -# to it. Set GRADLE_USER_HOME to the same value in the workflow. -ENV GRADLE_USER_HOME=/opt/gradle-home - -WORKDIR /warmup - -# Only the build definition first, on purpose: this layer is invalidated by a dependency change, not by -# every edit to the Kotlin sources. -COPY gradle/ gradle/ -COPY gradlew settings.gradle.kts build.gradle.kts gradle.properties* ./ -COPY package.json package-lock.json ./ - -# Downloads the Gradle distribution itself. Kept separate from the warm-up below so a network problem here -# is distinguishable from a build problem there. -RUN ./gradlew --no-daemon --version - -# The sources, needed because the warm-up below compiles. Filtered by /.dockerignore, so this is a few MB -# of Kotlin and resources rather than the 2 GB it used to be with node_modules and build/ swept in. -COPY . . - -# THE WARM-UP, and the reason it is `testClasses` rather than `dependencies`. -# -# It used to be `./gradlew dependencies --configuration compileClasspath > /dev/null 2>&1 || true`, and that -# command does NOT warm this cache. It resolves dependency METADATA; it never triggers the artifact -# transform that EXTRACTS the IntelliJ Platform, which is where the several GB actually are. Measured: that -# command leaves caches/*/transforms at 179 MB with no extracted IDE in it. The image looked warm and every -# job re-downloaded and re-extracted the platform — invisibly, because of the redirect and the `|| true`. -# -# `testClasses` compiles main and test sources, so it resolves AND extracts everything `test`, `detekt`, -# `spotlessCheck`, `buildPlugin` and the CodeQL Kotlin build need. -# -# No `> /dev/null`, and no `|| true`. A warm-up that fails must fail the image build. The old form could not -# report anything: the whole point of this image is the cache, so "the cache step failed but the image is -# fine" is not a state worth being able to reach. -RUN ./gradlew --no-daemon testClasses - -# npm dependencies for the frontend tests. `npm ci` needs package-lock.json, which is why it is copied above. -# -# What is baked is the npm CACHE, not `node_modules`, and the distinction is the whole point: the cleanup -# step below wipes /warmup, so a baked node_modules would be deleted moments after being built — the warm-up -# would look like it worked and buy nothing. `node_modules` also MUST match the package-lock.json of whatever -# commit CI checks out, not the one that happened to be current when the image was cut, so keeping it would -# be wrong even if it survived. The cache is version-addressed and therefore safe to reuse: `npm ci` in CI -# rebuilds node_modules from it without touching the network. -ENV npm_config_cache=/opt/npm-cache -# `node_modules` is removed in the SAME layer that creates it. It is scaffolding — the cache above is what -# survives — and a `rm` in a later layer would not reclaim the space, only hide it. -RUN npm ci --no-audit --no-fund \ - && rm -rf /warmup/node_modules - -RUN rm -rf /warmup/* /warmup/.[!.]* 2>/dev/null || true -WORKDIR /workspace diff --git a/.github/ci-image/jvm-test.Dockerfile b/.github/ci-image/jvm-test.Dockerfile new file mode 100644 index 00000000..371d561f --- /dev/null +++ b/.github/ci-image/jvm-test.Dockerfile @@ -0,0 +1,141 @@ +# jvm-test — the CI image for every job that runs Gradle. Built ON TOP of node-test. +# +# WHO USES IT +# `JVM tests`, `Static analysis` and `Plugin verifier` in ci.yml, `CodeQL (java-kotlin)`, the weekly drift +# check, and the release gate in release.yml. +# +# WHY IT IS BUILT FROM node-test RATHER THAN FROM fedora +# Two reasons, and the second is the one that matters. +# +# 1. No duplication. The base package list and the npm cache warm-up are written once, in +# node-test.Dockerfile. Two standalone files would have to keep them in step by hand, and the failure +# mode of that is not a build error — it is two images that quietly disagree about the Node version. +# 2. It needs Node anyway. Three of the jobs above run Gradle AND npm in a single job: `Static analysis` +# (detekt and spotless, then eslint and prettier), `drift` (`npm install` then `checkDrift`), and the +# release gate (`npm test` then `test verifyPlugin`). Splitting Node out would mean splitting those jobs +# in two, and a new job is a whole extra image pull — the exact cost this segmentation exists to remove. +# +# The registry stores the shared layers ONCE, so this is not a second copy of the small image. A job that +# needs neither image runs on a bare runner and is not served from here at all — `Build plugin` is the +# example: it downloads an artifact and runs `unzip`, and used to pull GB to do it. +# +# WHAT IS DELIBERATELY NOT IN HERE: THE VERIFIER'S IDEs +# Baking what `verifyPlugin` downloads once made the single image 38.1 GB, 29.1 GB of it extracted IDEs. +# The verifier is their only consumer and runs ONLY on a pull request from develop into main — a handful of +# times a month. Paying 29 GB on every job's pull, permanently, to save ten minutes on the rarest job is the +# wrong side of that trade by two orders of magnitude. There is a second reason that would have bitten +# silently: the verifier resolves IDEs from the EAP/RC channels, so the set MOVES, and the day JetBrains +# publishes a new build the baked copies stop matching and Gradle downloads the new one anyway. +# +# BUILDING — node-test FIRST, since this image starts from it. From the repository ROOT, so /.dockerignore +# applies: +# +# V=v1.0.0 +# docker build -f .github/ci-image/node-test.Dockerfile -t ghcr.io/OWNER/node-test:$V . +# docker build -f .github/ci-image/jvm-test.Dockerfile -t ghcr.io/OWNER/jvm-test:$V \ +# --build-arg NODE_IMAGE=ghcr.io/OWNER/node-test:$V . +# docker push ghcr.io/OWNER/node-test:$V +# docker push ghcr.io/OWNER/jvm-test:$V +# +# The tag is `vMAJOR.MINOR.PATCH`, never `latest`. Bumping it is a commit: change the tag here and in every +# workflow that references it, so the two move together in one reviewable diff. Both images share a version +# because this one is derived from that one — they are not independently versionable. + +# Declared before FROM so it can be used there. The default names this repository's own package; a fork +# overrides it with --build-arg rather than editing the file. +ARG NODE_IMAGE=ghcr.io/serialexperimentslainnnn/node-test:v1.0.0 +FROM ${NODE_IMAGE} + +# NB there is no dnf tuning here and that is not an omission: `max_parallel_downloads=20` and +# `fastestmirror=True` were written into /etc/dnf/dnf.conf by node-test, and this image starts from its +# filesystem — so the JDK transaction below already runs with them. Adding the lines again would append a +# SECOND copy of each key to dnf.conf rather than overriding anything. +# +# Temurin, not Fedora's OpenJDK. +# +# Fedora 44 no longer packages java-21-openjdk — it has moved on to a newer LTS — and the JDK version is not +# ours to float: build.gradle.kts pins the toolchain to 21 because the IDE runs on JBR 21, which is the +# ceiling. Building on 25 would produce class files no target IDE can load. Adoptium's repository is the +# same source the `setup-java` action uses on the hosted runners, so the image and the pipeline compile +# against the same JDK rather than two different builds of "21". +# +# There is deliberately no `dnf-plugins-core`: nothing here calls `dnf config-manager` — the repo file is +# written with `printf` — and `curl` is already in the base image, so installing it dragged in a ~150 MB +# Python stack to run a command nobody ran. +# +# `python3` is EXPLICIT, and that is a correctness requirement rather than a convenience: `bin/fake-claude`, +# the deterministic stand-in the integration tests drive a real ClaudeSession against, is a +# `#!/usr/bin/env python3` script. It used to arrive only as a transitive dependency of that unused package +# — a load-bearing dependency held up by an accident. +# +# `zip` is added here rather than in node-test because only the Gradle side packages archives. +RUN curl -fsSL https://packages.adoptium.net/artifactory/api/gpg/key/public \ + -o /etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium \ + && rpm --import /etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium \ + && printf '%s\n' \ + '[Adoptium]' \ + 'name=Adoptium' \ + 'baseurl=https://packages.adoptium.net/artifactory/rpm/fedora/$releasever/$basearch' \ + 'enabled=1' \ + 'gpgcheck=1' \ + 'gpgkey=file:///etc/pki/rpm-gpg/RPM-GPG-KEY-Adoptium' \ + > /etc/yum.repos.d/adoptium.repo \ + && dnf -y --setopt=install_weak_deps=False --setopt=tsflags=nodocs install \ + temurin-21-jdk \ + python3 \ + zip \ + && dnf clean all \ + && rm -rf /var/cache/dnf \ + && rm -rf /usr/share/locale + +# JAVA_HOME is resolved rather than hardcoded: the exact path carries the package's build number and would +# silently break on the next base-image bump. The symlink keeps the ENV below stable across rebuilds. +RUN JH="$(dirname "$(dirname "$(readlink -f "$(command -v javac)")")")" \ + && echo "JAVA_HOME=$JH" >> /etc/environment \ + && ln -sfn "$JH" /opt/java-21 \ + && "$JH/bin/java" -version +ENV JAVA_HOME=/opt/java-21 +ENV PATH="${JAVA_HOME}/bin:${PATH}" + +# Gradle writes here, and the path MUST match GRADLE_USER_HOME in the workflow. If they diverge, the warm +# cache below is invisible and every run silently re-resolves what this image already has. +ENV GRADLE_USER_HOME=/opt/gradle-home + +# The npm cache is inherited from node-test — `npm_config_cache=/opt/npm-cache` and the warmed store are +# already in the layers below this one, so `Static analysis`, `drift` and the release gate get it for free. +WORKDIR /warmup + +# The build definition first, on purpose: this layer is invalidated by a dependency change, not by every +# edit to the Kotlin sources. +COPY gradle/ gradle/ +COPY gradlew settings.gradle.kts build.gradle.kts gradle.properties* ./ + +# Downloads the Gradle distribution itself. Kept separate from the warm-up below so a network problem here +# is distinguishable from a build problem there. +RUN ./gradlew --no-daemon --version + +# The sources, needed because the warm-up below compiles. Filtered by /.dockerignore, so this is ~3 MB of +# Kotlin and resources rather than the 2 GB it was with node_modules, build/ and .git swept in. +COPY . . + +# THE WARM-UP, and the reason it is `testClasses` rather than `dependencies`. +# +# It used to be `./gradlew dependencies --configuration compileClasspath > /dev/null 2>&1 || true`, and that +# command does NOT warm this cache. It resolves dependency METADATA; it never triggers the artifact +# transform that EXTRACTS the IntelliJ Platform, which is where the several GB actually are. Measured: that +# command leaves caches/*/transforms at 179 MB with no extracted IDE in it. The image looked warm and every +# job re-downloaded and re-extracted the platform — invisibly, because of the redirect and the `|| true`. +# +# `testClasses` compiles main and test sources, so it resolves AND extracts everything the Gradle jobs need. +# +# No `> /dev/null`, and no `|| true`. A warm-up that fails must fail the image build: the whole point of +# this image is the cache, so "the cache step failed but the image is fine" is not a state worth being able +# to reach. Verified with the network disabled — `testClasses` compiles offline in 33s from this cache. +RUN ./gradlew --no-daemon testClasses + +# The sources were only ever scaffolding; keeping them would ship a stale copy of the repository inside the +# image, which someone would eventually mistake for the real one. This does not reclaim the space (layers +# are additive) — it prevents the confusion. +RUN rm -rf /warmup/* /warmup/.[!.]* 2>/dev/null || true + +WORKDIR /workspace diff --git a/.github/ci-image/node-test.Dockerfile b/.github/ci-image/node-test.Dockerfile new file mode 100644 index 00000000..2e42aa54 --- /dev/null +++ b/.github/ci-image/node-test.Dockerfile @@ -0,0 +1,79 @@ +# node-test — the small CI image: Node, npm, and the warm npm cache. Nothing else. +# +# WHO USES IT +# `Frontend tests` and `Dependency audit` in ci.yml. Both jobs are `npm ci` followed by one npm command, +# and both used to run on the full image: an 8-second vitest run spent 1m05s pulling a JDK, a Gradle +# distribution and 3.4 GB of extracted IntelliJ Platform it never opened. +# +# WHY IT IS A SEPARATE FILE RATHER THAN A STAGE +# Deliberate: each image is built and published on its own, so neither can grow because the other needed +# something. The cost is that the package list below is duplicated in jvm-test.Dockerfile — that duplication +# is the trade, and it is the thing to check when either file changes. +# +# WHY THE PULL CANNOT SIMPLY BE CACHED — the question this split exists to answer. +# A `container:` job pulls its image in `Initialize containers`, which runs BEFORE the first step of the +# job. There is no point at which an `actions/cache` step could run first, and every job starts on a fresh +# runner with no shared layer cache. So the image download is not cacheable at all; the only lever is how +# much each job has to download. Hence this file. +# +# BUILDING — from the repository ROOT, so /.dockerignore applies: +# +# docker build -f .github/ci-image/node-test.Dockerfile \ +# -t ghcr.io/OWNER/node-test:v1.0.0 . +# docker push ghcr.io/OWNER/node-test:v1.0.0 +# +# The tag is `vMAJOR.MINOR.PATCH`, never `latest`: a floating tag makes "which image was that job green on?" +# unanswerable, and this repository's standard is to pin. Bumping it is a commit — change the tag here and +# in every workflow that references it, so the two move together in one reviewable diff. +FROM fedora:44 + +# Parallel downloads: dnf defaults to 3, and the link is not the bottleneck. +RUN echo "max_parallel_downloads=20" >> /etc/dnf/dnf.conf \ + && echo "fastestmirror=True" >> /etc/dnf/dnf.conf + +# `git-core` rather than `git`: actions/checkout clones, fetches and checks out, and git-core provides +# /usr/bin/git for all of that. The `git` metapackage adds Perl tooling, git-core-doc and perl-libs — +# ~32 MB nothing in this pipeline invokes. +# +# `which`/`findutils`/`procps-ng` are assumed present by various actions; Fedora's base image is minimal +# enough not to ship them. `install_weak_deps=False` drops recommended-but-unused packages, `tsflags=nodocs` +# drops the documentation inside the ones we do want. +# `upgrade --refresh` before the install, in the SAME layer: the `fedora:44` tag is a moving snapshot that +# can be weeks behind, and a CI image is exactly where you do not want to be running last month's openssl. +# Refreshing first also means the install below resolves against current metadata rather than whatever was +# cached in the base layer. +# +# The cost, stated rather than discovered: this makes the build non-reproducible — the same Dockerfile +# yields different bytes on different days. That is acceptable HERE and only here, because the image is +# pinned by an explicit `vX.Y.Z` tag that CI references. What CI runs is frozen; what a rebuild produces is +# not, and bumping the tag is the deliberate act that moves it. +RUN dnf -y upgrade --refresh \ + && dnf -y --setopt=install_weak_deps=False --setopt=tsflags=nodocs install \ + nodejs npm \ + git-core unzip tar which findutils procps-ng ca-certificates \ + && dnf clean all \ + && rm -rf /var/cache/dnf \ + # gettext catalogues: translated CLI messages for tools this image only runs non-interactively under a + # C locale. NOT /usr/lib/locale, which is the locale DEFINITIONS glibc resolves against. + && rm -rf /usr/share/locale + +WORKDIR /warmup + +COPY package.json package-lock.json ./ + +# What is baked is the npm CACHE, not `node_modules`, and the distinction is the whole point: `node_modules` +# MUST match the package-lock.json of whatever commit CI checks out, not the one current when the image was +# cut. The cache is version-addressed and therefore safe to reuse — `npm ci` in CI rebuilds node_modules +# from it without touching the network. +# +# `node_modules` is removed in the SAME layer that creates it: a `rm` in a later layer would not reclaim the +# space, only hide it. Measured at 478 MB. +ENV npm_config_cache=/opt/npm-cache +RUN npm ci --no-audit --no-fund \ + && rm -rf /warmup/node_modules + +# The lockfile was scaffolding for the cache; keeping it would ship a stale copy inside the image that +# someone would eventually mistake for the real one. +RUN rm -rf /warmup/* /warmup/.[!.]* 2>/dev/null || true + +WORKDIR /workspace diff --git a/.github/rulesets/develop.json b/.github/rulesets/develop.json index c808a0df..503fb892 100644 --- a/.github/rulesets/develop.json +++ b/.github/rulesets/develop.json @@ -45,12 +45,27 @@ "parameters": { "strict_required_status_checks_policy": false, "do_not_enforce_on_create": false, + "_comment": [ + "CodeQL gates develop as well as main. It is affordable here in a way the other main-only", + "checks are not: codeql.yml already triggers on every pull request into develop, so the run", + "happens either way — requiring it only decides whether anyone has to look at the result.", + "A SAST finding is also the class of defect worth catching before it is merged rather than", + "at the release door, because by then it is mixed in with everything else on the branch.", + "The contexts are the jobs' DISPLAY names. Renaming a job in codeql.yml does not fail this", + "gate — it silently stops applying it." + ], "required_status_checks": [ { "context": "JVM tests" }, { "context": "Frontend tests" + }, + { + "context": "CodeQL (java-kotlin)" + }, + { + "context": "CodeQL (javascript-typescript)" } ] } diff --git a/.github/rulesets/main.json b/.github/rulesets/main.json index 0e7f67c0..56958592 100644 --- a/.github/rulesets/main.json +++ b/.github/rulesets/main.json @@ -78,6 +78,9 @@ }, { "context": "Build plugin" + }, + { + "context": "No bot PRs pending on develop" } ] } diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8dd2dd26..dcc3c2b7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -69,7 +69,7 @@ jobs: # the trade is explicit: refreshing what CI has cached now means rebuilding and pushing the image, # which is a deliberate act rather than something that drifts between runs. container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 # The package stays PRIVATE and is pulled with the run's own GITHUB_TOKEN — no new secret, nothing to # rotate, and access dies with the job. This requires the package to have been granted Read access to # THIS repository (package settings -> Manage Actions access): `packages: read` widens what the token @@ -123,7 +123,7 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 20 container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 # The package stays PRIVATE and is pulled with the run's own GITHUB_TOKEN — no new secret, nothing to # rotate, and access dies with the job. `packages: read` is granted per job below; without it the pull # fails with a 401 that reads like a wrong image name rather than a permission problem. @@ -195,7 +195,10 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 10 container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + # `node-test`, not `jvm-test`: this job is `npm ci` and then vitest. On the single combined image it + # pulled a JDK, a Gradle distribution and 3.4 GB of extracted IntelliJ Platform it never opened — + # 1m05s of container init for 8 seconds of work. + image: ghcr.io/serialexperimentslainnnn/node-test:v1.0.0 # The package stays PRIVATE and is pulled with the run's own GITHUB_TOKEN — no new secret, nothing to # rotate, and access dies with the job. `packages: read` is granted per job below; without it the pull # fails with a 401 that reads like a wrong image name rather than a permission problem. @@ -205,10 +208,7 @@ jobs: permissions: contents: read packages: read - env: - # MUST match GRADLE_USER_HOME in .github/ci-image/Dockerfile. If these diverge, the warmed caches - # baked into the image are invisible and every run silently re-downloads what the image already has. - GRADLE_USER_HOME: /opt/gradle-home + # No GRADLE_USER_HOME here: there is no Gradle in this image and nothing in this job invokes it. steps: - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: @@ -241,7 +241,8 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 10 container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + # `node-test`: this job is `npm ci` and two `npm audit` invocations. Nothing here touches the JVM. + image: ghcr.io/serialexperimentslainnnn/node-test:v1.0.0 # The package stays PRIVATE and is pulled with the run's own GITHUB_TOKEN — no new secret, nothing to # rotate, and access dies with the job. `packages: read` is granted per job below; without it the pull # fails with a 401 that reads like a wrong image name rather than a permission problem. @@ -251,10 +252,6 @@ jobs: permissions: contents: read packages: read - env: - # MUST match GRADLE_USER_HOME in .github/ci-image/Dockerfile. If these diverge, the warmed caches - # baked into the image are invisible and every run silently re-downloads what the image already has. - GRADLE_USER_HOME: /opt/gradle-home # Same door. NB this is the check that judges exactly what a Dependabot pull request changes, so it no # longer runs on the PR that proposes the bump — only once that bump is on develop, and again before it # can reach main. Nothing ships un-audited; the finding simply arrives one merge later. @@ -276,6 +273,63 @@ jobs: - name: Audit the full tree (informational) run: npm audit || true + # Release readiness: refuse to promote develop -> main while a bot still has work in flight. + # + # A release is a claim that `develop` is a finished state. An open pull request from Claude or from + # Dependabot is the opposite of that claim: it is a change someone intended to be in this release, + # sitting one click away from being in it. Merging past it does not lose the work, it does something + # worse — it ships a version whose CHANGELOG was written as if that work had landed. For Dependabot + # specifically it also means shipping with a known dependency update sitting unmerged, which is the + # one class of pending change a security advisory can be written about. + # + # This is a status check and NOT a ruleset entry because it cannot be one: a GitHub ruleset can require + # a check, a signature or an approval, and has no vocabulary for "no other pull request exists". The + # gate is therefore this job, and `.github/rulesets/main.json` requires it by DISPLAY name. + # + # THE COST, stated rather than discovered: Dependabot's resting state is "has something open" — this + # repository's history shows long runs of them. So this gate will block releases until that queue is + # drained, and draining it becomes a release step. That is the intended trade (nothing ships alongside + # an un-merged dependency bump), but it is the kind of gate that gets bypassed if the queue is ignored + # for weeks. If it starts being routinely in the way, the fix is to merge Dependabot more often — not + # to widen the filter. + bot-work-in-flight: + name: No bot PRs pending on develop + runs-on: ubuntu-latest + timeout-minutes: 5 + # Not in the CI image on purpose: this job needs `gh`, which the image does not install, and needs + # nothing the image does provide. A bare runner ships `gh` and starts instantly. + permissions: + contents: read + pull-requests: read + # Only at the release door. On a pull request into develop this would be self-referential. + if: github.event_name == 'pull_request' && github.base_ref == 'main' + steps: + - name: Fail if Claude or Dependabot has open pull requests into develop + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euo pipefail + # The login is matched by PATTERN because the exact one depends on how each integration is + # installed: the REST API renders an app author as `app/` (`app/dependabot` is what this + # repository's history shows), while a bot user appears as `[bot]`. Anchored rather than + # a bare substring, so a human whose username merely contains "claude" or "dependabot" is not + # caught by a release gate. + pending=$(gh pr list --repo "$GITHUB_REPOSITORY" --base develop --state open \ + --json number,title,url,author \ + --jq '[.[] | select(.author.login + | ascii_downcase + | test("^app/(claude|dependabot)$|^(claude|dependabot)(\\[bot\\])?$"))]') + + count=$(printf '%s' "$pending" | jq 'length') + if [ "$count" -eq 0 ]; then + echo "no bot pull requests open against develop — clear to promote." + exit 0 + fi + + echo "::error::$count bot pull request(s) still open against develop. Merge or close them before releasing." + printf '%s' "$pending" | jq -r '.[] | " #\(.number) \(.author.login) \(.title)\n \(.url)"' + exit 1 + # The IntelliJ Plugin Verifier: the ONLY thing that catches a *binary* incompatibility across the # declared 251 → 263.* range. Compiling against 252 proves nothing about 262 — that asymmetry is # exactly how the 4.4.1 /login regression shipped. It downloads several full IDEs, hence the timeout. @@ -287,7 +341,7 @@ jobs: # Same image as every other job, and it does NOT carry the IDEs this job downloads — see the note at # the top of .github/ci-image/Dockerfile. Baking them made the image 38.1 GB, which every job paid for # on its own runner, to save ten minutes on the one job that runs least often. - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 # The package stays PRIVATE and is pulled with the run's own GITHUB_TOKEN — no new secret, nothing to # rotate, and access dies with the job. `packages: read` is granted per job below; without it the pull # fails with a 401 that reads like a wrong image name rather than a permission problem. @@ -362,14 +416,11 @@ jobs: name: Build plugin runs-on: ubuntu-latest timeout-minutes: 10 - container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base - credentials: - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} + # NO container, deliberately. This job downloads an artifact and runs `unzip`, `grep` and `ls` over it — + # it does not build anything despite the name, and it used to pull GB of JDK, Gradle and extracted + # IntelliJ Platform to do it. `unzip` is on the bare runner, and the job now starts instantly. permissions: contents: read - packages: read needs: [verify] steps: - name: Fetch the verified distributable diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 957289f5..955b7885 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -28,18 +28,19 @@ concurrency: # around a `credentials:` block. # # java-kotlin needs a JDK and a full Gradle resolution of the IntelliJ Platform, so it runs in -# cc-ci:base and inherits the warm GRADLE_USER_HOME. It used to provision the JDK with +# jvm-test and inherits the warm GRADLE_USER_HOME. It used to provision the JDK with # setup-java and resolve the platform from cold on every run. # javascript-typescript is `build-mode: none`. It needs no JDK, no Gradle and no npm install — the -# scanner reads the sources. Putting it in the image would add a 6.5 GB pull to a job that -# would use none of it, making it strictly slower. It stays on the bare runner. +# scanner reads the sources. Putting it in an image would add a pull to a job that would +# use none of it, making it strictly slower. It stays on the bare runner. jobs: analyze-kotlin: name: CodeQL (java-kotlin) runs-on: ubuntu-latest timeout-minutes: 45 container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + # `jvm-test`: the manual build is `./gradlew classes`, which resolves the whole IntelliJ Platform. + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 credentials: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} @@ -69,6 +70,90 @@ jobs: # users, and the extra precision cost is a few minutes on a free runner. queries: security-extended + # Give CodeQL's build tracer the filename Fedora's loader asks for. + # + # `Initialize CodeQL` exports + # LD_PRELOAD=/tools/linux64/${LIB}_${PLATFORM}_trace.so + # where `$LIB` and `$PLATFORM` are glibc dynamic string tokens that ld.so expands at load time — one + # variable covering several ABIs. Measured locally: Fedora 44's loader expands them to + # `lib64_x86_64`, and Ubuntu 24.04's does not resolve that form at all. CodeQL is built and tested on + # Ubuntu runners, so the shipped filename matches Ubuntu's expansion and Fedora asks for a name that + # is not in the bundle. The result is the (misleadingly calm) line + # + # ERROR: ld.so: object '.../${LIB}_${PLATFORM}_trace.so' from LD_PRELOAD + # cannot be preloaded (cannot open shared object file): ignored + # + # on every run since this job moved into a container. Compare github/codeql-action#1113, where the + # same message was noise and the real failure was elsewhere — which is exactly why it is worth + # removing rather than tolerating: a permanent ERROR in a security gate's log trains you to skim past + # the one that matters. + # + # The `ls` is not decoration. It is the evidence that the bundle still has the layout this assumes; + # deliberately not guarded with `|| true`, so a future CodeQL release that moves these files fails + # here loudly instead of silently going back to the old behaviour. + # Point LD_PRELOAD at a path that resolves on BOTH sides of the container boundary. + # + # THE ACTUAL CAUSE, after three wrong guesses. `Initialize CodeQL` runs INSIDE this container and + # writes to $GITHUB_ENV: + # + # LD_PRELOAD=/__t/CodeQL//x64/codeql/tools/linux64/${LIB}_${PLATFORM}_trace.so + # + # `/__t` is the name the tool cache has INSIDE the container. On the host the same directory is + # /opt/hostedtoolcache — which is exactly what this job used to export back when it ran on a bare + # runner, and the reason the message never appeared there. But $GITHUB_ENV is consumed by the runner + # process, which lives on the HOST, so the runner's own helpers start with an LD_PRELOAD naming a + # path that does not exist from where they stand, and ld.so logs + # + # ERROR: ld.so: object '.../${LIB}_${PLATFORM}_trace.so' from LD_PRELOAD + # cannot be preloaded (cannot open shared object file): ignored + # + # What it is NOT, each ruled out by measurement rather than argument: not a missing Fedora package + # (the bundle ships the full matrix — lib/lib64/lib32/x86_64-linux-gnu × x86_64/haswell/i686/xeon_phi + # — and `lib64_x86_64_trace.so` is present in the container with mode 0755); not a glibc difference + # (Fedora 44's loader expands the tokens and loads the real tracer correctly — verified inside this + # exact image, `AT_PLATFORM: x86_64`, all dependencies satisfied); and not a broken database (the + # tracing that matters happens inside the container, where the path is valid, which is why the scan + # succeeds regardless). + # + # THE FIX. /opt/hostedtoolcache is real on the host, so making it resolve in here as well gives one + # string that both sides can open. Verified in this image before being written here. + # + # Residual: the ERROR still appears once, at the start of THIS step — the rewrite cannot take effect + # before the step that performs it. `Build (Kotlin)` and `Analyze`, the steps that actually run the + # compiler, get the corrected value. + - name: Make CodeQL's LD_PRELOAD resolve on the host as well as in the container + run: | + set -euo pipefail + preload="${LD_PRELOAD:-}" + [ -n "$preload" ] || { + echo "::error::LD_PRELOAD is unset — CodeQL tracing was never initialised, so this step is" + echo "::error::patching a problem that no longer exists in the shape it was written for." + exit 1 + } + + # Only rewrite the in-container name. A self-hosted runner whose tool cache is somewhere else + # leaves this untouched rather than being handed a path invented for GitHub's hosted images. + case "$preload" in + /__t/*) ;; + *) echo "LD_PRELOAD is not under /__t ($preload) — nothing to rewrite."; exit 0 ;; + esac + + mkdir -p /opt + [ -e /opt/hostedtoolcache ] || ln -s /__t /opt/hostedtoolcache + + new=${preload/#\/__t\//\/opt\/hostedtoolcache\/} + + # Prove the rewritten path resolves HERE before handing it to every later step; the host half is + # the native directory and needs no proving. Tokens substituted only for this check — the value + # exported below keeps them, because the loader is what expands them. + probe=${new//'${LIB}'/lib64} + probe=${probe//'${PLATFORM}'/x86_64} + [ -e "$probe" ] || { echo "::error::rewritten path does not resolve: $probe"; exit 1; } + + echo "LD_PRELOAD=$new" >> "$GITHUB_ENV" + echo "was: $preload" + echo "now: $new" + # Manual build rather than autobuild: autobuild guesses, and this project's build resolves the # whole IntelliJ Platform. `classes` compiles main + resources without running tests twice. - name: Build (Kotlin) diff --git a/.github/workflows/drift.yml b/.github/workflows/drift.yml index 93cef912..29cebb18 100644 --- a/.github/workflows/drift.yml +++ b/.github/workflows/drift.yml @@ -33,7 +33,8 @@ jobs: runs-on: ubuntu-latest timeout-minutes: 30 container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + # `jvm-test`: this job runs `npm install` and the global claude CLI install AND `./gradlew checkDrift`. + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 credentials: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 0ffd1078..cd5adfea 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,15 +1,32 @@ # Release — sign the plugin and publish it to the JetBrains Marketplace. # # This is the only workflow that can reach real users, so it is the most constrained one in the repo. +# +# THE SEQUENCE, which is the part that is easy to get subtly wrong: +# +# PR develop -> main -> tests -> merge -> tag + GitHub Release -> Marketplace, FROM that tag +# +# The tag is cut before the artifact is built, and the build then runs from the tag rather than from +# `main`. That ordering is the point: the tag is the identity of the release (ADR 0001 §3), and `main` +# is a moving ref — a merge landing between `guard` and `publish` would otherwise be silently included +# in a release named after a different tree. +# +# It is NOT two workflows chained by the tag push, and that is a constraint rather than a preference: +# a tag pushed with the GITHUB_TOKEN does not create a workflow run +# (https://docs.github.com/en/actions/concepts/security/github_token — the recursion guard). Chaining +# would need a PAT, a GitHub App or a deploy key, i.e. a long-lived write credential, to buy nothing: +# the same ordering is achievable inside one run by checking the tag out. +# # Three independent gates have to line up before anything is published: # -# 1. TAG. It runs on a `vX.Y.Z` tag and nothing else. The tag is the identity of the artifact -# (ADR 0001 §3) — the same input must always mean the same bytes. -# 2. LINEAGE. The tagged commit must be reachable from `main`. Tagging a feature branch, or a -# develop commit that never went through a PR into main, aborts the run. `main` is -# protected and only accepts PRs, so "reachable from main" IS "was reviewed and merged". -# 3. HUMAN. Publishing lives in the `marketplace` GitHub Environment with a required reviewer. -# Marketplace publication cannot be undone; a version is out the moment it is out. +# 1. VERSION. `build.gradle.kts` is the single source of truth and the tag is derived from it, so the +# two can never disagree. An existing tag means "already released" and the run stops. +# 2. LINEAGE. The commit must be reachable from `main`. Tagging a feature branch, or a develop commit +# that never went through a PR into main, aborts the run. `main` is protected and only +# accepts PRs, so "reachable from main" IS "was reviewed and merged". +# 3. HUMAN. Everything irreversible lives in the `marketplace` GitHub Environment with a required +# reviewer — including the tag, which is why cutting it early is safe. Marketplace +# publication cannot be undone; a version is out the moment it is out. # # Gate 2 is the one worth arguing about, so: it is not decoration. Without it, anyone who can push a # tag can publish from any code, and the PR review that gate 3 assumes has happened becomes optional. @@ -107,7 +124,8 @@ jobs: # the branch was green on. Provisioning the JDK and Node here from separate actions meant the release # gate could pass or fail on a toolchain the pull request never saw. container: - image: ghcr.io/serialexperimentslainnnn/cc-ci:base + # `jvm-test`: this gate runs `npm ci`, `npm test` AND `./gradlew test verifyPlugin` in one job. + image: ghcr.io/serialexperimentslainnnn/jvm-test:v1.0.0 credentials: username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} @@ -159,10 +177,107 @@ jobs: id-token: write # OIDC identity for the attestation attestations: write # write the provenance record steps: + # fetch-depth: 0 because this job CREATES a tag and has to push it — a shallow clone has no + # object graph to tag from. - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: + fetch-depth: 0 persist-credentials: false + # --- the tag comes FIRST, and everything below is built FROM it ----------------------------- + # + # The order is the point of this job. The tag is the identity of the release (ADR 0001 §3), so it + # is cut before the artifact exists and the artifact is then produced from that exact ref — rather + # than publishing first and stamping a tag on afterwards, which makes the tag a label for something + # already gone out. + # + # This is safe to do this early ONLY because the whole job is behind the `marketplace` environment: + # nothing here runs until a human approves, so a tag can no longer appear for a release nobody + # authorised. What it can still do is outlive a FAILED publish, and published tags are immutable + # here. That is deliberate and the recovery is to re-run this job on the existing tag: the tag step + # is a no-op when the ref already exists, and `guard` only blocks a *new* run for an + # already-released version. + - name: Import the CI signing key + run: | + printf '%s' "${{ secrets.GPG_SIGNING_KEY }}" | gpg --batch --import + fpr=$(gpg --list-secret-keys --with-colons | awk -F: '/^fpr:/ {print $10; exit}') + [ -n "$fpr" ] || { echo "::error::GPG_SIGNING_KEY did not import — is it truncated?"; exit 1; } + echo "GPG_FPR=$fpr" >> "$GITHUB_ENV" + + # Signed with the CI key, NOT the maintainer's YubiKey — which cannot sign inside a runner, and whose + # non-exportability is exactly what makes it worth trusting. The chain still terminates in hardware + # because the CI key is certified by it. The claims therefore shift, and SECURITY.md says so: the tag + # attests "this workflow released these bytes", and the human authorisation lives in the two gates + # around it — the reviewed PR into main, and the required approval on this environment. + - name: Create and sign the release tag + if: github.ref_type != 'tag' + env: + GPG_PASSPHRASE: ${{ secrets.GPG_SIGNING_PASSPHRASE }} + TAG: ${{ needs.guard.outputs.tag }} + TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + # git cannot pass gpg the loopback flags it needs in a headless runner, so it gets a wrapper that + # supplies them. The passphrase travels in the environment, never in argv, where `ps` would see it. + printf '#!/bin/sh\nexec gpg --batch --pinentry-mode loopback --passphrase "$GPG_PASSPHRASE" "$@"\n' \ + > /tmp/gpg-loopback + chmod +x /tmp/gpg-loopback + + # A bot identity, not a person: this tag is not a human's assertion and must not look like one. + # The noreply address is required by git and is not anyone's mailbox. + git config user.name 'github-actions[bot]' + git config user.email 'github-actions[bot]@users.noreply.github.com' + git config gpg.program /tmp/gpg-loopback + git config user.signingkey "$GPG_FPR" + + git tag -s "$TAG" -m "Release $TAG — published by the release workflow from $GITHUB_SHA" + git verify-tag "$TAG" # never push a signature we have not checked ourselves + git push "https://x-access-token:${TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "refs/tags/$TAG" + + # Build from the TAG, not from whatever `main` happens to be. On this workflow's primary path the two + # are the same commit, and checking the tag out anyway is what makes that a fact rather than a race: + # `main` is a moving ref, and a merge landing between the guard job and this one would otherwise be + # silently included in a release named after a different tree. + - name: Check out the tag being released + env: + TAG: ${{ needs.guard.outputs.tag }} + run: | + git checkout --detach "refs/tags/$TAG" + echo "building from $TAG -> $(git rev-parse HEAD)" + + # The GitHub Release, created as a DRAFT before anything is published. Draft rather than final + # because it has no assets yet — a release that exists with nothing attached is a broken download + # link for however long the build takes, and it is visible the whole time. + - name: Create the draft GitHub Release + env: + GH_TOKEN: ${{ github.token }} + TAG: ${{ needs.guard.outputs.tag }} + run: | + # The newest section of RELEASE_NOTES.md: from the first "## v" heading to the next one. Same + # source build.gradle.kts reads for the Marketplace "What's New" panel, so they cannot drift. + awk '/^## v/{if(seen)exit; seen=1} seen' RELEASE_NOTES.md > /tmp/notes.md + # NB the heredoc body stays indented to this block's level: YAML strips the common indentation, + # so the emitted markdown is flush-left. An unindented line here (a bare `---`, say) would end + # the block scalar and be read as a YAML document separator. + cat >> /tmp/notes.md <<'EOF' + + --- + + **Verifying this release.** Both the `.asc` files and the tag are signed by the project's **CI + signing key** (`docs/ci-signing-key.asc`), which is itself certified by the maintainer's hardware + key — so the chain terminates in a key that has never been on a computer. + + What the signatures do NOT assert is that a human pressed a button: the release is cut + automatically from `main`. That claim rests on the two gates around it — `main` accepts only + reviewed pull requests, and publication requires an approval on a protected environment. + + ```sh + gpg --import docs/ci-signing-key.asc + gpg --verify claude-code-native-*.zip.asc # these bytes came from this workflow + git verify-tag # this workflow cut this release from main + ``` + EOF + gh release create "$TAG" --draft --title "$TAG" --notes-file /tmp/notes.md --verify-tag + - uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0 with: distribution: temurin @@ -212,13 +327,9 @@ jobs: subject-path: dist/${{ steps.artifact.outputs.name }} # --- GPG-sign the exact bytes that were published ------------------------------------------- - - name: Import the CI signing key - run: | - printf '%s' "${{ secrets.GPG_SIGNING_KEY }}" | gpg --batch --import - fpr=$(gpg --list-secret-keys --with-colons | awk -F: '/^fpr:/ {print $10; exit}') - [ -n "$fpr" ] || { echo "::error::GPG_SIGNING_KEY did not import — is it truncated?"; exit 1; } - echo "GPG_FPR=$fpr" >> "$GITHUB_ENV" - + # The key is already imported: it was needed above to sign the tag, and it is the same key by + # design — one CI signing key backs both claims, and `docs/ci-signing-key.asc` is the single + # public half a user needs to check either of them. - name: Sign the artifact env: PASSPHRASE: ${{ secrets.GPG_SIGNING_PASSPHRASE }} @@ -236,74 +347,21 @@ jobs: for f in "$NAME" "$NAME.sha256"; do gpg --verify "$f.asc" "$f"; done sha256sum -c "$NAME.sha256" - # --- Cut the tag, signed, AFTER the release was approved and actually published ------------- + # --- Attach the artifacts and take the release out of draft --------------------------------- # - # Deliberately last, and deliberately not in `guard`. Creating it earlier would mean a tag exists for - # a version that was never published (a failed build, a declined approval), and published tags are - # immutable here — so the next attempt would be blocked by a tag naming a release that does not exist. - # Cutting it here makes the tag mean "this was published", which is the only claim it can honestly make - # when the version, not the tag, is the input. + # Last, and only now: the draft became a real release the moment it has the four files a user is + # told to verify — the signed zip, its checksum, and a detached signature for each. Undrafting + # earlier would publish a release whose download links 404 for the length of a build. # - # Signed with the CI key, NOT the maintainer's YubiKey — which cannot sign inside a runner, and whose - # non-exportability is exactly what makes it worth trusting. The chain still terminates in hardware - # because the CI key is certified by it. The claims therefore shift, and SECURITY.md says so: the tag - # now attests "this workflow published these bytes", and the human authorisation lives in the two gates - # that remain — the reviewed PR into main, and the required approval on the `marketplace` environment. - - name: Create and sign the release tag - if: github.ref_type != 'tag' - env: - GPG_PASSPHRASE: ${{ secrets.GPG_SIGNING_PASSPHRASE }} - TAG: ${{ needs.guard.outputs.tag }} - TOKEN: ${{ secrets.GITHUB_TOKEN }} - run: | - # git cannot pass gpg the loopback flags it needs in a headless runner, so it gets a wrapper that - # supplies them. The passphrase travels in the environment, never in argv, where `ps` would see it. - printf '#!/bin/sh\nexec gpg --batch --pinentry-mode loopback --passphrase "$GPG_PASSPHRASE" "$@"\n' \ - > /tmp/gpg-loopback - chmod +x /tmp/gpg-loopback - - # A bot identity, not a person: this tag is not a human's assertion and must not look like one. - # The noreply address is required by git and is not anyone's mailbox. - git config user.name 'github-actions[bot]' - git config user.email 'github-actions[bot]@users.noreply.github.com' - git config gpg.program /tmp/gpg-loopback - git config user.signingkey "$GPG_FPR" - - git tag -s "$TAG" -m "Release $TAG — published by the release workflow from $GITHUB_SHA" - git verify-tag "$TAG" # never push a signature we have not checked ourselves - git push "https://x-access-token:${TOKEN}@github.com/${GITHUB_REPOSITORY}.git" "refs/tags/$TAG" - ls -la - - # --- GitHub Release ------------------------------------------------------------------------- - - name: Create the GitHub Release + # `--clobber` so re-running this job on an existing tag replaces the assets instead of failing on + # a name collision. That is the documented recovery path when a publish fails after the tag was + # already cut, and it must not require deleting anything by hand. + - name: Attach the artifacts and publish the release env: GH_TOKEN: ${{ github.token }} + TAG: ${{ needs.guard.outputs.tag }} run: | - # The newest section of RELEASE_NOTES.md: from the first "## v" heading to the next one. Same - # source build.gradle.kts reads for the Marketplace "What's New" panel, so they cannot drift. - awk '/^## v/{if(seen)exit; seen=1} seen' RELEASE_NOTES.md > /tmp/notes.md - # NB the heredoc body stays indented to this block's level: YAML strips the common indentation, - # so the emitted markdown is flush-left. An unindented line here (a bare `---`, say) would end - # the block scalar and be read as a YAML document separator. - cat >> /tmp/notes.md <<'EOF' - - --- - - **Verifying this release.** Both the `.asc` files and the tag are signed by the project's **CI - signing key** (`docs/ci-signing-key.asc`), which is itself certified by the maintainer's hardware - key — so the chain terminates in a key that has never been on a computer. - - What the signatures do NOT assert is that a human pressed a button: the release is cut - automatically from `main`. That claim rests on the two gates around it — `main` accepts only - reviewed pull requests, and publication requires an approval on a protected environment. - - ```sh - gpg --import docs/ci-signing-key.asc - gpg --verify claude-code-native-*.zip.asc # these bytes came from this workflow - git verify-tag # this workflow cut this release from main - ``` - EOF - gh release create "${{ needs.guard.outputs.tag }}" dist/* \ - --title "${{ needs.guard.outputs.tag }}" \ - --notes-file /tmp/notes.md \ - --verify-tag + gh release upload "$TAG" dist/* --clobber + gh release edit "$TAG" --draft=false + echo "released $TAG with:" + gh release view "$TAG" --json assets --jq '.assets[].name' diff --git a/scripts/apply-rulesets.sh b/scripts/apply-rulesets.sh index b2a35010..d82b254a 100755 --- a/scripts/apply-rulesets.sh +++ b/scripts/apply-rulesets.sh @@ -43,7 +43,15 @@ for file in .github/rulesets/*.json; do # The JSON files carry `_comment` keys explaining the non-obvious choices — chiefly why the required # approval count is 0 on a single-maintainer repo. Those keys are documentation, not API fields, so # they are stripped here rather than risking a 422 on an unrecognised property. - body=$(jq 'walk(if type == "object" then del(._comment) else . end)' "$file") + # + # Every key with the `_comment` PREFIX, not just the exact name. Two comments cannot share one object + # under the exact-match version, so the moment a second annotation is needed in the same block the + # obvious move is to call it `_comment_` — which then sails through this filter and gets + # rejected by the API as an unrecognised property. The 422 does not name the offending key, so the + # failure reads as "the ruleset is wrong" rather than "the comment leaked". Observed, not hypothetical. + body=$(jq 'walk(if type == "object" + then with_entries(select(.key | startswith("_comment") | not)) + else . end)' "$file") if [ -n "$id" ]; then echo "updating '$name' (id $id)…"