diff --git a/.github/PULL_REQUEST_TEMPLATE/promotion.md b/.github/PULL_REQUEST_TEMPLATE/promotion.md index 9d488e9c..eec969d6 100644 --- a/.github/PULL_REQUEST_TEMPLATE/promotion.md +++ b/.github/PULL_REQUEST_TEMPLATE/promotion.md @@ -31,9 +31,10 @@ The `Promotion gate` check directly depends on every constituent below and fails | Exact-candidate clean install and startup | `Clean install and startup` | Pending | | Real objects install, update, unload, delete | `Real objects install, update, unload, delete` | Pending | | Aggregate | `Promotion gate` | Pending | +| Semantic version and notes | `Release plan` | Pending | - [ ] The candidate SHA has not changed since every required check completed. -- [ ] The `Guard main branch source` and `Promotion gate` required contexts pass. +- [ ] The `Guard main branch source`, `Promotion gate`, and `Release plan` required contexts pass. - [ ] The complete file and commit compare contains only reviewed work. - [ ] PR conversations, review summaries, and all review threads have been read; actionable findings are addressed and required code-owner approvals exist. @@ -47,6 +48,10 @@ List each unresolved issue and its disposition. Write `None` only after checking Summarize behavioral changes. If no migration is required, state why. +### Release plan + +Review the `Release Plan` workflow comment. Confirm the proposed semantic tag and deterministic notes describe the complete candidate, or confirm that the workflow reports a no-op. If the version is wrong, change the Conventional Commit history on `next` through a reviewed pull request before merging this promotion. + ### Public-contract follow-ups Link documentation and consumer follow-ups identified by the public-contract impact check. Write `None` only with a rationale. @@ -102,4 +107,4 @@ Never force-push either persistent branch. Roll back with a reviewed revert comm ## Stable consumption boundary -Merging this promotion updates the Git-consumed stable `main` ref. It does not create a semantic tag or GitHub release. Any later tag is a separately approved action with its own release notes and exact-ref validation. +Merging this promotion updates the Git-consumed stable `main` ref and authorizes the reviewed release plan. When releasable commits exist, publication waits until every required workflow succeeds on the exact merge SHA, then creates the annotated tag and GitHub release automatically. A no-op plan creates neither. The signed manual tag path remains available for recovery. diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml new file mode 100644 index 00000000..fbc034d9 --- /dev/null +++ b/.github/workflows/benchmark.yml @@ -0,0 +1,100 @@ +--- +name: Benchmark + +# Non-blocking performance evidence (#553): every pull request into next is +# measured against its base, and the promotion pull request into main is +# measured against main. A regression above the thresholds is flagged in the +# job summary and the artifact; it never fails the job (ADR-0009: observed, +# not gated). Hosted runners vary, so compare only within one run, where the +# A/A control row shows the noise floor. + +on: + pull_request: + branches: [next, main] + paths: + - "zi.zsh" + - "lib/**" + - "benchmarks/**" + - "tests/fixtures/package-manifests/**" + - ".github/workflows/benchmark.yml" + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + cancel-in-progress: true + +jobs: + compare: + name: Candidate versus baseline + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - name: Check out candidate + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + # A fork's head commit does not exist in this repository, so name the + # head repository explicitly; the checkout is read-only either way. + repository: ${{ github.event.pull_request.head.repo.full_name || github.repository }} + ref: ${{ github.event.pull_request.head.sha || github.sha }} + path: candidate + persist-credentials: false + + - name: Check out baseline + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.event.pull_request.base.sha || github.sha }} + path: baseline + persist-credentials: false + + - name: Install Zsh and jq + run: sudo apt-get update && sudo apt-get install -yq zsh jq + + - name: Measure baseline, candidate, and an A/A control + shell: bash + run: | + set -uo pipefail + # One invocation measures all three variants: within each round + # they run in a rotating order, with alternate cycles reversed, so + # every variant takes every position and drift on the runner is not + # correlated with a variant. The control is the baseline measured a + # second time. run.zsh exits 1 when a workload fails functionally, + # after writing every report, so that status is accepted here and + # the comparison step renders and returns it; usage and dependency + # errors (2) still fail this step. The step runs under the errexit + # that `shell: bash` inherits, so the status is captured on the + # command's own failure branch, which errexit does not act on. + status=0 + zsh candidate/benchmarks/run.zsh \ + --variant baseline=baseline --variant candidate=candidate --variant control=baseline \ + --output-dir results || status=$? + if [[ "$status" -gt 1 ]]; then exit "$status"; fi + + - name: Compare and publish the summary + shell: bash + run: | + set -euo pipefail + status=0 + zsh candidate/benchmarks/compare.zsh --baseline results/baseline.json --candidate results/candidate.json \ + --control results/control.json --output benchmark-comparison.json --markdown benchmark-comparison.md || status=$? + cat benchmark-comparison.md >> "$GITHUB_STEP_SUMMARY" + flagged="$(jq -r '.flagged | join(", ")' benchmark-comparison.json)" + if [[ -n "$flagged" ]]; then + echo "::notice title=Benchmark flag::Cases above the regression thresholds, for review: ${flagged}" + fi + # A functional failure means a workload no longer runs; that is a + # defect, not a timing question, and it fails the job. + exit "$status" + + - name: Upload benchmark evidence + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2 + with: + name: benchmark-${{ github.event.pull_request.head.sha || github.sha }} + path: | + results/ + benchmark-comparison.json + benchmark-comparison.md + retention-days: 90 diff --git a/.github/workflows/promotion-readiness.yml b/.github/workflows/promotion-readiness.yml index a824616d..5e725af1 100644 --- a/.github/workflows/promotion-readiness.yml +++ b/.github/workflows/promotion-readiness.yml @@ -23,8 +23,8 @@ jobs: with: ref: ${{ github.event.pull_request.head.sha }} - - name: Install Zsh and archive tools - run: sudo apt-get update && sudo apt-get install -yq zsh zip unzip + - name: Install Zsh, jq, and archive tools + run: sudo apt-get update && sudo apt-get install -yq zsh jq zip unzip - name: Check and compile Zsh sources shell: bash @@ -68,9 +68,18 @@ jobs: if: ${{ hashFiles('tests/snippet-directory-mirror.zsh') != '' }} run: zsh -f tests/snippet-directory-mirror.zsh + - name: Test release planning + run: zsh -f tests/release-plan.zsh + + - name: Test promotion release verification + run: zsh -f tests/promotion-release-verification.zsh + + - name: Test idempotent promotion publication + run: zsh -f tests/promotion-release-publication.zsh + zd: name: ZD integration - uses: z-shell/zd/.github/workflows/test-native.yml@01c3477e48c0c31bb7225986bb1bac4270e151ff # z-shell/zd#121 + uses: z-shell/zd/.github/workflows/test-native.yml@5d160597c909a23f03b7a07ff9d43793a106f3e8 # z-shell/zd#123 with: zi_repo: ${{ github.event.pull_request.head.repo.full_name }} zi_ref: ${{ github.event.pull_request.head.sha }} diff --git a/.github/workflows/release-plan.yml b/.github/workflows/release-plan.yml new file mode 100644 index 00000000..23c87a17 --- /dev/null +++ b/.github/workflows/release-plan.yml @@ -0,0 +1,75 @@ +--- +name: Release Plan + +on: + pull_request: + branches: [main] + types: [opened, reopened, synchronize, ready_for_review] + workflow_dispatch: {} + +permissions: + contents: read + pull-requests: write + +concurrency: + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.run_id }} + cancel-in-progress: true + +jobs: + plan: + name: Release plan + runs-on: ubuntu-latest + steps: + - name: Check out the candidate + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ github.event.pull_request.head.sha || github.sha }} + fetch-depth: 0 + fetch-tags: true + persist-credentials: false + + - name: Install Zsh + run: sudo apt-get update && sudo apt-get install -yq zsh + + - name: Compute release plan + id: release + if: "${{ github.event_name == 'workflow_dispatch' || (github.event.pull_request.head.repo.full_name == github.repository && github.event.pull_request.head.ref == 'next') }}" + env: + RELEASE_NOTES_FILE: ${{ runner.temp }}/release-notes.md + RELEASE_PLAN_OUTPUT: ${{ runner.temp }}/release-plan.env + RELEASE_PLAN_BODY: ${{ runner.temp }}/release-plan.md + run: | + zsh -f scripts/release-plan.zsh HEAD > "$RELEASE_PLAN_BODY" + cat "$RELEASE_PLAN_BODY" >> "$GITHUB_STEP_SUMMARY" + cat "$RELEASE_PLAN_OUTPUT" >> "$GITHUB_OUTPUT" + + - name: Record non-promotion result + if: "${{ github.event_name == 'pull_request' && (github.event.pull_request.head.repo.full_name != github.repository || github.event.pull_request.head.ref != 'next') }}" + run: echo 'This pull request is not an internal next-to-main promotion; automatic publication does not apply.' >> "$GITHUB_STEP_SUMMARY" + + - name: Update promotion pull request + if: "${{ github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && github.event.pull_request.head.ref == 'next' }}" + env: + GH_TOKEN: ${{ github.token }} + PR_NUMBER: ${{ github.event.pull_request.number }} + RELEASE_PLAN_BODY: ${{ runner.temp }}/release-plan.md + run: | + set -euo pipefail + marker='' + body="${RUNNER_TEMP}/release-plan-comment.md" + { + echo "$marker" + cat "$RELEASE_PLAN_BODY" + echo + echo '_Merging this reviewed promotion authorizes publication after every required workflow succeeds on the exact merge SHA._' + } > "$body" + comment_id="$(gh api --paginate \ + "repos/${GITHUB_REPOSITORY}/issues/${PR_NUMBER}/comments" \ + --jq ".[] | select(.user.login == \"github-actions[bot]\" and (.body | contains(\"${marker}\"))) | .id" | head -n 1)" + if [[ -n "$comment_id" ]]; then + gh api --method PATCH "repos/${GITHUB_REPOSITORY}/issues/comments/${comment_id}" \ + -F body=@"$body" >/dev/null + else + gh api --method POST "repos/${GITHUB_REPOSITORY}/issues/${PR_NUMBER}/comments" \ + -F body=@"$body" >/dev/null + fi diff --git a/.github/workflows/release-prepare.yml b/.github/workflows/release-prepare.yml deleted file mode 100644 index cd15c777..00000000 --- a/.github/workflows/release-prepare.yml +++ /dev/null @@ -1,21 +0,0 @@ ---- -name: Release Prepare - -on: - push: - branches: [main] - -permissions: - contents: read - issues: write - models: read - -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: false - -jobs: - propose: - uses: z-shell/.github/.github/workflows/release-prepare.yml@6f3d88335ca0ae77b795ec2883b4402b51f15c6a # main - with: - signed_tag: true diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index b4ca3044..4e34b01c 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -4,20 +4,24 @@ name: Release on: push: tags: ["v*.*.*"] + workflow_run: + workflows: [Zsh, ZD Integration, CodeQL, Trunk Code Quality] + types: [completed] -permissions: - actions: read - contents: write +permissions: {} concurrency: - group: ${{ github.workflow }}-${{ github.ref }} + group: release-${{ github.event.workflow_run.head_sha || github.ref_name }} cancel-in-progress: false jobs: - publish: - name: Verify and publish - if: github.repository == 'z-shell/zi' + manual: + name: Verify and publish recovery tag + if: github.event_name == 'push' && github.repository == 'z-shell/zi' runs-on: ubuntu-latest + permissions: + actions: read + contents: write steps: - name: Check out the tagged commit uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -49,3 +53,55 @@ jobs: --title "Zi $TAG" \ --generate-notes \ --latest + + automatic: + name: Publish reviewed promotion + if: "${{ github.event_name == 'workflow_run' && github.repository == 'z-shell/zi' && github.event.workflow_run.head_branch == 'main' }}" + runs-on: ubuntu-latest + permissions: + actions: read + contents: write + pull-requests: read + steps: + - name: Check out the trusted main branch + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: refs/heads/main + fetch-depth: 0 + fetch-tags: true + + - name: Install Zsh + run: sudo apt-get update && sudo apt-get install -yq zsh + + - name: Verify promotion and exact-SHA validation + id: verify + env: + GH_TOKEN: ${{ github.token }} + PROMOTION_SHA: ${{ github.event.workflow_run.head_sha }} + run: zsh -f scripts/verify-promotion-release.zsh + + - name: Compute the authorized release plan + id: plan + if: steps.verify.outputs.ready == 'true' + env: + RELEASE_NOTES_FILE: ${{ runner.temp }}/release-notes.md + RELEASE_TARGET: ${{ github.event.workflow_run.head_sha }} + run: | + RELEASE_PLAN_OUTPUT="$GITHUB_OUTPUT" \ + zsh -f scripts/release-plan.zsh \ + "$RELEASE_TARGET" >> "$GITHUB_STEP_SUMMARY" + + - name: Record no-op promotion + if: steps.verify.outputs.ready == 'true' && steps.plan.outputs.release != 'true' + run: echo 'The reviewed promotion contains no releasable Conventional Commits; no tag or release was created.' >> "$GITHUB_STEP_SUMMARY" + + - name: Create annotated tag and release + if: steps.verify.outputs.ready == 'true' && steps.plan.outputs.release == 'true' + env: + GH_TOKEN: ${{ github.token }} + RELEASE_NOTES_FILE: ${{ runner.temp }}/release-notes.md + RELEASE_TAG: ${{ steps.plan.outputs.tag }} + RELEASE_TARGET: ${{ github.event.workflow_run.head_sha }} + run: | + zsh -f scripts/publish-promotion-release.zsh >> "$GITHUB_STEP_SUMMARY" + echo "Authorized by reviewed promotion #${{ steps.verify.outputs.promotion_pr }}." >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/zd-integration.yml b/.github/workflows/zd-integration.yml index 0b9f6d72..ddcf725f 100644 --- a/.github/workflows/zd-integration.yml +++ b/.github/workflows/zd-integration.yml @@ -4,10 +4,6 @@ name: ZD Integration on: push: branches: [main, next] - paths: - - ".github/workflows/zd-integration.yml" - - "zi.zsh" - - "lib/**" pull_request: paths: - ".github/workflows/zd-integration.yml" @@ -24,7 +20,7 @@ permissions: jobs: zd-test: - uses: z-shell/zd/.github/workflows/test-native.yml@01c3477e48c0c31bb7225986bb1bac4270e151ff # z-shell/zd#121 + uses: z-shell/zd/.github/workflows/test-native.yml@5d160597c909a23f03b7a07ff9d43793a106f3e8 # z-shell/zd#123 with: zi_repo: ${{ github.event.pull_request.head.repo.full_name || github.repository }} zi_ref: ${{ github.event_name == 'pull_request' && github.event.pull_request.head.sha || github.sha }} diff --git a/.github/workflows/zsh-n.yml b/.github/workflows/zsh-n.yml index d21eaa1f..63c3294e 100644 --- a/.github/workflows/zsh-n.yml +++ b/.github/workflows/zsh-n.yml @@ -6,58 +6,21 @@ on: branches: - main - next - paths: - - "zi.zsh" - - "lib/**" - - "contracts/package-manifest-v1.json" - - "scripts/validate-package-manifest.py" - - "tests/**" - - ".github/workflows/*.yml" - - "tests/annex-unregister.zsh" - - "tests/atinit-deferred-marker.zsh" - - "tests/ci-registration.zsh" - - "tests/archive-extraction.zsh" - - "tests/completion-refresh.zsh" - - "tests/disk-ice-resolution.zsh" - - "tests/home-preparation.zsh" - - "tests/hook-ownership.zsh" - - "tests/ice-tokenizer.zsh" - - "tests/load-object-status.zsh" - - "tests/message-formatting.zsh" - - "tests/package-manifest-contract.zsh" - - "tests/package-manifest-fixtures.zsh" - - "tests/fixtures/package-manifests/**" - - "scripts/refresh-package-manifests.zsh" - - "tests/package-manifest-parsing.zsh" - - "tests/path-resolution.zsh" - - "tests/parallel-update.zsh" - - "tests/plugin-autoload-fpath-scope.zsh" - - "tests/plugin-autoload-ice.zsh" - - "tests/nested-load-state.zsh" - - "tests/pack-service-first-install.zsh" - - "tests/plugin-autoload-ownership.zsh" - - "tests/plugin-standard-callbacks.zsh" - - "tests/release-tag-verification.zsh" - - "tests/scheduler-idle.zsh" - - "tests/fixtures/plugin-standard-callbacks/**" - - "tests/self-update-reload.zsh" - - "tests/snippet-directory-mirror.zsh" - - "tests/snippet-update-status.zsh" - - "tests/source-hygiene.zsh" - - "tests/subst-nesting.zsh" - - "tests/unload-hook-dispatch.zsh" - - "tests/unload-ownership-contracts.zsh" - - "tests/version-reporting.zsh" pull_request: paths: - "zi.zsh" - "lib/**" - "contracts/package-manifest-v1.json" - "scripts/validate-package-manifest.py" + - "scripts/release-plan.zsh" + - "scripts/verify-promotion-release.zsh" + - "scripts/publish-promotion-release.zsh" - "tests/**" - ".github/workflows/*.yml" - "tests/annex-unregister.zsh" - "tests/atinit-deferred-marker.zsh" + - "tests/benchmark-harness.zsh" + - "benchmarks/**" - "tests/ci-registration.zsh" - "tests/archive-extraction.zsh" - "tests/completion-refresh.zsh" @@ -204,6 +167,50 @@ jobs: - name: Test release tag verification run: zsh -f tests/release-tag-verification.zsh + release-plan: + name: Release Plan + runs-on: ubuntu-latest + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install Zsh + run: sudo apt update && sudo apt-get install -yq zsh + - name: Test release planning + run: zsh -f tests/release-plan.zsh + + promotion-release-verification: + name: Promotion Release Verification + runs-on: ubuntu-latest + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install Zsh and jq + run: sudo apt update && sudo apt-get install -yq zsh jq + - name: Test promotion release verification + run: zsh -f tests/promotion-release-verification.zsh + + promotion-release-publication: + name: Promotion Release Publication + runs-on: ubuntu-latest + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install Zsh + run: sudo apt update && sudo apt-get install -yq zsh + - name: Test idempotent promotion publication + run: zsh -f tests/promotion-release-publication.zsh + + benchmark-harness: + name: Benchmark Harness + runs-on: ubuntu-latest + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + - name: Install Zsh and jq + run: sudo apt update && sudo apt-get install -yq zsh jq + - name: Test the benchmark runner and comparer + run: zsh -f tests/benchmark-harness.zsh + version-reporting: name: Version reporting runs-on: ubuntu-latest diff --git a/AGENTS.md b/AGENTS.md index 5f78ff70..6f94dfd1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -14,6 +14,7 @@ Zi is the canonical Zsh plugin manager for the organization. Changes can affect - A successful promotion needs no routine back-merge. Merge a `main` hotfix forward into `next` before ordinary development continues. - `hotfix-*` branches may target `main` directly. - Keep `delete_branch_on_merge` disabled because `next` is persistent. +- A pull request merged into `next` leaves its issue open: GitHub closes issues only from the default branch. Link the issue for the Development sidebar (a closing keyword, or `addCloseIssueReferences` when the link is missing) and close the accumulated issues by hand when `next` is promoted to `main`; do not close them early. ## Before merging diff --git a/benchmarks/README.md b/benchmarks/README.md new file mode 100644 index 00000000..744d13c0 --- /dev/null +++ b/benchmarks/README.md @@ -0,0 +1,40 @@ +# Zi benchmarks + +Deterministic, network-free measurements of the paths a user pays for on every shell start: sourcing `zi.zsh`, loading plugins, queueing turbo tasks, parsing ices, reading package manifests, and unloading. The suite exists to make a performance change visible in the pull request that causes it (#553); it does not gate merges (ADR-0009: coverage and performance are observed, not gated). + +## Cases + +| Case | Measured region | +| --- | --- | +| `source-fresh-home` | `source zi.zsh` in a home no sample has used before (directories are created, nothing is cached) | +| `source-reused-home` | `source zi.zsh` in the variant's home kept from the previous sample (prepared directories and completion state persist) | +| `light-load-10` | `zi light` of ten fixture plugins | +| `load-10` | `zi load` (tracked) of ten fixture plugins | +| `turbo-10` | ten `wait'0'` tasks queued, then `@zi-scheduler burst` | +| `ice-200` | 200 `zi ice` calls carrying 13 ices each | +| `manifest-21` | the 21 vendored package manifests through `.zi-read-package-manifest` | +| `unload-10` | `zi unload` of ten tracked plugins | + +Each sample starts a fresh `zsh -f` with an isolated home and reports floating-point `SECONDS` elapsed milliseconds of the measured region only; the timer needs no module, so the source cases include the `zsh/datetime` load that `zi.zsh` performs. All variants of one invocation are measured together: within a round they run in a rotating order, and alternate rotation cycles run reversed, so every variant takes every position equally often and runner drift is not correlated with a variant; cases rotate the same way across rounds. Zi does not compile itself on source, so the two source cases differ by persisted home state, not by bytecode. Every run also records health data that costs nothing extra: function and parameter counts after source and after loading, `.zwc` presence, `zsh -n` and `zcompile` durations, and the line counts of the hot files. + +## Run it + +```sh +zsh benchmarks/run.zsh --variant baseline=../zi-at-main --variant candidate=. \ + --variant control=../zi-at-main --output-dir results +zsh benchmarks/compare.zsh --baseline results/baseline.json --candidate results/candidate.json \ + --control results/control.json --output comparison.json --markdown comparison.md +``` + +Defaults are 5 warmups and 30 samples; `--case NAME` selects a subset, and `source-reused-home` needs at least one warmup so that its first measured sample times a home a previous sample has sourced. `run.zsh` writes one JSON report per variant and exits 1, after writing every report, when a workload fails functionally. `compare.zsh` reports median, p95, min and the sample count per variant, the median and p95 deltas, and an A/A control column from the second baseline measurement so the noise floor is visible next to the real comparison; a control measured under other settings (Zsh version, architecture, sample or warmup count) is rejected. A failed sample records the child's last diagnostic line, or what its exit status means when it wrote nothing. A case whose median regresses over 10% or whose p95 regresses over 15% is flagged for review; flags never fail, and they are null when the two reports are not comparable (different Zsh version, architecture, sample or warmup count). A functional failure on either side invalidates that case and exits 1, because timing a broken behaviour is meaningless. A case a checkout cannot run because it predates the API the case exercises is a different thing: `case.zsh` exits 6, `run.zsh` records the case as `unsupported` for that variant with the reason and still exits 0, and `compare.zsh` renders the row without a delta. A baseline without the API (or both sides without it) is unsupported, not failed, and does not exit 1; a candidate without an API its baseline has is recorded as a failure, because that is a removal. The manifest case is named for its inventory: `run.zsh` pins the 21 repository names and requires `tests/fixtures/package-manifests/repositories.txt` to list exactly them with a snapshot each, and each sample re-asserts the count; replacing one repository with another is rejected even though the count is unchanged. A changed inventory is a different workload and needs a renamed or versioned case and an updated pinned list, not a drifting number or a silently swapped set. Snapshot contents are not pinned, so refreshing a vendored manifest verbatim keeps the inventory. + +## Read results carefully + +- Hosted runners differ in hardware and load. Compare only within one run, and only when `comparable` is true (same Zsh version, architecture, and sample count). +- The fixture plugins are tiny; the cases isolate Zi's own overhead, not a real configuration's plugin bodies. +- `source-reused-home` is the everyday startup cost; `source-fresh-home` adds the first-run directory preparation. +- `tests/benchmark-harness.zsh` proves the runner and comparer: shape, A/A, a synthetic 30% regression flagged, a functional failure invalidated, a control-only failure rendered, a repeated `--case` rejected, a checkout without the manifest reader reported unsupported (and a candidate that lost it reported as a failure), a failed postcondition recorded with a one-line reason, a swapped manifest inventory rejected, a pattern given as `--case` rejected as a usage error, a pipe inside a reason kept in one cell, a failed health probe recorded as null counts beside the measured cases, an incompatible control rejected, an integral sample recorded as a number, and six cells in every table row. + +## Where it runs + +`.github/workflows/benchmark.yml` compares every pull request into `next` with its base and the promotion pull request into `main` with `main`, writes the table to the job summary, and keeps the JSON and Markdown as an artifact for 90 days. Per-release results are committed by a maintainer under `benchmarks/results/--/` (zpmod's naming) after a release, never by automation. diff --git a/benchmarks/case.zsh b/benchmarks/case.zsh new file mode 100644 index 00000000..babc78a4 --- /dev/null +++ b/benchmarks/case.zsh @@ -0,0 +1,99 @@ +#!/usr/bin/env zsh +# -*- mode: zsh; sh-indentation: 2; indent-tabs-mode: nil; sh-basic-offset: 2; -*- +# vim: ft=zsh sw=2 ts=2 et +# +# case.zsh -- one benchmark sample, run by run.zsh in a fresh `zsh -f`. +# Reads BENCH_CHECKOUT, BENCH_FIXTURES, BENCH_MANIFESTS and BENCH_CASE from the +# environment and prints the elapsed milliseconds of the measured region. +# The measured region excludes process start and setup, which is why every +# case sets up in the same process it measures in. The timer is floating-point +# SECONDS, which needs no module: zi.zsh loads zsh/datetime when sourced, so a +# timer built on that module would pay for the load before the source cases +# start measuring. +# +# Exit 6 means the checkout lacks the API the case exercises; run.zsh records +# that as unsupported for the variant, not as a failure. Every other non-zero +# status is a failure of the workload. + +emulate -R zsh +setopt extended_glob + +typeset -gAH ZI +ZI[BIN_DIR]=$BENCH_CHECKOUT +typeset -F 6 SECONDS +typeset -F6 t0 t1 +zle() { return 1; } +start() { t0=$SECONDS; } +mark() { t1=$SECONDS; } +# Always three decimals: arithmetic substitution prints an integral float as +# "250." with no digits after the point, which run.zsh rejects as a sample and +# which is not a JSON number. +elapsed() { printf '%.3f\n' $(( (t1 - t0) * 1000 )); } +stop() { mark; elapsed; } +load_zi() { builtin source "$BENCH_CHECKOUT/zi.zsh" || exit 3; .zi-prepare-home || exit 3; } +plugins() { local i; for i in {1..10}; do print -r -- "$BENCH_FIXTURES/plugins/p$i"; done; } +# Every fixture plugin defines a function, an alias, and a global parameter. +# All thirty artifacts prove that every plugin ran, and that an unload +# removed every kind of state it owns; checking one kind, or only the last +# plugin, would accept a partial workload or a partial cleanup. +all_loaded() { + local i + for i in {1..10}; do + (( $+functions[p${i}_fn] && $+aliases[p${i}_alias] && $+parameters[P${i}_PARAM] )) || return 1 + done + return 0 +} +none_loaded() { + local i + for i in {1..10}; do + (( $+functions[p${i}_fn] || $+aliases[p${i}_alias] || $+parameters[P${i}_PARAM] )) && return 1 + done + return 0 +} + +case $BENCH_CASE in + source-fresh-home|source-reused-home) + # Zi does not compile itself on source; the difference between these two + # is the persisted home state (prepared directories, completion dump) + # that the reused home carries over from the previous sample. + # zi.zsh does not propagate a failed .zi-prepare-home, so a successful + # source can leave the home unprepared; only a ready home is timed. + start; builtin source "$BENCH_CHECKOUT/zi.zsh" || exit 3; mark + [[ -n ${ZI[HOME_READY]} ]] || exit 4 + elapsed ;; + light-load-10) + load_zi; start; for p in $(plugins); do zi light %"$p" || exit 3; done; stop + all_loaded || exit 4 ;; + load-10) + load_zi; start; for p in $(plugins); do zi load %"$p" || exit 3; done; stop + all_loaded || exit 4 ;; + turbo-10) + load_zi; start; for p in $(plugins); do zi wait'0' lucid for %"$p" || exit 3; done; @zi-scheduler burst || exit 3; stop + all_loaded || exit 4 ;; + ice-200) + load_zi; start; for i in {1..200}; do zi ice wait'1' lucid depth'1' atinit'true' atload'true' pick'x' as'program' from'gh-r' mv'a -> b' id-as'x' compile'y' nocompile blockf || exit 3; done; stop ;; + manifest-21) + load_zi; builtin source "$BENCH_CHECKOUT/lib/zsh/install.zsh" || exit 3 + # A checkout that predates the manifest reader cannot run this workload. + # That is a capability of the checkout, not a defect in it: exit 6 tells + # run.zsh to record the case as unsupported for this variant instead of + # failed, and compare.zsh decides whether the asymmetry matters. + (( $+functions[.zi-read-package-manifest] )) || { print -u2 -r -- "checkout does not define .zi-read-package-manifest"; exit 6 } + local -a files; files=( $BENCH_MANIFESTS/*.json(N) ) + # Exactly the declared inventory, or the timing is not comparable (#555). + (( $#files == ${BENCH_MANIFEST_COUNT:-0} )) || exit 5 + start; for f in $files; do local -A info ices; local -a names; .zi-read-package-manifest "$(<$f)" default info names ices || exit 4; unset info ices names; done; stop ;; + unload-10) + load_zi; for p in $(plugins); do zi load %"$p" || exit 3; done + all_loaded || exit 4 + start; for p in $(plugins); do zi unload %"$p" -q || exit 3; done; stop + none_loaded || exit 4 ;; + symbols) + # Same setup as a measured case: a failed home preparation exits 3, so + # run.zsh records null counts instead of counts from a half-prepared home. + load_zi + integer f1=$#functions p1=$#parameters + for p in $(plugins); do zi load %"$p" || exit 3; done + print -r -- "$f1 $p1 $#functions $#parameters" ;; + *) exit 2 ;; +esac diff --git a/benchmarks/compare.zsh b/benchmarks/compare.zsh new file mode 100644 index 00000000..7a880628 --- /dev/null +++ b/benchmarks/compare.zsh @@ -0,0 +1,136 @@ +#!/usr/bin/env zsh +# -*- mode: zsh; sh-indentation: 2; indent-tabs-mode: nil; sh-basic-offset: 2; -*- +# vim: ft=zsh sw=2 ts=2 et +# +# compare.zsh -- join two run.zsh reports into a baseline-versus-candidate +# report, flag regressions above the thresholds, and render Markdown. +# +# Timing never fails the comparison: a flag marks a case for review, which is +# the organization's stance for hosted-runner evidence (observed, not gated). +# A functional failure on either side does invalidate that case. +# +# A case a checkout cannot run because it lacks the API the case exercises is +# reported by run.zsh as unsupported for that variant. A baseline without the +# API is expected when the candidate adds it, so that row is unsupported, not +# failed, and does not exit 1. A candidate without an API the baseline has is +# a removal, and that row is a failure. +# +# Usage: +# zsh benchmarks/compare.zsh --baseline FILE --candidate FILE --output FILE +# [--markdown FILE] [--control FILE] +# [--median-threshold PERCENT] [--p95-threshold PERCENT] +# +# --control is a second run of the baseline (an A/A sample) whose deltas show +# the noise floor next to the real comparison. +# +# Exit codes: +# 0 comparison written (flags, if any, are in the report) +# 1 a case failed functionally on either side +# 2 usage or dependency error + +emulate -LR zsh +setopt extended_glob pipe_fail no_unset + +typeset baseline='' candidate='' control='' output='' markdown='' +typeset -F1 median_threshold=10 p95_threshold=15 +die() { print -u2 -r -- "compare.zsh: $1"; exit ${2:-2}; } +while (( $# )); do + case "$1" in + --baseline) [[ -n ${2-} ]] || die "--baseline needs a value"; baseline=$2; shift 2 ;; + --candidate) [[ -n ${2-} ]] || die "--candidate needs a value"; candidate=$2; shift 2 ;; + --control) [[ -n ${2-} ]] || die "--control needs a value"; control=$2; shift 2 ;; + --output) [[ -n ${2-} ]] || die "--output needs a value"; output=$2; shift 2 ;; + --markdown) [[ -n ${2-} ]] || die "--markdown needs a value"; markdown=$2; shift 2 ;; + --median-threshold) [[ ${2-} == <->(.<->|) ]] || die "--median-threshold needs a number"; median_threshold=$2; shift 2 ;; + --p95-threshold) [[ ${2-} == <->(.<->|) ]] || die "--p95-threshold needs a number"; p95_threshold=$2; shift 2 ;; + --help|-h) print -r -- "usage: ${0:t} --baseline FILE --candidate FILE --output FILE [--markdown FILE] [--control FILE]"; exit 0 ;; + *) die "unknown argument: $1" ;; + esac +done +[[ -r $baseline && -r $candidate && -n $output ]] || die "--baseline, --candidate and --output are required" +(( $+commands[jq] )) || die "required command not found: jq" + +# The three reports must describe the same case set, or a missing case would +# be silently dropped (candidate-only) or fail deep inside jq (baseline-only). +typeset -a case_sets +case_sets=( "$(jq -c '.cases | keys' "$baseline")" "$(jq -c '.cases | keys' "$candidate")" ) +[[ -n $control ]] && case_sets+=( "$(jq -c '.cases | keys' "$control")" ) +[[ ${#${(u)case_sets}} -eq 1 ]] || die "the reports do not cover the same cases: ${(j: versus :)case_sets}" + +# The control is the baseline measured again, so it must come from the same +# source revision and the same settings; the control rows reuse the +# baseline-versus-candidate comparability and would otherwise show another +# revision, or an incompatible run, as the noise floor. +if [[ -n $control ]]; then + typeset baseline_identity control_identity + baseline_identity=$(jq -c '[.source_revision, .environment.zsh_version, .environment.architecture, .workload.samples, .workload.warmups]' "$baseline") + control_identity=$(jq -c '[.source_revision, .environment.zsh_version, .environment.architecture, .workload.samples, .workload.warmups]' "$control") + [[ $baseline_identity == "$control_identity" ]] || die "the control is not a second run of the baseline (source revision, Zsh version, architecture, samples, warmups): ${baseline_identity} versus ${control_identity}" +fi + +# jq does the arithmetic so the report is one deterministic document. +typeset -a control_args +[[ -n $control ]] && control_args=( --slurpfile control "$control" ) || control_args=( --argjson control '[null]' ) +jq -n --slurpfile b "$baseline" --slurpfile c "$candidate" "${control_args[@]}" \ + --argjson mt "$median_threshold" --argjson pt "$p95_threshold" ' + def pct(a; b): if a == null or b == null or b == 0 then null else ((a - b) * 100 / b) end; + ($b[0]) as $B | ($c[0]) as $C | ($control[0]) as $K | + ($B.environment.zsh_version == $C.environment.zsh_version + and $B.environment.architecture == $C.environment.architecture + and $B.workload.samples == $C.workload.samples + and $B.workload.warmups == $C.workload.warmups) as $comparable | + # Flags are meaningful only between comparable reports; otherwise every + # flag is null and the summary says why. + def row(base; cand): + # A baseline that lacks the API is unsupported (with or without the + # candidate); a candidate that lacks an API the baseline runs is a removal + # and therefore a failure. A candidate failure always stays a failure. + if (base.unsupported? // null) != null and (cand.failure? // null) == null then + {unsupported: {baseline: base.unsupported, candidate: (cand.unsupported? // null)}} + elif (cand.unsupported? // null) != null then + {failure: {baseline: (base.failure? // null), candidate: ("removed: " + cand.unsupported)}} + elif (base.failure? // null) != null or (cand.failure? // null) != null then + {failure: {baseline: (base.failure? // null), candidate: (cand.failure? // null)}} + else + {results: {baseline: base, candidate: cand}, + change: {median_delta_ms: (cand.median - base.median), median_delta_percent: pct(cand.median; base.median), + p95_delta_ms: (cand.p95 - base.p95), p95_delta_percent: pct(cand.p95; base.p95)}} + | .flag = (if $comparable then ((.change.median_delta_percent > $mt) or (.change.p95_delta_percent > $pt)) else null end) + end; + {schema_version: 1, + captured_at: (now | todate), + thresholds: {median_percent: $mt, p95_percent: $pt, policy: "flag for review, never fail on timing; flags are null when the reports are not comparable"}, + baseline: {label: $B.label, source_revision: $B.source_revision, environment: $B.environment, workload: $B.workload}, + candidate: {label: $C.label, source_revision: $C.source_revision, environment: $C.environment, workload: $C.workload}, + comparable: $comparable, + health: {baseline: $B.health, candidate: $C.health}, + cases: ($B.cases | keys | map(. as $k | {($k): row($B.cases[$k]; $C.cases[$k])}) | add), + control: (if $K == null then null else ($B.cases | keys | map(. as $k | {($k): row($B.cases[$k]; $K.cases[$k])}) | add) end)} + | .flagged = [.cases | to_entries[] | select(.value.flag == true) | .key] + | .unsupported = [.cases | to_entries[] | select(.value.unsupported != null) | .key] + | .failed = ([.cases | to_entries[] | select(.value.failure != null) | .key] + + (if .control == null then [] else [.control | to_entries[] | select(.value.failure != null) | .key] end) | unique) +' > "$output" || die "could not build the comparison" 1 +jq -e . "$output" >/dev/null || die "comparison is not valid JSON" 1 + +if [[ -n $markdown ]]; then + { + print -r -- "## Zi benchmark: candidate versus baseline" + print + print -r -- "Baseline \`$(jq -r .baseline.source_revision "$output" | cut -c1-7)\` ($(jq -r .baseline.label "$output")) versus candidate \`$(jq -r .candidate.source_revision "$output" | cut -c1-7)\` ($(jq -r .candidate.label "$output")); $(jq -r .baseline.workload.samples "$output") samples after $(jq -r .baseline.workload.warmups "$output") warmups; $(jq -r .candidate.environment.zsh_version "$output") on $(jq -r .candidate.environment.cpu "$output"). Comparable: $(jq -r .comparable "$output"). Flags mark a median regression over $(jq -r .thresholds.median_percent "$output")% or a p95 regression over $(jq -r .thresholds.p95_percent "$output")%; they never fail the job.$( [[ $(jq -r .comparable "$output") == true ]] || print -n " The reports are not comparable (Zsh version, architecture, sample or warmup counts differ), so no case is flagged." )" + print + print -r -- "| Case | Baseline median / p95 ms | Candidate median / p95 ms | Median delta | p95 delta | A/A control median delta |" + print -r -- "| --- | --- | --- | --- | --- | --- |" + # A reason is child output and may contain a pipe; escaped, it stays in + # its own cell instead of splitting the row. + jq -r 'def cell: tojson | gsub("\\|"; "\\|"); .cases | to_entries[] | . as $e | if .value.failure then "| \(.key) | failure | failure | \(.value.failure | cell) | | " elif .value.unsupported then "| \(.key) | unsupported | \(if .value.unsupported.candidate != null then "unsupported" else "not compared" end) | \(.value.unsupported.baseline | cell) | | " else "| \(.key)\(if .value.flag then " (flag)" else "" end) | \(.value.results.baseline.median) / \(.value.results.baseline.p95) | \(.value.results.candidate.median) / \(.value.results.candidate.p95) | \(.value.change.median_delta_percent | . * 10 | round / 10)% | \(.value.change.p95_delta_percent | . * 10 | round / 10)% | " end' "$output" | while IFS= read -r line; do + case_name=${${line#| }%% *} + ctrl=$(jq -r --arg k "$case_name" '.control[$k] | if . == null then "n/a" elif .failure != null then "failure: \(.failure.candidate // .failure.baseline | gsub("\\|"; "\\|"))" elif .unsupported != null then "unsupported" elif (.change.median_delta_percent | type) == "number" then (.change.median_delta_percent * 10 | round / 10 | tostring) + "%" else "n/a" end' "$output") + print -r -- "${line}${ctrl} |" + done + print + print -r -- "Health (baseline to candidate): functions after source $(jq -r .health.baseline.functions_after_source "$output") to $(jq -r .health.candidate.functions_after_source "$output"), parameters $(jq -r .health.baseline.parameters_after_source "$output") to $(jq -r .health.candidate.parameters_after_source "$output"), zi.zsh $(jq -r '.health.baseline["lines:zi.zsh"]' "$output") to $(jq -r '.health.candidate["lines:zi.zsh"]' "$output") lines, zcompile $(jq -r '.health.baseline["zcompile_ms:zi.zsh"] | . * 10 | round / 10' "$output") to $(jq -r '.health.candidate["zcompile_ms:zi.zsh"] | . * 10 | round / 10' "$output") ms." + } > "$markdown" +fi +print -r -- "wrote $output: $(jq -r '.flagged | length' "$output") flagged, $(jq -r '.failed | length' "$output") failed, $(jq -r '.unsupported | length' "$output") unsupported" +(( $(jq -r '.failed | length' "$output") == 0 )) diff --git a/benchmarks/fixtures/plugins/p1/p1.plugin.zsh b/benchmarks/fixtures/plugins/p1/p1.plugin.zsh new file mode 100644 index 00000000..297cd16b --- /dev/null +++ b/benchmarks/fixtures/plugins/p1/p1.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 1: one function, one alias, one parameter. +p1_fn() { :; } +alias p1_alias=true +typeset -g P1_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p10/p10.plugin.zsh b/benchmarks/fixtures/plugins/p10/p10.plugin.zsh new file mode 100644 index 00000000..cb6adb7a --- /dev/null +++ b/benchmarks/fixtures/plugins/p10/p10.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 10: one function, one alias, one parameter. +p10_fn() { :; } +alias p10_alias=true +typeset -g P10_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p2/p2.plugin.zsh b/benchmarks/fixtures/plugins/p2/p2.plugin.zsh new file mode 100644 index 00000000..5910c48e --- /dev/null +++ b/benchmarks/fixtures/plugins/p2/p2.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 2: one function, one alias, one parameter. +p2_fn() { :; } +alias p2_alias=true +typeset -g P2_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p3/p3.plugin.zsh b/benchmarks/fixtures/plugins/p3/p3.plugin.zsh new file mode 100644 index 00000000..a5d90912 --- /dev/null +++ b/benchmarks/fixtures/plugins/p3/p3.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 3: one function, one alias, one parameter. +p3_fn() { :; } +alias p3_alias=true +typeset -g P3_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p4/p4.plugin.zsh b/benchmarks/fixtures/plugins/p4/p4.plugin.zsh new file mode 100644 index 00000000..beb76c68 --- /dev/null +++ b/benchmarks/fixtures/plugins/p4/p4.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 4: one function, one alias, one parameter. +p4_fn() { :; } +alias p4_alias=true +typeset -g P4_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p5/p5.plugin.zsh b/benchmarks/fixtures/plugins/p5/p5.plugin.zsh new file mode 100644 index 00000000..952d4b61 --- /dev/null +++ b/benchmarks/fixtures/plugins/p5/p5.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 5: one function, one alias, one parameter. +p5_fn() { :; } +alias p5_alias=true +typeset -g P5_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p6/p6.plugin.zsh b/benchmarks/fixtures/plugins/p6/p6.plugin.zsh new file mode 100644 index 00000000..55207e5f --- /dev/null +++ b/benchmarks/fixtures/plugins/p6/p6.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 6: one function, one alias, one parameter. +p6_fn() { :; } +alias p6_alias=true +typeset -g P6_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p7/p7.plugin.zsh b/benchmarks/fixtures/plugins/p7/p7.plugin.zsh new file mode 100644 index 00000000..abf514ca --- /dev/null +++ b/benchmarks/fixtures/plugins/p7/p7.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 7: one function, one alias, one parameter. +p7_fn() { :; } +alias p7_alias=true +typeset -g P7_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p8/p8.plugin.zsh b/benchmarks/fixtures/plugins/p8/p8.plugin.zsh new file mode 100644 index 00000000..c9fe76a7 --- /dev/null +++ b/benchmarks/fixtures/plugins/p8/p8.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 8: one function, one alias, one parameter. +p8_fn() { :; } +alias p8_alias=true +typeset -g P8_PARAM=1 diff --git a/benchmarks/fixtures/plugins/p9/p9.plugin.zsh b/benchmarks/fixtures/plugins/p9/p9.plugin.zsh new file mode 100644 index 00000000..a4a232d0 --- /dev/null +++ b/benchmarks/fixtures/plugins/p9/p9.plugin.zsh @@ -0,0 +1,4 @@ +# Deterministic benchmark fixture plugin 9: one function, one alias, one parameter. +p9_fn() { :; } +alias p9_alias=true +typeset -g P9_PARAM=1 diff --git a/benchmarks/run.zsh b/benchmarks/run.zsh new file mode 100644 index 00000000..696cd0df --- /dev/null +++ b/benchmarks/run.zsh @@ -0,0 +1,297 @@ +#!/usr/bin/env zsh +# -*- mode: zsh; sh-indentation: 2; indent-tabs-mode: nil; sh-basic-offset: 2; -*- +# vim: ft=zsh sw=2 ts=2 et +# +# run.zsh -- measure Zi checkouts on deterministic, network-free workloads. +# +# Every sample starts a fresh `zsh -f` with an isolated home, so nothing leaks +# between samples or variants. Several variants (for example baseline, +# candidate, and a second baseline as the A/A control) are measured in the +# same invocation: within a round the variants run in a rotating order, with +# alternate rotation cycles reversed, so every variant takes every position +# equally often and runner load, cache, and thermal drift are not correlated +# with a variant. Cases rotate the same way across rounds. Output is one JSON +# document per variant; compare.zsh joins two of them. +# +# Usage: +# zsh benchmarks/run.zsh --variant LABEL=DIR [--variant LABEL=DIR]... --output-dir DIR +# [--warmups N] [--samples N] [--case NAME]... +# +# Requires: zsh with zsh/datetime, jq. +# +# Exit codes: +# 0 every case produced the requested samples, or a variant reported a case +# unsupported because its checkout lacks the API the case exercises (the +# report records that per case; compare.zsh decides whether it matters) +# 1 a case failed functionally (its samples are discarded and the report +# records the failure; timing of a broken behaviour is meaningless) +# 2 usage or dependency error + +emulate -LR zsh +setopt extended_glob pipe_fail no_unset + +typeset output_dir='' +integer warmups=5 samples=30 +typeset -a wanted labels +typeset -A dirs +typeset -a all_cases +all_cases=( source-fresh-home source-reused-home light-load-10 load-10 turbo-10 ice-200 manifest-21 unload-10 ) + +usage() { print -r -- "usage: ${0:t} --variant LABEL=DIR [--variant LABEL=DIR]... --output-dir DIR [--warmups N] [--samples N] [--case NAME]..."; } +die() { print -u2 -r -- "run.zsh: $1"; exit ${2:-2}; } + +while (( $# )); do + case "$1" in + --variant) + [[ ${2-} == ?*=?* ]] || die "--variant needs LABEL=DIR" + [[ ${2%%=*} == [[:alnum:]_-]## ]] || die "variant label must be alphanumeric: ${2%%=*}" + (( ${+dirs[${2%%=*}]} )) && die "duplicate variant label: ${2%%=*}" + labels+=( "${2%%=*}" ); dirs[${2%%=*}]=${${2#*=}:A}; shift 2 ;; + --output-dir) [[ -n ${2-} ]] || die "--output-dir needs a value"; output_dir=$2; shift 2 ;; + --warmups) [[ ${2-} == <-> ]] || die "--warmups needs an integer"; warmups=$2; shift 2 ;; + --samples) [[ ${2-} == <-> && ${2-} -ge 2 ]] || die "--samples needs an integer of at least 2"; samples=$2; shift 2 ;; + --case) [[ -n ${2-} ]] || die "--case needs a value"; wanted+=( "$2" ); shift 2 ;; + --help|-h) usage; exit 0 ;; + *) usage >&2; die "unknown argument: $1" ;; + esac +done +(( $#labels )) || die "at least one --variant LABEL=DIR is required" +typeset label +for label in "${labels[@]}"; do [[ -r ${dirs[$label]}/zi.zsh ]] || die "variant ${label}: no zi.zsh under ${dirs[$label]}"; done +[[ -n $output_dir ]] || die "--output-dir is required" +command mkdir -p -- "$output_dir" || die "could not create $output_dir" +(( $+commands[jq] )) || die "required command not found: jq" +zmodload zsh/datetime || die "zsh/datetime is required" + +typeset -a cases +if (( $#wanted )); then + # (Ie) matches the requested name literally: a pattern such as source-* + # is not a case name and must be a usage error here, not a workload failure + # in case.zsh. + for c in "${wanted[@]}"; do (( ${all_cases[(Ie)$c]} )) || die "unknown case: $c"; done + # A repeated selection would run the case twice per round and write the + # same report key twice, so the sample count would no longer be the truth. + typeset -a seen + for c in "${wanted[@]}"; do + (( ${seen[(Ie)$c]} )) && die "--case $c given more than once" + seen+=( "$c" ) + done + cases=( "${wanted[@]}" ) +else + cases=( "${all_cases[@]}" ) +fi +# The reused-home case times a home the previous sample sourced. Without a +# warmup its first measured sample would be a fresh-home timing mixed into the +# reused-home statistics; other cases are indifferent to zero warmups. +(( warmups >= 1 || ${cases[(Ie)source-reused-home]} == 0 )) || die "source-reused-home needs at least one warmup: its first sample would time a home no sample has sourced" + +typeset here=${0:A:h} fixtures=${0:A:h}/fixtures manifests=${0:A:h:h}/tests/fixtures/package-manifests +[[ -d $manifests ]] || die "vendored manifests not found at $manifests" +# The manifest workload is only comparable to earlier results when it reads +# exactly the inventory the case was named for: the 21 repositories pinned +# below, each with its snapshot, and no unlisted snapshot. The names are +# pinned, not only the count, because replacing one repository with another +# keeps the count and still changes the workload. A different inventory is a +# different workload: rename or version the case and its baseline, and update +# this list with it, instead of letting the contents drift under one name. +# Snapshot contents are not pinned; refreshing a vendored manifest verbatim +# keeps the inventory. +typeset -a manifest_inventory +manifest_inventory=( any-gem any-node apr asciidoctor brew-completions dircolors-material doctoc ecs-cli firefox-dev fzf fzy github-issues github-issues-srv ls_colors nb pyenv remark subversion system-completions zsh zsh-bin ) +typeset -a manifest_listed manifest_files unpinned unlisted +typeset manifest_name +while IFS= read -r manifest_name; do + [[ -n $manifest_name ]] && manifest_listed+=( "$manifest_name" ) +done < "$manifests/repositories.txt" || die "could not read $manifests/repositories.txt" +# (Ie) matches the name literally; a repository name is data, not a pattern. +for manifest_name in "${manifest_listed[@]}"; do + (( ${manifest_inventory[(Ie)$manifest_name]} )) || unpinned+=( "$manifest_name" ) +done +for manifest_name in "${manifest_inventory[@]}"; do + (( ${manifest_listed[(Ie)$manifest_name]} )) || unlisted+=( "$manifest_name" ) +done +[[ $#unpinned -eq 0 && $#unlisted -eq 0 && ${(j:,:)${(o)manifest_listed}} == ${(j:,:)${(o)manifest_inventory}} ]] || + die "manifest-21 is pinned to a fixed inventory and repositories.txt differs (listed but not pinned: ${(j:, :)unpinned:-none}; pinned but not listed: ${(j:, :)unlisted:-none}). A changed inventory is a different workload: rename or version the case and update the pinned list in run.zsh with it." +manifest_files=( "$manifests"/*.json(N) ) +for manifest_name in "${manifest_inventory[@]}"; do + [[ -r $manifests/$manifest_name.json ]] || die "repositories.txt lists $manifest_name but $manifest_name.json is missing" +done +for manifest_name in "${manifest_files[@]}"; do + (( ${manifest_inventory[(Ie)${manifest_name:t:r}]} )) || die "${manifest_name:t} is not listed in repositories.txt" +done +integer manifest_count=$#manifest_inventory + +typeset work +work=$(command mktemp -d "${TMPDIR:-/tmp}/zi-benchmark.XXXXXXXX") || die "could not create a work directory" +trap 'command rm -rf -- "$work"' EXIT INT TERM +# The reused-home case keeps one home per variant across samples; every +# other case gets a fresh one. +for label in "${labels[@]}"; do command mkdir -p -- "$work/reused-$label"; done + +# reason : the child's last stderr line, or what its exit +# status means when it wrote nothing. case.zsh exits 3 when setup or load +# fails, 4 when the workload ran but left the wrong state, and 5 when the +# manifest inventory did not match the declared count; a bare "exit 4:" in +# a report would say nothing about the failed workload. +reason() { + local line + line=$(command tail -n 1 -- "$2" 2>/dev/null) + if [[ -n $line ]]; then print -r -- "$line"; return; fi + case $1 in + 2) print -r -- "unknown case or missing dependency" ;; + 3) print -r -- "setup or load failed" ;; + 4) print -r -- "postcondition failed: the workload did not leave the expected state" ;; + 5) print -r -- "manifest inventory did not match the declared count" ;; + *) print -r -- "no diagnostic" ;; + esac +} + +# one_sample : prints elapsed milliseconds, "fail ", or +# "unsupported " when the checkout lacks the API the case exercises +# (case.zsh exit 6). +one_sample() { + local label=$1 case=$2 home out rc + if [[ $case == source-reused-home ]]; then home=$work/reused-$label; else home=$(command mktemp -d "$work/s.XXXXXXXX"); fi + command mkdir -p -- "$home" + out=$(env -i PATH="$PATH" HOME="$home" ZDOTDIR="$home" TMPDIR="$work" \ + XDG_DATA_HOME="$home/data" XDG_CACHE_HOME="$home/cache" XDG_CONFIG_HOME="$home/config" \ + BENCH_CHECKOUT="${dirs[$label]}" BENCH_FIXTURES="$fixtures" BENCH_MANIFESTS="$manifests" \ + BENCH_MANIFEST_COUNT="$manifest_count" BENCH_CASE="$case" \ + zsh -f "$here/case.zsh" 2>"$home/stderr"); rc=$? + # Only a successful sample reports its timing. A case prints the elapsed + # time before checking its postcondition, so on failure stdout can hold a + # number that must not reach the report as the start of the reason. + if (( rc == 6 )); then + print -r -- "unsupported $(command tail -n 1 -- "$home/stderr" 2>/dev/null)" + elif (( rc != 0 )); then + print -r -- "fail exit ${rc}: $(reason "$rc" "$home/stderr")" + else + print -r -- "${out##*$'\n'}" + fi + [[ $case == source-reused-home ]] || command rm -rf -- "$home" +} + +# collected[label/case], failed[label/case] and unsupported[label/case] +typeset -A collected failed unsupported +integer round total=$(( warmups + samples )) +typeset -a order variant_order +typeset case value +# Balanced rotation: round r starts the case list one position later than +# round r-1, so over one cycle of $#cases rounds every case occupies every +# position; every second cycle runs in reverse direction as well. Variants +# rotate the same way. A plain reversal every round is not enough for three +# variants: the middle one would be measured second in every round and take +# every first-position warm-up effect for granted. +integer shift_by +for (( round = 1; round <= total; round++ )); do + shift_by=$(( (round - 1) % $#cases )) + order=( "${cases[@][shift_by+1,-1]}" "${cases[@][1,shift_by]}" ) + (( ((round - 1) / $#cases) % 2 )) && order=( "${(Oa)order[@]}" ) + shift_by=$(( (round - 1) % $#labels )) + variant_order=( "${labels[@][shift_by+1,-1]}" "${labels[@][1,shift_by]}" ) + (( ((round - 1) / $#labels) % 2 )) && variant_order=( "${(Oa)variant_order[@]}" ) + for case in "${order[@]}"; do + for label in "${variant_order[@]}"; do + (( ${+failed[$label/$case]} || ${+unsupported[$label/$case]} )) && continue + value=$(one_sample "$label" "$case") || value="fail exit $?" + # An unsupported case is settled on its first sample for that variant and + # is not retried; it is not a failure and does not count toward exit 1. + if [[ $value == unsupported* ]]; then unsupported[$label/$case]=${value#unsupported }; continue; fi + if [[ $value != <->(.<->|) ]]; then failed[$label/$case]=${value#fail }; continue; fi + (( round > warmups )) && collected[$label/$case]+="${value} " + done + done +done + +# Health data that costs nothing extra: sizes, compile durations, symbol +# deltas, measured once per variant outside the sampled rounds. +health_json() { # health_json