diff --git a/.github/workflows/engine-benchmarks.yml b/.github/workflows/engine-benchmarks.yml index f36230953..68275ef0c 100644 --- a/.github/workflows/engine-benchmarks.yml +++ b/.github/workflows/engine-benchmarks.yml @@ -27,6 +27,11 @@ on: - 'scripts/check_engine_json.sh' - '.github/workflows/engine-benchmarks.yml' workflow_dispatch: + inputs: + confirm_publish: + description: 'Publish this run to OpenBenchmarking.org (full-self-hosted lane only). Requires a one-time `phoronix-test-suite openbenchmarking-login` on the runner.' + type: boolean + default: false release: types: [published] @@ -94,8 +99,10 @@ jobs: full-self-hosted: # Full pack on the registered self-hosted runner. Only fires on release # publish or manual dispatch — never on PRs — so untrusted code cannot - # land on the maintainer machine. Results land as artifacts; this - # session does NOT upload to OpenBenchmarking.org (separate milestone). + # land on the maintainer machine. Results always land as artifacts; + # publishing to OpenBenchmarking.org is a separate, explicitly gated + # step below (workflow_dispatch with confirm_publish: true only — a + # release publish alone never auto-publishes to the public leaderboard). if: github.event_name == 'workflow_dispatch' || github.event_name == 'release' runs-on: [self-hosted, linux, x86_64, skainet-bench-linux-x86] timeout-minutes: 120 @@ -137,3 +144,15 @@ jobs: name: engine-full-records-${{ github.run_id }} path: out/engine/**/*.json retention-days: 90 + + - name: Publish to OpenBenchmarking.org + # Explicitly gated — only a manual workflow_dispatch with + # confirm_publish: true reaches this step. A release publish event + # alone runs the benchmarks and uploads artifacts above, but never + # this step, so a bad release tag can't silently publish public + # numbers. Requires a one-time `phoronix-test-suite + # openbenchmarking-login` on this runner — see + # docs/modules/ROOT/pages/contributing/benchmarks.adoc. + if: github.event_name == 'workflow_dispatch' && github.event.inputs.confirm_publish == 'true' + run: | + ./scripts/upload_openbenchmarking.sh "$(ls -td out/engine/*/ | head -1)" --confirm diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/install.sh index f62234037..0a179735f 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-bf16-matmul --out "$HERE/last-result.json" "$@" + run --scenario engine-bf16-matmul --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/test-definition.xml index 326c8b350..c06bea81a 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-bf16-matmul/test-definition.xml @@ -11,16 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-bf16-matmul - --smoke false --provider panama + --provider panama diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/install.sh index 0b8569cde..53d76be74 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-elementwise-add --out "$HERE/last-result.json" "$@" + run --scenario engine-elementwise-add --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/test-definition.xml index af2245aa1..39b3a8e16 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-elementwise-add/test-definition.xml @@ -11,30 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-elementwise-add - --smoke false + - diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/install.sh index fada05563..d2556e9e9 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/install.sh @@ -18,12 +18,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-fp32-gemm --out "$HERE/last-result.json" "$@" + run --scenario engine-fp32-gemm --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/test-definition.xml index ea950d668..a4fd9102d 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-fp32-gemm/test-definition.xml @@ -11,30 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-fp32-gemm - --smoke false + - diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/install.sh index 762c72ef1..8eac43e01 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-kernel-matmul --out "$HERE/last-result.json" "$@" + run --scenario engine-kernel-matmul --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/test-definition.xml index fa7c79449..31fc66a7b 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-kernel-matmul/test-definition.xml @@ -11,16 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-kernel-matmul - --smoke false --provider panama + --provider panama diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/install.sh index 11a7de280..9b5cbbbb0 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-q4-gemm --out "$HERE/last-result.json" "$@" + run --scenario engine-q4-gemm --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/test-definition.xml index e2c2316da..7663d421c 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-q4-gemm/test-definition.xml @@ -11,30 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-q4-gemm - --smoke false + - diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/install.sh index 834f9729d..01ffdfbe7 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-q8-matmul --out "$HERE/last-result.json" "$@" + run --scenario engine-q8-matmul --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/test-definition.xml index 3940d838c..7a22b28db 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-q8-matmul/test-definition.xml @@ -11,16 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-q8-matmul - --smoke false --provider panama + --provider panama diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/install.sh index 2a9c80382..f1f414044 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-reductions-mean --out "$HERE/last-result.json" "$@" + run --scenario engine-reductions-mean --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/test-definition.xml index 06d62de2b..4da8cdc56 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-mean/test-definition.xml @@ -11,30 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-reductions-mean - --smoke false + - diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/install.sh b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/install.sh index 1dd4db18a..9839b9df9 100755 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/install.sh +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/install.sh @@ -16,12 +16,15 @@ fi cp "$SKAINET_PUBLISH_JAR" "$DEST/skainet-engine-publish.jar" -cat > "$DEST/$SCENARIO" <<'WRAPPER' +cat > "$DEST/skainet-$SCENARIO" <<'WRAPPER' #!/usr/bin/env bash +# PTS only parses results from the file at $LOG_FILE (set by the PTS client), never +# from captured stdout directly — so the parseable result line must land there too. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" -exec java --enable-preview --add-modules jdk.incubator.vector \ +java --enable-preview --add-modules jdk.incubator.vector \ -jar "$HERE/skainet-engine-publish.jar" \ - run --scenario engine-reductions-sum --out "$HERE/last-result.json" "$@" + run --scenario engine-reductions-sum --out "$HERE/last-result.json" "$@" \ + | tee -a "${LOG_FILE:-/dev/stdout}" WRAPPER -chmod +x "$DEST/$SCENARIO" +chmod +x "$DEST/skainet-$SCENARIO" diff --git a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/test-definition.xml b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/test-definition.xml index 35acb5c5a..2fe900a51 100644 --- a/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/test-definition.xml +++ b/benchmarks/openbenchmarking/profiles/skainet-engine-reductions-sum/test-definition.xml @@ -11,30 +11,15 @@ 1.0.0 Linux, MacOSX - Library + Benchmark Processor - MIT - Testing + Free + Unverified SKaiNET Developers - skainet-engine-reductions-sum - --smoke false + - diff --git a/docs/modules/ROOT/pages/contributing/benchmarks.adoc b/docs/modules/ROOT/pages/contributing/benchmarks.adoc index 01a6e501d..69a2ed12d 100644 --- a/docs/modules/ROOT/pages/contributing/benchmarks.adoc +++ b/docs/modules/ROOT/pages/contributing/benchmarks.adoc @@ -73,10 +73,17 @@ The full lane currently runs on a Linux x86 host with an AVX2-capable CPU. macOS Arm64 and Linux Arm64 lanes are tracked as follow-ups. +Both triggers (`release`, `workflow_dispatch`) run the full lane and +upload its JSON records as a build artifact — that alone never +touches OpenBenchmarking.org. Actually publishing to the public +leaderboard is a separate, explicitly gated step; see +<>. + == Reproducing a public run locally -Prerequisites: JDK 21 or newer, Phoronix Test Suite (optional, only -required to validate the local PTS profiles). +Prerequisites: JDK 21 or newer, Phoronix Test Suite (required to +actually publish a run — see below; optional if you only want the raw +JSON records). [source,bash] ---- @@ -112,6 +119,64 @@ GH_RUNNER_TOKEN= Actions -> Runners> \ ./scripts/register_bench_runner.sh ---- +[#publishing-to-openbenchmarking-org] +== Publishing to OpenBenchmarking.org + +Publishing is deliberately a separate, human-confirmed step from +running the benchmarks — the full lane runs (and uploads its JSON +artifact) on every `release` and `workflow_dispatch`, but nothing +reaches the public leaderboard without an explicit opt-in each time. +See <> for why the OpenBenchmarking.org login itself +can't be automated either. + +One-time setup on the self-hosted bench host (also covered in +xref:contributing/register-bench-runner.adoc[Register a self-hosted +bench runner]): + +[source,bash] +---- +./scripts/install_pts.sh +phoronix-test-suite openbenchmarking-login # interactive; stores the account session locally +---- + +To publish a full-mode run: + +[source,bash] +---- +./scripts/run_engine_benchmarks.sh +./scripts/upload_openbenchmarking.sh out/engine/ # dry run — prints what it would do +./scripts/upload_openbenchmarking.sh out/engine/ --confirm # actually publishes +---- + +`upload_openbenchmarking.sh` validates the JSON records against the +schema, refuses to proceed if any record is `"unstable": true` (CoV +over `stability.cov_limit_percent`), re-runs the suite through PTS +itself (PTS needs its own native result file to upload — the JSON +harness output is a parallel, richer record, not a PTS result), then +uploads. It only performs the actual upload with `--confirm` (or +`CONFIRM_PUBLISH=yes`) — publishing is public and effectively +irreversible, so it's opt-in every time. + +In CI, this is the `full-self-hosted` job's last step, gated on a +manual `workflow_dispatch` run with `confirm_publish: true`. A +`release` publish event alone runs the benchmarks and uploads the +JSON artifact, but never triggers a publish to OpenBenchmarking.org — +that always requires an explicit, separate dispatch. + +[#why-manual-login] +[TIP] +.Why the OpenBenchmarking.org login stays manual +==== +`phoronix-test-suite openbenchmarking-login` does a live +username/password handshake against OpenBenchmarking.org and stores +the resulting session locally — there's no static API key or token to +hand the runner as a secret. Re-implementing that handshake in CI +would mean the workflow handling account credentials directly; a +one-time interactive login on a box only maintainers can reach is the +safer trade. The login persists across runs, so this is genuinely +one-time per bench host, not per publish. +==== + == Result record schema Every scenario emits a `BenchmarkRecord` JSON (schema version `1.0.0`) @@ -136,5 +201,6 @@ comparisons don't silently break. * {url-pts}[Phoronix Test Suite] * {url-obo}[OpenBenchmarking.org] +* xref:contributing/register-bench-runner.adoc[Register a self-hosted bench runner] — one-time host setup, including OpenBenchmarking.org login * xref:explanation/perf/jvm-cpu.adoc[JVM CPU backend notes] * xref:explanation/perf/simd-kernels.adoc[How SIMD kernels are built] diff --git a/docs/modules/ROOT/pages/contributing/index.adoc b/docs/modules/ROOT/pages/contributing/index.adoc index dfd71f99d..13bf39fe1 100644 --- a/docs/modules/ROOT/pages/contributing/index.adoc +++ b/docs/modules/ROOT/pages/contributing/index.adoc @@ -31,7 +31,7 @@ Concretely: |=== | Page | What it answers | xref:contributing/build-from-source.adoc[Build from source] | How to clone, build, and run the test suite locally. -| xref:contributing/benchmarks.adoc[Engine benchmark program] | The Phoronix Test Suite / OpenBenchmarking publication path: methodology, manifest, lanes, CI workflow, replay. +| xref:contributing/benchmarks.adoc[Engine benchmark program] | The Phoronix Test Suite / OpenBenchmarking publication path: methodology, manifest, lanes, CI workflow, replay, and the gated steps to actually publish a run to OpenBenchmarking.org. | xref:contributing/matmul-kernels.adoc[Reading the matmul benchmark] | What the published numbers mean — scalar vs. Panama vs. quantized regimes, roofline reasoning, and a checklist for interpreting any new measurement. | xref:contributing/register-bench-runner.adoc[Register a self-hosted bench runner] | The one-time operator setup that lights up the full-publish CI lane on a Linux x86 box. | xref:skeep:index.adoc[SKEEP proposal track] | How SKaiNET records long-lived API, DSL, runtime, compiler, storage, and compatibility proposals. diff --git a/docs/modules/ROOT/pages/contributing/register-bench-runner.adoc b/docs/modules/ROOT/pages/contributing/register-bench-runner.adoc index c41220a8a..492d44730 100644 --- a/docs/modules/ROOT/pages/contributing/register-bench-runner.adoc +++ b/docs/modules/ROOT/pages/contributing/register-bench-runner.adoc @@ -38,6 +38,10 @@ workflow's `runs-on:` clause can route the job to it. * Optional but recommended: Phoronix Test Suite, installed via `./scripts/install_pts.sh`. The benchmark workflow installs it itself if missing, but pre-installing avoids a per-run download. +* Only if this box will publish to OpenBenchmarking.org: a free + OpenBenchmarking.org account, and `phoronix-test-suite + openbenchmarking-login` run once, interactively, on this host (see + <>). [#nat-firewall] == Running behind NAT — what you don't need to configure @@ -160,6 +164,37 @@ Within ~30 seconds the runner journal should pick up the job. The full lane completes in 10–20 minutes; the published artifacts land at `engine-full-records-` on the workflow run page. +[#openbenchmarking-login] +=== 5. (Optional) Enable OpenBenchmarking.org publishing + +The full lane above always runs benchmarks and uploads a JSON +artifact — it never touches OpenBenchmarking.org by itself. Actually +publishing to the public leaderboard needs one more, separate +one-time step on this host, since the workflow itself is never +handed OpenBenchmarking.org credentials (see +xref:contributing/benchmarks.adoc#why-manual-login[why the login +stays manual]): + +[source,bash] +---- +phoronix-test-suite openbenchmarking-login +---- + +This prompts for your OpenBenchmarking.org username and password +once and stores the resulting session under +`~/.phoronix-test-suite/`. After that, publishing is a per-run +opt-in — trigger the workflow with `confirm_publish: true`: + +[source,bash] +---- +gh workflow run engine-benchmarks.yml --ref develop -f confirm_publish=true +---- + +Full details, including the local (non-CI) publish path via +`scripts/upload_openbenchmarking.sh`, are in +xref:contributing/benchmarks.adoc#publishing-to-openbenchmarking-org[Publishing +to OpenBenchmarking.org]. + == Optional hardening The default registration script runs the agent as the invoking user diff --git a/scripts/upload_openbenchmarking.sh b/scripts/upload_openbenchmarking.sh new file mode 100755 index 000000000..15bb06564 --- /dev/null +++ b/scripts/upload_openbenchmarking.sh @@ -0,0 +1,138 @@ +#!/usr/bin/env bash +# Publish a full-mode engine benchmark run to OpenBenchmarking.org via Phoronix Test Suite. +# +# One-time setup per bench host (not handled by this script — see +# docs/modules/ROOT/pages/contributing/benchmarks.adoc): +# 1. ./scripts/install_pts.sh +# 2. phoronix-test-suite openbenchmarking-login +# (interactive; stores the account session locally — this script never sees or +# handles OpenBenchmarking.org credentials) +# +# What this script does: +# 1. Validates a full-mode JSON run (from scripts/run_engine_benchmarks.sh) against the +# BenchmarkRecord schema and refuses to proceed if any record is "unstable": true +# (CoV > benchmarks/manifests/engine-release.yml's stability.cov_limit_percent). +# Re-run on a quiet, dedicated host if this gate fails. +# 2. Syncs + statically validates the local PTS profiles/suite (scripts/validate_pts_profiles.sh). +# 3. Re-runs the suite through PTS itself (`phoronix-test-suite batch-benchmark`) — PTS needs +# to produce its own native result to upload; the JSON from step 1 is a parallel, richer +# record kept for schema validation and historical diffing, not a PTS result file. Batch +# mode is configured here (non-interactively, upload OFF) rather than depending on a +# prior `phoronix-test-suite batch-setup` on the host. +# 4. Uploads the resulting PTS result to OpenBenchmarking.org — a public, essentially +# irreversible action, so it only runs with --confirm (or CONFIRM_PUBLISH=yes); without +# it, this prints what it would do and exits. +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "$0")/.." && pwd)" + +usage() { + sed -n '2,25p' "$0" + echo + echo "Usage: $0 dir> [--confirm]" +} + +JSON_DIR="${1:-}" +if [ -z "$JSON_DIR" ] || [ "$JSON_DIR" = "-h" ] || [ "$JSON_DIR" = "--help" ]; then + usage + exit "$([ -z "$JSON_DIR" ] && echo 2 || echo 0)" +fi + +CONFIRM="${CONFIRM_PUBLISH:-no}" +[ "${2:-}" = "--confirm" ] && CONFIRM="yes" + +RESULT_IDENTIFIER="${RESULT_IDENTIFIER:-skainet-engine-$(basename "$JSON_DIR")}" + +if ! command -v phoronix-test-suite >/dev/null 2>&1; then + echo "ERROR: phoronix-test-suite not installed. Run ./scripts/install_pts.sh first." >&2 + exit 2 +fi + +echo "== Stage 1: validate the JSON harness run at $JSON_DIR ==" +"$REPO_ROOT/scripts/check_engine_json.sh" "$JSON_DIR" + +UNSTABLE="$(python3 - "$JSON_DIR" <<'PY' +import glob, json, os, sys +d = sys.argv[1] +bad = [p for p in sorted(glob.glob(os.path.join(d, "*.json"))) if json.load(open(p)).get("unstable")] +print("\n".join(bad)) +PY +)" +if [ -n "$UNSTABLE" ]; then + echo "ERROR: unstable records present (CoV over the manifest's stability limit) —" >&2 + echo " re-run on a quiet, dedicated bench host before publishing:" >&2 + echo "$UNSTABLE" >&2 + exit 1 +fi +echo "Stability gate: OK — no unstable records." + +echo +echo "== Stage 2: sync + validate local PTS profiles and suite ==" +"$REPO_ROOT/scripts/validate_pts_profiles.sh" + +echo +echo "== Stage 3: configure PTS batch mode (non-interactive, upload OFF for this stage) ==" +python3 - <<'PY' +import os +import xml.etree.ElementTree as ET + +path = os.path.expanduser("~/.phoronix-test-suite/user-config.xml") +os.makedirs(os.path.dirname(path), exist_ok=True) + +if os.path.isfile(path): + tree = ET.parse(path) + root = tree.getroot() +else: + root = ET.Element("PhoronixTestSuite") + tree = ET.ElementTree(root) + +def ensure_path(parent, *tags): + for tag in tags: + child = parent.find(tag) + if child is None: + child = ET.SubElement(parent, tag) + parent = child + return parent + +batch = ensure_path(root, "Options", "BatchMode") +values = { + "Configured": "TRUE", + "SaveResults": "TRUE", + "OpenBrowser": "FALSE", + "UploadResults": "FALSE", # upload is Stage 4, explicit and gated — never a side effect here + "PromptForTestIdentifier": "FALSE", + "PromptForTestDescription": "FALSE", + "PromptSaveName": "FALSE", + "RunAllTestCombinations": "TRUE", # runs both panama/scalar for the 3 provider-optioned profiles +} +for tag, value in values.items(): + ensure_path(batch, tag).text = value + +tree.write(path, encoding="unicode", xml_declaration=False) +print(f"BatchMode configured in {path}") +PY + +echo +echo "== Stage 4: run the suite through PTS (produces the native result PTS can upload) ==" +echo "Result identifier: $RESULT_IDENTIFIER" +if [ "$CONFIRM" != "yes" ]; then + echo + echo "DRY RUN — not executing. This would run:" + echo " TEST_RESULTS_NAME=$RESULT_IDENTIFIER phoronix-test-suite batch-benchmark local/skainet-engine-suite" + echo " phoronix-test-suite upload-result $RESULT_IDENTIFIER" + echo + echo "Re-run with --confirm (or CONFIRM_PUBLISH=yes) to actually publish to OpenBenchmarking.org." + exit 0 +fi + +TEST_RESULTS_NAME="$RESULT_IDENTIFIER" \ +TEST_RESULTS_DESCRIPTION="SKaiNET engine suite — full mode, published from $JSON_DIR" \ + phoronix-test-suite batch-benchmark local/skainet-engine-suite < /dev/null + +echo +echo "== Stage 5: upload to OpenBenchmarking.org ==" +echo "Requires a one-time 'phoronix-test-suite openbenchmarking-login' on this host." +phoronix-test-suite upload-result "$RESULT_IDENTIFIER" + +echo +echo "Published: $RESULT_IDENTIFIER"