From dc83a09938a5706582660b8f57c9df0fe8a79d3d Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Thu, 1 Oct 2026 00:06:16 +0800 Subject: [PATCH 1/5] Select PR checks by affected consumers and unify the required gate --- .github/actions/node/action.yml | 24 ++++ .github/workflows/actionlint.yml | 18 ++- .github/workflows/api-acceptance.yml | 43 ++----- .github/workflows/check.yml | 154 ++++++++++++++++++------ .github/workflows/native.yml | 71 ++++++----- .github/workflows/release.yml | 13 +-- CONTRIBUTING.md | 4 +- Makefile | 9 +- docs/maintainers.md | 51 +++++--- scripts/ci_metrics.py | 76 ++++++++++++ scripts/ci_metrics_test.py | 28 +++++ scripts/ci_plan.py | 169 +++++++++++++++++++++++++++ scripts/ci_plan_test.py | 156 +++++++++++++++++++++++++ 13 files changed, 679 insertions(+), 137 deletions(-) create mode 100644 .github/actions/node/action.yml create mode 100644 scripts/ci_metrics.py create mode 100644 scripts/ci_metrics_test.py create mode 100644 scripts/ci_plan.py create mode 100644 scripts/ci_plan_test.py diff --git a/.github/actions/node/action.yml b/.github/actions/node/action.yml new file mode 100644 index 00000000..a319ca60 --- /dev/null +++ b/.github/actions/node/action.yml @@ -0,0 +1,24 @@ +name: Node and pnpm store +description: Set up the pinned package manager and cache downloaded packages. +inputs: + node-version: + description: Node.js version + default: '22' +runs: + using: composite + steps: + - uses: actions/setup-node@v6 + with: + node-version: ${{ inputs.node-version }} + package-manager-cache: false + - name: Install pinned pnpm + shell: bash + run: npm install --global pnpm@10.30.3 + - name: Locate pnpm store + id: store + shell: bash + run: echo "path=$(pnpm store path --silent)" >> "$GITHUB_OUTPUT" + - uses: actions/cache@v6 + with: + path: ${{ steps.store.outputs.path }} + key: pnpm-v1-${{ runner.os }}-${{ runner.arch }}-node${{ inputs.node-version }}-10.30.3-${{ hashFiles('pnpm-lock.yaml') }} diff --git a/.github/workflows/actionlint.yml b/.github/workflows/actionlint.yml index 9926bc7f..8ab64f0b 100644 --- a/.github/workflows/actionlint.yml +++ b/.github/workflows/actionlint.yml @@ -11,17 +11,11 @@ name: actionlint on: - push: - branches: [main] - paths: - - '.github/workflows/**' - pull_request: - paths: - - '.github/workflows/**' - -concurrency: - group: ${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true + workflow_call: + inputs: + ref: + type: string + required: true permissions: contents: read @@ -32,6 +26,8 @@ jobs: timeout-minutes: 5 steps: - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref }} - name: Download actionlint run: bash <(curl -sSf https://raw.githubusercontent.com/rhysd/actionlint/main/scripts/download-actionlint.bash) 1.7.12 - name: Run actionlint diff --git a/.github/workflows/api-acceptance.yml b/.github/workflows/api-acceptance.yml index 2502be03..2faff478 100644 --- a/.github/workflows/api-acceptance.yml +++ b/.github/workflows/api-acceptance.yml @@ -1,41 +1,19 @@ name: api-acceptance on: - push: - branches: [main] - paths: - - 'services/core/**' - - 'contracts/agents-api/**' - - 'packages/agents-client/**' - - 'internal/**' - - 'go.mod' - - 'go.sum' - - 'Makefile' - - 'scripts/build-core.sh' - - 'scripts/build-core-image-context.sh' - - 'deploy/distribution/Dockerfile' - - '.github/workflows/api-acceptance.yml' - pull_request: - paths: - - 'services/core/**' - - 'contracts/agents-api/**' - - 'packages/agents-client/**' - - 'internal/**' - - 'go.mod' - - 'go.sum' - - 'Makefile' - - 'scripts/build-core.sh' - - 'scripts/build-core-image-context.sh' - - 'deploy/distribution/Dockerfile' - - '.github/workflows/api-acceptance.yml' + workflow_call: + inputs: + ref: + type: string + required: true + container: + description: Verify the Core image as well as standalone commands + type: boolean + default: true permissions: contents: read -concurrency: - group: oac-core-${{ github.ref }} - cancel-in-progress: true - jobs: official-client: runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-24.04' || 'blacksmith-2vcpu-ubuntu-2404' }} @@ -56,6 +34,8 @@ jobs: --health-retries 10 steps: - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref }} - uses: actions/setup-go@v7 with: go-version-file: go.mod @@ -88,6 +68,7 @@ jobs: python services/core/tests/official_client.py go test ./services/core/internal/store -run '^(TestFunctionStateOfficialClientReadsAndLiveEvents|TestSavedReferenceRetryOfficialClient|TestAgentUpdateOfficialClient|TestAgentDeletionOfficialClient|TestSessionAgentFilterOfficialClient|TestSessionDeletionOfficialClient|TestEnvironmentInitialFailureOfficialClient|TestSelfHostedInitialCreationOfficialClient|TestSelfHostedCancellationOfficialClient|TestSelfHostedFunctionsOfficialClient|TestSelfHostedSteeringOfficialClient)$' -count=1 - name: Verify the distribution's Core image + if: inputs.container env: OAC_TEST_DATABASE_URL: postgres://agents_api:agents_api_test_only@127.0.0.1:${{ job.services.postgres.ports['5432'] }}/oac_ci_tests?sslmode=disable OAC_DEV_CORE_IMAGE: oac-core:ci diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index 26a91cb3..c0aa0ae5 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -7,6 +7,10 @@ on: description: Source commit to check required: false type: string + native-artifacts: + description: Retain native installers for a release build + type: boolean + default: false push: branches: [main] pull_request: @@ -19,7 +23,39 @@ concurrency: cancel-in-progress: true jobs: + plan: + runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} + timeout-minutes: 5 + outputs: + plan: ${{ steps.select.outputs.plan }} + jobs: ${{ steps.select.outputs.jobs }} + image: ${{ steps.select.outputs.image }} + steps: + - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref || github.sha }} + fetch-depth: 2 + - name: Select checks from the integrated change + id: select + env: + REQUESTED_REF: ${{ inputs.ref }} + run: python3 scripts/ci_plan.py plan + + hygiene: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'hygiene') + runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} + timeout-minutes: 5 + steps: + - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref || github.sha }} + - name: Verify names, documentation and CI selection invariants + run: make check-names check-docs check-ci + backend: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'backend') runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} timeout-minutes: 40 services: @@ -83,7 +119,9 @@ jobs: - name: Build execution daemon run: make build-daemon - tooling: + distribution: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'distribution') runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} timeout-minutes: 20 steps: @@ -105,48 +143,75 @@ jobs: path: | ~/.oac/cache/go-build ~/.oac/cache/go-mod - key: core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-tooling-${{ steps.source.outputs.revision }} + key: core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-distribution-${{ steps.source.outputs.revision }} restore-keys: | - core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-tooling- + core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-distribution- core-go-v1-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}- - uses: actions/setup-node@v6 with: node-version: '22' - - name: Install pinned pnpm - run: npm install --global pnpm@10.30.3 + - name: Verify Harness catalog and installer schema + run: make check-harness-catalog + - name: Test distribution, installer and console packaging + run: make check-distribution + + harness: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'harness') + runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} + timeout-minutes: 20 + steps: + - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref || github.sha }} + - uses: ./.github/actions/node + - name: Test and package Harness adapters + run: make check-claude-sdk check-mcode-harness + + example: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'example') + runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} + timeout-minutes: 15 + steps: + - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref || github.sha }} + - uses: ./.github/actions/node - name: Install dependencies and example acceptance browser run: | make node-deps pnpm exec playwright install --with-deps chrome - - name: Verify Harness catalog - run: make check-harness-catalog - - name: Verify repository names - run: make check-names - - name: Test distribution, installer and console packaging - run: make check-distribution - - name: Test and package Claude SDK adapter - run: make check-claude-sdk - name: Test and build optional example with browser acceptance run: make check-example - - name: Verify MiniMax companion scripts - run: make check-mcode-harness + - name: Upload browser failure evidence + if: failure() + uses: actions/upload-artifact@v6 + with: + name: example-acceptance-${{ github.run_attempt }} + path: | + example/*/playwright-report/ + example/*/test-results/ + retention-days: 7 + compression-level: 0 + if-no-files-found: ignore web: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'web') runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} timeout-minutes: 15 steps: - uses: actions/checkout@v7 with: ref: ${{ inputs.ref || github.sha }} - - uses: actions/setup-node@v6 - with: - node-version: '22' - - name: Install pinned pnpm - run: npm install --global pnpm@10.30.3 + - uses: ./.github/actions/node - name: Typecheck, test and build Web and clients run: make check-web-unit web-acceptance: + needs: [plan, web] + if: needs.web.result == 'success' && needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'web-acceptance') strategy: fail-fast: false matrix: @@ -157,11 +222,7 @@ jobs: - uses: actions/checkout@v7 with: ref: ${{ inputs.ref || github.sha }} - - uses: actions/setup-node@v6 - with: - node-version: '22' - - name: Install pinned pnpm - run: npm install --global pnpm@10.30.3 + - uses: ./.github/actions/node - name: Install dependencies and Web acceptance browser run: | make node-deps @@ -180,22 +241,41 @@ jobs: compression-level: 0 if-no-files-found: ignore - # Preserve the required check name, and fail closed for skipped/cancelled jobs. + api: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'api') + uses: ./.github/workflows/api-acceptance.yml + with: + ref: ${{ inputs.ref || github.sha }} + container: ${{ needs.plan.outputs.image == 'true' }} + + native: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'native') + uses: ./.github/workflows/native.yml + with: + ref: ${{ inputs.ref || github.sha }} + upload-artifacts: ${{ inputs.native-artifacts || false }} + + lint: + needs: plan + if: needs.plan.result == 'success' && contains(fromJSON(needs.plan.outputs.jobs || '[]'), 'lint') + uses: ./.github/workflows/actionlint.yml + with: + ref: ${{ inputs.ref || github.sha }} + + # Always report the required check, even when planning or a dependency fails. check: if: always() - needs: [backend, tooling, web, web-acceptance] + needs: [plan, hygiene, distribution, backend, harness, example, web, web-acceptance, api, native, lint] runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} timeout-minutes: 5 steps: - - name: Require every full-gate partition to succeed + - uses: actions/checkout@v7 + with: + ref: ${{ inputs.ref || github.sha }} + - name: Require every selected check to succeed env: + PLAN: ${{ needs.plan.outputs.plan }} RESULTS: ${{ toJSON(needs) }} - run: | - python3 - <<'PY' - import json, os, sys - results = json.loads(os.environ['RESULTS']) - failed = [name for name, job in results.items() if job['result'] != 'success'] - if failed: - sys.exit('Full gate did not pass: ' + ', '.join(failed)) - print('OpenAgentCore checks passed.') - PY + run: python3 scripts/ci_plan.py gate diff --git a/.github/workflows/native.yml b/.github/workflows/native.yml index 6b63cecd..8cc529cf 100644 --- a/.github/workflows/native.yml +++ b/.github/workflows/native.yml @@ -1,47 +1,33 @@ name: native-check on: - pull_request: - paths: - - 'apps/daemon/**' - - 'internal/**' - - 'contracts/agents-api/**' - - 'packages/claude-sdk-adapter/**' - - 'packages/mcode-harness/**' - - 'packages/tsconfig/**' - - 'scripts/build-native-installer*' - - 'scripts/native-harness-smoke.mjs' - - 'scripts/native-onboarding-smoke.mjs' - - 'services/core/internal/nativeinstaller/**' - - 'scripts/build-mcode-harness.sh' - - 'go.mod' - - 'go.sum' - - 'go.work' - - 'go.work.sum' - - 'package.json' - - 'pnpm-lock.yaml' - - 'pnpm-workspace.yaml' - - 'tsconfig.base.json' - - '.npmrc' - - '.github/workflows/native.yml' workflow_dispatch: + inputs: + upload-artifacts: + description: Retain the three native installers for packaging + type: boolean + default: true workflow_call: inputs: ref: type: string required: true + upload-artifacts: + type: boolean + default: false permissions: contents: read concurrency: - group: native-check-${{ github.ref }} + group: native-check-${{ github.workflow }}-${{ github.run_id }} cancel-in-progress: true jobs: platform: strategy: fail-fast: false + max-parallel: 2 matrix: include: - platform: linux-amd64 @@ -66,14 +52,16 @@ jobs: - uses: actions/setup-go@v7 with: go-version-file: go.mod - - uses: actions/setup-node@v6 + - uses: ./.github/actions/node with: node-version: '22.22.0' - name: Build the native daemon and test bundle boundaries + id: build run: | go build -ldflags "-X github.com/MiniMax-AI/OpenAgentCore/apps/daemon/internal/cli.Version=$(git rev-parse HEAD)" -o "$RUNNER_TEMP/oac-daemon${{ runner.os == 'Windows' && '.exe' || '' }}" ./apps/daemon/cmd/oac-daemon node --test scripts/build-native-installer.test.mjs - name: Native filesystem, authentication and process lifecycle + id: filesystem run: >- go test -race -count=1 ./internal/runtimefs @@ -82,6 +70,7 @@ jobs: ./apps/daemon/internal/daemonize ./apps/daemon/internal/agent/clirunner - name: Native Harness ownership and environment identity + id: harness run: >- go test -race -count=1 ./apps/daemon/internal/agent/codex @@ -89,6 +78,7 @@ jobs: ./apps/daemon/internal/agent/mcode -run 'TestJSONRPCClientOwnsToolDescendants|TestSelfHostedToolEnvironment' - name: Native capability snapshots, files and installation + id: files run: >- go test -race -count=1 ./apps/daemon/internal/localworkspace @@ -103,7 +93,6 @@ jobs: -run 'TestWindowsPackageManager|TestWindowsRuntimeMCP' - name: Build pinned native components run: | - npm install --global pnpm@10.30.3 pnpm install --frozen-lockfile --filter @oac/claude-sdk-adapter... pnpm --filter @oac/claude-sdk-adapter build --outDir "$RUNNER_TEMP/claude-compiled" pnpm --config.inject-workspace-packages=true --config.extend-node-path=false --filter @oac/claude-sdk-adapter deploy --prod "$RUNNER_TEMP/claude-runtime" @@ -130,25 +119,55 @@ jobs: MCODE_HARNESS_BUILD_DIR="$RUNNER_TEMP/minimax-runtime" \ bash scripts/build-mcode-harness.sh - name: Package and exercise CLI-only installation, additions and reuse + id: package run: | if [[ "$RUNNER_OS" == Windows ]]; then export CLAUDE_CODE_GIT_BASH_PATH='C:\Program Files\Git\bin\bash.exe' fi node scripts/build-native-installer-ci.mjs - name: Bootstrap, authenticate, install and connect natively + id: onboarding run: node scripts/native-onboarding-smoke.mjs - name: Verify native Harness protocols without model requests + id: protocol run: | if [[ "$RUNNER_OS" == Windows ]]; then export CLAUDE_CODE_GIT_BASH_PATH='C:\Program Files\Git\bin\bash.exe' fi node scripts/native-harness-smoke.mjs --codex-binary "$RUNNER_TEMP/native-installer/components/codex/bin/codex${{ runner.os == 'Windows' && '.exe' || '' }}" --claude-runtime "$RUNNER_TEMP/native-installer/components/claude" - name: Archive the native distribution + if: inputs.upload-artifacts run: | cd "$RUNNER_TEMP" tar -C native-installer -czf "oac-native-installer-${{ runner.os }}-${{ runner.arch }}.tar.gz" . - uses: actions/upload-artifact@v6 + if: inputs.upload-artifacts with: name: oac-native-installer-${{ runner.os }}-${{ runner.arch }} path: ${{ runner.temp }}/oac-native-installer-*.tar.gz if-no-files-found: error + retention-days: 7 + compression-level: 0 + - name: Record failed native check phases + if: failure() + env: + PHASES: ${{ toJSON(steps) }} + PLATFORM: ${{ matrix.platform }} + run: | + node --input-type=module <<'JS' + import { mkdirSync, writeFileSync } from 'node:fs'; + import { join } from 'node:path'; + const dir = join(process.env.RUNNER_TEMP, 'native-ci-diagnostics'); + mkdirSync(dir, { recursive: true }); + const phases = Object.fromEntries(Object.entries(JSON.parse(process.env.PHASES)) + .map(([name, step]) => [name, step.outcome])); + writeFileSync(join(dir, 'phases.json'), JSON.stringify({ platform: process.env.PLATFORM, phases }, null, 2)); + JS + - name: Retain native failure diagnostics + if: failure() + uses: actions/upload-artifact@v6 + with: + name: native-diagnostics-${{ matrix.platform }}-${{ github.run_attempt }} + path: ${{ runner.temp }}/native-ci-diagnostics/ + retention-days: 7 + if-no-files-found: ignore diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 141f1700..97f3c14d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -30,14 +30,10 @@ jobs: uses: ./.github/workflows/check.yml with: ref: ${{ inputs.ref || github.sha }} - - native: - uses: ./.github/workflows/native.yml - with: - ref: ${{ inputs.ref || github.sha }} + native-artifacts: true build: - needs: native + needs: check runs-on: ${{ vars.OAC_USE_GITHUB_RUNNERS == 'true' && 'ubuntu-22.04' || 'blacksmith-2vcpu-ubuntu-2204' }} timeout-minutes: 120 outputs: @@ -88,12 +84,9 @@ jobs: core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-release- core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-backend- core-go-v1-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}- - - uses: actions/setup-node@v6 - with: - node-version: '22' + - uses: ./.github/actions/node - name: Install build prerequisites run: | - npm install --global pnpm@10.30.3 sudo apt-get update sudo apt-get install -y build-essential pkg-config libssl-dev - name: Check release metadata and prepare pinned harnesses diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 4b18c3f7..4e890f91 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -99,9 +99,9 @@ Do not use `codex exec` as a substitute reviewer. Toolchain setup and focused commands are in [Develop OpenAgentCore](docs/development.md#set-up-a-checkout). CI coverage, caches and release publication are owned by the [maintainer guide](docs/maintainers.md#publish-a-version). -### Full gate +### Checks for a change -Run `make check` before completing code changes. The `check` target in the [Makefile](Makefile) is the authoritative list of gates. [Focused validation](docs/development.md#validate-a-change) selects checks for development; [live acceptance](#live-acceptance) qualifies native execution beyond fixtures and builds. +Run the checks for the changed behavior and its consumers before completion, including database migration and cross-component tests when affected. Use the [CI selection policy](docs/maintainers.md#continuous-integration) to determine the relevant groups; record what passed and any validation limits. A small follow-up needs its relevant checks, not another unrelated full run. The [Makefile](Makefile) retains `make check` as the complete local gate; main and release CI run all groups, and releases require the full gate. [Live acceptance](#live-acceptance) qualifies native execution beyond fixtures and builds. ### Test database diff --git a/Makefile b/Makefile index f0dea91d..b43b111f 100644 --- a/Makefile +++ b/Makefile @@ -8,7 +8,7 @@ SWAG_VERSION ?= v1.16.4 help: @printf '%s\n' 'make build-core Build standalone Core commands' 'make build-daemon Build the execution daemon' 'make check Run Core, persistence and runtime checks' 'See README.md for runtime prerequisites and deployment.' -check: check-harness-catalog check-names check-distribution check-database check-sqlc check-go check-microsandbox-provider check-core check-claude-sdk check-web check-example check-mcode-harness +check: check-ci check-harness-catalog check-names check-distribution check-database check-sqlc check-go check-microsandbox-provider check-core check-claude-sdk check-web check-example check-mcode-harness @printf 'OpenAgentCore checks passed.\n' .PHONY: generate-harness-catalog check-harness-catalog @@ -175,3 +175,10 @@ check-e2b-provider: .PHONY: check-sandbox-provider-contract check-sandbox-provider-contract: go test ./services/core/internal/sandbox/... -count=1 + +.PHONY: check-docs check-ci +check-docs: + PYTHONDONTWRITEBYTECODE=1 python3 scripts/core-distribution-manifest.test.py + +check-ci: + PYTHONDONTWRITEBYTECODE=1 python3 -m unittest discover -s scripts -p 'ci_*test.py' diff --git a/docs/maintainers.md b/docs/maintainers.md index 47c3c8c8..9d08b64a 100644 --- a/docs/maintainers.md +++ b/docs/maintainers.md @@ -39,7 +39,7 @@ A distribution carries the docs listed in `BUNDLED_DOCS` in `scripts/core-distri ### Native installers -Self-hosted machines install `oac-daemon` from per-platform native installers: Linux amd64, macOS arm64 and Windows amd64. Each is built on its own OS by the `native-check` workflow (`scripts/build-native-installer.mjs`, whose `pins` object fixes the Node.js and Harness versions) and uploaded as `oac-native-installer--.tar.gz`. For a local distribution, download the three artifacts from a `native-check` run on that exact commit (a manual run or the release run; pull-request runs build the merge commit and do not match), then assemble the catalog from that checkout: +Self-hosted machines install `oac-daemon` from per-platform native installers: Linux amd64, macOS arm64 and Windows amd64. Each is built on its own OS by the `native-check` workflow (`scripts/build-native-installer.mjs`, whose `pins` object fixes the Node.js and Harness versions) and verified on every selected native check. Manual packaging runs and release checks upload `oac-native-installer--.tar.gz` for seven days; ordinary PR and main checks do not upload successful packages. For a local distribution, download the three artifacts from a `native-check` run on that exact commit (a manual run or the release run; pull-request runs build the merge commit and do not match), then assemble the catalog from that checkout: ```sh node scripts/build-native-catalog.mjs INPUT_DIR OUTPUT_DIR @@ -109,7 +109,7 @@ The helper is written to `~/.oac/build/microsandbox-provider/oac-microsandbox-pr `make build-core` builds `oac-core`, `oac-core-migrate`, `oac-core-device`, `oac-core-environment-key` and `oac-node` into `${OAC_DEV_HOME:-$HOME/.oac}/build/oac-core` (`OAC_DEV_CORE_BUILD_DIR` selects another absolute directory). The build copies only the source set listed in `scripts/build-core.sh` (the Core service, its contracts, the shared packages it needs and the root Go module files) into a temporary context and builds with CGO disabled, read-only modules and trimmed paths. It needs no Node, Docker or other application. When Core gains a shared dependency, add that package to the list; never copy the whole repository to make it compile. -`make docker-build-core` builds the image `oac-core:dev` (`OAC_DEV_CORE_IMAGE` selects another name) from those five commands and the E2B helper. The base is the digest-pinned `debian:bookworm-slim` with CA certificates and the glibc runtime the helper needs; the default user is UID/GID 65532 and Core listens on `:8091`. The image is Linux amd64 only and is not pushed to a registry. Changes to the image or its build need `make check-core-container` in addition to `make check`: it runs the official-client suite against the image with a read-only root filesystem and needs Linux Docker, a non-root user, and the [test database and pinned SDK](../services/core/README.md#official-client-verification) of the service checks (`OAC_TEST_DATABASE_URL` naming an `oac_*_tests` database with the migrations applied, and `OAC_TEST_OFFICIAL_SDK_PYTHON`). +`make docker-build-core` builds the image `oac-core:dev` (`OAC_DEV_CORE_IMAGE` selects another name) from those five commands and the E2B helper. The base is the digest-pinned `debian:bookworm-slim` with CA certificates and the glibc runtime the helper needs; the default user is UID/GID 65532 and Core listens on `:8091`. The image is Linux amd64 only and is not pushed to a registry. Changes to the image or its build need `make check-core-container` in addition to the relevant source checks: it runs the official-client suite against the image with a read-only root filesystem and needs Linux Docker, a non-root user, and the [test database and pinned SDK](../services/core/README.md#official-client-verification) of the service checks (`OAC_TEST_DATABASE_URL` naming an `oac_*_tests` database with the migrations applied, and `OAC_TEST_OFFICIAL_SDK_PYTHON`). ## Publish a version @@ -122,7 +122,7 @@ git push origin v1.2.3 Tags use `vMAJOR.MINOR.PATCH`, optionally with a prerelease suffix such as `-rc.1` and build metadata such as `+build.1`. A prerelease suffix creates a GitHub prerelease. Pushing the tag is the release decision. Automated checks establish build and test results, not real-model qualification: assess live execution evidence before you push the tag. Model credentials and private certificate authorities never enter CI or release inputs, including acceptance images that contain them. -The workflow runs three jobs on the tagged commit: `check` (the full `make check` workflow), `native` (the `native-check` matrix) and `build`, which starts after `native` succeeds. `build` prepares the pinned Runtime inputs, assembles the native catalog and builds the distribution with the offline archive, and adds `deploy/install-release.sh` as `install.sh` with its checksum. The `release` job runs only after `check` and `build` succeed. It is the only job with `contents: write`. It verifies the archive checksums and the native installer checksums against the catalog, resolves the repository's current name from GitHub before any write (Actions can keep an old name after a rename), refuses an existing Release or draft for the tag, uploads everything to a new draft on `uploads.github.com` bound to that draft's ID without retrying failed uploads, confirms the tag still points at the built commit, and publishes that draft by its ID. Images ship as archives; no registry is pushed. Downloads are anonymous. +The workflow runs `check` on the tagged commit, including the full local gate, official-client and image acceptance, and the native matrix with its packaging artifacts enabled. `build` starts after `check` succeeds and reuses those native artifacts. `build` prepares the pinned Runtime inputs, assembles the native catalog and builds the distribution with the offline archive, and adds `deploy/install-release.sh` as `install.sh` with its checksum. The `release` job runs only after `check` and `build` succeed. It is the only job with `contents: write`. It verifies the archive checksums and the native installer checksums against the catalog, resolves the repository's current name from GitHub before any write (Actions can keep an old name after a rename), refuses an existing Release or draft for the tag, uploads everything to a new draft on `uploads.github.com` bound to that draft's ID without retrying failed uploads, confirms the tag still points at the built commit, and publishes that draft by its ID. Images ship as archives; no registry is pushed. Downloads are anonymous. `install.sh` resolves the latest stable release once, or the release named by `--version`, verifies the control archive and runs that bundle's installer; the [installation guide](getting-started/install.md#install) covers its use. @@ -144,26 +144,39 @@ With `draft_release=true` the result is an unpublished `build-` draft ## Continuous integration -| Workflow | Runs on | Covers | -| --- | --- | --- | -| `core-check` (`check.yml`) | Pushes to `main`, every pull request, releases | All `make check` checks in concurrent partitions, plus a daemon build; see the partitions below | -| `api-acceptance` | Pushes to `main` and pull requests that touch Core, its contracts, clients, shared Go code or build scripts | Standalone commands and migration, the pinned official client over HTTP, and the distribution's Core image | -| `native-check` (`native.yml`) | Pull requests that touch native sources, shared dependencies or packaging inputs; manual runs; releases | Daemon, process lifecycle, Harness protocols and the installer bundle on Linux, macOS and Windows; uploads the native installers | -| `actionlint` | Changes to workflows | Workflow syntax | -| `core-release` | Version tags and manual runs | See [Publish a version](#publish-a-version) | +Every PR runs `core-check` and reports the required status `check`. `scripts/ci_plan.py` owns the only path-to-check map. The planner compares the PR event's tested merge commit with its verified first parent, using NUL-delimited Git output with rename detection disabled so both old and new paths count. Its JSON plan and reasons appear in the run summary. Missing or inconsistent merge parents, unavailable diffs, empty changes, unknown files, CI changes and shared build/dependency inputs select the full gate. Deletions and mixed changes retain all affected groups. Main pushes and release calls always select every group. -Changes limited to Web or to documentation outside `contracts/agents-api` do not start `native-check`. A newer `core-check`, `api-acceptance` or `native-check` run on the same branch or pull request cancels the older one. +| Group | Checks and consumers | +| --- | --- | +| `hygiene` | Names, repository links, bundled documentation integrity, and CI planner/gate tests; runs for every change | +| `distribution` | Harness catalog and installer schema, install/apply/recovery/cleanup tests, release/download and bundle contracts, Go console tests and build; needs no pnpm install or browser | +| `backend` | Dedicated PostgreSQL guard, sqlc freshness, Runtime/shared Go tests, Linux microsandbox helper, standalone Core/service/client tests, daemon build | +| `harness` | Claude SDK tests and packaging, MiniMax companion scripts | +| `example` | Optional application typecheck, tests, build and isolated browser acceptance | +| `web` | TypeScript, Web/client tests and Web build | +| `web-acceptance` | Full Web browser suite in two isolated shards after Web unit/build success; each keeps one worker | +| `api` | Reusable official-client acceptance against standalone commands and migrations; image acceptance when image/build/helper inputs change, and in every full gate | +| `native` | Reusable Linux, macOS and Windows builds, filesystem/process/Harness checks and native installation; at most two platforms run concurrently | +| `lint` | Reusable actionlint check, including local composite actions | -The full gate starts these partitions concurrently: +Ordinary documentation runs hygiene only. Installer changes add distribution checks. Web changes add Web checks and both browser shards; Core/DB changes add backend and official-client acceptance. Shared contracts, SDKs, Runtime inputs and dependencies propagate to their consumers according to the planner. Generated catalog and protocol inputs include the installer, client and UI consumers. Do not duplicate path lists in reusable workflows or put a `paths` filter on the required workflow. -| Job | Checks | -| --- | --- | -| `backend` | Dedicated PostgreSQL guard, sqlc freshness, Runtime/shared Go tests, Linux microsandbox helper, standalone Core build and service/client tests, daemon build | -| `tooling` | Harness catalog, name guard, distribution/installer, Claude SDK packaging, optional example including browser acceptance, MiniMax companion scripts | -| `web` | TypeScript checks and Web/client tests, Web build | -| `web-acceptance` (two shards) | The complete Web Playwright suite, split by test files between two isolated runners | +The final `check` runs even when planning or a dependency fails. It requires a successful, valid plan, every selected job to be successful, and every unselected job to be skipped. Failure, cancellation, a missing job, an unexpected skip or an unexpected execution fails the gate. API/native reusable workflows are direct dependencies of this gate. A newer run on the same PR cancels its predecessor. Release checks run at their requested immutable ref; native packaging executes once inside those checks, and the distribution build waits for them. + +Browser jobs own separate fixtures and servers; increasing workers against the shared mutable fixture is unsafe. Failed browser jobs retain reports/traces for seven days. Native failure phase summaries are retained for seven days and detailed output stays in the Actions logs; credentials and temporary installation trees are not uploaded. Successful native archives are uploaded only for explicit manual packaging or releases, without recompressing the compressed archive. Release distribution artifacts retain their existing recovery policy; failed publication can reuse the original build as described above. + +The local Node composite action installs the pinned pnpm and caches its package store by lockfile, OS, architecture, Node version and pnpm version. It caches downloaded packages, not `node_modules`; installs remain frozen. Go partitions retain the existing module/compiler caches described under [publication](#publish-a-version). Cache hits seed work and never replace tests. The native concurrency limit reduces peak demand; it does not imply a reduction in total machine time. + +For a local change, inspect the selected groups and run their Makefile targets from the table and workflows: + +```sh +python3 scripts/ci_plan.py plan --base origin/main --head HEAD +make check-ci +``` + +`make check` remains the full local entry point with an unsharded Web suite. `make check-web-unit` and `make check-web-acceptance OAC_WEB_TEST_SHARD=1/2` expose the Web parts. The selection tests cover mixed changes, shared consumers, renames/deletions, unknown inputs, shallow merge checkouts and failed/cancelled/missing results. Changes to the map or workflow graph also require actionlint and replay of representative PR diffs; exercise real documentation, installer, Web and Core runs before relying on new selection rules. -Each check has its own named step. Only the backend job needs a database. Each browser job starts its own fixture and Web server, retaining one Playwright worker per runner so tests never share mutable fixtures across concurrent jobs. Failed Web shards upload their reports and traces for seven days. The final `check` job runs after every partition and succeeds only when all results are `success`; failed, cancelled or skipped jobs cannot produce a green required gate. Releases use this same workflow. Local `make check` still runs every check and the unsharded Web suite; `make check-web-unit` and `make check-web-acceptance` expose its Web parts. `OAC_WEB_TEST_SHARD=1/2` selects a shard for focused CI validation. +Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time including queueing, initial queue delay, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. ### CI runners and free allowance diff --git a/scripts/ci_metrics.py b/scripts/ci_metrics.py new file mode 100644 index 00000000..18d620d8 --- /dev/null +++ b/scripts/ci_metrics.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Report observed runner time and concurrency, without estimating billed cost.""" + +import argparse +from collections import Counter, defaultdict +from datetime import datetime +import json +import subprocess + + +def timestamp(value): + return datetime.fromisoformat(value.replace("Z", "+00:00")).timestamp() + + +def measure(run, jobs): + intervals = [] + platform_seconds = defaultdict(float) + outcomes = Counter() + for job in jobs: + outcomes[job.get("conclusion") or "unfinished"] += 1 + if not job.get("started_at") or not job.get("completed_at") or job.get("conclusion") == "skipped": + continue + start, end = timestamp(job["started_at"]), timestamp(job["completed_at"]) + if end < start: + raise ValueError("Job completed before it started") + if end > start: + intervals.extend(((start, 1), (end, -1))) + labels = ",".join(job.get("labels", [])) + platform = "macOS" if any(x in labels.lower() for x in ("macos", "darwin")) else "Windows" if "windows" in labels.lower() else "Linux" if any(x in labels.lower() for x in ("ubuntu", "linux")) else "unknown" + platform_seconds[platform] += end - start + active = peak = 0 + for _, delta in sorted(intervals): + active += delta + peak = max(peak, active) + starts = [time for time, delta in intervals if delta == 1] + ends = [time for time, delta in intervals if delta == -1] + finished = sum(outcomes[c] for c in ("success", "failure", "timed_out", "cancelled", "action_required", "startup_failure")) + return { + "run_id": run["id"], "attempt": run.get("run_attempt", 1), "head": run["head_sha"], + "status": run["status"], "conclusion": run.get("conclusion"), + "runner_minutes": round(sum(platform_seconds.values()) / 60, 2), + "platform_minutes": {p: round(seconds / 60, 2) for p, seconds in sorted(platform_seconds.items())}, + "elapsed_minutes": round((max(ends) - timestamp(run["created_at"])) / 60, 2) if ends else None, + "initial_queue_seconds": min(starts) - timestamp(run["created_at"]) if starts else None, + "peak_parallel_jobs": peak, "job_outcomes": dict(outcomes), + "failed_job_fraction": (outcomes["failure"] + outcomes["timed_out"] + outcomes["startup_failure"]) / finished if finished else None, + } + + +def api(path): + return json.loads(subprocess.check_output(["gh", "api", path])) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("runs", nargs="+", type=int) + parser.add_argument("--repo", default="MiniMax-AI/OpenAgentCore") + args = parser.parse_args() + reports = [] + for ident in args.runs: + root = f"repos/{args.repo}/actions/runs/{ident}" + run = api(root) + jobs = [] + page = 1 + while True: + batch = api(f"{root}/attempts/{run['run_attempt']}/jobs?per_page=100&page={page}")["jobs"] + jobs.extend(batch) + if len(batch) < 100: + break + page += 1 + reports.append(measure(run, jobs)) + print(json.dumps(reports, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/ci_metrics_test.py b/scripts/ci_metrics_test.py new file mode 100644 index 00000000..7a26e3c1 --- /dev/null +++ b/scripts/ci_metrics_test.py @@ -0,0 +1,28 @@ +import unittest +from ci_metrics import measure + + +class MetricsTests(unittest.TestCase): + def test_overlap_queue_platforms_and_failed_jobs(self): + run = {"id": 1, "head_sha": "a", "created_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "failure"} + def job(start, end, label, conclusion="success"): + return {"started_at": f"2026-09-30T00:{start}:00Z", "completed_at": f"2026-09-30T00:{end}:00Z", "labels": [label], "conclusion": conclusion} + jobs = [job("01", "03", "ubuntu-22.04"), job("02", "04", "macos-15", "failure"), job("03", "05", "windows-2025"), {"conclusion": "skipped"}] + report = measure(run, jobs) + self.assertEqual(report["runner_minutes"], 6) + self.assertEqual(report["elapsed_minutes"], 5) + self.assertEqual(report["initial_queue_seconds"], 60) + self.assertEqual(report["peak_parallel_jobs"], 2) + self.assertEqual(report["platform_minutes"], {"Linux": 2, "Windows": 2, "macOS": 2}) + self.assertEqual(report["failed_job_fraction"], 1 / 3) + + def test_incomplete_run_is_not_reported_as_a_pass(self): + run = {"id": 1, "head_sha": "a", "created_at": "2026-09-30T00:00:00Z", "status": "in_progress", "conclusion": None} + report = measure(run, [{"started_at": "2026-09-30T00:01:00Z", "completed_at": None}]) + self.assertIsNone(report["conclusion"]) + self.assertIsNone(report["elapsed_minutes"]) + self.assertEqual(report["job_outcomes"], {"unfinished": 1}) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/ci_plan.py b/scripts/ci_plan.py new file mode 100644 index 00000000..0f719729 --- /dev/null +++ b/scripts/ci_plan.py @@ -0,0 +1,169 @@ +#!/usr/bin/env python3 +"""Select PR checks from the tested merge diff; unknown inputs select the full gate.""" + +import argparse +import json +import os +from pathlib import Path, PurePosixPath +import subprocess + +JOBS = ("hygiene", "distribution", "backend", "harness", "example", "web", "web-acceptance", "api", "native", "lint") +# Rules accumulate: shared inputs exercise every declared consumer. This is the +# only authored path map; workflows consume the resulting plan. +RULES = ( + ((".github/", "scripts/ci_"), JOBS), + (("apps/web/", "playwright.config.ts"), ("web", "web-acceptance")), + (("services/web/",), ("distribution", "web", "web-acceptance")), + (("example/",), ("example",)), + (("services/core/",), ("backend", "api")), + (("services/core/internal/nativeinstaller/",), ("native", "distribution")), + (("services/core/deploy/", "services/core/tools/"), ("distribution",)), + (("apps/daemon/",), ("backend", "native")), + (("internal/",), ("backend", "api", "native", "distribution")), + (("internal/harnessconfig/",), ("web", "web-acceptance", "example", "harness")), + (("contracts/",), ("backend", "api", "native", "web", "web-acceptance", "example", "distribution")), + (("packages/agents-client/",), ("backend", "api", "web", "web-acceptance", "example")), + (("packages/claude-sdk-adapter/", "packages/mcode-harness/"), ("harness", "native", "backend", "distribution")), + (("packages/tsconfig/",), JOBS), + (("deploy/install/", "deploy/install-release.sh", "scripts/install-release.", "scripts/publish-core-release.", + "scripts/core-distribution-manifest.", "scripts/build-core-distribution.sh", "scripts/config-reference.py", + "scripts/build-web.sh"), ("distribution",)), + (("scripts/build-native-", "scripts/native-"), ("native", "backend", "distribution")), + (("scripts/build-core.sh", "scripts/build-core-image-context.sh", "deploy/distribution/"), ("backend", "api", "distribution", "native")), + (("scripts/build-e2b-provider.sh",), ("backend", "api", "distribution")), + (("scripts/build-claude", "scripts/check-claude", "scripts/build-mcode", "scripts/prepare-release-runtimes.sh"), + ("harness", "native", "backend", "distribution")), + (("scripts/build-agents-runtime.sh",), ("backend", "native", "distribution")), + (("scripts/generate-harness-catalog", "scripts/harness-catalog/", "scripts/openapi-split/", "scripts/patch-agents-openapi.py", + "scripts/extract-agents-api-upstream.py"), JOBS), + (("scripts/check-sqlc.py",), ("backend",)), + (("scripts/check-names", "scripts/name-allowlist.json"), ("hygiene",)), +) +FULL_INPUTS = {"Makefile", "go.mod", "go.sum", "go.work", "go.work.sum", "package.json", "pnpm-lock.yaml", + "pnpm-workspace.yaml", "tsconfig.base.json", ".npmrc", ".gitignore", ".gitattributes", ".dockerignore"} +IMAGE_INPUTS = ("scripts/build-core", "scripts/build-e2b-provider", "deploy/distribution/", "services/core/tools/e2b-provider/", + "services/core/deploy/e2b/") + + +def documentation(path): + p = PurePosixPath(path) + if p.name in {"README.md", "README.zh-CN.md", "AGENTS.md", "CONTRIBUTING.md", "LICENSE"}: + return True + return (path.startswith(("docs/", "contracts/")) and p.suffix in {".md", ".png", ".jpg", ".jpeg", ".svg", ".webp"}) or path in { + "apps/web/PRODUCT.md", "apps/web/DESIGN.md", "services/core/IMPLEMENTATION.md"} + + +def full(reason): + return {"version": 1, "jobs": list(JOBS), "image": True, "reasons": [reason]} + + +def select(paths): + if not paths: + return full("Empty diff; run the full gate") + jobs = {"hygiene"} + image = False + reasons = [] + for path in paths: + if not path or path.startswith("/") or ".." in PurePosixPath(path).parts: + return full("Invalid path in diff") + if path in FULL_INPUTS or path.startswith(".github/"): + return full(f"Shared build or CI input: {path}") + if documentation(path): + reasons.append(f"{path}: hygiene") + continue + matches = {job for prefixes, targets in RULES if path.startswith(prefixes) for job in targets} + if not matches: + return full(f"Unclassified input: {path}") + jobs.update(matches) + image = image or path.startswith(IMAGE_INPUTS) or matches == set(JOBS) + reasons.append(f"{path}: {', '.join(sorted(matches))}") + return {"version": 1, "jobs": [job for job in JOBS if job in jobs], "image": image, "reasons": reasons} + + +def git(*args): + return subprocess.run(["git", *args], check=True, capture_output=True).stdout + + +def changed_paths(base, head): + # Disable rename detection: both the deleted path and new path affect checks. + fields = git("diff", "--no-renames", "--name-status", "-z", base, head, "--").decode("utf-8").split("\0") + if fields.pop() != "" or len(fields) % 2: + raise ValueError("Malformed git diff") + if any(status not in {"A", "D", "M", "T"} for status in fields[::2]): + raise ValueError("Unresolved diff status") + return fields[1::2] + + +def event_plan(event_name, event, requested_ref=""): + if event_name != "pull_request" or requested_ref: + return full("Main, manual or reusable run: full gate") + try: + pr = event["pull_request"] + base, head = pr["base"]["sha"], pr["head"]["sha"] + # Checkout tests the event's merge commit, not an arbitrary head or ref. + # Its first-parent diff includes the integrated PR changes, with no API + # pagination/truncation or merge-base assumptions in a shallow clone. + commit = git("cat-file", "-p", "HEAD").decode("utf-8") + parents = [line[7:] for line in commit.split("\n\n", 1)[0].splitlines() if line.startswith("parent ")] + if parents != [base, head]: + raise ValueError("Checkout does not match the PR merge parents") + return select(changed_paths(base, "HEAD")) + except (KeyError, TypeError, ValueError, UnicodeError, subprocess.CalledProcessError) as err: + return full(f"Diff unavailable ({type(err).__name__}); full gate") + + +def validate_plan(plan): + if not isinstance(plan, dict) or type(plan.get("version")) is not int or plan.get("version") != 1 or type(plan.get("image")) is not bool: + raise ValueError("Invalid check plan") + selected = plan.get("jobs") + if not isinstance(selected, list) or any(not isinstance(j, str) for j in selected): + raise ValueError("Invalid selected jobs") + if len(selected) != len(set(selected)) or not set(selected) <= set(JOBS) or "hygiene" not in selected: + raise ValueError("Invalid selected jobs") + if plan["image"] and "api" not in selected: + raise ValueError("Image checks require API acceptance") + return set(selected) + + +def check_results(plan, needs): + selected = validate_plan(plan) + if set(needs) != set(JOBS) | {"plan"} or needs["plan"].get("result") != "success": + raise ValueError("Missing jobs or unsuccessful plan") + failed = [job for job in JOBS if needs[job].get("result") != ("success" if job in selected else "skipped")] + if failed: + raise ValueError("Check results do not match the plan: " + ", ".join(failed)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + plan_parser = sub.add_parser("plan") + plan_parser.add_argument("--base") + plan_parser.add_argument("--head", default="HEAD") + sub.add_parser("gate") + args = parser.parse_args() + if args.command == "gate": + check_results(json.loads(os.environ["PLAN"]), json.loads(os.environ["RESULTS"])) + print("All checks selected by the plan passed.") + return + if args.base: + plan = select(changed_paths(args.base, args.head)) + else: + try: + event = json.loads(Path(os.environ["GITHUB_EVENT_PATH"]).read_text()) + except (OSError, ValueError, KeyError): + event = {} + plan = event_plan(os.environ.get("GITHUB_EVENT_NAME"), event, os.environ.get("REQUESTED_REF", "")) + print(json.dumps(plan, indent=2)) + if output := os.environ.get("GITHUB_OUTPUT"): + with open(output, "a") as f: + f.write("plan=" + json.dumps(plan, separators=(",", ":")) + "\n") + f.write("jobs=" + json.dumps(plan["jobs"]) + "\n") + f.write("image=" + json.dumps(plan["image"]) + "\n") + if summary := os.environ.get("GITHUB_STEP_SUMMARY"): + with open(summary, "a") as f: + f.write("### Selected checks\n\n```json\n" + json.dumps(plan, indent=2) + "\n```\n") + + +if __name__ == "__main__": + main() diff --git a/scripts/ci_plan_test.py b/scripts/ci_plan_test.py new file mode 100644 index 00000000..ad9799cb --- /dev/null +++ b/scripts/ci_plan_test.py @@ -0,0 +1,156 @@ +import re +import os +from pathlib import Path +import subprocess +import tempfile +import unittest +from unittest.mock import patch + +import ci_plan as ci + + +class SelectionTests(unittest.TestCase): + def jobs(self, *paths): + return set(ci.select(paths)["jobs"]) + + def test_documents_only_need_repository_integrity(self): + for path in ("docs/maintainers.md", "README.md", "contracts/agents-api/admin-api.md", "docs/assets/logo.svg"): + self.assertEqual(self.jobs(path), {"hygiene"}) + + def test_installer_does_not_download_a_browser_or_run_database_tests(self): + self.assertEqual(self.jobs("deploy/install/install.py", "scripts/install-release.test.py"), {"hygiene", "distribution"}) + + def test_web_and_core_have_different_consumers(self): + self.assertEqual(self.jobs("apps/web/src/app.tsx"), {"hygiene", "web", "web-acceptance"}) + plan = ci.select(["services/core/internal/store/sessions.go"]) + self.assertEqual(set(plan["jobs"]), {"hygiene", "backend", "api"}) + self.assertFalse(plan["image"]) + + def test_image_and_native_inputs_keep_their_acceptance(self): + self.assertTrue(ci.select(["deploy/distribution/Dockerfile"])["image"]) + self.assertTrue(ci.select(["services/core/tools/e2b-provider/requirements.txt"])["image"]) + self.assertIn("native", self.jobs("services/core/internal/nativeinstaller/catalog.go")) + self.assertIn("native", self.jobs("scripts/build-native-installer.mjs")) + + def test_shared_protocol_and_catalog_propagate_to_consumers(self): + for path in ("contracts/agents-api/v1/session.go", "internal/harnessconfig/builtin/catalog.json"): + self.assertTrue({"backend", "api", "native", "web", "web-acceptance", "example", "distribution"} <= self.jobs(path)) + self.assertTrue({"web", "web-acceptance", "example", "api"} <= self.jobs("packages/agents-client/src/client.ts")) + + def test_unknown_dependencies_ci_and_empty_diffs_are_full(self): + for paths in ([], ["new-component/source.rs"], ["pnpm-lock.yaml"], ["go.sum"], ["Makefile"], + [".github/actions/node/action.yml"], ["scripts/ci_plan.py"], ["../outside"], ["/outside"]): + self.assertEqual(set(ci.select(paths)["jobs"]), set(ci.JOBS)) + self.assertTrue(ci.select(paths)["image"]) + + def test_mixed_changes_accumulate(self): + self.assertEqual(self.jobs("docs/maintainers.md", "deploy/install/install.py", "apps/web/src/app.tsx"), + {"hygiene", "distribution", "web", "web-acceptance"}) + + def test_installer_pr_300_replay(self): + self.assertEqual(self.jobs( + "deploy/install-release.sh", "deploy/install/README.md", "deploy/install/install.py", + "deploy/install/install_display.py", "deploy/install/test_install.py", "deploy/install/test_install_output.py", + "docs/getting-started/install.md", "scripts/install-release.test.py"), {"hygiene", "distribution"}) + + def test_workflow_graph_cannot_silently_omit_or_add_a_gate_dependency(self): + workflow = (Path(__file__).resolve().parents[1] / ".github/workflows/check.yml").read_text().split("jobs:\n", 1)[1] + jobs = set(re.findall(r"^ ([a-z-]+):$", workflow, re.M)) + self.assertEqual(jobs, set(ci.JOBS) | {"plan", "check"}) + gate = workflow.split(" check:\n", 1)[1] + dependencies = re.search(r"needs: \[(.+)\]", gate).group(1).split(", ") + self.assertEqual(set(dependencies), set(ci.JOBS) | {"plan"}) + + def test_unknown_markdown_is_not_assumed_to_be_documentation(self): + self.assertEqual(self.jobs("new-engine/system-prompt.md"), set(ci.JOBS)) + + def test_non_pr_events_always_run_full(self): + for event in ("push", "workflow_dispatch", "workflow_call"): + self.assertEqual(ci.event_plan(event, {})["jobs"], list(ci.JOBS)) + self.assertEqual(ci.event_plan("pull_request", {}, "release-sha")["jobs"], list(ci.JOBS)) + + def test_unavailable_or_wrong_merge_diff_runs_full(self): + self.assertEqual(ci.event_plan("pull_request", {})["jobs"], list(ci.JOBS)) + event = {"pull_request": {"base": {"sha": "a"}, "head": {"sha": "b"}}} + with patch.object(ci, "git", return_value=b"parent wrong\n\nmessage"): + self.assertEqual(ci.event_plan("pull_request", event)["jobs"], list(ci.JOBS)) + with patch.object(ci, "git", side_effect=subprocess.CalledProcessError(1, "git")): + self.assertEqual(ci.event_plan("pull_request", event)["jobs"], list(ci.JOBS)) + + +class GitDiffTests(unittest.TestCase): + def test_real_merge_in_shallow_clone_handles_deleted_renamed_and_odd_paths(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "source" + repo.mkdir() + def git(*args): + return subprocess.check_output(["git", "-C", str(repo), *args], stderr=subprocess.DEVNULL).decode().strip() + git("init", "-b", "main") + git("config", "user.email", "ci-test@example.invalid") + git("config", "user.name", "CI test") + (repo / "services/core").mkdir(parents=True) + (repo / "services/core/deleted.go").write_text("package example\n") + (repo / "docs").mkdir() + (repo / "docs/old.md").write_text("rename me\n") + git("add", "."); git("commit", "-m", "base") + base = git("rev-parse", "HEAD") + git("switch", "-c", "topic") + (repo / "services/core/deleted.go").unlink() + (repo / "apps/web").mkdir(parents=True) + (repo / "docs/old.md").rename(repo / "apps/web/renamed\nwith space.ts") + git("add", "."); git("commit", "-m", "change") + head = git("rev-parse", "HEAD") + git("switch", "main"); git("merge", "--no-ff", "topic", "-m", "merge") + clone = Path(tmp) / "shallow" + subprocess.run(["git", "clone", "--depth=2", repo.as_uri(), str(clone)], check=True, capture_output=True) + previous = Path.cwd() + try: + os.chdir(clone) + paths = ci.changed_paths(base, "HEAD") + self.assertEqual(set(paths), {"docs/old.md", "services/core/deleted.go", "apps/web/renamed\nwith space.ts"}) + plan = ci.event_plan("pull_request", {"pull_request": {"base": {"sha": base}, "head": {"sha": head}}}) + self.assertEqual(set(plan["jobs"]), {"hygiene", "backend", "api", "web", "web-acceptance"}) + (clone / "old.md").write_text("untracked content cannot change the diff\n") + self.assertEqual(paths, ci.changed_paths(base, "HEAD")) + finally: + os.chdir(previous) + + +class GateTests(unittest.TestCase): + def fixture(self): + plan = ci.select(["apps/web/src/app.tsx"]) + needs = {job: {"result": "success" if job in plan["jobs"] else "skipped"} for job in ci.JOBS} + needs["plan"] = {"result": "success"} + return plan, needs + + def test_only_deliberately_unselected_jobs_may_skip(self): + plan, needs = self.fixture() + ci.check_results(plan, needs) + for state in ("failure", "cancelled", "skipped", "", None): + with self.subTest(state=state), self.assertRaises(ValueError): + ci.check_results(plan, needs | {"web": {"result": state}}) + + def test_matrix_result_failure_is_not_hidden_by_other_jobs(self): + plan = ci.full("test") + needs = {job: {"result": "success"} for job in (*ci.JOBS, "plan")} + for name in ("native", "web-acceptance", "api"): + with self.subTest(name=name), self.assertRaises(ValueError): + ci.check_results(plan, needs | {name: {"result": "failure"}}) + + def test_plan_failure_missing_jobs_and_unexpected_execution_fail(self): + plan, needs = self.fixture() + for bad in ({}, {k: v for k, v in needs.items() if k != "native"}, needs | {"plan": {"result": "failure"}}, + needs | {"native": {"result": "failure"}}, needs | {"native": {"result": "success"}}): + with self.subTest(needs=bad), self.assertRaises(ValueError): + ci.check_results(plan, bad) + + def test_malformed_plan_cannot_turn_checks_off(self): + plan, needs = self.fixture() + for bad in (None, {}, plan | {"jobs": []}, plan | {"jobs": ["hygiene", "invented"]}, + plan | {"jobs": ["hygiene", "hygiene"]}, plan | {"image": True}, plan | {"image": "false"}): + with self.subTest(plan=bad), self.assertRaises(ValueError): + ci.check_results(bad, needs) + + +if __name__ == "__main__": + unittest.main() From 847755c5fca2c73a45113c4c57ebc3174ca47eed Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Thu, 1 Oct 2026 00:26:52 +0800 Subject: [PATCH 2/5] Preserve generated-contract checks and measure rerun attempts accurately --- .github/workflows/check.yml | 1 + .github/workflows/native.yml | 1 - docs/maintainers.md | 4 ++-- scripts/ci_metrics.py | 7 +++++-- scripts/ci_metrics_test.py | 15 +++++++++++++++ scripts/ci_plan.py | 17 ++++++++++++----- scripts/ci_plan_test.py | 24 ++++++++++++++++++++++++ 7 files changed, 59 insertions(+), 10 deletions(-) diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index c0aa0ae5..3691a013 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -146,6 +146,7 @@ jobs: key: core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-distribution-${{ steps.source.outputs.revision }} restore-keys: | core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-distribution- + core-go-v2-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}-tooling- core-go-v1-${{ runner.os }}-${{ runner.arch }}-${{ hashFiles('**/go.mod', '**/go.sum') }}- - uses: actions/setup-node@v6 with: diff --git a/.github/workflows/native.yml b/.github/workflows/native.yml index 8cc529cf..fa18b948 100644 --- a/.github/workflows/native.yml +++ b/.github/workflows/native.yml @@ -136,7 +136,6 @@ jobs: fi node scripts/native-harness-smoke.mjs --codex-binary "$RUNNER_TEMP/native-installer/components/codex/bin/codex${{ runner.os == 'Windows' && '.exe' || '' }}" --claude-runtime "$RUNNER_TEMP/native-installer/components/claude" - name: Archive the native distribution - if: inputs.upload-artifacts run: | cd "$RUNNER_TEMP" tar -C native-installer -czf "oac-native-installer-${{ runner.os }}-${{ runner.arch }}.tar.gz" . diff --git a/docs/maintainers.md b/docs/maintainers.md index 9d08b64a..24044cc6 100644 --- a/docs/maintainers.md +++ b/docs/maintainers.md @@ -159,7 +159,7 @@ Every PR runs `core-check` and reports the required status `check`. `scripts/ci_ | `native` | Reusable Linux, macOS and Windows builds, filesystem/process/Harness checks and native installation; at most two platforms run concurrently | | `lint` | Reusable actionlint check, including local composite actions | -Ordinary documentation runs hygiene only. Installer changes add distribution checks. Web changes add Web checks and both browser shards; Core/DB changes add backend and official-client acceptance. Shared contracts, SDKs, Runtime inputs and dependencies propagate to their consumers according to the planner. Generated catalog and protocol inputs include the installer, client and UI consumers. Do not duplicate path lists in reusable workflows or put a `paths` filter on the required workflow. +Ordinary documentation runs hygiene only; generated catalog files and configuration reference sections retain their distribution freshness checks. Installer changes add distribution checks. Web changes add Web checks and both browser shards; Core/DB changes add backend and official-client acceptance. Shared contracts, SDKs, Runtime inputs and dependencies propagate to their consumers according to the planner. Generated catalog and protocol inputs include the installer, client and UI consumers. Do not duplicate path lists in reusable workflows or put a `paths` filter on the required workflow. The final `check` runs even when planning or a dependency fails. It requires a successful, valid plan, every selected job to be successful, and every unselected job to be skipped. Failure, cancellation, a missing job, an unexpected skip or an unexpected execution fails the gate. API/native reusable workflows are direct dependencies of this gate. A newer run on the same PR cancels its predecessor. Release checks run at their requested immutable ref; native packaging executes once inside those checks, and the distribution build waits for them. @@ -176,7 +176,7 @@ make check-ci `make check` remains the full local entry point with an unsharded Web suite. `make check-web-unit` and `make check-web-acceptance OAC_WEB_TEST_SHARD=1/2` expose the Web parts. The selection tests cover mixed changes, shared consumers, renames/deletions, unknown inputs, shallow merge checkouts and failed/cancelled/missing results. Changes to the map or workflow graph also require actionlint and replay of representative PR diffs; exercise real documentation, installer, Web and Core runs before relying on new selection rules. -Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time including queueing, initial queue delay, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. +Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time and initial queue delay from that attempt’s start, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Only jobs executed in that attempt contribute; earlier attempts are not included. Missing rerun start timestamps leave elapsed and queue values unavailable. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. ### CI runners and free allowance diff --git a/scripts/ci_metrics.py b/scripts/ci_metrics.py index 18d620d8..393b5e7a 100644 --- a/scripts/ci_metrics.py +++ b/scripts/ci_metrics.py @@ -13,6 +13,8 @@ def timestamp(value): def measure(run, jobs): + # created_at belongs to the original run; run_started_at resets on reruns. + attempt_start = run.get("run_started_at") or (run["created_at"] if run.get("run_attempt", 1) == 1 else None) intervals = [] platform_seconds = defaultdict(float) outcomes = Counter() @@ -40,8 +42,8 @@ def measure(run, jobs): "status": run["status"], "conclusion": run.get("conclusion"), "runner_minutes": round(sum(platform_seconds.values()) / 60, 2), "platform_minutes": {p: round(seconds / 60, 2) for p, seconds in sorted(platform_seconds.items())}, - "elapsed_minutes": round((max(ends) - timestamp(run["created_at"])) / 60, 2) if ends else None, - "initial_queue_seconds": min(starts) - timestamp(run["created_at"]) if starts else None, + "elapsed_minutes": round((max(ends) - timestamp(attempt_start)) / 60, 2) if ends and attempt_start else None, + "initial_queue_seconds": min(starts) - timestamp(attempt_start) if starts and attempt_start else None, "peak_parallel_jobs": peak, "job_outcomes": dict(outcomes), "failed_job_fraction": (outcomes["failure"] + outcomes["timed_out"] + outcomes["startup_failure"]) / finished if finished else None, } @@ -60,6 +62,7 @@ def main(): for ident in args.runs: root = f"repos/{args.repo}/actions/runs/{ident}" run = api(root) + run = api(f"{root}/attempts/{run['run_attempt']}") jobs = [] page = 1 while True: diff --git a/scripts/ci_metrics_test.py b/scripts/ci_metrics_test.py index 7a26e3c1..afe4b674 100644 --- a/scripts/ci_metrics_test.py +++ b/scripts/ci_metrics_test.py @@ -23,6 +23,21 @@ def test_incomplete_run_is_not_reported_as_a_pass(self): self.assertIsNone(report["elapsed_minutes"]) self.assertEqual(report["job_outcomes"], {"unfinished": 1}) + def test_rerun_uses_its_own_start_and_only_its_jobs(self): + run = {"id": 1, "head_sha": "a", "run_attempt": 2, "created_at": "2026-09-29T00:00:00Z", + "run_started_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "success"} + jobs = [{"started_at": "2026-09-30T00:01:00Z", "completed_at": "2026-09-30T00:03:00Z", + "labels": ["ubuntu-22.04"], "conclusion": "success"}] + report = measure(run, jobs) + self.assertEqual(report["attempt"], 2) + self.assertEqual(report["runner_minutes"], 2) + self.assertEqual(report["elapsed_minutes"], 3) + self.assertEqual(report["initial_queue_seconds"], 60) + del run["run_started_at"] + report = measure(run, jobs) + self.assertIsNone(report["elapsed_minutes"]) + self.assertIsNone(report["initial_queue_seconds"]) + if __name__ == "__main__": unittest.main() diff --git a/scripts/ci_plan.py b/scripts/ci_plan.py index 0f719729..ccd3b952 100644 --- a/scripts/ci_plan.py +++ b/scripts/ci_plan.py @@ -43,6 +43,10 @@ "pnpm-workspace.yaml", "tsconfig.base.json", ".npmrc", ".gitignore", ".gitattributes", ".dockerignore"} IMAGE_INPUTS = ("scripts/build-core", "scripts/build-e2b-provider", "deploy/distribution/", "services/core/tools/e2b-provider/", "services/core/deploy/e2b/") +# Generated outputs retain freshness checks even when the file is documentation. +GENERATED_OUTPUTS = {"contracts/agents-api/harness-catalog.md", "packages/agents-client/src/harness-catalog.ts", + "services/core/internal/engine/catalog_generated.go", "docs/configuration.md", + "docs/getting-started/install-options.md"} def documentation(path): @@ -68,14 +72,17 @@ def select(paths): return full("Invalid path in diff") if path in FULL_INPUTS or path.startswith(".github/"): return full(f"Shared build or CI input: {path}") - if documentation(path): - reasons.append(f"{path}: hygiene") - continue - matches = {job for prefixes, targets in RULES if path.startswith(prefixes) for job in targets} + matches = {"hygiene"} if documentation(path) else { + job for prefixes, targets in RULES if path.startswith(prefixes) for job in targets} + if path in GENERATED_OUTPUTS: + matches.add("distribution") if not matches: return full(f"Unclassified input: {path}") + if not documentation(path) and path.startswith(IMAGE_INPUTS): + matches.add("api") + image = True jobs.update(matches) - image = image or path.startswith(IMAGE_INPUTS) or matches == set(JOBS) + image = image or matches == set(JOBS) reasons.append(f"{path}: {', '.join(sorted(matches))}") return {"version": 1, "jobs": [job for job in JOBS if job in jobs], "image": image, "reasons": reasons} diff --git a/scripts/ci_plan_test.py b/scripts/ci_plan_test.py index ad9799cb..98fab6d2 100644 --- a/scripts/ci_plan_test.py +++ b/scripts/ci_plan_test.py @@ -1,3 +1,4 @@ +import importlib.util import re import os from pathlib import Path @@ -32,6 +33,29 @@ def test_image_and_native_inputs_keep_their_acceptance(self): self.assertIn("native", self.jobs("services/core/internal/nativeinstaller/catalog.go")) self.assertIn("native", self.jobs("scripts/build-native-installer.mjs")) + def test_every_tracked_path_produces_a_valid_plan(self): + root = Path(__file__).resolve().parents[1] + paths = subprocess.check_output(["git", "ls-files", "-z"], cwd=root).decode().split("\0") + for path in filter(None, paths): + with self.subTest(path=path): + ci.validate_plan(ci.select([path])) + + def test_generated_outputs_keep_freshness_checks(self): + spec = importlib.util.spec_from_file_location("catalog_generator", Path(__file__).with_name("generate-harness-catalog.py")) + generator = importlib.util.module_from_spec(spec) + spec.loader.exec_module(generator) + with patch.object(generator, "go", side_effect=lambda source: source): + outputs = generator.render(generator.load_catalog(generator.ROOT / generator.CATALOG)) + for path in [*map(str, outputs), "deploy/install/harness_catalog.py", "docs/configuration.md", + "docs/getting-started/install-options.md"]: + with self.subTest(path=path): + self.assertIn("distribution", self.jobs(path)) + + def test_distribution_image_build_keeps_api_acceptance(self): + plan = ci.select(["scripts/build-core-distribution.sh"]) + self.assertTrue(plan["image"]) + self.assertEqual(set(plan["jobs"]), {"hygiene", "distribution", "api"}) + def test_shared_protocol_and_catalog_propagate_to_consumers(self): for path in ("contracts/agents-api/v1/session.go", "internal/harnessconfig/builtin/catalog.json"): self.assertTrue({"backend", "api", "native", "web", "web-acceptance", "example", "distribution"} <= self.jobs(path)) From 3d011888af64bd889778d0e0f2e4fa926abd1f77 Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Thu, 1 Oct 2026 00:30:29 +0800 Subject: [PATCH 3/5] Exclude reused jobs from partial rerun resource measurements --- docs/maintainers.md | 2 +- scripts/ci_metrics.py | 14 ++++++++++++-- scripts/ci_metrics_test.py | 11 +++++++---- 3 files changed, 20 insertions(+), 7 deletions(-) diff --git a/docs/maintainers.md b/docs/maintainers.md index 24044cc6..d451887a 100644 --- a/docs/maintainers.md +++ b/docs/maintainers.md @@ -176,7 +176,7 @@ make check-ci `make check` remains the full local entry point with an unsharded Web suite. `make check-web-unit` and `make check-web-acceptance OAC_WEB_TEST_SHARD=1/2` expose the Web parts. The selection tests cover mixed changes, shared consumers, renames/deletions, unknown inputs, shallow merge checkouts and failed/cancelled/missing results. Changes to the map or workflow graph also require actionlint and replay of representative PR diffs; exercise real documentation, installer, Web and Core runs before relying on new selection rules. -Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time and initial queue delay from that attempt’s start, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Only jobs executed in that attempt contribute; earlier attempts are not included. Missing rerun start timestamps leave elapsed and queue values unavailable. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. +Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time and initial queue delay from that attempt's start, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Only jobs executed in that attempt contribute; earlier attempts are not included. Failed-job reruns can carry earlier successful results: their outcomes appear separately and their old execution time is excluded. A missing rerun start timestamp stops measurement because reused jobs cannot be separated reliably. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. ### CI runners and free allowance diff --git a/scripts/ci_metrics.py b/scripts/ci_metrics.py index 393b5e7a..f4990c95 100644 --- a/scripts/ci_metrics.py +++ b/scripts/ci_metrics.py @@ -15,10 +15,19 @@ def timestamp(value): def measure(run, jobs): # created_at belongs to the original run; run_started_at resets on reruns. attempt_start = run.get("run_started_at") or (run["created_at"] if run.get("run_attempt", 1) == 1 else None) + if not attempt_start: + raise ValueError("Rerun start is unavailable; cannot separate reused jobs") + origin = timestamp(attempt_start) intervals = [] platform_seconds = defaultdict(float) outcomes = Counter() + reused_outcomes = Counter() for job in jobs: + # Failed-job reruns include successful prior jobs, relabeled with the + # new run_attempt but retaining their original execution timestamps. + if run.get("run_attempt", 1) > 1 and job.get("started_at") and timestamp(job["started_at"]) < origin: + reused_outcomes[job.get("conclusion") or "unfinished"] += 1 + continue outcomes[job.get("conclusion") or "unfinished"] += 1 if not job.get("started_at") or not job.get("completed_at") or job.get("conclusion") == "skipped": continue @@ -42,9 +51,10 @@ def measure(run, jobs): "status": run["status"], "conclusion": run.get("conclusion"), "runner_minutes": round(sum(platform_seconds.values()) / 60, 2), "platform_minutes": {p: round(seconds / 60, 2) for p, seconds in sorted(platform_seconds.items())}, - "elapsed_minutes": round((max(ends) - timestamp(attempt_start)) / 60, 2) if ends and attempt_start else None, - "initial_queue_seconds": min(starts) - timestamp(attempt_start) if starts and attempt_start else None, + "elapsed_minutes": round((max(ends) - origin) / 60, 2) if ends else None, + "initial_queue_seconds": min(starts) - origin if starts else None, "peak_parallel_jobs": peak, "job_outcomes": dict(outcomes), + "reused_job_outcomes": dict(reused_outcomes), "failed_job_fraction": (outcomes["failure"] + outcomes["timed_out"] + outcomes["startup_failure"]) / finished if finished else None, } diff --git a/scripts/ci_metrics_test.py b/scripts/ci_metrics_test.py index afe4b674..b568bff4 100644 --- a/scripts/ci_metrics_test.py +++ b/scripts/ci_metrics_test.py @@ -27,16 +27,19 @@ def test_rerun_uses_its_own_start_and_only_its_jobs(self): run = {"id": 1, "head_sha": "a", "run_attempt": 2, "created_at": "2026-09-29T00:00:00Z", "run_started_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "success"} jobs = [{"started_at": "2026-09-30T00:01:00Z", "completed_at": "2026-09-30T00:03:00Z", - "labels": ["ubuntu-22.04"], "conclusion": "success"}] + "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2}, + {"started_at": "2026-09-29T00:01:00Z", "completed_at": "2026-09-29T00:04:00Z", + "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2}] report = measure(run, jobs) self.assertEqual(report["attempt"], 2) self.assertEqual(report["runner_minutes"], 2) self.assertEqual(report["elapsed_minutes"], 3) self.assertEqual(report["initial_queue_seconds"], 60) + self.assertEqual(report["job_outcomes"], {"success": 1}) + self.assertEqual(report["reused_job_outcomes"], {"success": 1}) del run["run_started_at"] - report = measure(run, jobs) - self.assertIsNone(report["elapsed_minutes"]) - self.assertIsNone(report["initial_queue_seconds"]) + with self.assertRaises(ValueError): + measure(run, jobs) if __name__ == "__main__": From 5d563af1ed189cd20cb617f8bc00d43551d020ed Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Thu, 1 Oct 2026 00:34:18 +0800 Subject: [PATCH 4/5] Keep cancelled queue waits out of runner resource totals --- docs/maintainers.md | 2 +- scripts/ci_metrics.py | 8 ++++++-- scripts/ci_metrics_test.py | 20 +++++++++++++++++--- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/docs/maintainers.md b/docs/maintainers.md index d451887a..1091a55d 100644 --- a/docs/maintainers.md +++ b/docs/maintainers.md @@ -176,7 +176,7 @@ make check-ci `make check` remains the full local entry point with an unsharded Web suite. `make check-web-unit` and `make check-web-acceptance OAC_WEB_TEST_SHARD=1/2` expose the Web parts. The selection tests cover mixed changes, shared consumers, renames/deletions, unknown inputs, shallow merge checkouts and failed/cancelled/missing results. Changes to the map or workflow graph also require actionlint and replay of representative PR diffs; exercise real documentation, installer, Web and Core runs before relying on new selection rules. -Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time and initial queue delay from that attempt's start, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Only jobs executed in that attempt contribute; earlier attempts are not included. Failed-job reruns can carry earlier successful results: their outcomes appear separately and their old execution time is excluded. A missing rerun start timestamp stops measurement because reused jobs cannot be separated reliably. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. +Measure completed runs with `python3 scripts/ci_metrics.py RUN_ID ...`. It reports the latest attempt's summed runner minutes, elapsed time and initial queue delay from that attempt's start, peak concurrent jobs, platform breakdown and job outcomes/failure fraction. Only jobs assigned a runner in that attempt contribute machine time and execution concurrency; jobs cancelled while queued retain their outcome and wall time. Earlier attempts are not included. Failed-job reruns can carry earlier successful results: their outcomes appear separately and their old execution time is excluded. A missing rerun start timestamp stops measurement because reused jobs cannot be separated reliably. Keep run/head/attempt identities with comparisons, and report cancellations and unfinished runs separately. Raw runner minutes are not billed minutes; use each platform's published conversion and allowance rules before estimating cost. A small successful sample is not a long-term failure-rate estimate. Main impact selection or scheduled full runs are outside this policy. ### CI runners and free allowance diff --git a/scripts/ci_metrics.py b/scripts/ci_metrics.py index f4990c95..707e61fd 100644 --- a/scripts/ci_metrics.py +++ b/scripts/ci_metrics.py @@ -19,6 +19,7 @@ def measure(run, jobs): raise ValueError("Rerun start is unavailable; cannot separate reused jobs") origin = timestamp(attempt_start) intervals = [] + completions = [] platform_seconds = defaultdict(float) outcomes = Counter() reused_outcomes = Counter() @@ -34,6 +35,10 @@ def measure(run, jobs): start, end = timestamp(job["started_at"]), timestamp(job["completed_at"]) if end < start: raise ValueError("Job completed before it started") + completions.append(end) + # Jobs cancelled in the queue have timestamps but no assigned runner. + if not job.get("runner_id") and not job.get("runner_name"): + continue if end > start: intervals.extend(((start, 1), (end, -1))) labels = ",".join(job.get("labels", [])) @@ -44,14 +49,13 @@ def measure(run, jobs): active += delta peak = max(peak, active) starts = [time for time, delta in intervals if delta == 1] - ends = [time for time, delta in intervals if delta == -1] finished = sum(outcomes[c] for c in ("success", "failure", "timed_out", "cancelled", "action_required", "startup_failure")) return { "run_id": run["id"], "attempt": run.get("run_attempt", 1), "head": run["head_sha"], "status": run["status"], "conclusion": run.get("conclusion"), "runner_minutes": round(sum(platform_seconds.values()) / 60, 2), "platform_minutes": {p: round(seconds / 60, 2) for p, seconds in sorted(platform_seconds.items())}, - "elapsed_minutes": round((max(ends) - origin) / 60, 2) if ends else None, + "elapsed_minutes": round((max(completions) - origin) / 60, 2) if completions else None, "initial_queue_seconds": min(starts) - origin if starts else None, "peak_parallel_jobs": peak, "job_outcomes": dict(outcomes), "reused_job_outcomes": dict(reused_outcomes), diff --git a/scripts/ci_metrics_test.py b/scripts/ci_metrics_test.py index b568bff4..cc1d3a5a 100644 --- a/scripts/ci_metrics_test.py +++ b/scripts/ci_metrics_test.py @@ -6,7 +6,7 @@ class MetricsTests(unittest.TestCase): def test_overlap_queue_platforms_and_failed_jobs(self): run = {"id": 1, "head_sha": "a", "created_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "failure"} def job(start, end, label, conclusion="success"): - return {"started_at": f"2026-09-30T00:{start}:00Z", "completed_at": f"2026-09-30T00:{end}:00Z", "labels": [label], "conclusion": conclusion} + return {"started_at": f"2026-09-30T00:{start}:00Z", "completed_at": f"2026-09-30T00:{end}:00Z", "labels": [label], "conclusion": conclusion, "runner_id": 1} jobs = [job("01", "03", "ubuntu-22.04"), job("02", "04", "macos-15", "failure"), job("03", "05", "windows-2025"), {"conclusion": "skipped"}] report = measure(run, jobs) self.assertEqual(report["runner_minutes"], 6) @@ -27,9 +27,9 @@ def test_rerun_uses_its_own_start_and_only_its_jobs(self): run = {"id": 1, "head_sha": "a", "run_attempt": 2, "created_at": "2026-09-29T00:00:00Z", "run_started_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "success"} jobs = [{"started_at": "2026-09-30T00:01:00Z", "completed_at": "2026-09-30T00:03:00Z", - "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2}, + "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2, "runner_id": 1}, {"started_at": "2026-09-29T00:01:00Z", "completed_at": "2026-09-29T00:04:00Z", - "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2}] + "labels": ["ubuntu-22.04"], "conclusion": "success", "run_attempt": 2, "runner_id": 1}] report = measure(run, jobs) self.assertEqual(report["attempt"], 2) self.assertEqual(report["runner_minutes"], 2) @@ -41,6 +41,20 @@ def test_rerun_uses_its_own_start_and_only_its_jobs(self): with self.assertRaises(ValueError): measure(run, jobs) + def test_cancelled_queue_time_is_not_runner_time(self): + run = {"id": 1, "head_sha": "a", "created_at": "2026-09-30T00:00:00Z", "status": "completed", "conclusion": "cancelled"} + jobs = [{"started_at": "2026-09-30T00:01:00Z", "completed_at": "2026-09-30T00:03:00Z", + "labels": ["ubuntu-22.04"], "conclusion": "success", "runner_id": 1}, + {"started_at": "2026-09-30T00:00:00Z", "completed_at": "2026-09-30T00:04:00Z", + "labels": ["windows-2025"], "conclusion": "cancelled", "runner_id": 0, "runner_name": ""}] + report = measure(run, jobs) + self.assertEqual(report["runner_minutes"], 2) + self.assertEqual(report["platform_minutes"], {"Linux": 2}) + self.assertEqual(report["peak_parallel_jobs"], 1) + self.assertEqual(report["elapsed_minutes"], 4) + self.assertEqual(report["initial_queue_seconds"], 60) + self.assertEqual(report["job_outcomes"], {"success": 1, "cancelled": 1}) + if __name__ == "__main__": unittest.main() From ce80ab5979da4efb113dc5a3399e93253dac0da7 Mon Sep 17 00:00:00 2001 From: saladday <1203511142@qq.com> Date: Thu, 1 Oct 2026 00:37:16 +0800 Subject: [PATCH 5/5] Propagate shared Core fixtures to client and installer checks --- scripts/ci_plan.py | 3 +++ scripts/ci_plan_test.py | 8 ++++++++ 2 files changed, 11 insertions(+) diff --git a/scripts/ci_plan.py b/scripts/ci_plan.py index ccd3b952..7b76da22 100644 --- a/scripts/ci_plan.py +++ b/scripts/ci_plan.py @@ -16,6 +16,9 @@ (("services/web/",), ("distribution", "web", "web-acceptance")), (("example/",), ("example",)), (("services/core/",), ("backend", "api")), + (("services/core/internal/sandbox/testdata/node-diagnostics.json",), ("web", "web-acceptance", "example")), + (("services/core/internal/sandbox/testdata/deployment-contract.json", + "services/core/internal/sandbox/e2b/testdata/configuration-selectors.json"), ("distribution",)), (("services/core/internal/nativeinstaller/",), ("native", "distribution")), (("services/core/deploy/", "services/core/tools/"), ("distribution",)), (("apps/daemon/",), ("backend", "native")), diff --git a/scripts/ci_plan_test.py b/scripts/ci_plan_test.py index 98fab6d2..6f52cdd6 100644 --- a/scripts/ci_plan_test.py +++ b/scripts/ci_plan_test.py @@ -61,6 +61,14 @@ def test_shared_protocol_and_catalog_propagate_to_consumers(self): self.assertTrue({"backend", "api", "native", "web", "web-acceptance", "example", "distribution"} <= self.jobs(path)) self.assertTrue({"web", "web-acceptance", "example", "api"} <= self.jobs("packages/agents-client/src/client.ts")) + def test_core_fixtures_retain_client_and_installer_consumers(self): + self.assertTrue({"backend", "api", "web", "web-acceptance", "example"} <= self.jobs( + "services/core/internal/sandbox/testdata/node-diagnostics.json")) + for path in ("services/core/internal/sandbox/testdata/deployment-contract.json", + "services/core/internal/sandbox/e2b/testdata/configuration-selectors.json"): + with self.subTest(path=path): + self.assertTrue({"backend", "api", "distribution"} <= self.jobs(path)) + def test_unknown_dependencies_ci_and_empty_diffs_are_full(self): for paths in ([], ["new-component/source.rs"], ["pnpm-lock.yaml"], ["go.sum"], ["Makefile"], [".github/actions/node/action.yml"], ["scripts/ci_plan.py"], ["../outside"], ["/outside"]):