diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index ed47f6e7..ce937e78 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -138,10 +138,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" @@ -172,10 +170,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" @@ -323,13 +319,8 @@ jobs: run: | $runtimeDir = Join-Path $env:RUNNER_TEMP "ffmpeg-runtime" New-Item -ItemType Directory -Path $runtimeDir -Force | Out-Null - $archive = Join-Path $runtimeDir "ffmpeg-win.zip" - Invoke-WebRequest -Uri "https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-win64-gpl-8.1.zip" -OutFile $archive - $actualHash = (Get-FileHash -Algorithm SHA256 -Path $archive).Hash.ToLowerInvariant() - if ($actualHash -ne "4e40699fa864811312d6c1895ea4873beaf8475e1188563fee57fac860554601") { - throw "Unexpected FFmpeg archive hash: $actualHash" - } - Expand-Archive -LiteralPath $archive -DestinationPath $runtimeDir -Force + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python scripts/provision_pinned_runtime.py --platform win --output-root $runtimeDir $bundleDir = Get-ChildItem -LiteralPath $runtimeDir -Directory | Where-Object { $_.Name -like "ffmpeg-*" } | Select-Object -First 1 if (-not $bundleDir) { throw "Expanded FFmpeg bundle directory not found" } $ffmpeg = Join-Path $bundleDir.FullName "bin\\ffmpeg.exe" @@ -535,10 +526,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" diff --git a/.github/workflows/release-preflight.yml b/.github/workflows/release-preflight.yml index 3a9b16e5..dfb49ae7 100644 --- a/.github/workflows/release-preflight.yml +++ b/.github/workflows/release-preflight.yml @@ -30,10 +30,8 @@ jobs: sudo apt-get install -y --no-install-recommends openssl ripgrep mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" diff --git a/CHANGELOG.md b/CHANGELOG.md index 795267ee..38aafd40 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,44 @@ All notable changes to this project will be documented in this file. The format is based on Keep a Changelog, and this repository uses date-stamped release notes until a stricter semver/tagging policy is formalized. +## [1.3.0-rc.8] - 2026-09-29 + +Unpublished review candidate. Client 0.3.8 builds console onefile launchers +without forwarding process-group signals twice. The exact rc.7 Mac DMG exited +130 when its child alone received SIGINT during a held upload response, but a +terminal-style group SIGINT left the child waiting; the saved queue survived. +This packaging correction leaves Windows GUI Stop/Close on its own owned +cancellation event. Protocol 7.1 and the frozen suite remain unchanged. + +## [1.3.0-rc.7] - 2026-09-29 + +Unpublished review candidate. Client 0.3.7 gives terminal Publish/Retry a +scoped interrupt signal that cancels and reaps owned network work before it +returns exit 130. A held-response native fault exposed an unhandled interrupt +in rc.6; its queue remained durable, but the exit and traceback were wrong. +The measurement protocol, frozen suite and scientific checks are unchanged. +No public asset is promoted by this note. + +## [1.3.0-rc.6] - 2026-09-28 + +Unpublished review candidate for the September 28 client reliability work. +Client 0.3.6 bounds runtime preparation, carries cancellation and deadlines +through normal uploads, reconciles saved campaign evidence and queue receipts, +and gives Windows Overall and Batch progress durable attempt units that move +as warmups and measurements are recorded, including across checkpoints. +It keeps the guided menu reachable with local queue settings and makes console +output safe under Windows code pages that cannot print every Unicode symbol. +Protocol 7.1, the frozen suite bytes and scientific admission checks are +unchanged. Source and focused tests do not certify the native Medium, fault, +recovery or four-asset release gates; no public asset is promoted by this note. + +## [1.3.0-rc.5] - 2026-09-22 + +Published macOS resume and storage repair from source `b0607f0`. The Windows +GUI and console assets carried the rc.4 binaries; Linux carried an rc.2-era +binary. This release did not contain the later draft PR #22 client work, and +the four assets did not share one updated client source. + ## [1.3.0-rc.4] - 2026-09-22 Windows-focused repair with automatic recovery. A suite-cache extraction folder diff --git a/README.md b/README.md index f99b4c4a..c2a906e7 100644 --- a/README.md +++ b/README.md @@ -127,29 +127,38 @@ For authoritative V7 submissions, the client also uploads the encoded benchmark ## Client downloads and candidate commands -Published **1.2.0 / client/0.2.0** downloads: [Windows GUI](https://github.com/oliverdougherC/Encoding_Database/releases/download/1.2.0/encodingdb-client-windows.exe), -[Windows console](https://github.com/oliverdougherC/Encoding_Database/releases/download/1.2.0/encodingdb-client-windows-console.exe), -[Linux](https://github.com/oliverdougherC/Encoding_Database/releases/download/1.2.0/encodingdb-client-linux), -and [macOS](https://github.com/oliverdougherC/Encoding_Database/releases/download/1.2.0/encodingdb-client-macos). -These historical builds do not implement the corrected campaign interface below -and cannot establish the protocol 7.1 epoch. Their unsigned/notarization and -translated-helper limits remain in the published release notes. - -Application candidate `b3ef24a` now has physical Windows seven-clip software and +The current public download is [1.3.0-rc.5](https://github.com/oliverdougherC/Encoding_Database/releases/tag/1.3.0-rc.5). +Verify its `SHA256SUMS` before use. Its macOS DMG contains the September 22 +resume repair; the Windows GUI/console binaries were carried forward from +rc.4, and the Linux archive from an rc.2-era build. These assets do not share +one corrected client source. Draft PR #22 contains later unpublished candidate +work and is not part of any public download. The September 28 failed Medium +sweeps and current readiness limits are recorded in +[the original incident ledger](docs/collection-readiness/reliability-20260928/ORIGINAL-INCIDENTS.md). +Per-clip assets in PR #22 remain staged with `published=false`, so a cold +public guided run can still require the full roughly 1.51 GB frozen suite pack. + +Published **1.2.0 / client/0.2.0** downloads remain available as historical +assets in their [release](https://github.com/oliverdougherC/Encoding_Database/releases/tag/1.2.0). +They do not implement the protocol 7.1 campaign interface. + +Historical application candidate `b3ef24a` has physical Windows seven-clip software and NVENC evidence: 63 VALID attempts, 42 measured runs and 21 stable groups across three campaigns, plus four controlled recovery/failure cases. Hosted Windows -seven-clip console acceptance also passed; final GUI acceptance remains pending. +seven-clip console acceptance also passed; GUI acceptance was pending at that +checkpoint. The [Mac package smoke](docs/collection-readiness/mac-build-b3ef24a/README.md) passed with native ARM64 helpers and a **macOS 27.0** runtime floor; signing is ad hoc, with no Developer ID signature or notarization. Both b3 Mac seven-clip campaigns finished: 46 artifacts verified, with two unstable software groups and six -GPU-suspect hardware measurements retained. Publication remains unverified. +GPU-suspect hardware measurements retained. Publication was unverified at that +checkpoint. The Linux candidate build, migrations, -trusted TLS and isolated restore passed; final native acceptance is now running -after the capacity trial. These results do not establish collection readiness or validated +trusted TLS and isolated restore passed; final native acceptance was still +running after the capacity trial. These results do not establish collection readiness or validated PL. See the [current evidence and open gates](docs/FINAL_RELEASE_HANDOFF.md). -From the corrected source checkout with client requirements installed and a +From a source checkout with client requirements installed and a compatible local staging server, run one clip without publication: ```bash @@ -280,9 +289,10 @@ V7 artifact authorization uses `ARTIFACT_UPLOAD_SECRET` only on the server to si ## Version identities -- Published project release: `1.3.0-rc.4` / `2026-09-22` in `release.json`; client/0.3.3, protocol `7.1`. +- Published project release: `1.3.0-rc.5` / `2026-09-22`, protocol `7.1`. Its macOS and Windows assets identify `client/0.3.3`; the carried Linux asset identifies `client/0.3.1` in retained native envelopes. +- Unpublished review candidate: proposed `1.3.0-rc.8` / `client/0.3.8`, protocol `7.1`; native acceptance and promotion remain open. The rc.6 and rc.7 builds are retained as diagnostic evidence. - Historical project release: `1.2.0` / `client/0.2.0`, protocol `7.0` (historical timing). -- Candidate client implementation/minimum version: `client/0.3.0`. +- Minimum accepted client implementation version: `client/0.3.0`. - Candidate benchmark protocol version: `7.1`, timer boundary `ffmpeg-process-v1`. - PL formula version: `7.0`. - Test-suite version: EncodingDB Test Suite v1 (`encodingdb-test-suite-v1`). diff --git a/client/ENCODINGDB_TEST_SUITE_V1.md b/client/ENCODINGDB_TEST_SUITE_V1.md index 88264131..9533f1be 100644 --- a/client/ENCODINGDB_TEST_SUITE_V1.md +++ b/client/ENCODINGDB_TEST_SUITE_V1.md @@ -18,7 +18,9 @@ python3 scripts/test_suite_drift_check.py The materializer verifies archive, metadata, notices and every reference, then installs matching client/server resources. Run it before native builds or Docker builds. Frozen references have no synthetic fallback; `build_test_suite_v1.py --rewrite-manifest` refuses a frozen suite. -Packaged clients obtain the archive beside the executable or through the manifest download URLs, with verified cache reuse and atomic extraction. Explicit overrides: `ENCODINGDB_SUITE_CACHE_DIR`, `ENCODINGDB_SUITE_PACK_PATH`, `ENCODINGDB_SUITE_PACK_URL`, and `ENCODINGDB_QUICK_CLIP_ID`. +Packaged clients reuse hash-verified cached clips and prefer the separately addressable clips declared in `clip-distribution.json` when those assets are published. The checked-in per-clip inventory is staged but **not published** yet; current public clients still use the external 1.51 GB pack. A full-pack fallback states its transfer size before downloading. `ENCODINGDB_ALLOW_FULL_PACK=0` refuses that fallback, and `ENCODINGDB_SUITE_CLIP_BASE_URL` selects a staged per-clip host for testing. Other overrides: `ENCODINGDB_SUITE_CACHE_DIR`, `ENCODINGDB_SUITE_PACK_PATH`, `ENCODINGDB_SUITE_PACK_URL`, and `ENCODINGDB_QUICK_CLIP_ID`. + +`scripts/prepare_client_suite_distribution.py --clip-bundle-out DIR` stages flat, unique release-asset filenames with exactly the frozen clip bytes and bound license notices. Staging does not upload them or activate public URLs. Publication and actual download verification belong to the release gate in PLA-90. Per-clip attributions, modification notices and license texts travel inside the pack and are hash-bound to its inventory. Third-party CC BY media is not relabeled CC0 or covered by Apache-2.0. diff --git a/client/artifacts.py b/client/artifacts.py index 90b9bf43..ffc77ebf 100644 --- a/client/artifacts.py +++ b/client/artifacts.py @@ -1,9 +1,27 @@ import hashlib import json import os -from typing import Any, Dict, Optional - -from .network import SubmitError, _load_requests, retry_after_seconds +import time +from typing import Any, Callable, Dict, Optional + +from .network import (SubmissionCancelled, SubmitError, _bounded_error_body, + _event_cancelled, _load_requests, _read_response_body, + _run_cancellable, retry_after_seconds) + +# C12: bounded, cooperatively cancellable artifact transport. Each phase runs +# its blocking HTTP call in a daemon worker so a cancel_event (threading.Event +# or duck-typed `is_set()`) is observed within ~50 ms even while the call is +# socket-blocked; the worker itself carries a socket inactivity timeout at +# most the phase budget, so a stalled peer cannot pin a worker beyond it. +# Worst-case no-cancel chain: create 30 s + auth 30 s + upload 120 s = 180 s, +# vs the old 60+60+300 s with no cancellation input at all. +CREATE_TIMEOUT_SECONDS = 30.0 +AUTH_TIMEOUT_SECONDS = 30.0 +UPLOAD_TIMEOUT_SECONDS = 120.0 + +# Progress callback signature: (phase, sent_bytes, total_bytes). GUI-safe: +# ints only, no server data. +ProgressCallback = Callable[[str, int, int], None] AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND = "authoritative-artifact-run-v1" AUTHORITATIVE_ANALYZER_VERSION = "authoritative-analysis/v2" @@ -180,17 +198,138 @@ def build_artifact_submission_payload( return submission +class _PhaseBudget: + """Monotonic wall-clock budget for one transport phase (C12). + + R03: an external caller `deadline` (monotonic) tightens the phase budget; + the effective deadline is the earlier of phase-budget and caller deadline. + `seconds` stays the configured budget for messages.""" + + def __init__(self, seconds: float, *, deadline: Optional[float] = None, + phase: str = "phase") -> None: + self.seconds = max(0.05, float(seconds)) + self.deadline = time.monotonic() + self.seconds + if deadline is not None: + self.deadline = min(self.deadline, deadline) + if self.deadline <= time.monotonic(): + raise SubmitError(f"{phase} exceeded caller deadline before start", + retryable=True) + + def remaining(self) -> float: + return self.deadline - time.monotonic() + + def timeout(self, phase: str) -> float: + """Socket inactivity timeout clamped to the remaining budget. + + A stalled transaction therefore self-limits to at most the phase + budget (or the caller deadline) even if no cancel ever arrives.""" + remaining = self.remaining() + if remaining <= 0: + raise SubmitError(f"{phase} exceeded {self.seconds:g}s phase budget", retryable=True) + return max(0.1, min(self.seconds, remaining)) + + +def _phase_guard(budget: _PhaseBudget, cancel_event: Optional[Any], phase: str) -> None: + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + if budget.remaining() <= 0: + raise SubmitError(f"{phase} exceeded {budget.seconds:g}s phase budget", retryable=True) + + +def _safe_request_error(exc: Exception) -> SubmitError: + """Sanitized retryable wrapper: exception type only, never str(exc) (may + embed URLs, upload tokens or bodies).""" + return SubmitError(f"artifact transport failed: {type(exc).__name__}", retryable=True) + + +class _UploadBody: + """Sized streaming PUT body with cancel/deadline checks and progress (C12). + + Exposes `__len__` so requests keeps the original Content-Length framing + (no chunked-encoding wire change) while urllib3 still sends it chunk by + chunk; each chunk pull re-checks cancel and the phase budget, so a GUI + Stop ends the upload at the next chunk boundary (or, if the server stops + reading and the socket send blocks, within the armed send timeout — + bounded by the upload phase budget).""" + + def __init__(self, path: str, *, phase: str, budget: _PhaseBudget, + cancel_event: Optional[Any], progress: Optional[ProgressCallback]) -> None: + self._path = path + self._phase = phase + self._budget = budget + self._cancel_event = cancel_event + self._progress = progress + self._total = os.path.getsize(path) + + def __len__(self) -> int: + return self._total + + def __iter__(self): + sent = 0 + chunk_size = 262144 + with open(self._path, "rb") as handle: + while True: + _phase_guard(self._budget, self._cancel_event, self._phase) + try: + chunk = handle.read(chunk_size) + except Exception as exc: + raise _safe_request_error(exc) from exc + if not chunk: + break + sent += len(chunk) + if self._progress is not None: + try: + self._progress(self._phase, sent, self._total) + except Exception: + pass + yield chunk + + +def _read_json_bounded(response: Any, *, phase: str, budget: _PhaseBudget, + cancel_event: Optional[Any]) -> Any: + """Parse a JSON response body under the actual deadline with a size cap + (R04): the shared reader consumes at most the cap bytes, stops at the + real remaining deadline/cancel, and closes the response.""" + text = _read_response_body(_load_requests(), response, cancel_event, + budget.deadline, + max_bytes=1 << 22) + try: + return json.loads(text) + except Exception: + return None + + def submit_artifact_submission( base_url: str, submission: Dict[str, Any], *, retries: int = 3, + cancel_event: Optional[Any] = None, + progress: Optional[ProgressCallback] = None, + create_seconds: float = CREATE_TIMEOUT_SECONDS, + auth_seconds: float = AUTH_TIMEOUT_SECONDS, + upload_seconds: float = UPLOAD_TIMEOUT_SECONDS, + deadline: Optional[float] = None, ) -> Dict[str, Any]: + """Create run → authorize → upload, each phase bounded and cancellable. + + `cancel_event` is a threading.Event (or duck-typed `is_set()`); each + blocking phase runs in a daemon worker so cancel is observed within + ~50 ms even mid-socket-wait, and immediately between phases/chunks. The + worker's own socket inactivity timeout never exceeds its phase budget, so + nothing outlives create 30 s + auth 30 s + upload 120 s even with no + cancel. `deadline` (monotonic) is a real caller deadline: it tightens + every phase budget and every response-body read (R03/R04). Every + response is closed on every path. On cancel/timeout the error is + retryable, so the durable spool keeps the entry and its localHash; an + accepted-but-lost response replays idempotently against the same run. + Progress: `progress(phase, sent_bytes, total_bytes)` during upload only. + """ artifact_path = str(submission.get("artifactPath") or "").strip() if not artifact_path: raise SubmitError("artifact submission missing artifactPath", retryable=False) if not os.path.exists(artifact_path): - raise SubmitError(f"artifact missing at {artifact_path}", retryable=False) + raise SubmitError("artifact missing on disk", retryable=False) run_create = submission.get("runCreate") if not isinstance(run_create, dict): @@ -204,81 +343,123 @@ def submit_artifact_submission( # The durable spool owns retry scheduling, deadlines and backoff. One network # transaction per entry avoids hot retries across multiple clients. for attempt in range(1, 2): + if _event_cancelled(cancel_event): + raise SubmissionCancelled("run create") + create_budget = _PhaseBudget(create_seconds, deadline=deadline, + phase="run create") + create_response = None try: - create_response = requests.post( - create_url, - json=run_create, - timeout=60, - allow_redirects=False, - ) + create_response = _run_cancellable( + lambda: requests.post( + create_url, + json=run_create, + timeout=create_budget.timeout("run create"), + allow_redirects=False, + stream=True, + ), + phase="run create", cancel_event=cancel_event, + deadline=create_budget.deadline, bound_seconds=create_budget.seconds) + except SubmissionCancelled: + raise except Exception as exc: - last_error = SubmitError(str(exc), retryable=True) - continue - - if create_response.status_code in (429,) or create_response.status_code >= 500: - last_error = SubmitError( - f"run create failed ({create_response.status_code})", - retryable=True, - status_code=create_response.status_code, - body=create_response.text or "", - retry_after=retry_after_seconds(create_response.headers), - ) + last_error = _safe_request_error(exc) continue - if create_response.status_code >= 400: - raise SubmitError( - f"run create rejected ({create_response.status_code})", - retryable=False, - status_code=create_response.status_code, - body=create_response.text or "", - retry_after=retry_after_seconds(create_response.headers), - ) try: - create_json = create_response.json() - except Exception as exc: - last_error = SubmitError(f"run create response invalid JSON: {exc}", retryable=True) + if create_response.status_code in (429,) or create_response.status_code >= 500: + last_error = SubmitError( + f"run create failed ({create_response.status_code})", + retryable=True, + status_code=create_response.status_code, + body=_bounded_error_body(requests, create_response, cancel_event, + create_budget.deadline), + retry_after=retry_after_seconds(create_response.headers), + ) + continue + if create_response.status_code >= 400: + raise SubmitError( + f"run create rejected ({create_response.status_code})", + retryable=False, + status_code=create_response.status_code, + body=_bounded_error_body(requests, create_response, cancel_event, + create_budget.deadline), + retry_after=retry_after_seconds(create_response.headers), + ) + create_json = _read_json_bounded(create_response, phase="run create", + budget=create_budget, cancel_event=cancel_event) + finally: + try: + create_response.close() + except Exception: + pass + if not isinstance(create_json, dict): + last_error = SubmitError("run create response invalid JSON", retryable=True) continue - benchmark_run = create_json.get("benchmarkRun") if isinstance(create_json, dict) else None + benchmark_run = create_json.get("benchmarkRun") run_id = str((benchmark_run or {}).get("id") or "").strip() if not run_id: last_error = SubmitError("run create response missing benchmarkRun.id", retryable=True) continue - artifact = create_json.get("artifact") if isinstance(create_json, dict) else None + artifact = create_json.get("artifact") artifact_state = str((artifact or {}).get("storageState") or "").strip().upper() - analyses = create_json.get("analyses") if isinstance(create_json, dict) else None + analyses = create_json.get("analyses") if artifact_state in ("RETAINED", "VERIFIED") and isinstance(analyses, list) and analyses: return create_json - auth_response = requests.post( - f"{base_url.rstrip('/')}/v7/benchmark-runs/{run_id}/artifacts/ENCODED/upload-authorizations", - json={ - "sha256": run_create["artifact"]["sha256"], - "byteSize": run_create["artifact"]["byteSize"], - "contentType": auth_content_type, - }, - timeout=60, - allow_redirects=False, - ) - if auth_response.status_code in (429,) or auth_response.status_code >= 500: - last_error = SubmitError( - f"upload authorization failed ({auth_response.status_code})", - retryable=True, - status_code=auth_response.status_code, - body=auth_response.text or "", - retry_after=retry_after_seconds(auth_response.headers), - ) + if _event_cancelled(cancel_event): + raise SubmissionCancelled("upload authorization") + auth_budget = _PhaseBudget(auth_seconds, deadline=deadline, + phase="upload authorization") + auth_response = None + try: + auth_response = _run_cancellable( + lambda: requests.post( + f"{base_url.rstrip('/')}/v7/benchmark-runs/{run_id}/artifacts/ENCODED/upload-authorizations", + json={ + "sha256": run_create["artifact"]["sha256"], + "byteSize": run_create["artifact"]["byteSize"], + "contentType": auth_content_type, + }, + timeout=auth_budget.timeout("upload authorization"), + allow_redirects=False, + stream=True, + ), + phase="upload authorization", cancel_event=cancel_event, + deadline=auth_budget.deadline, bound_seconds=auth_budget.seconds) + except SubmissionCancelled: + raise + except Exception as exc: + last_error = _safe_request_error(exc) continue - if auth_response.status_code >= 400: - raise SubmitError( - f"upload authorization rejected ({auth_response.status_code})", - retryable=False, - status_code=auth_response.status_code, - body=auth_response.text or "", - retry_after=retry_after_seconds(auth_response.headers), - ) - auth_json = auth_response.json() + try: + if auth_response.status_code in (429,) or auth_response.status_code >= 500: + last_error = SubmitError( + f"upload authorization failed ({auth_response.status_code})", + retryable=True, + status_code=auth_response.status_code, + body=_bounded_error_body(requests, auth_response, cancel_event, + auth_budget.deadline), + retry_after=retry_after_seconds(auth_response.headers), + ) + continue + if auth_response.status_code >= 400: + raise SubmitError( + f"upload authorization rejected ({auth_response.status_code})", + retryable=False, + status_code=auth_response.status_code, + body=_bounded_error_body(requests, auth_response, cancel_event, + auth_budget.deadline), + retry_after=retry_after_seconds(auth_response.headers), + ) + auth_json = _read_json_bounded(auth_response, phase="upload authorization", + budget=auth_budget, cancel_event=cancel_event) + finally: + try: + auth_response.close() + except Exception: + pass if not isinstance(auth_json, dict): last_error = SubmitError("upload authorization response invalid JSON", retryable=True) continue @@ -290,40 +471,68 @@ def submit_artifact_submission( last_error = SubmitError("upload authorization missing token", retryable=True) continue + if _event_cancelled(cancel_event): + raise SubmissionCancelled("artifact upload") + upload_budget = _PhaseBudget(upload_seconds, deadline=deadline, + phase="artifact upload") + upload_response = None try: - with open(artifact_path, "rb") as handle: - upload_response = requests.put( + upload_response = _run_cancellable( + lambda: requests.put( f"{base_url.rstrip('/')}/v7/artifact-uploads/{token}", - data=handle, + data=_UploadBody(artifact_path, phase="artifact upload", + budget=upload_budget, cancel_event=cancel_event, + progress=progress), headers={"Content-Type": auth_content_type}, - timeout=300, + timeout=upload_budget.timeout("artifact upload"), allow_redirects=False, - ) + stream=True, + ), + phase="artifact upload", cancel_event=cancel_event, + deadline=upload_budget.deadline, bound_seconds=upload_budget.seconds) + except (SubmissionCancelled, SubmitError): + raise except Exception as exc: - last_error = SubmitError(str(exc), retryable=True) + # A cancel raised inside the streaming body can surface wrapped by + # the HTTP stack; unwrap so the SubmissionCancelled contract holds. + cause = exc.__cause__ or exc.__context__ + while cause is not None: + if isinstance(cause, SubmissionCancelled): + raise cause from exc + cause = getattr(cause, "__cause__", None) or getattr(cause, "__context__", None) + last_error = _safe_request_error(exc) continue - - if upload_response.status_code in (429,) or upload_response.status_code >= 500: - last_error = SubmitError( - f"artifact upload failed ({upload_response.status_code})", - retryable=True, - status_code=upload_response.status_code, - body=upload_response.text or "", - retry_after=retry_after_seconds(upload_response.headers), - ) - continue - if upload_response.status_code >= 400: - raise SubmitError( - f"artifact upload rejected ({upload_response.status_code})", - retryable=False, - status_code=upload_response.status_code, - body=upload_response.text or "", - retry_after=retry_after_seconds(upload_response.headers), - ) try: - return upload_response.json() - except Exception as exc: - last_error = SubmitError(f"artifact upload response invalid JSON: {exc}", retryable=True) + if upload_response.status_code in (429,) or upload_response.status_code >= 500: + last_error = SubmitError( + f"artifact upload failed ({upload_response.status_code})", + retryable=True, + status_code=upload_response.status_code, + body=_bounded_error_body(requests, upload_response, cancel_event, + upload_budget.deadline), + retry_after=retry_after_seconds(upload_response.headers), + ) + continue + if upload_response.status_code >= 400: + raise SubmitError( + f"artifact upload rejected ({upload_response.status_code})", + retryable=False, + status_code=upload_response.status_code, + body=_bounded_error_body(requests, upload_response, cancel_event, + upload_budget.deadline), + retry_after=retry_after_seconds(upload_response.headers), + ) + upload_json = _read_json_bounded(upload_response, phase="artifact upload", + budget=upload_budget, cancel_event=cancel_event) + finally: + try: + upload_response.close() + except Exception: + pass + if not isinstance(upload_json, dict): + last_error = SubmitError("artifact upload response invalid JSON", retryable=True) + continue + return upload_json if last_error is not None: raise last_error diff --git a/client/campaign.py b/client/campaign.py index 0bb8ac29..9ee1fdbd 100644 --- a/client/campaign.py +++ b/client/campaign.py @@ -6,13 +6,14 @@ import re import secrets import math +import sys import time import subprocess import threading from contextvars import ContextVar from contextlib import contextmanager from pathlib import Path -from typing import Any +from typing import Any, Optional from . import config from .console_policy import hidden_console_kwargs @@ -20,39 +21,193 @@ ScheduledRun, ValidityReason, ValidityResult) +PREPARATION_HEARTBEAT_FILENAME = "preparation.json" +PREPARATION_STALL_SECONDS = 300.0 +PROBE_PROCESS_TIMEOUT_SECONDS = 60.0 +_ACQUISITION_FENCE_BUDGET_SECONDS = 120.0 +JOURNAL_REOPEN_BUDGET_SECONDS = 900.0 + + +def _default_heartbeat_path(): + """Persistent substage trail: packaged clients always, scripts by env override. + + Auto-enabling only when frozen keeps unit tests and repo scripts from + writing into a developer-visible state directory while the packaged + client — the surface that produced the unattended 'never got past + preparing runtime' incident — always leaves a postmortem trail.""" + explicit = str(os.environ.get("ENCODINGDB_PREPARATION_HEARTBEAT") or "").strip() + if explicit: + if explicit.lower() in ("0", "false", "off", "none"): + return None + return os.path.abspath(explicit) + if not bool(getattr(sys, "frozen", False)): + return None + return os.path.join(config.default_client_state_dir(), PREPARATION_HEARTBEAT_FILENAME) + + +class PreparationTimeout(RuntimeError): + """A preparation stage exhausted its finite wall-clock budget or went silent. + + Distinct from KeyboardInterrupt (operator cancellation) and from + MeasurementBudgetExceeded (a resumable pause): the storage, cache, or + runtime a stage touched is unresponsive; nothing measured was lost and + the operator can retry after freeing that resource.""" + def __init__(self, stage, seconds, kind): + self.stage = str(stage) + self.budget_seconds = float(seconds) + self.kind = kind + limit = (f"exceeded its {self.budget_seconds:g}s wall-clock budget" + if kind == "budget" else + f"went silent for {self.budget_seconds:g}s") + super().__init__(f"preparation stage '{self.stage}' {limit}; the storage, cache, " + "or runtime it touched is unresponsive — retry after freeing that resource") + + +class PreparationBudgetExceeded(PreparationTimeout): + def __init__(self, stage, seconds): + super().__init__(stage, seconds, "budget") + + +class PreparationStalled(PreparationTimeout): + def __init__(self, stage, seconds): + super().__init__(stage, seconds, "stall") + + +@dataclasses.dataclass +class _StageFrame: + name: str + budget_seconds: float + deadline: float + stall_seconds: float + stall_deadline: float + parent: Optional["_StageFrame"] = None + parent_deadline_remaining: float = 0.0 + parent_stall_remaining: float = 0.0 + + _ACQUISITION_READER = None _ACQUISITION_READER_LOCK = threading.Lock() - _PREPARATION = ContextVar("encodingdb_preparation", default=None) class PreparationScope: - """Cancellation and throttled progress, without a measurement deadline.""" - def __init__(self, cancel_event=None, progress=None): + """Cancellation, throttled progress, per-stage wall-clock budgets and a + persistent substage/heartbeat trail — never a measurement deadline.""" + def __init__(self, cancel_event=None, progress=None, *, heartbeat_path=None, stall_seconds=None): self.cancel_event = cancel_event self.progress = progress self.last_key = None self.last_update = 0.0 + self.stall_seconds = PREPARATION_STALL_SECONDS if stall_seconds is None else max(1.0, float(stall_seconds)) + self.heartbeat_path = _default_heartbeat_path() if heartbeat_path is None else heartbeat_path + self.started_monotonic = time.monotonic() + self.started_wall = time.time() + self.stage_stack = [] def check(self): if self.cancel_event is not None and self.cancel_event.is_set(): raise KeyboardInterrupt + if not self.stage_stack: + return + frame = self.stage_stack[-1] + now = time.monotonic() + if now >= frame.deadline: + raise PreparationBudgetExceeded(frame.name, frame.budget_seconds) + if now >= frame.stall_deadline: + raise PreparationStalled(frame.name, frame.stall_seconds) def report(self, stage, **details): self.check() now = time.monotonic() - key = (stage, details.get("path"), details.get("clipId")) - if self.progress and (key != self.last_key or now - self.last_update >= 1.0): + if self.stage_stack: + frame = self.stage_stack[-1] + frame.stall_deadline = now + frame.stall_seconds + substage = details.pop("substage", None) or stage + key = (substage, details.get("path"), details.get("clipId")) + if key != self.last_key or now - self.last_update >= 1.0: self.last_key, self.last_update = key, now - self.progress(stage, **details) + self._write_heartbeat(substage, "progress", now, details) + if self.progress: + self.progress(stage, substage=substage, **details) self.check() + def _write_heartbeat(self, stage, status, now, details=None, frame=None): + if self.heartbeat_path is None: + return + if frame is None: + frame = self.stage_stack[-1] if self.stage_stack else None + payload = {"schemaVersion": 1, "pid": os.getpid(), "status": status, + "stage": frame.name if frame else stage, "substage": stage, + "startedAt": self.started_wall, + "elapsedSeconds": round(now - self.started_monotonic, 3), + "heartbeatAt": time.time(), + "stageBudgetSeconds": frame.budget_seconds if frame else None, + "stageStallSeconds": frame.stall_seconds if frame else None} + for key in ("path", "clipId", "completedBytes", "totalBytes", "completed", "total", "message"): + if details is not None and details.get(key) is not None: + payload[key] = details[key] + try: + atomic_json(Path(self.heartbeat_path), payload) + except OSError: + pass # A heartbeat trail must never break preparation itself. + + @contextmanager + def stage(self, name, budget_seconds, stall_seconds=None, transitions=True): + """Bound the innermost active stage; nesting suspends the outer deadline. + + A child stage consumes only its own budget: the parent's deadline and + stall timers freeze while the child runs and resume with their saved + remaining time when the child exits on any path. A body that blocks + past its own deadline without a single checkpoint is reported as an + overrun (aborted heartbeat, PreparationBudgetExceeded) instead of a + silent success; operator cancellation still wins over late return.""" + budget = float(budget_seconds) + if not math.isfinite(budget) or budget <= 0: + raise ValueError("preparation stage budget must be positive and finite") + stall = self.stall_seconds if stall_seconds is None else float(stall_seconds) + stall = min(max(1.0, stall), budget) + now = time.monotonic() + frame = _StageFrame(str(name), budget, now + budget, stall, now + stall) + if self.stage_stack: + parent = self.stage_stack[-1] + frame.parent = parent + frame.parent_deadline_remaining = max(0.0, parent.deadline - now) + frame.parent_stall_remaining = max(0.0, parent.stall_deadline - now) + self.stage_stack.append(frame) + if transitions: + self._write_heartbeat(name, "started", now, frame=frame) + try: + self.check() + yield frame + except BaseException: + self._retire(frame, "aborted", transitions) + raise + late = time.monotonic() >= frame.deadline + self._retire(frame, "aborted" if late else "completed", transitions) + if self.cancel_event is not None and self.cancel_event.is_set(): + raise KeyboardInterrupt + if late: + raise PreparationBudgetExceeded(frame.name, frame.budget_seconds) + + def _retire(self, frame, status, transitions): + """Pop a finished frame, resume its frozen parent timers, trail the outcome.""" + self.stage_stack.pop() + now = time.monotonic() + if frame.parent is not None: + frame.parent.deadline = now + frame.parent_deadline_remaining + frame.parent.stall_deadline = now + frame.parent_stall_remaining + if transitions: + self._write_heartbeat(frame.name, status, now, frame=frame) + @contextmanager def activate(self): token = _PREPARATION.set(self) try: self.check() - wait_for_owned_acquisition() + # The fence bounds only the handoff from an older canceled network + # reader; the preparation body below runs under its own stage budgets. + with self.stage("acquisition-fence", _ACQUISITION_FENCE_BUDGET_SECONDS, transitions=False): + wait_for_owned_acquisition() yield self finally: _PREPARATION.reset(token) @@ -64,6 +219,30 @@ def preparation_progress(stage, **details): scope.report(stage, **details) +def preparation_heartbeat(*, substage=None, **details): + """Progress pulse for the open stage, if any; silent outside stages. + + Long silent computes (binary hashing) emit through this so stall + watchdogs see forward motion and the persistent trail names the exact + file being hashed; cancel checks stay unconditional at call sites. + Gating on an open stage keeps measurement-phase hashing (journal save) + from being mislabelled as a preparation stage in the GUI.""" + scope = _PREPARATION.get() + if scope is not None and scope.stage_stack: + scope.report(scope.stage_stack[-1].name, substage=substage, **details) + + +@contextmanager +def preparation_stage(name, budget_seconds, stall_seconds=None, transitions=True): + """Stage budget that degrades to a no-op outside a preparation scope.""" + scope = _PREPARATION.get() + if scope is None: + yield None + return + with scope.stage(name, budget_seconds, stall_seconds, transitions) as frame: + yield frame + + def check_preparation_cancelled(): scope = _PREPARATION.get() if scope is not None: @@ -159,8 +338,13 @@ def run_measurement_process(*args, **kwargs): return subprocess.run(*args, **kwargs) check_measurement_budget() timeout = kwargs.pop("timeout", None) - if timeout is None and budget is not None: - timeout = 60 + if timeout is None: + # A scoped validation probe always owns a finite wall-clock deadline: + # the measurement allowance clamps it further when one is active, and + # preparation-only callers fall back to the probe default so a hung + # ffmpeg/ffprobe can never pin preparation forever. + timeout = (min(60, budget.remaining_seconds()) if budget is not None + else PROBE_PROCESS_TIMEOUT_SECONDS) check = kwargs.pop("check", False) command = args[0] if args else kwargs.get("args") deadline = time.monotonic() + timeout if timeout is not None else None @@ -283,30 +467,38 @@ def __init__(self, queue_dir: str, campaign_id: str, manifest: dict, max_storage # supplied an explicit limit. Honor their resolved policy here, including # deliberate increases or decreases, without changing campaign identity. atomic_json(budget_path, {"schemaVersion": 1, "maxStorageMb": int(max_storage_mb)}) - for path in sorted(self.root.glob("attempt-*.json")): - record = load_record(json.loads(path.read_text())) - info = record.metadata.get("info") or {} - artifact = info.get("artifactPath") - if artifact and not info.get("error"): - candidate = Path(artifact) - if candidate.is_file(): - if self.root.resolve() not in candidate.resolve().parents: + # Reopen hashes every retained artifact on resumable storage; a hung + # disk must fail the stage with a name instead of pinning preparation + # open forever, and operator cancel must land between hash chunks. + with preparation_stage("journal-reopen", JOURNAL_REOPEN_BUDGET_SECONDS): + for path in sorted(self.root.glob("attempt-*.json")): + record = load_record(json.loads(path.read_text())) + info = record.metadata.get("info") or {} + artifact = info.get("artifactPath") + if artifact and not info.get("error"): + candidate = Path(artifact) + if candidate.is_file(): + if self.root.resolve() not in candidate.resolve().parents: + raise ValueError("Journal artifact missing or outside owned campaign") + if self.hash_file(candidate) != info.get("artifactSha256"): + raise ValueError("Journal artifact changed; cannot resume") + elif record.schedule.phase == "warmup": + if not self._warmup_released(record): + raise ValueError("Warmup artifact missing without verified release evidence") + elif not self.accepted_receipt(record): raise ValueError("Journal artifact missing or outside owned campaign") - if self.hash_file(candidate) != info.get("artifactSha256"): - raise ValueError("Journal artifact changed; cannot resume") - elif record.schedule.phase == "warmup": - if not self._warmup_released(record): - raise ValueError("Warmup artifact missing without verified release evidence") - elif not self.accepted_receipt(record): - raise ValueError("Journal artifact missing or outside owned campaign") - self.records[record.schedule.execution_order] = record + self.records[record.schedule.execution_order] = record @staticmethod def hash_file(path): digest = hashlib.sha256() + completed = 0 with open(path, "rb") as handle: for chunk in iter(lambda: handle.read(1024 * 1024), b""): + check_preparation_cancelled() digest.update(chunk) + completed += len(chunk) + preparation_heartbeat(path=str(path), completedBytes=completed) return digest.hexdigest() def _receipt_path(self, execution_order: int) -> Path: diff --git a/client/config.py b/client/config.py index 1cdc1548..346c1efe 100644 --- a/client/config.py +++ b/client/config.py @@ -68,7 +68,11 @@ def default_queue_dir() -> str: # Batch aggregation for Small/Full multi-run flows _BATCH_ACTIVE: bool = False _BATCH_START_TS: float = 0.0 -_BATCH_COMPLETED_COUNT: int = 0 + +# Durable end-of-run campaign view (journal + spool), recomputed by +# run_benchmark_batch on every exit path. Transient per-segment counters were +# replaced by this because they conflated measured, queued and confirmed work. +_BATCH_LEDGER: Optional[Dict[str, Any]] = None # New-attempt heartbeat across checkpoint segments; resumed attempts do not count. _BATCH_ATTEMPTS_RECORDED: int = 0 diff --git a/client/encoders.py b/client/encoders.py index ada82b66..c32dbc62 100644 --- a/client/encoders.py +++ b/client/encoders.py @@ -6,6 +6,8 @@ from typing import Optional, Dict, List, Tuple from . import config +from .campaign import (PreparationBudgetExceeded, PreparationTimeout, + run_measurement_process, preparation_stage, preparation_progress) from .console_policy import hidden_console_kwargs from .hardware import detect_hardware @@ -87,9 +89,16 @@ def is_hardware_encoder_name(encoder: str) -> bool: def exec_ok(cmd: List[str]) -> bool: try: - subprocess.run(cmd, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, - **hidden_console_kwargs()) + with preparation_stage("encoder-version-probe", 60): + preparation_progress("encoder-version-probe") + run_measurement_process(cmd, check=True, stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, timeout=30, + **hidden_console_kwargs()) return True + except subprocess.TimeoutExpired as exc: + raise PreparationBudgetExceeded("encoder-version-probe", 30) from exc + except PreparationTimeout: + raise except Exception: return False @@ -98,9 +107,16 @@ def ensure_ffmpeg_and_ffprobe() -> Tuple[bool, Optional[str]]: if not exec_ok([config.ffmpeg_exe(), "-version"]) or not exec_ok([config.ffprobe_exe(), "-version"]): return False, None try: - out = subprocess.run([config.ffmpeg_exe(), "-version"], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, - **hidden_console_kwargs()) + with preparation_stage("encoder-version-probe", 60): + preparation_progress("encoder-version-probe") + out = run_measurement_process([config.ffmpeg_exe(), "-version"], + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + text=True, timeout=30, **hidden_console_kwargs()) version_line = (out.stdout or "").splitlines()[0] if out.stdout else "" + except subprocess.TimeoutExpired as exc: + raise PreparationBudgetExceeded("encoder-version-probe", 30) from exc + except PreparationTimeout: + raise except Exception: version_line = "" return True, version_line @@ -115,11 +131,13 @@ def _get_encoder_set() -> set: if _ENCODER_LIST_CACHE is not None: return _ENCODER_LIST_CACHE try: - out = subprocess.run( - [config.ffmpeg_exe(), "-hide_banner", "-encoders"], - stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, - **hidden_console_kwargs(), - ) + with preparation_stage("encoder-list-probe", 60): + preparation_progress("encoder-list-probe") + out = run_measurement_process( + [config.ffmpeg_exe(), "-hide_banner", "-encoders"], + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, + timeout=30, **hidden_console_kwargs(), + ) names: set = set() for line in (out.stdout or "").splitlines(): # Encoder lines start with a flags field (e.g. " V..... libx264") @@ -128,6 +146,10 @@ def _get_encoder_set() -> set: names.add(parts[1]) _ENCODER_LIST_CACHE = names return names + except subprocess.TimeoutExpired as exc: + raise PreparationBudgetExceeded("encoder-list-probe", 30) from exc + except PreparationTimeout: + raise except Exception: return set() @@ -173,12 +195,23 @@ def is_hardware_encoder_usable(encoder: str) -> bool: elif enc == "hevc_videotoolbox": cmd += ["-tag:v", "hvc1"] cmd += ["-an", out_path] - proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, timeout=8, - **hidden_console_kwargs()) + with preparation_stage(f"encoder-usable-{enc}", 30): + preparation_progress("encoder-usable", encoder=enc) + proc = run_measurement_process(cmd, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, text=True, + timeout=8, **hidden_console_kwargs()) ok = (proc.returncode == 0) and os.path.exists(out_path) and os.path.getsize(out_path) > 0 with config._GLOBAL_STATE_LOCK: config._ENCODER_USABLE_CACHE[enc] = bool(ok) return bool(ok) + except subprocess.TimeoutExpired: + # A missing or wedged optional device is unavailable for this plan; + # the eight-second probe deadline still bounds its process lifetime. + with config._GLOBAL_STATE_LOCK: + config._ENCODER_USABLE_CACHE[enc] = False + return False + except PreparationTimeout: + raise except Exception: with config._GLOBAL_STATE_LOCK: config._ENCODER_USABLE_CACHE[enc] = False @@ -187,9 +220,16 @@ def is_hardware_encoder_usable(encoder: str) -> bool: def has_libvmaf() -> bool: try: - out = subprocess.run([config.ffmpeg_exe(), "-hide_banner", "-filters"], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, - **hidden_console_kwargs()) + with preparation_stage("filter-list-probe", 60): + preparation_progress("filter-list-probe") + out = run_measurement_process([config.ffmpeg_exe(), "-hide_banner", "-filters"], + stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, + text=True, timeout=30, **hidden_console_kwargs()) return "libvmaf" in (out.stdout or "") + except subprocess.TimeoutExpired as exc: + raise PreparationBudgetExceeded("filter-list-probe", 30) from exc + except PreparationTimeout: + raise except Exception: return False diff --git a/client/main.py b/client/main.py index 3d7bd1bc..d3062fc3 100644 --- a/client/main.py +++ b/client/main.py @@ -1,13 +1,16 @@ import argparse from functools import wraps import dataclasses +import hashlib import json import math import os import re +import signal import subprocess import sys import tempfile +import threading import time import secrets from contextlib import nullcontext @@ -59,14 +62,18 @@ ) from .artifacts import ( AUTHORITATIVE_ANALYZER_VERSION, + AUTH_TIMEOUT_SECONDS, + CREATE_TIMEOUT_SECONDS, + UPLOAD_TIMEOUT_SECONDS, build_artifact_submission_payload, build_environment_bootstrap, build_payload_hash, build_recipe_bootstrap, ) -from .network import fetch_baseline_rows, check_compatibility -from .campaign import (CampaignJournal, active_collection, atomic_json, directory_bytes, physical_source_id, journal_path, - PreparationScope, preparation_progress, check_preparation_cancelled, +from .network import SubmitError, SubmissionCancelled, fetch_baseline_rows, check_compatibility +from .campaign import (CampaignJournal, active_collection, atomic_json, directory_bytes, physical_source_id, journal_path, load_record, + PreparationScope, PreparationTimeout, preparation_progress, preparation_stage, + check_preparation_cancelled, MeasurementBudget, MeasurementBudgetExceeded, check_measurement_budget, measurement_timeout, run_measurement_process) from .identity import selected_device from .protocol import ( @@ -74,27 +81,39 @@ EncodeOutcome, EncodeTiming, EnvironmentSnapshot, + EnvironmentThresholds, ProtocolConfig, RecipeSpec, StructuralExpectation, + StructuralTolerance, execute_protocol_campaign, campaign_result_from_records, generate_campaign_id, ) +from .recovery_projection import project_attempt_groups +from .publication_result import failure_info, failure_text from .spool import ( + campaign_queue_summary, + collector_publication_scope, cleanup_spool, count_pending_entries, + drain_committed_receipts, + local_hash_for_payload, + host_phase_busy, + host_phase_hold, inspect_spool, + publication_lock_busy, + queue_recovery_summary, replay_spool, spool_payload, SpoolCapacityError, submit_spooled_path, ) + from .stats import should_skip_submission from .suite import ( PreparedSuiteClip, REQUIRED_CONTENT_CLASSES, - ensure_suite, ensure_suite_clip, get_clip, get_default_quick_clip, @@ -109,9 +128,14 @@ print_info, print_success, print_warning, print_error, print_batch_summary, ) -CLIENT_VERSION = "client/0.3.3" +CLIENT_VERSION = "client/0.3.8" # UI/package patches do not change the server's frozen protocol 7.1 contract. PROTOCOL_MINIMUM_CLIENT_VERSION = "client/0.3.0" +ACTIVE_PUBLICATION_DEADLINE_SECONDS = ( + CREATE_TIMEOUT_SECONDS + AUTH_TIMEOUT_SECONDS + UPLOAD_TIMEOUT_SECONDS +) +SOURCE_CLIP_BUDGET_SECONDS = 3600.0 +SOURCE_CONTRACT_BUDGET_SECONDS = 600.0 PUBLICATION_CONSENT_VERSION = 1 PUBLICATION_CONSENT_FILENAME = "publication-consent.json" @@ -155,6 +179,57 @@ def _has_publication_consent() -> bool: return int(payload.get("version") or 0) == PUBLICATION_CONSENT_VERSION +def publication_endpoint_fingerprint(base_url: str) -> str: + """Bind saved continuation to an endpoint without persisting credentials.""" + return hashlib.sha256(str(base_url).strip().rstrip("/").encode("utf-8")).hexdigest() + + +def publication_intent_state(queue_dir: str, campaign_id: str) -> Dict[str, Any]: + """Read one explicit, consented saved-publication continuation intent.""" + root = journal_path(queue_dir, campaign_id) + try: + value = json.loads((root / "publication-intent.json").read_text(encoding="utf-8")) + except (OSError, ValueError): + return {"active": False} + if (not isinstance(value, dict) or value.get("schemaVersion") != 1 + or value.get("campaignId") != campaign_id): + return {"active": False, "error": "saved continuation intent is invalid"} + return value + + +def _save_publication_intent(queue_dir: str, campaign_id: str, base_url: str, + *, active: bool, status: str, + failure: Optional[Any] = None) -> None: + root = journal_path(queue_dir, campaign_id) + old = publication_intent_state(queue_dir, campaign_id) + now = time.time() + attempts = int(old.get("attempts") or 0) + (1 if active and status != "working" else 0) + next_at = now + if active and status != "working": + next_at = now + min(300, 30 * (2 ** min(attempts, 4))) + try: + due = campaign_queue_summary(queue_dir, campaign_id).get("nextAttemptAt") + if due: + next_at = max(next_at, float(due)) + except (OSError, ValueError, TypeError): + pass + payload = { + "schemaVersion": 1, + "campaignId": campaign_id, + "baseUrlFingerprint": publication_endpoint_fingerprint(base_url), + "createdAt": float(old.get("createdAt") or now), + "updatedAt": now, + "nextAttemptAt": next_at, + "attempts": attempts, + "active": bool(active), + "status": str(status)[:40], + } + if failure: + payload["lastFailure"] = failure_info( + failure, operation="publish_saved", campaign_id=campaign_id) + atomic_json(root / "publication-intent.json", payload) + + def _store_publication_consent() -> None: path = _publication_consent_path() os.makedirs(os.path.dirname(path), exist_ok=True) @@ -308,6 +383,11 @@ def progress(stage, **details): try: with PreparationScope(kwargs.get("cancel_event"), progress).activate(): return function(*args, **kwargs) + except PreparationTimeout as exc: + message = str(exc) + print_error(message) + _emit_event(sink, "run_error", scope="preparation", code=2, message=message) + return 2 except KeyboardInterrupt: print_info("Preparation or collection interrupted; retained downloads and campaign records can be resumed.") _emit_event(sink, "run_interrupted", scope="preparation") @@ -344,14 +424,21 @@ def _preparation_runtime_integrity(event_sink=None): return 0 +def _prepare_source_clip(clip: Any) -> PreparedSuiteClip: + """Bound acquisition and complete hash/media validation for one frozen clip.""" + with preparation_stage(f"source-{clip.clip_id}", SOURCE_CLIP_BUDGET_SECONDS): + preparation_progress("source", clipId=clip.clip_id) + return ensure_suite_clip(clip) + + def _prepare_quick_suite_clip() -> PreparedSuiteClip: manifest = load_default_suite_manifest() - return ensure_suite_clip(get_default_quick_clip(manifest)) + return _prepare_source_clip(get_default_quick_clip(manifest)) def _prepare_full_suite() -> List[PreparedSuiteClip]: manifest = load_default_suite_manifest() - prepared = ensure_suite(manifest) + prepared = [_prepare_source_clip(clip) for clip in manifest.clips] if not has_general_pl_coverage(prepared): raise RuntimeError("suite coverage is incomplete; General PL requires all declared content classes") return prepared @@ -359,7 +446,7 @@ def _prepare_full_suite() -> List[PreparedSuiteClip]: def _prepare_named_suite_clip(clip_id: str) -> PreparedSuiteClip: manifest = load_default_suite_manifest() - return ensure_suite_clip(get_clip(manifest, clip_id)) + return _prepare_source_clip(get_clip(manifest, clip_id)) def _suite_identity_note(clip: PreparedSuiteClip) -> str: @@ -650,6 +737,123 @@ def _emit_counters( ) +def _durable_campaign_ledger(queue_dir: str, campaign_id: str, journal: Any, + local_only: bool) -> Dict[str, Any]: + """Durable end-of-run campaign view: measured vs saved vs queued vs confirmed. + + Recomputed from journal + spool on every run_benchmark_batch exit so the + end screen never reports in-memory optimism as confirmed work. An accepted + receipt is server-confirmed (analysis pending), a local submission file is + saved work only in local-only mode, and queue entries are pending uploads. + """ + measured = uploaded = saved_local = unpublishable = 0 + groups: Dict[str, Dict[str, int]] = {} + legacy_reconcile: list = [] + try: + records = list(journal.records.values()) + except Exception: + records = [] + for record in records: + try: + if record.schedule.phase != "measured": + continue + recipe_id = str(record.schedule.recipe_id) + order = int(record.schedule.execution_order) + except Exception: + continue + measured += 1 + group = groups.setdefault(recipe_id, {"records": 0, "confirmed": 0}) + group["records"] += 1 + if record.skipped_before_encode or record.timing is None or record.overall_validity.state == "invalid": + unpublishable += 1 + continue + try: + accepted = journal.accepted_receipt(record) is not None + except Exception: + accepted = False + if accepted: + uploaded += 1 + group["confirmed"] += 1 + continue + if local_only and (journal.root / f"submission-{order:06d}.json").exists(): + saved_local += 1 + group["confirmed"] += 1 + continue + if not local_only: + legacy_reconcile.append((order, group)) + # Legacy Windows/Linux campaigns hold immutable envelopes plus queue + # receipts (valid backend run ids) but no journal accepted markers - + # uploads were committed before journal self-publish existed. Reconcile + # by payload hash: the envelope is the immutable identity, so a receipt + # counts only when its hash matches the envelope exactly, the response + # names a real run, and no terminal verdict overrides it. + if legacy_reconcile: + receipts: Dict[str, str] = {} + try: + for receipt_file in (Path(queue_dir) / "receipts").glob("*.json"): + try: + receipt = json.loads(receipt_file.read_text()) + except (OSError, ValueError): + continue + if not isinstance(receipt, dict) or receipt.get("localHash") != receipt_file.stem: + continue + response = receipt.get("response") + run = response.get("benchmarkRun") if isinstance(response, dict) else None + run_id = str(run.get("id") or "").strip() if isinstance(run, dict) else "" + if run_id: + receipts[receipt_file.stem] = run_id + except OSError: + receipts = {} + try: + terminal_hashes = {f.stem for f in (Path(queue_dir) / "terminal").glob("*.json")} + except OSError: + terminal_hashes = set() + for order, group in legacy_reconcile: + envelope = journal.root / f"submission-{order:06d}.json" + try: + payload = json.loads(envelope.read_text()) + except (OSError, ValueError): + continue + if not isinstance(payload, dict): + continue + try: + local_hash = local_hash_for_payload(payload) + except Exception: + continue + if local_hash in receipts and local_hash not in terminal_hashes: + uploaded += 1 + group["confirmed"] += 1 + try: + queue = campaign_queue_summary(queue_dir, campaign_id) + except Exception: + queue = {} + projected = project_attempt_groups(journal.root, campaign_id) + finished_groups = set(projected["finishedGroupIds"]) + planned_groups = int(projected["plannedGroups"] or 0) + try: + saved_manifest = json.loads((journal.root / "manifest.json").read_text(encoding="utf-8")) + minimum_measured = int((saved_manifest.get("protocolConfig") or {}).get("minimum_measured_runs", 2)) + except (OSError, ValueError, TypeError, AttributeError): + minimum_measured = 2 + return { + "campaignId": campaign_id, + "measuredAttempts": measured, + "uploaded": uploaded, + "savedLocal": saved_local, + "unpublishable": unpublishable, + "queued": int(queue.get("pendingEntries") or 0), + "terminalFailures": int(queue.get("terminalEntries") or 0), + "groupsTotal": planned_groups or len(groups), + "groupsFinished": len(finished_groups), + "requiredMeasured": planned_groups * minimum_measured, + "optionalMeasured": sum(max(0, g["records"] - minimum_measured) + for g in groups.values()), + "groupsConfirmed": sum(1 for recipe_id, group in groups.items() + if recipe_id in finished_groups and group["records"] > 0 + and group["confirmed"] == group["records"]), + } + + def _format_vmaf_model_unavailable(context: Dict[str, Any]) -> str: reason = str(context.get("reason") or "unknown") model_id = str(context.get("metricModelId") or "unknown-model") @@ -988,14 +1192,15 @@ def _build_protocol_recipe_specs( contract_key = _source_contract_cache_key(effective_input) source_probe = _SOURCE_CONTRACT_CACHE.get(contract_key) if source_probe is None: - preparation_progress( - "source-contract", - clipId=(prepared_clip.clip_id if isinstance(prepared_clip, PreparedSuiteClip) - else os.path.basename(effective_input)), - completed=contracts_done + 1, - total=contract_total, - ) - source_probe = _probe_artifact_contract(effective_input) + with preparation_stage("source-contract", SOURCE_CONTRACT_BUDGET_SECONDS): + preparation_progress( + "source-contract", + clipId=(prepared_clip.clip_id if isinstance(prepared_clip, PreparedSuiteClip) + else os.path.basename(effective_input)), + completed=contracts_done + 1, + total=contract_total, + ) + source_probe = _probe_artifact_contract(effective_input) _SOURCE_CONTRACT_CACHE[contract_key] = source_probe contracts_done += 1 source_duration = source_probe.duration_s @@ -1156,13 +1361,21 @@ def _replay_pending_uploads( api_key: str, retries: int, use_token: bool, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, ) -> int: + operation = {} + if cancel_event is not None: + operation["cancel_event"] = cancel_event + if deadline is not None: + operation["deadline"] = deadline stats = replay_spool( queue_dir, base_url=base_url, api_key=api_key, retries=max(1, retries), use_token=use_token, + **operation, ) if stats.submitted: print_info(f"Submitted {stats.submitted} queued payload(s).") @@ -1232,6 +1445,810 @@ def _retire_uploaded_artifact(journal_root, record: Any, artifact_sha256: str, b pass +def _safe_failure_info(exc: BaseException) -> Dict[str, Any]: + """Structured safe failure fields; raw server bodies and secrets never cross.""" + return failure_info(exc, operation="saved_publication") + + +def _submit_failure_fields(status: str, message: str) -> Dict[str, Any]: + """Safe cause/category/recovery fields for submit_result events (C04/C05). + + Distinct categories for the outcomes an operator acts on differently: + 429 rate limiting, 5xx upstream failure, permanent protocol rejection, + expired retries, missing bytes, corrupt queue files. Parses the status + code from the transport's own bounded messages (`network`/`artifacts` + embed `(NNN)`); raw server bodies never cross the event boundary.""" + text = (message or "").strip() + match = re.search(r"\((\d{3})\)", text) or re.match(r"(?:server_error|submit failed) (\d{3})", text) + code = int(match.group(1)) if match else 0 + lowered = text.lower() + if code == 429 or "rate limited" in lowered: + return {"errorCategory": "rate_limited", + "safeReason": "Server rate limited the upload (429)" if code else text[:200] or "server rate limited the upload", + "recoveryAction": "No action needed: honoring Retry-After, the durable queue retries when due"} + if 500 <= code <= 599 or lowered.startswith("server_error"): + return {"errorCategory": "server_error", + "safeReason": f"Transient upstream failure ({code})" if code else text[:200] or "transient upstream failure", + "recoveryAction": "No action needed: the payload is retained with the server's Retry-After and retried idempotently"} + if code and 400 <= code < 500 and code != 429: + return {"errorCategory": "protocol_rejected", + "safeReason": f"Server rejected the submission ({code})", + "recoveryAction": "Terminal: verify client/suite versions before republishing; the server will reject identical evidence again"} + if text == "retry_deadline_expired": + return {"errorCategory": "expired", + "safeReason": "Retry deadline expired before the server accepted the upload", + "recoveryAction": "Inspect expired saved evidence; do not restart its retry deadline"} + if text in ("missing_spooled_artifact", "missing_artifact") or "missing" in lowered: + return {"errorCategory": "unavailable_source", + "safeReason": text[:200] or "Queued artifact bytes are no longer on disk", + "recoveryAction": "Inspect retained artifact and queue copies; keep completed measurements unchanged"} + if text.startswith("corrupt"): + return {"errorCategory": "corrupt_queue", + "safeReason": "Queue file could not be parsed and was moved to dead-letter", + "recoveryAction": "Inspect the affected dead-letter entry; preserve unrelated saved work"} + if "rejected" in lowered: + return {"errorCategory": "protocol_rejected", + "safeReason": text[:200] or "server rejected the submission", + "recoveryAction": "Inspect the terminal verdict and client/suite versions; preserve rejected evidence"} + if status == "queued": + return {"errorCategory": "transient", + "safeReason": text[:200] or "upload deferred; scheduled for retry", + "recoveryAction": "No action needed: the durable queue retries when due"} + return {"errorCategory": "publication_failed", + "safeReason": text[:200] or "upload attempt failed", + "recoveryAction": "Retained for idempotent retry; use --upload-only or Publish saved when due"} + + +def _reconstruct_saved_submissions( + *, + queue_dir: str, + campaign_id: str, + max_storage_mb: int, + cancel_event: Optional[Any] = None, +) -> Dict[str, Any]: + """Materialize submission envelopes for COMPLETE groups lacking them (C09). + + A controlled stop or storage failure can end a campaign AFTER a group's + measured artifacts are durable in the journal but BEFORE the immutable + `submission-*.json` envelopes were written. Publishing must then REBUILD the + exact envelope from retained attempt records - never encode, never download + a source, never add missing repetitions, never invent an accepted receipt. + The journal reopen hash-verifies every retained member, so the rebuilt + payload is byte-faithful to what the live path would have written; frozen + manifest IDs (physicalSourceId) are preserved over current-machine values. + Groups still unfinished (checkpoint could extend them) or individually + invalid records stay exactly as unfinished as they are now. Returns counters + with `failure` set only when publishable evidence cannot be honestly + rebuilt - never a silent drop.""" + outcome = {"reconstructed": 0, "skippedExisting": 0, "skippedAccepted": 0, + "skippedIncomplete": 0, "skippedInvalid": 0, + "cancelled": False, "failure": None} + root = journal_path(queue_dir, campaign_id) + manifest_path = root / "manifest.json" + if not manifest_path.is_file(): + return outcome # legacy journal without plan identity: nothing to reconstruct + try: + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + if not isinstance(manifest, dict): + raise ValueError("journal manifest is not an object") + if str(manifest.get("protocolVersion") or "") != config.BENCHMARK_PROTOCOL_VERSION: + raise ValueError("saved protocol identity differs from this client; use the original compatible client") + saved_client_version = str(manifest.get("clientVersion") or "") + if not saved_client_version: + raise ValueError("saved client identity is missing; missing envelopes need the original compatible client") + if saved_client_version != CLIENT_VERSION: + raise ValueError("saved client identity differs from this client; use the original client version") + from .identity import runtime_identity + saved_runtime = manifest.get("runtime") + if saved_runtime is not None and saved_runtime != runtime_identity(): + raise ValueError("saved runtime identity differs from this client; use the original runtime") + pc = dict(manifest.get("protocolConfig") or {}) + protocol_config = ProtocolConfig( + version=str(pc.get("version") or manifest.get("protocolVersion") or ""), + warmup_runs=int(pc.get("warmup_runs", 1)), + minimum_measured_runs=int(pc.get("minimum_measured_runs", 2)), + stability_threshold_ratio=float(pc.get("stability_threshold_ratio", 0.03)), + max_adaptive_repeats=int(pc.get("max_adaptive_repeats", 2)), + environment=EnvironmentThresholds(**dict(pc.get("environment") or {})), + structural_tolerance=StructuralTolerance(**dict(pc.get("structural_tolerance") or {}))) + hardware = HardwareInfo(**{ + key: value for key, value in dict(manifest.get("hardware") or {}).items() + if key in HardwareInfo.__dataclass_fields__}) + except (ValueError, TypeError, OSError) as exc: + outcome["failure"] = f"journal evidence cannot be reopened: {exc}"[:200] + return outcome + # Reopen attempt cells individually and read-only. A corrupt unrelated + # record cannot block an otherwise complete group or rewrite the saved + # budget/manifest merely because the operator chose Publish. + projected = project_attempt_groups(root, campaign_id) + excluded = {item["path"] for item in projected["corruptEntries"]} + record_index: Dict[int, Any] = {} + outcome["corruptEntries"] = list(projected["corruptEntries"]) + for path in sorted(root.glob("attempt-*.json")): + if path.name in excluded: + continue + try: + record = load_record(json.loads(path.read_text(encoding="utf-8"))) + order = int(path.stem.removeprefix("attempt-")) + if (record.schedule.campaign_id != campaign_id + or record.schedule.execution_order != order): + raise ValueError("attempt identity mismatch") + record_index[order] = record + except (OSError, ValueError, TypeError, KeyError, AttributeError) as exc: + outcome["corruptEntries"].append({"path": path.name, "reason": str(exc)[:120]}) + records = [record_index[order] for order in sorted(record_index)] + if not records: + return outcome + recipe_ids: List[str] = [] + for record in records: + if record.schedule.recipe_id not in recipe_ids: + recipe_ids.append(record.schedule.recipe_id) + try: + campaign_result = campaign_result_from_records( + campaign_id=campaign_id, config=protocol_config, + seed=int(manifest.get("seed") or 0), + recipes=[RecipeSpec(recipe_id=recipe_id, expectation=StructuralExpectation()) + for recipe_id in recipe_ids], + records=record_index) + except (TypeError, ValueError, KeyError) as exc: + outcome["failure"] = f"campaign evidence cannot be projected: {exc}"[:200] + return outcome + # Same completeness rule as the live checkpoint path: only recipes the + # scheduler considers final (stable or at the attempt cap) may publish, and + # the group receipt carries exactly the counted timing members. + unfinished = set(getattr(campaign_result, "unfinished_recipes", frozenset())) + groups = _completed_measurement_groups(campaign_result) + ffmpeg_ok, ffmpeg_version = ensure_ffmpeg_and_ffprobe() + if not ffmpeg_ok: + outcome["failure"] = "ffmpeg/ffprobe unavailable; toolchain identity cannot be reconstructed faithfully" + return outcome + for record in records: + if _is_cancelled(cancel_event): + outcome["cancelled"] = True + return outcome + if record.schedule.phase != "measured": + continue + envelope_path = root / f"submission-{record.schedule.execution_order:06d}.json" + if envelope_path.is_file(): + outcome["skippedExisting"] += 1 + continue + if record.schedule.execution_order in projected["acceptedOrders"]: + outcome["skippedAccepted"] += 1 + continue # an accepted receipt already authorizes this attempt + if record.schedule.recipe_id in unfinished: + outcome["skippedIncomplete"] += 1 # a later segment could still extend it + continue + info = dict(record.metadata.get("info") or {}) + if (record.skipped_before_encode or record.timing is None + or record.overall_validity.state == "invalid" + or not info or info.get("error") is not None): + outcome["skippedInvalid"] += 1 # the live path never submitted these + continue + suite_clip_data = record.metadata.get("suiteClip") + try: + prepared_clip = (PreparedSuiteClip(**suite_clip_data) + if isinstance(suite_clip_data, dict) else suite_clip_data) + artifact_path = str(info.get("artifactPath") or "") + if not artifact_path or not os.path.isfile(artifact_path): + raise OSError("retained artifact bytes are missing") + artifact = Path(artifact_path).resolve() + if root.resolve() not in artifact.parents: + raise ValueError("retained artifact is outside its owned campaign") + if CampaignJournal.hash_file(artifact) != info.get("artifactSha256"): + raise ValueError("retained artifact hash differs from the durable attempt") + artifact_probe = probe_video_stream_metrics(artifact_path) + execution_identity_payload = build_execution_identity_payload( + hardware=hardware, + artifact_info=info, + ffmpeg_version=ffmpeg_version, + client_version=CLIENT_VERSION, + benchmark_protocol_version=config.BENCHMARK_PROTOCOL_VERSION, + ) + run_create = _build_authoritative_run_create_request( + prepared_clip=prepared_clip, + recipe_id=record.schedule.recipe_id, + record=record, + info=info, + metrics=dict(record.metadata.get("metrics") or {}), + hardware=hardware, + source_probe=dict(record.metadata.get("sourceProbe") or {}), + artifact_probe=artifact_probe, + ffmpeg_version=ffmpeg_version, + client_version=CLIENT_VERSION, + execution_identity_payload=execution_identity_payload, + protocol_config=protocol_config, + measurement_group=groups.get(record.schedule.recipe_id), + ) + # Frozen plan identity wins over this machine's current values: the + # campaign was requested and sourced under the manifest IDs. + physical_frozen = str(manifest.get("physicalSourceId") or "") + if physical_frozen and run_create.get("physicalSourceId") != physical_frozen: + run_create["physicalSourceId"] = physical_frozen + run_create["payloadHash"] = build_payload_hash(run_create) + submission = build_artifact_submission_payload( + artifact_path=artifact_path, + media_container=artifact_probe.get("containerFormat"), + run_create=run_create, + ) + except (OSError, ValueError, TypeError, KeyError, RuntimeError, AttributeError) as exc: + outcome["failure"] = ( + f"completed group {record.schedule.recipe_id} cannot be reconstructed: {exc}"[:200]) + return outcome + atomic_json(envelope_path, submission) + outcome["reconstructed"] += 1 + return outcome + + +def publish_saved_campaign( + *, + queue_dir: str, + campaign_id: str, + base_url: str, + api_key: str, + max_storage_mb: Optional[int] = None, + retries: int = 1, + use_token: bool = False, + interactive: bool = False, + continue_when_open: bool = False, + cancel_event: Optional[Any] = None, + event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, +) -> Tuple[int, Dict[str, Any]]: + """Publish a saved campaign's completed groups with ZERO encodes (C06/C09/C11). + + Never-encode contract: complete groups whose immutable `submission-*.json` + envelopes already exist are admitted directly; groups whose measured attempts + are durable but whose envelopes never materialized (controlled stop or disk + cap before envelope creation) are REBUILT from retained journal records - + no encoder, suite clip or input source is ever touched, and unfinished + groups stay unfinished. The saved storage budget is restored unless the + caller overrides it; committed receipts drain before any new staging (C10); + accepted groups are skipped by journal receipt and by queue receipt + identity, so replays never re-upload (C07). Publication holds the + host-wide phase lock and defers while any collector measures on this host, + including through a different queue directory (C11). + + Returns (exit_code, info). Exit codes match --upload-only: 0 done, 10 + deferred/pending, 1 terminal failures or corrupt queue evidence. `info` + carries reconciled counters and safe failure fields only.""" + info: Dict[str, Any] = {"campaignId": campaign_id, "status": "working", + "admitted": 0, "skippedAccepted": 0, "terminal": 0, + "deferredReason": None, "failure": None} + if interactive and not _ensure_interactive_publication_consent(queue_dir=queue_dir): + info.update(status="consent_declined", deferredReason="consent_required") + return 10, info + try: + root = journal_path(queue_dir, campaign_id) + except ValueError as exc: + info.update(status="blocked", failure=_safe_failure_info(exc)) + return 1, info + if not root.is_dir(): + info.update(status="blocked", failure=failure_info( + "saved campaign journal was not found", operation="publish_saved", + campaign_id=campaign_id, category="corrupt_evidence", + retryable=False)) + return 1, info + if continue_when_open: + if not _has_publication_consent(): + info.update(status="consent_declined", deferredReason="consent_required") + return 10, info + try: + _save_publication_intent(queue_dir, campaign_id, base_url, + active=True, status="working") + except OSError as exc: + info.update(status="blocked", failure=failure_info( + exc, operation="publication_intent", campaign_id=campaign_id)) + return 1, info + if max_storage_mb is None: + try: + budget = json.loads((root / "budget.json").read_text(encoding="utf-8")) + max_storage_mb = int(budget.get("maxStorageMb") or 0) or 2048 + except (OSError, ValueError, TypeError): + max_storage_mb = 2048 + info["maxStorageMb"] = int(max_storage_mb) + try: + # C11: kernel-backed host phase lock on top of per-queue ownership, so + # two different queue directories on one host can never publish while + # a collector times measurements. Crash releases the flock; nothing is + # ever deleted as "stale". Inspection failure fails closed (defer). + with host_phase_hold("publication"): + rc, result = _publish_saved_campaign_gated( + queue_dir=queue_dir, campaign_id=campaign_id, base_url=base_url, + api_key=api_key, max_storage_mb=int(max_storage_mb), retries=retries, + use_token=use_token, cancel_event=cancel_event, event_sink=event_sink) + except SpoolCapacityError as exc: + info.update(status="deferred", deferredReason="measurement_exclusion", + failure=_safe_failure_info(exc)) + _emit_event(event_sink, "publication_deferred", + **{k: v for k, v in info.items() if k != "failure"}, + reason=info["deferredReason"]) + rc, result = 10, info + if continue_when_open: + active = rc == 10 and result.get("status") not in ("cancelled", "consent_declined") + try: + _save_publication_intent( + queue_dir, campaign_id, base_url, active=active, + status=str(result.get("status") or "pending"), + failure=result.get("failure"), + ) + except OSError as exc: + result.update(status="deferred", deferredReason="intent_persistence_failed", + failure=failure_info(exc, operation="publication_intent", + campaign_id=campaign_id)) + return 10, result + return rc, result + + +def _stage_and_replay_saved_envelopes( + *, + queue_dir: str, + paths: List[Path], + base_url: str, + api_key: str, + max_storage_mb: int, + retries: int, + use_token: bool, + cancel_event: Optional[Any], + info: Dict[str, Any], +) -> None: + """Interleave admission, due replay and retirement until stable. + + A near-full campaign can use one-artifact headroom: drain a staged prefix, + let the queue retire only receipted bytes, then retry the unvisited suffix. + Each pass either visits a new envelope or confirms an upload; no progress + exits with a visible deferred/pending reason instead of spinning. + """ + index = 0 + rounds = 0 + max_rounds = max(2, len(paths) + count_pending_entries(queue_dir) + 1) + while rounds < max_rounds: + rounds += 1 + capacity_blocked = False + while index < len(paths): + if _is_cancelled(cancel_event): + info.update(status="cancelled", deferredReason="cancelled", + pending=count_pending_entries(queue_dir)) + return + path = paths[index] + receipt = path.with_name(f"{path.stem}.accepted.json") + if receipt.is_file(): + info["skippedAccepted"] += 1 + index += 1 + continue + try: + saved = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(saved, dict): + raise ValueError("saved envelope is not an object") + admission = {} + if cancel_event is not None: + admission["cancel_event"] = cancel_event + admission["deadline"] = time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS + spooled_path, entry = spool_payload( + queue_dir, saved, max_storage_mb=max_storage_mb, **admission) + except SpoolCapacityError as exc: + capacity_blocked = True + info.update(deferredReason="storage_or_exclusion", + failure=_safe_failure_info(exc)) + break + except SubmissionCancelled: + info.update(status="cancelled", deferredReason="cancelled", + pending=count_pending_entries(queue_dir)) + return + except SubmitError as exc: + if exc.retryable: + info.update(status="deferred", deferredReason="publication_deadline", + failure=_safe_failure_info(exc), + unadmitted=len(paths) - index, + pending=count_pending_entries(queue_dir)) + return + info["terminal"] += 1 + index += 1 + continue + except (OSError, ValueError, TypeError) as exc: + info["terminal"] += 1 + info.setdefault("entryFailures", []).append( + failure_info(exc, operation="saved_envelope", campaign_id=str(path.parent.name))) + index += 1 + continue + if entry.get("terminal") is True: + info["terminal"] += 1 + elif Path(spooled_path).parent.name == "receipts": + info["skippedAccepted"] += 1 + else: + info["admitted"] += 1 + index += 1 + try: + stats = replay_spool( + queue_dir, base_url=base_url, api_key=api_key, + retries=max(1, int(retries)), use_token=use_token, + cancel_event=cancel_event, + ) + except SpoolCapacityError as exc: + info.update(status="deferred", deferredReason="measurement_exclusion", + failure=_safe_failure_info(exc), + unadmitted=len(paths) - index, + pending=count_pending_entries(queue_dir)) + return + for key, value in (("submitted", stats.submitted), ("retained", stats.retained), + ("deadLettered", stats.dead_lettered), ("corrupt", stats.corrupt), + ("deferred", stats.deferred), ("cancelled", stats.cancelled)): + info[key] = int(info.get(key) or 0) + int(value) + if stats.submitted: + info["drainedResiduals"] = int(info.get("drainedResiduals") or 0) + drain_committed_receipts(queue_dir) + if stats.cancelled or _is_cancelled(cancel_event): + info.update(status="cancelled", deferredReason="cancelled", + pending=count_pending_entries(queue_dir)) + return + if capacity_blocked: + if stats.submitted: + continue + info.update(status="deferred", unadmitted=len(paths) - index, + deferredReason="storage_or_exclusion", + pending=count_pending_entries(queue_dir)) + return + if index < len(paths): + continue + if stats.submitted and count_pending_entries(queue_dir): + continue # More due entries may sit beyond replay_spool's window. + break + if index < len(paths): + info.update(status="deferred", unadmitted=len(paths) - index, + deferredReason="publication_budget") + info["pending"] = count_pending_entries(queue_dir) + + +def _publish_saved_campaign_gated( + *, + queue_dir: str, + campaign_id: str, + base_url: str, + api_key: str, + max_storage_mb: int, + retries: int = 1, + use_token: bool = False, + cancel_event: Optional[Any] = None, + event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, +) -> Tuple[int, Dict[str, Any]]: + """Publish body that runs only while the host publication phase is held.""" + info: Dict[str, Any] = {"campaignId": campaign_id, "status": "working", + "admitted": 0, "skippedAccepted": 0, "terminal": 0, + "unadmitted": 0, + "deferredReason": None, "failure": None, + "maxStorageMb": int(max_storage_mb), "pending": 0} + root = journal_path(queue_dir, campaign_id) + try: + marker = root / "campaign-complete.json" + if marker.exists(): + result = json.loads(marker.read_text()) + info["campaignFailures"] = bool(result.get("skipped") or result.get("failed")) + # Drain crash-residual receipted entries BEFORE staging (C10): committed + # acceptances own no new bytes and free budget for genuinely new work. + info["drainedResiduals"] = drain_committed_receipts(queue_dir) + before = campaign_recovery_state(queue_dir, campaign_id) or {} + accepted_before = int(before.get("acceptedUploads") or 0) + existing = [path for path in sorted(root.glob("submission-*.json")) + if not path.name.endswith(".accepted.json")] + _stage_and_replay_saved_envelopes( + queue_dir=queue_dir, paths=existing, base_url=base_url, + api_key=api_key, max_storage_mb=max_storage_mb, retries=retries, + use_token=use_token, cancel_event=cancel_event, info=info) + if info["status"] in ("cancelled", "deferred"): + return 10, info + # Reconstruct genuinely missing envelopes only AFTER intact saved + # payloads have had their independent publication chance. A runtime or + # journal problem in one group cannot strand the existing envelopes. + projected = project_attempt_groups(root, campaign_id) + missing_orders = [ + order for order in projected["candidateOrders"] + if not (root / f"submission-{order:06d}.json").is_file() + and not (root / f"submission-{order:06d}.accepted.json").is_file() + ] + recon = ({"reconstructed": 0, "failure": None, "cancelled": False} + if not missing_orders else _reconstruct_saved_submissions( + queue_dir=queue_dir, campaign_id=campaign_id, + max_storage_mb=int(max_storage_mb), cancel_event=cancel_event)) + info["reconstructedGroups"] = int(recon.get("reconstructed", 0)) + info["reconstruction"] = {k: v for k, v in recon.items() if k != "reconstructed"} + if recon.get("failure"): + info.update(status="blocked", failure=failure_info( + str(recon["failure"]), operation="reconstruct", campaign_id=campaign_id)) + info["submitted"] = max(0, int((campaign_recovery_state(queue_dir, campaign_id) or {}).get("acceptedUploads") or 0) - accepted_before) + return 1, info + if recon.get("cancelled"): + info.update(status="cancelled", deferredReason="cancelled") + return 10, info + if recon.get("reconstructed"): + newly_materialized = [ + root / f"submission-{order:06d}.json" for order in missing_orders + if (root / f"submission-{order:06d}.json").is_file() + ] + _stage_and_replay_saved_envelopes( + queue_dir=queue_dir, paths=newly_materialized, base_url=base_url, + api_key=api_key, max_storage_mb=max_storage_mb, retries=retries, + use_token=use_token, cancel_event=cancel_event, info=info) + except Exception as exc: # unreadable journal root: honest stop, never encode + info.update(status="blocked", failure=_safe_failure_info(exc)) + return 1, info + after = campaign_recovery_state(queue_dir, campaign_id) or {} + info["submitted"] = max(0, int(after.get("acceptedUploads") or 0) - accepted_before) + info["selectedPending"] = int(after.get("logicalPendingUploads") or 0) + info["unavailableSources"] = int(after.get("unavailableSources") or 0) + info["pending"] = count_pending_entries(queue_dir) + if info["status"] in ("cancelled", "deferred"): + return 10, info + if info.get("corrupt") or info.get("deadLettered") or info["terminal"] or info.get("campaignFailures"): + info["status"] = "terminal_failures" + return 1, info + if info["unavailableSources"]: + info.update(status="blocked", failure=failure_info( + "retained artifact bytes are missing from saved work", + operation="publish_saved", campaign_id=campaign_id, + category="corrupt_evidence", retryable=False)) + return 1, info + if info["unadmitted"]: + info.update(status="deferred", deferredReason=info["deferredReason"] or "storage_or_exclusion") + return 10, info + if info["selectedPending"]: + info.update(status="pending", deferredReason=info["deferredReason"] or "uploads_pending") + return 10, info + info["status"] = "published" + _emit_event(event_sink, "publication_complete", campaignId=campaign_id, + submitted=info.get("submitted", 0)) + return 0, info + + +def retry_due_uploads( + *, + queue_dir: str, + base_url: str, + api_key: str, + retries: int = 1, + use_token: bool = False, + cancel_event: Optional[Any] = None, +) -> Tuple[int, Dict[str, Any]]: + """Retry DUE queued uploads without encodes and without touching campaigns (C12). + + Bounded (due-first window, time budget) and cancellable; ambiguous network + outcomes stay durable for idempotent retry. Exit codes match publish_saved_campaign.""" + drained = drain_committed_receipts(queue_dir) + try: + stats = replay_spool(queue_dir, base_url=base_url, api_key=api_key, + retries=max(1, int(retries)), use_token=use_token, + cancel_event=cancel_event) + except SpoolCapacityError as exc: + return 10, {"status": "deferred", "deferredReason": "measurement_exclusion", + "drainedResiduals": drained, "failure": _safe_failure_info(exc)} + pending = count_pending_entries(queue_dir) + result = {"status": "pending" if pending else "published", + "drainedResiduals": drained, "submitted": stats.submitted, + "retained": stats.retained, "deadLettered": stats.dead_lettered, + "corrupt": stats.corrupt, "deferred": stats.deferred, + "cancelled": stats.cancelled, "pending": pending} + if stats.corrupt or stats.dead_lettered: + result["status"] = "terminal_failures" + return 1, result + if pending: + return 10, result + return 0, result + + +def campaign_recovery_state(queue_dir: str, campaign_id: str) -> Optional[Dict[str, Any]]: + """Project one campaign's logical work from attempts, envelopes and queue. + + One saved envelope and its spool copy are one upload. Complete attempt + groups remain discoverable before envelopes exist; accepted receipts win + even when an older journal lacks its self-published marker. This is read + only and never asks FFmpeg or the original source to prepare. + """ + try: + root = journal_path(queue_dir, campaign_id) + except ValueError: + return None + if not root.is_dir(): + return None + try: + journal_bytes = directory_bytes(str(root)) + except OSError: + journal_bytes = None + state: Dict[str, Any] = {"campaignId": campaign_id, "journalBytes": journal_bytes} + try: + state["savedBudgetMb"] = int(json.loads((root / "budget.json").read_text()).get("maxStorageMb") or 0) + except (OSError, ValueError, TypeError): + state["savedBudgetMb"] = None + state["complete"] = (root / "campaign-complete.json").is_file() + state["publicationIntent"] = publication_intent_state(queue_dir, campaign_id) + projected = project_attempt_groups(root, campaign_id) + state["plannedGroups"] = projected["plannedGroups"] + state["attempts"] = projected["attempts"] + if state["attempts"]: + try: + saved_manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8")) + if not str((saved_manifest or {}).get("clientVersion") or "").strip(): + state["measurementBlocked"] = ( + "saved client identity is missing; use the original compatible client to resume") + except (OSError, ValueError, TypeError, AttributeError): + state["measurementBlocked"] = "saved plan identity is unreadable" + state["completedGroups"] = len(projected["finishedGroupIds"]) + state["incompleteGroups"] = len(projected["incompleteGroupIds"]) + corrupt_entries = list(projected["corruptEntries"]) + if projected["failure"]: + corrupt_entries.append({"path": "manifest.json", "reason": projected["failure"]}) + accepted_orders = set(projected["acceptedOrders"]) + pending_orders = set(projected["candidateOrders"]) + unavailable_orders = set(projected["unavailableOrders"]) + terminal_orders = set() + envelope_orders = set() + queue_root = Path(queue_dir) + for path in sorted(root.glob("submission-*.json")): + if path.name.endswith(".accepted.json"): + continue + try: + order = int(path.stem.removeprefix("submission-")) + saved = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(saved, dict): + raise ValueError("saved envelope is not an object") + local_hash = local_hash_for_payload(saved) + except (OSError, ValueError, TypeError) as exc: + corrupt_entries.append({"path": path.name, "reason": str(exc)[:120]}) + continue + envelope_orders.add(order) + if order in accepted_orders: + pending_orders.discard(order) + unavailable_orders.discard(order) + continue + receipt_path = queue_root / "receipts" / f"{local_hash}.json" + try: + receipt = json.loads(receipt_path.read_text(encoding="utf-8")) + response = receipt.get("response") if isinstance(receipt, dict) else None + run = response.get("benchmarkRun") if isinstance(response, dict) else None + run_id = str(run.get("id") or "").strip() if isinstance(run, dict) else "" + receipted = (isinstance(receipt, dict) + and receipt.get("localHash") == local_hash and bool(run_id)) + except (OSError, ValueError): + receipted = False + if receipted: + accepted_orders.add(order) + pending_orders.discard(order) + unavailable_orders.discard(order) + continue + if (queue_root / "terminal" / f"{local_hash}.json").is_file(): + terminal_orders.add(order) + pending_orders.discard(order) + unavailable_orders.discard(order) + continue + artifact = str(saved.get("artifactPath") or "") + available = bool(artifact and os.path.isfile(artifact)) + queued_path = queue_root / f"{local_hash}.json" + if queued_path.is_file(): + try: + entry = json.loads(queued_path.read_text(encoding="utf-8")) + staged = entry.get("payload") if isinstance(entry, dict) else None + staged_path = str(staged.get("artifactPath") or "") if isinstance(staged, dict) else "" + available = available or bool(staged_path and os.path.isfile(staged_path)) + except (OSError, ValueError): + corrupt_entries.append({"path": queued_path.name, "reason": "queued entry is unreadable"}) + if available: + pending_orders.add(order) + unavailable_orders.discard(order) + else: + pending_orders.discard(order) + unavailable_orders.add(order) + pending_orders.difference_update(accepted_orders | terminal_orders) + unavailable_orders.difference_update(accepted_orders | terminal_orders) + state["savedEnvelopes"] = len(envelope_orders) + state["acceptedUploads"] = len(accepted_orders) + state["pendingUploads"] = len(pending_orders) + state["logicalPendingUploads"] = len(pending_orders) + state["unavailableSources"] = len(unavailable_orders) + state["corruptEntries"] = corrupt_entries + queue_view = campaign_queue_summary(queue_dir, campaign_id) + state["queuePending"] = queue_view["pendingEntries"] + state["queueDue"] = queue_view["dueEntries"] + state["queueAccepted"] = queue_view["acceptedReceipts"] + state["queueTerminal"] = queue_view["terminalEntries"] + state["nextAttemptAt"] = queue_view["nextAttemptAt"] + actions: List[Dict[str, str]] = [] + if queue_view["terminalEntries"] or terminal_orders: + actions.append({"action": "review_terminal", + "why": "a rejected or expired upload is terminal evidence; inspect dead-letter before deciding"}) + if state["logicalPendingUploads"]: + actions.append({"action": "publish_saved", + "why": "completed measured work can publish with zero encodes"}) + if state["unavailableSources"] or corrupt_entries: + actions.append({"action": "inspect_evidence", + "why": "inspect blocked saved entries; keep other completed work available"}) + if not state["complete"]: + if state.get("measurementBlocked"): + actions.append({"action": "inspect_evidence", + "why": state["measurementBlocked"]}) + else: + actions.append({"action": "resume", + "why": "campaign never reached its completion marker; resume continues saved plan"}) + if queue_view["nextAttemptAt"]: + actions.append({"action": "wait", + "why": f"earliest Retry-After arrives at {int(queue_view['nextAttemptAt'])}"}) + state["actions"] = actions + return state + + +def recovery_state(queue_dir: str, campaign_id: str = "") -> Dict[str, Any]: + """Documented cross-process recovery projection (C06) for CLI/GUI consumers. + + Read-only: consent, live-collection exclusion, queue counters and per- + campaign journal state, reconciled. Safe strings only - no server bodies, + tokens or credentials.""" + state: Dict[str, Any] = { + "queueDir": str(queue_dir), + "publicationConsent": _has_publication_consent(), + "activeCollection": None, + "publication": queue_recovery_summary(queue_dir), + "publicationLockBusy": publication_lock_busy(queue_dir), + "campaigns": [], + } + try: + state["activeCollection"] = active_collection(queue_dir) + except OSError as exc: + state["activeCollectionError"] = str(exc)[:200] + wanted = [campaign_id] if campaign_id else [ + name for name in sorted(os.listdir(os.path.join(queue_dir, "campaigns"))) + if name.startswith("campaign-") + ] if os.path.isdir(os.path.join(queue_dir, "campaigns")) else [] + for cid in wanted: + view = campaign_recovery_state(queue_dir, cid) + if view is not None: + state["campaigns"].append(view) + return state + + +def _print_recovery_state(queue_dir: str, campaign_id: str) -> None: + state = recovery_state(queue_dir, campaign_id) + pub = state.get("publication") or {} + print_info(f"Queue: {state['queueDir']}") + print_info(f"Publication consent: {'granted' if state['publicationConsent'] else 'NOT granted (publishing stays local)'}") + if state.get("activeCollection"): + print_warning("A measurement is running on this queue; publication work defers until it checkpoints.") + elif state.get("publicationLockBusy"): + print_warning("Another publisher currently owns this queue; retry shortly.") + print_info( + f"Queue: {pub.get('pendingEntries', 0)} pending ({pub.get('dueEntries', 0)} due), " + f"{pub.get('acceptedReceipts', 0)} accepted receipt(s), {pub.get('terminalEntries', 0)} terminal" + ) + if pub.get("nextAttemptAt"): + print_info(f"Earliest scheduled retry: {time.strftime('%Y-%m-%d %H:%M', time.localtime(int(pub['nextAttemptAt'])))}") + for view in state.get("campaigns", []): + print_info( + f"{view['campaignId']}: {view['completedGroups']} completed group(s), " + f"{view['acceptedUploads']} accepted, {view['pendingUploads']} pending upload, " + f"{view['unavailableSources']} unavailable source(s), " + f"{'complete' if view['complete'] else 'incomplete'}" + ) + for act in view.get("actions", []): + print_info(f" -> {act['action']}: {act['why']}") + if not state.get("campaigns"): + print_info("No retained campaigns in this queue.") + + +def _report_recovery_result(info: Dict[str, Any]) -> None: + status = str(info.get("status") or "unknown") + if status == "published": + print_success(f"Published saved campaign {info.get('campaignId')}: {info.get('submitted', 0)} upload(s), " + f"{info.get('skippedAccepted', 0)} already accepted.") + elif status == "consent_declined": + print_warning("Publication needs your consent; nothing was uploaded.") + elif status == "cancelled": + print_warning("Publishing cancelled; accepted uploads remain recorded and retries stay durable.") + elif status == "deferred": + reason = (failure_text(info["failure"]) if info.get("failure") + else str(info.get("deferredReason") or "storage or exclusion")) + print_warning(f"Publishing deferred: {reason}") + elif status == "terminal_failures": + print_warning(f"{info.get('terminal', 0)} terminal and {info.get('corrupt', 0)} corrupt entr(ies); " + "inspect dead-letter before retrying.") + elif status == "blocked": + print_error(failure_text(info.get("failure") or "campaign journal unavailable")) + else: + print_info(f"Publishing status: {status} ({info.get('selectedPending', 0)} campaign upload(s) still pending)") + + def _submit_payload_with_spool( *, queue_dir: str, @@ -1241,8 +2258,16 @@ def _submit_payload_with_spool( retries: int, use_token: bool, max_storage_mb: int = 2048, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, ) -> Tuple[str, str, int]: - path, _entry = spool_payload(queue_dir, payload, max_storage_mb=max_storage_mb) + operation = {} + if cancel_event is not None: + operation["cancel_event"] = cancel_event + if deadline is not None: + operation["deadline"] = deadline + path, _entry = spool_payload(queue_dir, payload, max_storage_mb=max_storage_mb, + **operation) status, message = submit_spooled_path( path, queue_dir=queue_dir, @@ -1250,23 +2275,17 @@ def _submit_payload_with_spool( api_key=api_key, retries=max(1, retries), use_token=use_token, + **operation, ) return status, message, count_pending_entries(queue_dir) def _has_direct_single_run_intent(raw_args: List[str]) -> bool: """Return True when CLI args explicitly request direct single-run execution.""" - direct_flags = ( - "--codec", - "--presets", - "--crf", - "--submit", - "--no-submit", - "--use-token", - "--retries", - "--queue-dir", - "--batch-size", - ) + # Storage and submission policy configure either flow; they do not choose + # a single recipe. Treating --queue-dir/--no-submit as single-run intent + # bypassed the guided sweep and failed only after source acquisition. + direct_flags = ("--codec", "--presets", "--crf", "--target-bitrate-kbps", "--submit") for token in raw_args: for flag in direct_flags: if token == flag or token.startswith(flag + "="): @@ -1283,14 +2302,14 @@ def _prepare_sweep_clips(clip_policy: str) -> List[PreparedSuiteClip]: """Resolve the frozen suite clips a sweep mode covers, in stable order.""" manifest = load_default_suite_manifest() if clip_policy == sweep_plan.CLIP_POLICY_QUICK: - return [ensure_suite_clip(get_default_quick_clip(manifest))] + return [_prepare_source_clip(get_default_quick_clip(manifest))] if clip_policy == sweep_plan.CLIP_POLICY_CLASSES: prepared: List[PreparedSuiteClip] = [] for content_class in REQUIRED_CONTENT_CLASSES: clip = next((c for c in manifest.clips if c.canonical_content_class == content_class), None) if clip is None: raise RuntimeError(f"EncodingDB Test Suite v1 is missing the {content_class} clip") - prepared.append(ensure_suite_clip(clip)) + prepared.append(_prepare_source_clip(clip)) return prepared return _prepare_full_suite() @@ -1468,9 +2487,9 @@ def run_sweep_mode( print_info(f"Active limits: storage budget {storage_mb} MB, attempts cap {attempts_cap}.") config._BATCH_ACTIVE = True config._BATCH_START_TS = time.perf_counter() - config._BATCH_COMPLETED_COUNT = 0 + config._BATCH_LEDGER = None config._BATCH_ATTEMPTS_RECORDED = 0 - total_submitted = 0 + ledger: Dict[str, Any] = {} try: segment = 0 while True: @@ -1499,7 +2518,7 @@ def run_sweep_mode( cancel_event=cancel_event, plan_metadata=plan_metadata, ) - total_submitted += int(getattr(config, "_BATCH_COMPLETED_COUNT", 0)) + ledger = getattr(config, "_BATCH_LEDGER", None) or {} if rc != 11 or explicit_duration or _is_cancelled(cancel_event): if rc == 11 and _is_cancelled(cancel_event): rc = 130 @@ -1513,7 +2532,7 @@ def run_sweep_mode( print("Safety segment limit reached; campaign remains saved and continues on the next start.", file=sys.stderr) break - config._BATCH_COMPLETED_COUNT = 0 + config._BATCH_LEDGER = None print_info(f"Time checkpoint reached; continuing the campaign from retained evidence (segment {segment + 1}).") _emit_event(event_sink, "campaign_checkpoint_continue", segment=segment + 1) elapsed_sec = max(0.0, time.perf_counter() - config._BATCH_START_TS) @@ -1522,12 +2541,19 @@ def run_sweep_mode( end_status = ("complete" if rc == 0 else "paused" if rc in (10, 11) else "interrupted" if rc == 130 else "failed") - print_end_screen(total_submitted, elapsed_sec, status=end_status, - recovery=None if rc == 0 else - ("Retained campaign saved; start this mode again to continue it." - if rc in (10, 11, 130) else - "Nothing was marked complete; fix the error above and start again — " - "the retained campaign continues from its journal.")) + confirmed = int(ledger.get("uploaded") or 0) + int(ledger.get("savedLocal") or 0) + if rc == 0: + recovery = None + elif rc in (10, 11, 130): + recovery = "Retained campaign saved; start this mode again to continue it." + elif confirmed > 0: + recovery = (f"{confirmed} attempt(s) already confirmed or saved; fix the cause and start " + "again — the retained campaign continues from its journal.") + else: + recovery = ("Nothing was marked complete; fix the error above and start again — " + "the retained campaign continues from its journal.") + print_end_screen(int(ledger.get("uploaded") or 0), elapsed_sec, status=end_status, + recovery=recovery, ledger=ledger or None) try: if os.name == "nt" and (bool(getattr(base_args, "pause_on_exit", False)) or bool(getattr(sys, "frozen", False))): input("Press Enter to exit...") @@ -1548,11 +2574,14 @@ def sweep_plan_label(encoder: str) -> str: def active_collection_guard(queue_dir: str, event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, *, scope: str = "batch") -> Optional[int]: - """Refusal code when this queue already hosts a live collection, else None. + """Refusal code when this queue already hosts a live collection or a live + publication pass, else None. A live collector holds the measurement lock for its whole campaign: its checkpoints continue automatically, so a second run must wait for it to - finish or stop/cancel it first - never "resume over" a checkpoint.""" + finish or stop/cancel it first - never "resume over" a checkpoint. The + reverse direction holds too: while a publisher drains this queue, starting + a collector would corrupt timing through the same disk (C11).""" try: active = active_collection(str(queue_dir)) except OSError as exc: @@ -1561,6 +2590,38 @@ def active_collection_guard(queue_dir: str, _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) return 6 if active is None: + try: + publisher_busy = publication_lock_busy(str(queue_dir)) + except OSError as exc: + message = f"Cannot verify publication exclusion: {exc}" + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + if publisher_busy: + # C11 reverse direction: a publisher is draining/staging this queue's + # bytes right now. Its replay is time-bounded; starting a collector + # mid-drain would corrupt measurement timing through the same disk. + message = ("A publication pass currently owns this queue; it ends within its time budget. " + "Wait for it to finish, then start the collection.") + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + try: + # C11 cross-queue: a publisher or collector owns the HOST-wide phase + # through a different queue directory; a non-blocking probe defers + # this start, and an uninspectable lock fails closed (exit 6). + if host_phase_busy(): + message = ("A publication or measurement pass owns this host right now " + "(possibly through another queue directory); it is time-bounded. " + "Wait for it to finish, then start the collection.") + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + except OSError as exc: + message = f"Cannot verify host phase exclusion: {exc}" + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 return None who = f" (campaign {active['campaignId']}, PID {active['pid']})" if active.get("campaignId") else "" message = (f"Another collection is actively running in this queue{who}. Its checkpoints continue " @@ -1572,7 +2633,39 @@ def active_collection_guard(queue_dir: str, +def _collector_publication(function): + """Mark this process as the host's live collector for its whole batch (C11). + + The collector's own checkpoint uploads legitimately run while it holds the + measurement lock and the host measurement phase; the exclusion probes refuse + every OTHER publisher, in-process re-entry (checkpoint upload) passes. The + kernel phase lock is the atomic arbiter across queue directories: a held + lock refuses this collector with exit 6 instead of letting two hosts' worth + of I/O overlap through different queues.""" + @wraps(function) + def wrapped(*args, **kwargs): + queue_dir = str(getattr(kwargs.get("args"), "queue_dir", "") or "") + if not queue_dir: + with collector_publication_scope(): + return function(*args, **kwargs) + try: + hold = host_phase_hold("measurement") + hold.__enter__() + except SpoolCapacityError as exc: + message = str(exc) + print(message, file=sys.stderr) + _emit_event(kwargs.get("event_sink"), "run_error", scope="batch", code=6, message=message) + return 6 + try: + with collector_publication_scope(): + return function(*args, **kwargs) + finally: + hold.__exit__(None, None, None) + return wrapped + + @_preparation_operation +@_collector_publication def run_benchmark_batch( *, hardware: HardwareInfo, @@ -1652,6 +2745,28 @@ def run_benchmark_batch( "tasks": [{"encoder": t["encoder"], "preset": t["preset"], "crf": t.get("crf"), "rateControl": t.get("rateControl"), "clipId": t["suiteClip"].clip_id} for t in tasks], } + existing_manifest = journal_path(args.queue_dir, campaign_id) / "manifest.json" + if existing_manifest.is_file(): + try: + prior_version = str(json.loads(existing_manifest.read_text()).get("clientVersion") or "") + except (OSError, ValueError, TypeError, AttributeError): + prior_version = "" # CampaignJournal reports an unreadable manifest below. + if not prior_version and any(existing_manifest.parent.glob("attempt-*.json")): + message = ("Saved client identity is missing for measured attempts; " + "use the original compatible client to resume measurement. " + "Intact saved envelopes can still be published without encoding.") + _emit_event(event_sink, "run_error", scope="batch", code=6, message=message) + print(message, file=sys.stderr) + return 6 + if prior_version and prior_version != client_version: + message = f"Saved campaign requires {prior_version}; this client is {client_version}. Use the original client." + _emit_event(event_sink, "run_error", scope="batch", code=6, message=message) + print(message, file=sys.stderr) + return 6 + if prior_version: + manifest["clientVersion"] = prior_version + else: + manifest["clientVersion"] = client_version if isinstance(plan_metadata, dict): manifest.update(plan_metadata) try: @@ -1671,7 +2786,7 @@ def run_benchmark_batch( typical = observed[len(observed) // 2] if any(observed) else 12 * 1024 * 1024 needed = pending_attempts * typical if needed > remaining_bytes: - print_info(f"Planned attempts could need up to ≈{needed // (1024 * 1024)} MB of retention while " + print_info(f"Planned attempts could need up to about {needed // (1024 * 1024)} MB of retention while " f"this campaign's remaining allowance is {remaining_bytes // (1024 * 1024)} MB. " f"Completed groups upload and retire automatically at checkpoints; if the volume " f"allows, --max-storage-mb raises the allowance.") @@ -1687,6 +2802,8 @@ def run_benchmark_batch( api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) if not getattr(args, 'no_submit', False): baseline_rows = fetch_baseline_rows(base_url) @@ -1696,15 +2813,74 @@ def run_benchmark_batch( submitted_count = 0 skipped_count = 0 failed_count = 0 + locally_complete_count = 0 + measured_cap = int(protocol_config.minimum_measured_runs + protocol_config.max_adaptive_repeats) + groups_total = len(recipe_specs) + # Both bars show measurement work, not publication. Seed from every + # durable attempt so warmups move the batch bar and resume never rewinds. + recorded_orders = set(journal.records) + warmups_done = sum(record.schedule.phase == "warmup" for record in journal.records.values()) + measured_done = sum(record.schedule.phase == "measured" for record in journal.records.values()) + local_only_run = bool(getattr(args, "no_submit", False)) + done_before = len(recorded_orders) + attempts_done = done_before + batch_done = 0 + per_recipe_cap = int(protocol_config.warmup_runs) + measured_cap + declared_attempts = max(1, groups_total * per_recipe_cap) + batch_attempts_total = max(1, declared_attempts - done_before) + declared_batches_total = batch_attempts_total + progress: Optional[Any] = None + + def _emit_progress() -> None: + total_n = max(1, declared_attempts) + done_n = max(0, min(attempts_done, total_n)) + batch_total_n = max(1, batch_attempts_total) + batch_done_n = max(0, min(batch_done, batch_total_n)) + _emit_event( + event_sink, + "campaign_progress", + scope="batch", + campaignId=campaign_id, + unit="durable-attempt", + done=done_n, + total=total_n, + batchDone=batch_done_n, + batchTotal=batch_total_n, + groupsTotal=groups_total, + warmupsDone=warmups_done, + measuredDone=measured_done, + ) + if progress is not None: + progress.set_progress(done=done_n, total=total_n, + batch_done=batch_done_n, batch_total=batch_total_n) + + def _record_attempt(record: Any) -> None: + nonlocal attempts_done, batch_done, warmups_done, measured_done + order = int(record.schedule.execution_order) + if order not in recorded_orders: + recorded_orders.add(order) + attempts_done += 1 + batch_done += 1 + if record.schedule.phase == "warmup": + warmups_done += 1 + elif record.schedule.phase == "measured": + measured_done += 1 + _emit_progress() + _emit_event( event_sink, "run_start", scope="batch", totalTasks=total_tasks, + totalGroups=groups_total, + declaredAttempts=declared_attempts, totalBatches=total_batches, workers=workers, noSubmit=bool(getattr(args, "no_submit", False)), maxDurationMinutes=duration_minutes, + campaignId=campaign_id, + progressUnit="durable-attempt", + doneTotal=done_before, protocol={ "version": protocol_config.version, "warmupRuns": protocol_config.warmup_runs, @@ -1724,26 +2900,28 @@ def run_benchmark_batch( def _batch_status(stage: str, index: int, codec: str = "", preset: str = "") -> str: label = f"{codec} {preset}".strip() - stats = f"ok={submitted_count} skip={skipped_count} queue={queued_count} fail={failed_count}" + stats = f"ok={submitted_count} local={locally_complete_count} skip={skipped_count} queue={queued_count} fail={failed_count}" total = max(1, total_tasks) if label: return f"{stage} {index}/{total}: {label} | {stats}" return f"{stage} {index}/{total} | {stats}" + _emit_progress() try: with journal.measurement_lock(), nullcontext(str(journal.root)) as batch_dir, \ BatchRunDashboard(total_tasks=total_tasks, total_batches=total_batches, hardware=hardware) as progress: print_info(f"Batch 1/{total_batches}: {len(recipe_specs)} protocol recipe(s)") - progress.start_batch(batch_no=1, batch_size=total_tasks) + progress.start_batch(batch_no=1, batch_size=declared_batches_total) progress.set_description(_batch_status("Batch 1/1 preparing", 1)) _emit_event( event_sink, "batch_start", batchNo=1, totalBatches=total_batches, - batchSize=len(recipe_specs), - processedTotal=processed_total, + batchDeclaredAttempts=declared_batches_total, + campaignId=campaign_id, ) + _emit_progress() def _task_from_recipe(recipe: RecipeSpec) -> Dict[str, Any]: return { @@ -1935,6 +3113,7 @@ def _encode_protocol_run(schedule: Any, recipe: RecipeSpec) -> EncodeOutcome: def _journal_attempt(record: Any) -> None: journal.save(record) + _record_attempt(record) if record.schedule.execution_order not in recorded_before: with config._GLOBAL_STATE_LOCK: config._BATCH_ATTEMPTS_RECORDED += 1 @@ -1991,6 +3170,18 @@ def _journal_attempt(record: Any) -> None: for record in recipe_result.runs: if record.schedule.phase == "measured": measured_records.append((recipe, record)) + # Optional adaptive slots disappear only after a group is terminal. + # Remaining unfinished groups retain their frozen maximum; totals + # shrink rather than pretending unused optional encodes were done. + record_counts: Dict[str, int] = {} + for recorded in journal.records.values(): + key = str(recorded.schedule.recipe_id) + record_counts[key] = record_counts.get(key, 0) + 1 + remaining = sum(max(0, per_recipe_cap - record_counts.get(recipe_id, 0)) + for recipe_id in unfinished_recipes) + declared_attempts = max(1, attempts_done + remaining) + batch_attempts_total = max(1, declared_attempts - done_before) + _emit_progress() for recipe, record in measured_records: if not getattr(args, "local_metrics", False): @@ -2058,7 +3249,9 @@ def _journal_attempt(record: Any) -> None: if _is_cancelled(cancel_event): raise KeyboardInterrupt if journal.accepted_receipt(record) is not None: - continue # Faithful accepted receipt from an earlier segment. + # Publication is already reconciled in the durable ledger; + # it does not change measurement progress. + continue task = _task_from_recipe(recipe) info = dict(record.metadata.get("info") or {}) codec_label = str(info.get('encoderUsed') or task['encoder']) @@ -2098,6 +3291,7 @@ def _journal_attempt(record: Any) -> None: progress.update_counters( submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count, + locally=locally_complete_count, ) _emit_event( event_sink, @@ -2119,8 +3313,9 @@ def _journal_attempt(record: Any) -> None: failed=failed_count, ) processed_total += 1 + # Invalid evidence remains a durable measurement attempt + # in the bars, but never advances publication counters. progress.advance(description=_batch_status("Completed", processed_total, codec_label, preset_label)) - _emit_event(event_sink, "task_complete", scope="batch", processed=processed_total, total=total_tasks) continue prepared_clip = task.get("suiteClip") @@ -2263,6 +3458,7 @@ def _journal_attempt(record: Any) -> None: progress.update_counters( submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count, + locally=locally_complete_count, ) _emit_event( event_sink, @@ -2315,8 +3511,28 @@ def _journal_attempt(record: Any) -> None: else: atomic_json(local_path, authoritative_submission) if args.no_submit: - _emit_event(event_sink, "submit_result", status="locally_complete", campaignId=campaign_id) + # Local-only is a completed unit of work: the global counters and + # task_complete must advance exactly like a submitted path (C04/C05). + # 'submitted' stays the upload count honestly: zero here. + locally_complete_count += 1 completed_count_local += 1 + processed_total += 1 + progress.advance(description=_batch_status("Locally complete", processed_total)) + _emit_event( + event_sink, + "submit_result", + index=next_index, + total=total_tasks, + status="locally_complete", + campaignId=campaign_id, + recipeId=recipe.recipe_id, + repetitionIndex=record.schedule.repetition_index, + executionOrder=record.schedule.execution_order, + ) + _emit_event(event_sink, "task_complete", scope="batch", + processed=processed_total, total=total_tasks) + _emit_counters(event_sink, submitted=submitted_count, skipped=skipped_count, + queued=queued_count, failed=failed_count) continue status, error_text, queued_count = _submit_payload_with_spool( queue_dir=args.queue_dir, @@ -2326,6 +3542,8 @@ def _journal_attempt(record: Any) -> None: api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) if status == "submitted": submitted_count += 1 @@ -2354,6 +3572,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="queued", error=error_text, + **_submit_failure_fields("queued", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2369,6 +3588,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="failed", error=error_text, + **_submit_failure_fields("failed", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2388,6 +3608,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="failed", error=error_text, + **_submit_failure_fields("failed", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2396,6 +3617,7 @@ def _journal_attempt(record: Any) -> None: progress.update_counters( submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count, + locally=locally_complete_count, ) _emit_counters( event_sink, @@ -2404,13 +3626,8 @@ def _journal_attempt(record: Any) -> None: queued=queued_count, failed=failed_count, ) - if float(payload.get('fps', 0.0)) > 0.0 and int(payload.get('fileSizeBytes', 0)) > 0: completed_count_local += 1 - if config._BATCH_ACTIVE: - with config._GLOBAL_STATE_LOCK: - config._BATCH_COMPLETED_COUNT += 1 - processed_total += 1 progress.advance(description=_batch_status("Completed", processed_total, str(payload['codec']), str(payload['preset']))) _emit_event(event_sink, "task_complete", scope="batch", processed=processed_total, total=total_tasks) @@ -2465,8 +3682,14 @@ def _journal_attempt(record: Any) -> None: return 6 except KeyboardInterrupt: print_warning("Batch run interrupted by user.") + _emit_progress() _emit_event(event_sink, "run_interrupted", scope="batch", processed=processed_total, total=total_tasks) return 130 + finally: + # Durable truth for the end screen: recomputed from journal + spool on + # every exit path, including cancel and budget pause. + config._BATCH_LEDGER = _durable_campaign_ledger( + args.queue_dir, campaign_id, journal, local_only_run) atomic_json(journal.root / "campaign-complete.json", {"campaignId": campaign_id, "skipped": skipped_count, "failed": failed_count}) elapsed_seconds = max(0.0, time.perf_counter() - run_started_at) @@ -2476,6 +3699,7 @@ def _journal_attempt(record: Any) -> None: "totalBatches": total_batches, "completed": completed_count_local, "submitted": submitted_count, + "locallyComplete": locally_complete_count, "skipped": skipped_count, "queued": queued_count, "failed": failed_count, @@ -2490,6 +3714,7 @@ def _journal_attempt(record: Any) -> None: totalBatches=total_batches, completed=completed_count_local, submitted=submitted_count, + locallyComplete=locally_complete_count, skipped=skipped_count, queued=queued_count, failed=failed_count, @@ -2785,6 +4010,8 @@ def run_legacy_diagnostic( api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) submitted_count = 0 skipped_count = 0 @@ -2895,9 +4122,6 @@ def run_legacy_diagnostic( if float(payload.get("fps", 0.0)) > 0.0 and int(payload.get("fileSizeBytes", 0)) > 0: completed_count += 1 - if config._BATCH_ACTIVE: - with config._GLOBAL_STATE_LOCK: - config._BATCH_COMPLETED_COUNT += 1 if args.no_submit: print_info(f"Dry-run: not submitting preset={effective_preset}") @@ -2926,6 +4150,8 @@ def run_legacy_diagnostic( api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) if status == "submitted": queued_count = _replay_pending_uploads( @@ -2934,6 +4160,8 @@ def run_legacy_diagnostic( api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) submitted_count += 1 print_success("Submitted Results") @@ -2942,12 +4170,12 @@ def run_legacy_diagnostic( elif status == "retained": print_warning(f"Queued for retry: {effective_preset} ({message})") progress.advance(description=f"{effective_preset} (queued)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="queued", preset=effective_preset, error=message) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="queued", preset=effective_preset, error=message, **_submit_failure_fields("queued", message)) else: failed_count += 1 print(f"Failed to submit {effective_preset}: {message}", file=sys.stderr) progress.advance(description=f"{effective_preset} (failed)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=message) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=message, **_submit_failure_fields("failed", message)) except SpoolCapacityError as exc: print_warning(f"Upload deferred: {exc}") return 10 @@ -2956,7 +4184,7 @@ def run_legacy_diagnostic( print(f"Failed to submit {effective_preset}: {e}", file=sys.stderr) queued_count = count_pending_entries(args.queue_dir) progress.advance(description=f"{effective_preset} (failed)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=str(e)) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=str(e), **_submit_failure_fields("failed", str(e))) _emit_counters(event_sink, submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count) _emit_event(event_sink, "task_complete", scope="single", processed=task_index, total=len(combos), preset=effective_preset) @@ -2967,6 +4195,8 @@ def run_legacy_diagnostic( api_key=args.api_key, retries=max(1, args.retries), use_token=use_token, + cancel_event=cancel_event, + deadline=time.monotonic() + ACTIVE_PUBLICATION_DEADLINE_SECONDS, ) _emit_counters(event_sink, submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count) elapsed_sec = max(0.0, time.perf_counter() - benchmark_start_ts) @@ -3043,7 +4273,34 @@ def _incomplete_campaigns(queue_dir: str) -> List[Tuple[str, float]]: continue found.append((name, os.path.getmtime(entry))) found.sort(key=lambda item: item[1], reverse=True) - return found[:5] + return found + except Exception: + return [] + + +def _publishable_campaigns(queue_dir: str) -> List[Tuple[str, float, int]]: + """All campaigns with logical finished work available for zero-encode Publish.""" + try: + root = os.path.join(queue_dir, "campaigns") + if not os.path.isdir(root): + return [] + found: List[Tuple[str, float, int]] = [] + for name in os.listdir(root): + entry = os.path.join(root, name) + if not os.path.isdir(entry): + continue + try: + state = campaign_recovery_state(queue_dir, name) + except (OSError, ValueError): + continue # A damaged sibling must not hide other saved work. + if state is None: + continue + pending = int(state.get("logicalPendingUploads") or 0) + if pending <= 0: + continue + found.append((name, os.path.getmtime(entry), pending)) + found.sort(key=lambda item: item[1], reverse=True) + return found except Exception: return [] @@ -3141,6 +4398,34 @@ def interactive_menu_flow(parser: argparse.ArgumentParser, base_args: argparse.N subprocess.run(["stty", "sane"], check=False) except Exception: pass + # Saved envelopes require neither a measurement runtime nor source media. + # Offer their normal recovery actions before any FFmpeg/encoder discovery. + saved_root = Path(base_args.queue_dir) / "campaigns" + if saved_root.is_dir() and any(path.is_dir() for path in saved_root.iterdir()): + early_actions: List[Tuple[str, Optional[str]]] = [] + early_labels: List[str] = [] + for campaign_id, _mtime, pending in _publishable_campaigns(base_args.queue_dir): + early_actions.append(("publish", campaign_id)) + early_labels.append( + f"Publish saved work for {campaign_id} ({pending} upload(s), no encoding)") + early_actions.extend([("recovery", None), ("continue", None), ("exit", None)]) + early_labels.extend(["Show saved-work status", "Continue to contributions", "Exit"]) + early_choice = prompt_choice("Saved work is available", early_labels, default_index=0) + early_action, early_campaign = early_actions[early_choice] + if early_action == "publish": + rc, info = publish_saved_campaign( + queue_dir=base_args.queue_dir, campaign_id=str(early_campaign), + base_url=base_args.base_url, api_key=base_args.api_key, + retries=max(1, int(getattr(base_args, "retries", 1) or 1)), + use_token=bool(getattr(base_args, "use_token", False)), + interactive=True) + _report_recovery_result(info) + return rc + if early_action == "recovery": + _print_recovery_state(base_args.queue_dir, "") + return 0 + if early_action == "exit": + return 0 ffmpeg_ok, _ffmpeg_version = ensure_ffmpeg_and_ffprobe() if not ffmpeg_ok: print_error("ffmpeg/ffprobe were not found in PATH. Install ffmpeg (https://ffmpeg.org/download.html), then start EncodingDB again.") @@ -3185,8 +4470,13 @@ def interactive_menu_flow(parser: argparse.ArgumentParser, base_args: argparse.N f"{_guided_mode_label(mode, previews[mode], per_recipe_min, per_recipe_max)}]" ) actions.append(("sweep", mode)) + for campaign_id, _mtime, pending in _publishable_campaigns(base_args.queue_dir): + option_labels.append(f"Publish saved uploads for {campaign_id} ({pending} completed group(s), no re-encode)") + actions.append(("publish", campaign_id)) option_labels.append("Advanced: configure one recipe yourself (encoder, preset, native quality or bitrate)") actions.append(("single", None)) + option_labels.append("Show recovery status (read-only: consent, exclusion, queue and campaign counters)") + actions.append(("recovery", None)) option_labels.append("Exit") actions.append(("exit", None)) default_index = next((index for index, (action, _payload) in enumerate(actions) if action == "sweep"), 0) @@ -3197,6 +4487,18 @@ def interactive_menu_flow(parser: argparse.ArgumentParser, base_args: argparse.N if action == "resume": base_args.resume_campaign = payload return _resume_campaign(base_args, interactive=True) + if action == "publish": + rc, info = publish_saved_campaign( + queue_dir=base_args.queue_dir, campaign_id=str(payload), + base_url=base_args.base_url, api_key=base_args.api_key, + retries=max(1, int(getattr(base_args, "retries", 1) or 1)), + use_token=bool(getattr(base_args, "use_token", False)), + interactive=True) + _report_recovery_result(info) + return rc + if action == "recovery": + _print_recovery_state(base_args.queue_dir, "") + return 0 if action == "sweep": if not confirm_benchmark_readiness(): print("Aborted by user. Please close other programs and try again.") @@ -3243,6 +4545,12 @@ def build_arg_parser() -> argparse.ArgumentParser: p.add_argument("--campaign", choices=("quick", "full"), default="quick", help="One clip (quick) or all seven clips (full), with the selected recipe") p.add_argument("--resume-campaign", default="", help="Resume a retained campaign ID, preserving completed attempts") p.add_argument("--upload-only", action="store_true", help="Retry due queued uploads without encoding") + p.add_argument("--publish-saved", default="", metavar="CAMPAIGN_ID", + help="Publish a saved campaign's completed groups with zero encodes " + "(accepted uploads are skipped; nothing unfinished is materialized)") + p.add_argument("--recovery-status", action="store_true", + help="Print a read-only JSON recovery projection (consent, exclusion, queue and " + "campaign counters with concrete next actions), then exit") p.add_argument("--local-metrics", action="store_true", help="Run optional local quality diagnostics after all measurements") p.add_argument("--max-attempts", type=int, default=100, action=_ExplicitBudgetAction, help="Maximum planned warmup/measured encodes (default 100; guided sweeps size the cap " @@ -3270,6 +4578,14 @@ def run_windows_gui_flow(args: argparse.Namespace) -> int: def main(argv: List[str]) -> int: + # Packaged Windows console output can be CP1252 even when logs are + # redirected. Keep status/error reporting alive for Unicode paths and + # messages instead of losing the campaign to UnicodeEncodeError. + for stream in (sys.stdout, sys.stderr): + try: + stream.reconfigure(errors="backslashreplace") + except (AttributeError, OSError, ValueError): + pass if len(argv) > 1 and argv[1].startswith('--multiprocessing-fork'): return 0 @@ -3301,6 +4617,15 @@ def main(argv: List[str]) -> int: return 10 if not args.queue_status: return 0 + if args.recovery_status: + print(json.dumps(recovery_state(args.queue_dir, args.publish_saved or args.resume_campaign), + indent=2, sort_keys=True)) + return 0 + if args.publish_saved: + if args.resume_campaign and args.resume_campaign != args.publish_saved: + parser.error("--publish-saved and --resume-campaign name different campaigns") + args.resume_campaign = args.publish_saved + args.upload_only = True if args.queue_status: _print_queue_status(args.queue_dir) return 0 @@ -3320,33 +4645,49 @@ def main(argv: List[str]) -> int: if args.upload_only: if args.no_submit: parser.error("--upload-only cannot be combined with --no-submit") + cancel_event = threading.Event() + previous_sigint = None try: check_compatibility(args.base_url, CLIENT_VERSION) - campaign_failures = False - publication_deferred = False + if threading.current_thread() is threading.main_thread(): + previous_sigint = signal.getsignal(signal.SIGINT) + signal.signal(signal.SIGINT, lambda _signum, _frame: cancel_event.set()) + storage_mb = (int(args.max_storage_mb) if bool(getattr(args, "max_storage_mb_explicit", False)) + else None) if args.resume_campaign: - root = journal_path(args.queue_dir, args.resume_campaign) - marker = root / "campaign-complete.json" - if marker.exists(): - result = json.loads(marker.read_text()) - campaign_failures = bool(result.get("skipped") or result.get("failed")) - for path in sorted(root.glob("submission-*.json")): - try: - _spooled_path, entry = spool_payload(args.queue_dir, json.loads(path.read_text()), - max_storage_mb=args.max_storage_mb) - except SpoolCapacityError as exc: - print_warning(f"Upload deferred: {exc}") - publication_deferred = True - break - if entry.get("terminal") is True: - campaign_failures = True - print_warning(f"Retained upload is terminal ({entry.get('lastError') or 'terminal_upload'}); see {_spooled_path}.") - stats = replay_spool(args.queue_dir, base_url=args.base_url, api_key=args.api_key, - retries=1, use_token=False) - return 1 if campaign_failures or stats.dead_lettered or stats.corrupt else (10 if publication_deferred or count_pending_entries(args.queue_dir) else 0) + rc, info = publish_saved_campaign( + queue_dir=args.queue_dir, campaign_id=args.resume_campaign, + base_url=args.base_url, api_key=args.api_key, + max_storage_mb=storage_mb, retries=max(1, args.retries), + cancel_event=cancel_event) + else: + rc, info = retry_due_uploads(queue_dir=args.queue_dir, base_url=args.base_url, + api_key=args.api_key, retries=max(1, args.retries), + cancel_event=cancel_event) + if cancel_event.is_set() or info.get("status") == "cancelled": + print_warning("Upload cancelled; saved work remains recoverable.") + return 130 + for warning in (("Retained uploads are terminal; inspect the dead-letter before retrying." + if info.get("deadLettered") else ""), + (f"Corrupt queue files moved to dead-letter: {info.get('corrupt')}." + if info.get("corrupt") else ""), + (f"Saved campaign has terminal or skipped work: {info.get('terminal', 0)} entry(ies)." + if info.get("terminal") else "")): + if warning: + print_warning(warning) + if info.get("deferredReason") == "storage_or_exclusion": + print_warning(f"Upload deferred: {(info.get('failure') or {}).get('reason') or 'storage budget'}") + return rc + except KeyboardInterrupt: + cancel_event.set() + print_warning("Upload cancelled; saved work remains recoverable.") + return 130 except Exception as exc: print(f"Upload deferred: {exc}", file=sys.stderr) return 10 + finally: + if previous_sigint is not None: + signal.signal(signal.SIGINT, previous_sigint) if args.resume_campaign: return _resume_campaign(args) if getattr(args, "v7_suite_clip", ""): diff --git a/client/network.py b/client/network.py index 32819082..b934418c 100644 --- a/client/network.py +++ b/client/network.py @@ -1,14 +1,169 @@ +import contextvars import hashlib import json import re +import socket import sys +import threading import time import warnings -from typing import Optional, Dict, Any, List +from typing import Any, Callable, Dict, List, Optional from urllib.parse import urljoin from . import config +# C12: default wall-clock bound on one submit() transaction chain (token fetches, +# POSTs, redirects and waits). Callers may pass a shorter/longer value. +SUBMIT_TRANSACTION_SECONDS = 90.0 + +# R05: owned-worker lifecycle. Transport workers are registered process-wide; +# an operation that owns a phase/exclusion scope reaps its workers on release +# (bounded sync join up to WORKER_QUIESCE_SYNC_JOIN_SECONDS). If a worker is +# still mid-I/O past that window, the scope's exclusion ownership is retained +# (deferred releaser) until the worker is quiescent, so a cancelled or +# timed-out worker can never perform I/O after ownership moved on. +WORKER_QUIESCE_SYNC_JOIN_SECONDS = 1.0 + +# R04: hard cap on bytes CONSUMED from an error/aux response body (the +# retained text is additionally truncated by SubmitError). Consumption stops +# at the cap, the actual deadline, or cancellation — then the response is +# closed to interrupt any blocked read and return the connection. +ERROR_BODY_HARD_CAP_BYTES = 65536 +JSON_BODY_HARD_CAP_BYTES = 1 << 22 + + +_OWNED_WORKERS: set = set() +_OWNED_WORKERS_LOCK = threading.Lock() +_ACTIVE_WORKER_PHASE: contextvars.ContextVar = contextvars.ContextVar( + "encodingdb_owned_worker_phase", default=None) + + +def owned_worker_census() -> int: + """Process-wide count of live transport workers (ownership census, R05).""" + with _OWNED_WORKERS_LOCK: + return sum(1 for record in _OWNED_WORKERS if record.alive) + + +class _WorkerRecord: + """One owned transport worker: thread + completion latch.""" + + def __init__(self, phase: str) -> None: + self.phase = phase + self.thread: Optional[threading.Thread] = None + self.done = threading.Event() + # Set by the caller exactly when it abandons the call (cancel or + # deadline). The worker checks it after func() returns so a response + # that will never be delivered is closed by the worker itself. + self.abandoned = False + + @property + def alive(self) -> bool: + return not self.done.is_set() + + +class WorkerGroup: + """Workers owned by one operation scope (R05). + + Release semantics: the scope holder calls `end_owned_worker_phase` which + reaps owned workers with a bounded sync join (WORKER_QUIESCE_SYNC_JOIN_ + SECONDS). If any worker is still mid-I/O past that window it is handed to + a background releaser that joins it (its own socket timeout bounds the + wait) and only then invokes the scope's release callback — so exclusion + ownership is retained, never released, while an owned worker can still + perform I/O.""" + + def __init__(self) -> None: + self._lock = threading.Lock() + self._records: List[_WorkerRecord] = [] + self.token: Optional[Any] = None + + def _register(self, record: _WorkerRecord) -> None: + with self._lock: + self._records.append(record) + + def await_quiescence(self, sync_join_seconds: float) -> bool: + """True when every owned worker finished within the join window.""" + deadline = time.monotonic() + max(0.0, float(sync_join_seconds)) + while True: + with self._lock: + pending = [record for record in self._records if not record.done.is_set()] + if not pending: + return True + for record in pending: + record.thread.join(max(0.001, deadline - time.monotonic())) + if time.monotonic() >= deadline: + with self._lock: + return not any(record.done.is_set() is False + for record in self._records) + + def stragglers(self) -> List[_WorkerRecord]: + with self._lock: + return [record for record in self._records if not record.done.is_set()] + + def _release_when_quiescent(self, records: List[_WorkerRecord], + release: Callable[[], None]) -> None: + try: + for record in records: + # Bounded by the worker's own socket timeout: an abandoned + # worker cannot outlive its phase budget against a stalled peer. + record.thread.join() + finally: + release() + + +def begin_owned_worker_phase() -> Optional[WorkerGroup]: + """Open an owned-worker phase on the calling thread (R05). + + Returns the group to close with `end_owned_worker_phase`, or None when an + outer phase is already active in this thread — nested operations + (collector checkpoint uploads inside a held batch, per-entry submits + inside a replay pass) register with the outer phase, which owns + quiescence.""" + if _ACTIVE_WORKER_PHASE.get() is not None: + return None + group = WorkerGroup() + group.token = _ACTIVE_WORKER_PHASE.set(group) + return group + + +def end_owned_worker_phase(group: Optional[WorkerGroup], + release: Callable[[], None]) -> bool: + """Close an owned-worker phase; run `release` only once owned workers are + quiescent. Returns True when released synchronously, False when the + release was deferred to a background joiner (exclusion stays held). + + The phase contextvar is always reset on the path that owns it, so a + later phase on the same thread opens a fresh group instead of silently + attaching to a closed one (which would release exclusion while its own + cancelled worker was still mid-I/O).""" + if group is None: + release() + return True + try: + if group.await_quiescence(WORKER_QUIESCE_SYNC_JOIN_SECONDS): + release() + return True + stragglers = group.stragglers() + threading.Thread(target=group._release_when_quiescent, + args=(stragglers, release), + name="encodingdb-worker-releaser", daemon=True).start() + return False + finally: + _ACTIVE_WORKER_PHASE.reset(group.token) + + +def _register_worker(record: _WorkerRecord) -> None: + with _OWNED_WORKERS_LOCK: + _OWNED_WORKERS.add(record) + group = _ACTIVE_WORKER_PHASE.get() + if group is not None: + group._register(record) + + +def _unregister_worker(record: _WorkerRecord) -> None: + with _OWNED_WORKERS_LOCK: + _OWNED_WORKERS.discard(record) + def _load_requests(): with warnings.catch_warnings(): @@ -21,6 +176,16 @@ def _load_requests(): class SubmitError(RuntimeError): + """Transport failure with structured fields (status_code, retry_after). + + C04/C12: public text is bounded and redacted — never raw server bodies, + tokens or URLs. Raw response bodies stay private in `_server_body` for + diagnostics only; they must not reach exception text or GUI events. + """ + + _MAX_MESSAGE_CHARS = 300 + _MAX_BODY_CHARS = 4096 + def __init__( self, message: str, @@ -30,15 +195,146 @@ def __init__( body: str = "", retry_after: float = 0.0, ) -> None: - super().__init__(message) + super().__init__(str(message or "")[: self._MAX_MESSAGE_CHARS]) self.retryable = retryable self.status_code = status_code - self.body = body + self._server_body = str(body or "")[: self._MAX_BODY_CHARS] self.retry_after = retry_after -def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: - """Fetch and solve a one-time token for exactly one POST attempt.""" +class SubmissionCancelled(SubmitError): + """Cooperative cancellation (C12). Retryable by contract: the durable spool + keeps the entry and its localHash identity so replay is idempotent.""" + + def __init__(self, phase: str) -> None: + super().__init__(f"submission cancelled during {phase}", retryable=True) + self.phase = phase + + +def _event_cancelled(cancel_event: Optional[Any]) -> bool: + if cancel_event is None: + return False + try: + return bool(cancel_event.is_set()) + except Exception: + return False + + +def _run_cancellable(func: Callable[[], Any], *, phase: str, + cancel_event: Optional[Any], deadline: Optional[float], + bound_seconds: float, + poll_seconds: float = 0.05) -> Any: + """Run a blocking HTTP call in a daemon worker so cooperative cancellation + is observed within ~poll_seconds even while the call is socket-blocked. + + The worker always carries a socket inactivity timeout at most the phase + budget, so an abandoned worker cannot outlive that budget against a + stalled peer. R05: the worker is registered in the process-wide census and + in the caller thread's WorkerGroup (when one is active), so the scope that + owns exclusion rights reaps it on release — or retains ownership until it + is quiescent. If cancel wins the race the connection may have been fully + sent (ambiguous outcome); the durable spool keeps the entry and replays it + idempotently.""" + outcome: List[Any] = [] + record = _WorkerRecord(phase) + handoff = threading.Lock() + _register_worker(record) + def _close_late(value: Any) -> None: + closer = getattr(value, "close", None) + if callable(closer): + try: + closer() + except Exception: + pass + + def _worker() -> None: + try: + value = func() + outcome.append(("ok", value)) + except BaseException as exc: # noqa: BLE001 — re-raised in caller + outcome.append(("err", exc)) + finally: + try: + # Closing an abandoned response is still owned transport I/O. + # Keep the worker live until it finishes so the host phase + # cannot release its exclusion lock during close(). + with handoff: + if record.abandoned and outcome and outcome[0][0] == "ok": + _close_late(outcome[0][1]) + finally: + record.done.set() + _unregister_worker(record) + record.thread = threading.Thread(target=_worker, name=f"encodingdb-{phase}", + daemon=True) + record.thread.start() + try: + while not record.done.wait(poll_seconds): + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + remaining = _remaining_seconds(deadline) + if remaining is not None and remaining <= 0: + raise SubmitError(f"{phase} exceeded {bound_seconds:g}s wall-clock bound", + retryable=True) + except BaseException: + # R05: this call is abandoned. Under `handoff` either the worker has + # not produced a response yet (flag closes any late one when it + # lands) or it already has (closed here, now). The owning WorkerGroup + # reaps the thread before exclusion is released. + with handoff: + record.abandoned = True + if outcome and outcome[0][0] == "ok": + _close_late(outcome[0][1]) + raise + kind, value = outcome[0] + if kind == "err": + raise value + return value + + +def _remaining_seconds(deadline: Optional[float]) -> Optional[float]: + if deadline is None: + return None + return deadline - time.monotonic() + + +def _check_transaction(cancel_event: Optional[Any], deadline: Optional[float], + phase: str, bound: float) -> None: + """Cooperative cancel + wall-clock bound check between blocking steps.""" + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + remaining = _remaining_seconds(deadline) + if remaining is not None and remaining <= 0: + raise SubmitError(f"{phase} exceeded {bound:g}s wall-clock bound", retryable=True) + + +def _bounded_wait(seconds: float, cancel_event: Optional[Any], deadline: Optional[float], + phase: str, bound: float) -> None: + """Backoff sleep that wakes on cancellation and respects the deadline.""" + remaining = _remaining_seconds(deadline) + if remaining is not None: + seconds = min(seconds, max(0.0, remaining)) + wait = getattr(cancel_event, "wait", None) + if callable(wait): + try: + if wait(seconds): + raise SubmissionCancelled(phase) + except SubmissionCancelled: + raise + except Exception: + if seconds > 0: + time.sleep(seconds) + elif seconds > 0: + time.sleep(seconds) + _check_transaction(cancel_event, deadline, phase, bound) + + +def _get_submit_token_headers(requests: Any, base_url: str, *, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> Dict[str, str]: + """Fetch and solve a one-time token for exactly one POST attempt. + + C12: cancellation- and deadline-aware — each endpoint GET and the PoW loop + check cancel/deadline so a Stop does not wait out the full 30 s solve.""" headers: Dict[str, str] = {} try: base = base_url.rstrip('/') @@ -49,17 +345,26 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: ] token_resp = None for endpoint in endpoints: + _check_transaction(cancel_event, deadline, "token fetch", SUBMIT_TRANSACTION_SECONDS) try: - response = requests.get(endpoint, timeout=10, verify=config.REQUESTS_VERIFY) + remaining = _remaining_seconds(deadline) + timeout = 10 if remaining is None else max(0.1, min(10.0, remaining)) + response = requests.get(endpoint, timeout=timeout, verify=config.REQUESTS_VERIFY) if response.status_code == 200: token_resp = response break + response.close() + except SubmissionCancelled: + raise except Exception: continue if token_resp is None: return headers - token_data = token_resp.json() or {} + try: + token_data = token_resp.json() or {} + finally: + token_resp.close() token = str(token_data.get('token') or '') if not re.fullmatch(r"[0-9a-f]{32}", token): return headers @@ -81,6 +386,14 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: print(f" Solving Proof-of-Work (difficulty={difficulty})...", end='', flush=True) while nonce < max_iters: if nonce % 10000 == 0: + if _event_cancelled(cancel_event): + print(" cancelled") + raise SubmissionCancelled("proof-of-work") + remaining_pow = _remaining_seconds(deadline) + if remaining_pow is not None and remaining_pow <= 0: + print(" deadline reached") + raise SubmitError("proof-of-work exceeded transaction bound", + retryable=True) elapsed_pow = time.time() - pow_start if elapsed_pow > pow_timeout: print(f" timeout after {elapsed_pow:.1f}s") @@ -95,18 +408,163 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: nonce += 1 print(f" exhausted {max_iters} iterations without solution") return {} + except SubmissionCancelled: + raise + except SubmitError: + raise except Exception as exc: try: - print(f"token fetch error: {exc}", file=sys.stderr) + print(f"token fetch error: {type(exc).__name__}", file=sys.stderr) except Exception: pass return {} -def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: int = 3, backoff_seconds: float = 1.0, use_token: Optional[bool] = None) -> None: +def _read_response_body(requests: Any, response: Any, cancel_event: Optional[Any], + deadline: Optional[float], + max_bytes: int = ERROR_BODY_HARD_CAP_BYTES) -> str: + """Consume an error/aux body under the ACTUAL deadline (R04). + + Consumption stops at the hard byte cap, the caller's real remaining + deadline, or cancellation — whichever comes first — and the response is + always closed. A separate monitor thread closes the socket when a chunk + read blocks past the deadline/cancel (iter_content only checks between + chunks, so a drip-fed or silent peer would otherwise pin the read). + Raises SubmissionCancelled on cancel and SubmitError (retryable) on + deadline; other socket errors end the read with what was consumed.""" + state = {"cancelled": False, "deadline": False} + done = threading.Event() + + def _close() -> None: + # Interrupt a peer-blocked read immediately: response.close() alone only + # returns after the socket timeout (the kernel recv is already parked). + # shutdown(SHUT_RDWR) on the live socket wakes it in ~0 ms; close then + # releases the connection. Every step is best-effort — a finished or + # half-torn-down connection raising here is normal. + raw = getattr(response, "raw", None) + sock = getattr(getattr(raw, "connection", None), "sock", None) + if sock is None: + orig = getattr(raw, "_original_response", None) + fp = getattr(orig, "fp", None) + candidate = getattr(getattr(fp, "raw", None), "_sock", None) + sock = candidate if isinstance(candidate, socket.socket) else None + if sock is not None: + try: + sock.shutdown(socket.SHUT_RDWR) + except OSError: + pass + try: + response.close() + except Exception: + pass + + def _watch() -> None: + poll = 0.05 + while not done.wait(poll): + if _event_cancelled(cancel_event): + state["cancelled"] = True + _close() + return + remaining = _remaining_seconds(deadline) + if remaining is not None and remaining <= 0: + state["deadline"] = True + _close() + return + + watcher = None + if cancel_event is not None or deadline is not None: + watcher = threading.Thread(target=_watch, + name="encodingdb-error-body-watch", daemon=True) + watcher.start() + chunks: List[bytes] = [] + total = 0 + overshoot = 0 + try: + # R04: hard consumed-byte cap. Request exactly the remaining budget so + # a compliant producer never reads past it, and stop consuming the + # moment the cap is crossed even if a producer yields more than asked. + iterator = iter(response.iter_content(chunk_size=max(1, max_bytes))) + while total < max_bytes: + if state["cancelled"]: + break + try: + chunk = next(iterator) + except StopIteration: + break + if not chunk: + continue + total += len(chunk) + if total > max_bytes: + overshoot = total - max_bytes + chunk = chunk[:max_bytes - (total - len(chunk))] + chunks.append(bytes(chunk)) + except Exception: + pass + finally: + done.set() + _close() + if watcher is not None: + watcher.join(0.5) + if state["cancelled"]: + raise SubmissionCancelled("response read") + if state["deadline"]: + raise SubmitError("response read exceeded remaining deadline", retryable=True) + text = b"".join(chunks).decode("utf-8", "replace") + if overshoot: + text += f"...[truncated {overshoot} bytes]" + return text + + + + + + +def _bounded_error_body(requests: Any, response: Any, cancel_event: Optional[Any], + deadline: Optional[float], + max_bytes: int = ERROR_BODY_HARD_CAP_BYTES) -> str: + """Error body for a structured error that is already decided from headers. + + Deadline exhaustion while reading discards the body (""), preserving the + status/retry_after verdict; cancellation still propagates so the caller + records a cancelled phase.""" + try: + return _read_response_body(requests, response, cancel_event, deadline, + max_bytes=max_bytes) + except SubmissionCancelled: + raise + except SubmitError: + return "" + + +def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: int = 3, + backoff_seconds: float = 1.0, use_token: Optional[bool] = None, + cancel_event: Optional[Any] = None, + transaction_seconds: float = SUBMIT_TRANSACTION_SECONDS, + deadline: Optional[float] = None) -> None: + """POST a payload with bounded, cancellable retries (C12). + + `cancel_event` (threading.Event-like) and `transaction_seconds` cap the + whole attempt chain on a monotonic wall clock: every step between blocking + calls checks both, and each socket step gets at most the remaining budget + as its inactivity timeout. `deadline` (monotonic) lets a caller impose a + tighter real deadline than the transaction budget; the effective deadline + is the earlier of the two. Cancellation raises SubmissionCancelled + (retryable) so the durable spool keeps identity. Every response object is + closed on every path (R04).""" requests = _load_requests() url = f"{base_url.rstrip('/')}/submit" payload_to_send: Dict[str, Any] = dict(payload) + bound_deadline = time.monotonic() + max(1.0, float(transaction_seconds)) + deadline = bound_deadline if deadline is None else min(deadline, bound_deadline) + + def step_timeout() -> float: + remaining = _remaining_seconds(deadline) + if remaining is None: + return 30 + if remaining <= 0: + raise SubmitError(f"submit exceeded {transaction_seconds:g}s wall-clock bound", + retryable=True) + return max(0.1, min(30.0, remaining)) base_headers: Dict[str, str] = {"Content-Type": "application/json"} if use_token is None: @@ -117,10 +575,12 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i attempt = 1 last_hmac_timestamp = 0 while attempt <= retries: + _check_transaction(cancel_event, deadline, "submit", transaction_seconds) body = json.dumps(payload_to_send, separators=(",", ":")) headers = dict(base_headers) if use_token: - headers.update(_get_submit_token_headers(requests, base_url)) + headers.update(_get_submit_token_headers(requests, base_url, + cancel_event=cancel_event, deadline=deadline)) if secret: import hmac ts = max(int(time.time()), last_hmac_timestamp + 1) @@ -128,15 +588,21 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i sig = hmac.new(secret.encode("utf-8"), f"{ts}.".encode("utf-8") + body.encode("utf-8"), hashlib.sha256).hexdigest() headers["x-signature"] = sig headers["x-timestamp"] = str(ts) + resp_for_close: List[Any] = [] try: - r = requests.post(url, data=body, timeout=30, headers=headers, verify=config.REQUESTS_VERIFY, allow_redirects=False) + r = _run_cancellable( + lambda: requests.post(url, data=body, timeout=step_timeout(), headers=headers, verify=config.REQUESTS_VERIFY, allow_redirects=False, stream=True), + phase="submit", cancel_event=cancel_event, deadline=deadline, + bound_seconds=transaction_seconds) + resp_for_close.append(r) if 300 <= r.status_code < 400: loc = r.headers.get('Location') or r.headers.get('location') if loc: redirect_url = urljoin(url, loc) redirect_headers = dict(base_headers) if use_token: - redirect_headers.update(_get_submit_token_headers(requests, base_url)) + redirect_headers.update(_get_submit_token_headers(requests, base_url, + cancel_event=cancel_event, deadline=deadline)) if secret: import hmac redirect_ts = max(int(time.time()), last_hmac_timestamp + 1) @@ -144,30 +610,42 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i redirect_sig = hmac.new(secret.encode("utf-8"), f"{redirect_ts}.".encode("utf-8") + body.encode("utf-8"), hashlib.sha256).hexdigest() redirect_headers["x-signature"] = redirect_sig redirect_headers["x-timestamp"] = str(redirect_ts) - r = requests.post(redirect_url, data=body, timeout=30, headers=redirect_headers, verify=config.REQUESTS_VERIFY, allow_redirects=False) + _check_transaction(cancel_event, deadline, "submit redirect", transaction_seconds) + prev = r + r = _run_cancellable( + lambda: requests.post(redirect_url, data=body, timeout=step_timeout(), headers=redirect_headers, verify=config.REQUESTS_VERIFY, allow_redirects=False, stream=True), + phase="submit redirect", cancel_event=cancel_event, deadline=deadline, + bound_seconds=transaction_seconds) + try: + prev.close() + except Exception: + pass + resp_for_close.append(r) if r.status_code == 429: - try: - ra = r.headers.get('Retry-After') - delay = float(ra) if ra and str(ra).replace('.', '', 1).isdigit() else (backoff_seconds * attempt * 2) - except Exception: + # The server's own wait is decided from headers BEFORE the + # body: a hostile body must not erase retry_after (C08/R04). + delay = retry_after_seconds(r.headers) + if delay <= 0: delay = backoff_seconds * attempt * 2 if attempt >= retries: raise SubmitError( f"submit rate limited ({r.status_code})", retryable=True, status_code=r.status_code, - body=(r.text or ""), + body=_bounded_error_body(requests, r, cancel_event, deadline), + retry_after=delay, # durable spool keeps the server's own wait (C08) ) - time.sleep(max(0.5, delay)) + _bounded_wait(max(0.5, delay), cancel_event, deadline, "submit retry wait", transaction_seconds) attempt += 1 continue if r.status_code >= 500: + error_body = _bounded_error_body(requests, r, cancel_event, deadline) if attempt >= retries: raise SubmitError( f"server_error {r.status_code}", retryable=True, - status_code=r.status_code, - body=(r.text or ""), + body=error_body, + retry_after=retry_after_seconds(r.headers), ) raise RuntimeError(f"server_error {r.status_code}") if r.status_code >= 400: @@ -175,10 +653,12 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i f"submit rejected ({r.status_code})", retryable=False, status_code=r.status_code, - body=(r.text or ""), + body=_bounded_error_body(requests, r, cancel_event, deadline), ) r.raise_for_status() return + except SubmissionCancelled: + raise except Exception as e: retryable = True status_code: Optional[int] = None @@ -186,33 +666,45 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i if isinstance(e, SubmitError): retryable = e.retryable status_code = e.status_code - body = e.body + body = e._server_body if attempt == retries: + # C04: never echo server bodies; status/content-type only — + # no resp.content drain (R04: unbounded body consumption). + status_for_print = status_code + detail = "" try: _req = _load_requests() if isinstance(e, _req.HTTPError) and getattr(e, 'response', None) is not None: resp = e.response - try: - err_text = resp.text - except Exception: - err_text = "" - sent_token = 'x-ingest-token' in headers - sent_nonce = 'x-ingest-nonce' in headers - print(f"submit error body ({resp.status_code}): {err_text}\n(sent_token={sent_token}, sent_nonce={sent_nonce})", file=sys.stderr) + status_for_print = resp.status_code + detail = f" content_type={resp.headers.get('content-type', '')}" except Exception: pass + sent_token = 'x-ingest-token' in headers + sent_nonce = 'x-ingest-nonce' in headers + print( + f"submit failed (status={status_for_print}{detail}; sent_token={sent_token}, sent_nonce={sent_nonce})", + file=sys.stderr, + ) if isinstance(e, SubmitError): raise + # Never put exception str() (may embed URLs/tokens) straight into + # the message: type + safe shape only. raise SubmitError( - str(e), + f"submit failed: {type(e).__name__}", retryable=True, status_code=status_code, - body=body, ) from e if isinstance(e, SubmitError) and not retryable: raise - time.sleep(backoff_seconds * attempt) + _bounded_wait(backoff_seconds * attempt, cancel_event, deadline, "submit retry wait", transaction_seconds) attempt += 1 + finally: + for resp in resp_for_close: + try: + resp.close() + except Exception: + pass def fetch_baseline_rows(base_url: str) -> List[Dict[str, Any]]: @@ -228,13 +720,16 @@ def fetch_baseline_rows(base_url: str) -> List[Dict[str, Any]]: requests = _load_requests() url = f"{base_url.rstrip('/')}/query?limit=500" r = requests.get(url, timeout=15, verify=config.REQUESTS_VERIFY) - if r.status_code == 200: - data = r.json() - if isinstance(data, list): - with config._GLOBAL_STATE_LOCK: - config._BASELINE_ROWS_CACHE = data - config._BASELINE_ROWS_CACHE_TS = time.time() - return data + try: + if r.status_code == 200: + data = r.json() + if isinstance(data, list): + with config._GLOBAL_STATE_LOCK: + config._BASELINE_ROWS_CACHE = data + config._BASELINE_ROWS_CACHE_TS = time.time() + return data + finally: + r.close() except Exception: pass @@ -248,10 +743,13 @@ def check_compatibility(base_url: str, client_version: str) -> Dict[str, Any]: from .suite import load_suite_pack_metadata, SUITE_VERSION response = _load_requests().get(f"{base_url.rstrip('/')}/v7/compatibility", timeout=10, verify=config.REQUESTS_VERIFY, allow_redirects=False) - if response.status_code != 200: - raise SubmitError(f"Compatibility endpoint returned {response.status_code}", retryable=True, - status_code=response.status_code) - contract = response.json() + try: + if response.status_code != 200: + raise SubmitError(f"Compatibility endpoint returned {response.status_code}", retryable=True, + status_code=response.status_code) + contract = response.json() + finally: + response.close() def version(value): return tuple(int(part) for part in str(value).removeprefix("client/").split(".")) if (contract.get("protocolVersion") != config.BENCHMARK_PROTOCOL_VERSION diff --git a/client/publication_result.py b/client/publication_result.py new file mode 100644 index 00000000..41ad2140 --- /dev/null +++ b/client/publication_result.py @@ -0,0 +1,120 @@ +"""Safe, typed failure details shared by publication producers and UIs.""" + +import errno +import re +from typing import Any, Dict, Optional, TypedDict + + +class PublicationFailure(TypedDict): + category: str + operation: str + retryable: bool + reason: str + nextAction: str + campaignId: str + runId: str + statusCode: Optional[int] + + +_ACTIONS = { + "rate_limited": "Keep saved work; retry after the server's delay.", + "server_error": "Keep saved work; retry when the server is available.", + "network": "Keep saved work; retry when the connection returns.", + "storage": "Free working space, then publish the saved work again.", + "publication_deferred": "Wait for the active work or storage constraint, then retry saved publication.", + "incompatible": "Use the original compatible client for missing envelopes; intact envelopes remain publishable.", + "corrupt_evidence": "Inspect the affected saved entry; other completed work remains available.", + "rejected": "Inspect the terminal server verdict and preserve the rejected evidence.", + "expired": "Inspect expired saved evidence; do not restart its retry deadline.", + "cancelled": "Restart Publish saved when ready; recorded receipts remain durable.", + "unexpected": "Inspect the saved campaign and keep its evidence unchanged.", +} + + +def _safe_reason(value: Any) -> str: + text = str(value or "").replace("\n", " ").replace("\r", " ") + text = re.sub(r"https?://\S+", "[endpoint]", text, flags=re.IGNORECASE) + text = re.sub(r"(?i)\bauthorization\s*[:=]\s*bearer\s+\S+", + "[credential redacted]", text) + text = re.sub(r"(?i)\b(api[_ -]?key|authorization|token)\b\s*[:=]\s*\S+", + "[credential redacted]", text) + text = re.sub(r"(?i)\bbearer\s+\S+", "[credential redacted]", text) + text = re.sub(r"(? PublicationFailure: + """Normalize every producer to one bounded, non-secret result shape.""" + from .network import SubmitError, SubmissionCancelled + from .spool import SpoolCapacityError + + raw = cause if isinstance(cause, dict) else {} + if isinstance(cause, SubmissionCancelled): + detected = "cancelled" + elif isinstance(cause, SubmitError): + code = int(cause.status_code or 0) + detected = ("rate_limited" if code == 429 else "server_error" if code >= 500 + else "rejected" if not cause.retryable else "network") + elif isinstance(cause, SpoolCapacityError): + detected = "publication_deferred" + elif isinstance(cause, OSError) and cause.errno in (errno.ENOSPC, errno.EDQUOT): + detected = "storage" + elif isinstance(cause, FileNotFoundError): + detected = "corrupt_evidence" + elif isinstance(cause, PermissionError): + detected = "storage" + elif isinstance(cause, (ConnectionError, TimeoutError, OSError)): + detected = "network" + elif isinstance(cause, (ValueError, TypeError, KeyError)): + detected = "corrupt_evidence" + else: + text = str(raw.get("reason") or raw.get("safeReason") or + ("publication failed" if isinstance(cause, dict) else cause)).lower() + if any(word in text for word in ("runtime identity differs", "client identity differs", "protocol identity differs")): + detected = "incompatible" + elif "retry_deadline_expired" in text or "retry deadline expired" in text: + detected = "expired" + elif any(word in text for word in ("journal", "evidence", "artifact")): + detected = "corrupt_evidence" + else: + detected = "unexpected" + chosen = str(category or raw.get("category") or detected) + if chosen not in _ACTIONS: + chosen = "unexpected" + can_retry = (bool(raw.get("retryable")) if "retryable" in raw else + bool(retryable) if retryable is not None else + bool(cause.retryable) if isinstance(cause, SubmitError) else + chosen in {"rate_limited", "server_error", "network", "storage", "publication_deferred", "cancelled"}) + reason = _safe_reason(raw.get("reason") or raw.get("safeReason") or + ("Publication could not continue." if isinstance(cause, dict) else cause)) + raw_status = raw.get("statusCode") if isinstance(cause, dict) else ( + cause.status_code if isinstance(cause, SubmitError) else None) + try: + status_code = int(raw_status) + if status_code < 100 or status_code > 599: + status_code = None + except (TypeError, ValueError): + status_code = None + return PublicationFailure( + category=chosen, + operation=str(operation)[:60], + retryable=can_retry, + reason=reason, + nextAction=_ACTIONS[chosen], + campaignId=str(campaign_id or raw.get("campaignId") or "")[:80], + runId=str(run_id or raw.get("runId") or "")[:80], + statusCode=status_code, + ) + + +def failure_text(value: Any) -> str: + failure = failure_info(value, operation="saved_publication") + return f"{failure['reason']} {failure['nextAction']}" diff --git a/client/recovery_projection.py b/client/recovery_projection.py new file mode 100644 index 00000000..f3bf4181 --- /dev/null +++ b/client/recovery_projection.py @@ -0,0 +1,146 @@ +"""Read-only projection of durable attempt journals for normal recovery UI.""" + +import json +import math +from collections import Counter +from pathlib import Path +from typing import Any, Dict + +from .campaign import load_record +from .protocol import ( + EnvironmentThresholds, + ProtocolConfig, + RecipeSpec, + StructuralExpectation, + StructuralTolerance, + campaign_result_from_records, +) + + +def protocol_config_from_manifest(manifest: Dict[str, Any]) -> ProtocolConfig: + saved = dict(manifest.get("protocolConfig") or {}) + return ProtocolConfig( + version=str(saved.get("version") or manifest.get("protocolVersion") or ""), + warmup_runs=int(saved.get("warmup_runs", 1)), + minimum_measured_runs=int(saved.get("minimum_measured_runs", 2)), + stability_threshold_ratio=float(saved.get("stability_threshold_ratio", 0.03)), + max_adaptive_repeats=int(saved.get("max_adaptive_repeats", 2)), + environment=EnvironmentThresholds(**dict(saved.get("environment") or {})), + structural_tolerance=StructuralTolerance(**dict(saved.get("structural_tolerance") or {})), + ) + + +def project_attempt_groups(root: Path, campaign_id: str) -> Dict[str, Any]: + """Find finalizable groups without changing a journal or requiring FFmpeg. + + A damaged attempt is reported and skipped individually. This projection + only identifies candidate evidence: publication still verifies exact + retained bytes and provenance before creating an immutable envelope. + """ + result: Dict[str, Any] = { + "plannedGroups": 0, + "attempts": 0, + "finishedGroupIds": [], + "acceptedOrders": [], + "candidateOrders": [], + "unavailableOrders": [], + "incompleteGroupIds": [], + "corruptEntries": [], + "failure": None, + } + try: + manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8")) + if not isinstance(manifest, dict): + raise ValueError("manifest is not an object") + config = protocol_config_from_manifest(manifest) + if config.version not in ("7.0", "7.1"): + raise ValueError("unsupported saved protocol") + if config.warmup_runs < 0 or config.minimum_measured_runs < 1 or config.max_adaptive_repeats < 0: + raise ValueError("invalid saved repetition plan") + result["plannedGroups"] = len(manifest.get("tasks") or []) + except (OSError, ValueError, TypeError, KeyError) as exc: + result["failure"] = f"saved plan cannot be read: {exc}"[:200] + return result + + records = {} + for path in sorted(root.glob("attempt-*.json")): + try: + order = int(path.stem.removeprefix("attempt-")) + record = load_record(json.loads(path.read_text(encoding="utf-8"))) + if (record.schedule.campaign_id != campaign_id + or record.schedule.execution_order != order + or order in records + or record.schedule.phase not in ("warmup", "measured") + or not record.schedule.recipe_id + or not isinstance(record.metadata, dict)): + raise ValueError("attempt identity or phase mismatch") + if record.timing is not None: + elapsed = float(record.timing.elapsed_s) + if not math.isfinite(elapsed) or elapsed <= 0: + raise ValueError("attempt timing is not positive and finite") + records[order] = record + except (OSError, ValueError, TypeError, KeyError, AttributeError) as exc: + result["corruptEntries"].append({"path": path.name, "reason": str(exc)[:120]}) + result["attempts"] = len(records) + recipe_ids = sorted({record.schedule.recipe_id for record in records.values()}) + if not recipe_ids: + return result + if result["plannedGroups"] and len(recipe_ids) > result["plannedGroups"]: + result["failure"] = "saved attempts contain more groups than the frozen plan" + try: + projected = campaign_result_from_records( + campaign_id=campaign_id, + config=config, + seed=int(manifest.get("seed") or 0), + recipes=[RecipeSpec(recipe_id=recipe_id, expectation=StructuralExpectation()) + for recipe_id in recipe_ids], + records=records, + ) + except (ValueError, TypeError, KeyError, ArithmeticError) as exc: + result["failure"] = f"saved attempt groups cannot be projected: {exc}"[:200] + return result + root_resolved = root.resolve() + for recipe in projected.recipe_results: + phase_counts = Counter(record.schedule.phase for record in recipe.runs) + if (recipe.recipe_id in projected.unfinished_recipes + or phase_counts["warmup"] < config.warmup_runs + or phase_counts["measured"] < config.minimum_measured_runs): + result["incompleteGroupIds"].append(recipe.recipe_id) + continue + result["finishedGroupIds"].append(recipe.recipe_id) + for record in recipe.runs: + if record.schedule.phase != "measured" or record.skipped_before_encode: + continue + info = record.metadata.get("info") or {} + if not isinstance(info, dict): + result["corruptEntries"].append({ + "path": f"attempt-{record.schedule.execution_order:06d}.json", + "reason": "attempt info is not an object", + }) + continue + if (record.timing is None or record.overall_validity.state == "invalid" + or info.get("error")): + continue + order = record.schedule.execution_order + receipt_path = root / f"submission-{order:06d}.accepted.json" + if receipt_path.is_file(): + try: + receipt = json.loads(receipt_path.read_text(encoding="utf-8")) + if (receipt.get("schemaVersion") == 1 + and receipt.get("executionOrder") == order + and receipt.get("recipeId") == recipe.recipe_id + and receipt.get("artifactPath") == info.get("artifactPath") + and receipt.get("artifactSha256") == info.get("artifactSha256") + and str(receipt.get("benchmarkRunId") or "").strip()): + result["acceptedOrders"].append(order) + continue + raise ValueError("accepted marker does not match attempt") + except (OSError, ValueError, TypeError, AttributeError) as exc: + result["corruptEntries"].append({"path": receipt_path.name, "reason": str(exc)[:120]}) + artifact = Path(str(info.get("artifactPath") or "")) + try: + available = artifact.is_file() and root_resolved in artifact.resolve().parents + except (OSError, ValueError): + available = False + result["candidateOrders" if available else "unavailableOrders"].append(order) + return result diff --git a/client/resources/test_suite_v1/clip-distribution.json b/client/resources/test_suite_v1/clip-distribution.json new file mode 100644 index 00000000..2cffbf70 --- /dev/null +++ b/client/resources/test_suite_v1/clip-distribution.json @@ -0,0 +1,215 @@ +{ + "schemaVersion": 1, + "suiteId": "encodingdb-test-suite", + "suiteVersion": "encodingdb-test-suite-v1", + "manifestVersion": 2, + "source": "staged-unpublished", + "releaseTag": null, + "distribution": { + "baseUrl": "", + "clipUrlOverrideEnv": "ENCODINGDB_SUITE_CLIP_BASE_URL", + "published": false + }, + "manifest": { + "sha256": "e0fa76d96f75f5e88c3c95ff452d155159ba1ac0c282e80e4c60245658309a76", + "byteSize": 15791 + }, + "clips": { + "athletic-action-1080p24-final": { + "fileName": "athletic-action-1080p24-final.mkv", + "license": "CC-BY-3.0", + "assets": [ + { + "role": "clip", + "path": "athletic-action-1080p24-final/athletic-action-1080p24-final.mkv", + "downloadName": "athletic-action-1080p24-final--athletic-action-1080p24-final.mkv", + "sha256": "1e06fe0315d0cb90247a3dae2258327989e657496247ca8974ddd8c2431755de", + "byteSize": 139815565 + }, + { + "role": "notice", + "path": "athletic-action-1080p24-final/notices/athletic-action-1080p24-final.txt", + "downloadName": "athletic-action-1080p24-final--athletic-action-1080p24-final.txt", + "sha256": "b7c8d6954e9a5ead857aa2b31a1a31a9ba24951fae0cbb924827463fb2870338", + "byteSize": 1435 + }, + { + "role": "license", + "path": "athletic-action-1080p24-final/notices/CC-BY-3.0.txt", + "downloadName": "athletic-action-1080p24-final--CC-BY-3.0.txt", + "sha256": "e6bc9e9c474700b708f568bac9e5a8a9bcb2b1dad53442f5ba449fcb848b8e76", + "byteSize": 19467 + } + ] + }, + "natural-detail-1080p24-final": { + "fileName": "natural-detail-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "natural-detail-1080p24-final/natural-detail-1080p24-final.mkv", + "downloadName": "natural-detail-1080p24-final--natural-detail-1080p24-final.mkv", + "sha256": "cfcc51d48372138f6f66b3286fec5a533a808650098e40f119011060d8e51225", + "byteSize": 283872120 + }, + { + "role": "notice", + "path": "natural-detail-1080p24-final/notices/natural-detail-1080p24-final.txt", + "downloadName": "natural-detail-1080p24-final--natural-detail-1080p24-final.txt", + "sha256": "d353a009802973bbfb8910c7440440c3ae59d981b0f9c306515c2d10c671de66", + "byteSize": 1388 + }, + { + "role": "license", + "path": "natural-detail-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "natural-detail-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "film-grain-1080p24-final": { + "fileName": "film-grain-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "film-grain-1080p24-final/film-grain-1080p24-final.mkv", + "downloadName": "film-grain-1080p24-final--film-grain-1080p24-final.mkv", + "sha256": "67d3d2f5a4f8c617f223077e7071aaee625f014950126e0e93a16b27e578d603", + "byteSize": 336898554 + }, + { + "role": "notice", + "path": "film-grain-1080p24-final/notices/film-grain-1080p24-final.txt", + "downloadName": "film-grain-1080p24-final--film-grain-1080p24-final.txt", + "sha256": "046f2cb4547c0223dd072b5109e03394d80418fa915f909fbe4d24a9006ee49c", + "byteSize": 1165 + }, + { + "role": "license", + "path": "film-grain-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "film-grain-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "dark-gradients-1080p24-final": { + "fileName": "dark-gradients-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "dark-gradients-1080p24-final/dark-gradients-1080p24-final.mkv", + "downloadName": "dark-gradients-1080p24-final--dark-gradients-1080p24-final.mkv", + "sha256": "3377e6927fdd256633961520ace19f6b5b1494f689841ed3a29831e48e0d27c3", + "byteSize": 317829621 + }, + { + "role": "notice", + "path": "dark-gradients-1080p24-final/notices/dark-gradients-1080p24-final.txt", + "downloadName": "dark-gradients-1080p24-final--dark-gradients-1080p24-final.txt", + "sha256": "decd3e519d768ec76400ac6de3a7445c68b0df26e9b29adab55fb5847e634c4e", + "byteSize": 1179 + }, + { + "role": "license", + "path": "dark-gradients-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "dark-gradients-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "animation-1080p24-final": { + "fileName": "animation-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "animation-1080p24-final/animation-1080p24-final.mkv", + "downloadName": "animation-1080p24-final--animation-1080p24-final.mkv", + "sha256": "d70c4d9e85e88c4369b6391a21f0388a4a3a7b72e1b89f545e5897b4d5d820ea", + "byteSize": 173819800 + }, + { + "role": "notice", + "path": "animation-1080p24-final/notices/animation-1080p24-final.txt", + "downloadName": "animation-1080p24-final--animation-1080p24-final.txt", + "sha256": "74d6dce61b31a0039262021ae5bdda18d9e8ab61825000f442afa86d143b9ea9", + "byteSize": 1110 + }, + { + "role": "license", + "path": "animation-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "animation-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "screen-text-1080p24-final": { + "fileName": "screen-text-1080p24-final.mkv", + "license": "CC0-1.0", + "assets": [ + { + "role": "clip", + "path": "screen-text-1080p24-final/screen-text-1080p24-final.mkv", + "downloadName": "screen-text-1080p24-final--screen-text-1080p24-final.mkv", + "sha256": "3c024a4aceb4ad09f8fa8cf51b6a4aba9be68460acb3f6a24a223e287bc05c8c", + "byteSize": 35313873 + }, + { + "role": "notice", + "path": "screen-text-1080p24-final/notices/screen-text-1080p24-final.txt", + "downloadName": "screen-text-1080p24-final--screen-text-1080p24-final.txt", + "sha256": "b8e04df7bb96a8f23cde80b4b52a551b05313242623924f5dafba930cc834092", + "byteSize": 1016 + }, + { + "role": "license", + "path": "screen-text-1080p24-final/notices/CC0-1.0.txt", + "downloadName": "screen-text-1080p24-final--CC0-1.0.txt", + "sha256": "a2010f343487d3f7618affe54f789f5487602331c0a8d03f49e9a7c547cf0499", + "byteSize": 7048 + }, + { + "role": "license", + "path": "screen-text-1080p24-final/notices/IBM-Plex-Mono-OFL.txt", + "downloadName": "screen-text-1080p24-final--IBM-Plex-Mono-OFL.txt", + "sha256": "7e6b2818edbd8f6a01ae80641cc8f16a51080d08fb4e532be3a0b6f74adb07da", + "byteSize": 4456 + } + ] + }, + "talking-head-1080p24-final": { + "fileName": "talking-head-1080p24-final.mkv", + "license": "CC-BY-3.0", + "assets": [ + { + "role": "clip", + "path": "talking-head-1080p24-final/talking-head-1080p24-final.mkv", + "downloadName": "talking-head-1080p24-final--talking-head-1080p24-final.mkv", + "sha256": "ac85d1350e5e668d5c0798b25fd7de16f8da05e831f396680a8f0f0ab05bcaf3", + "byteSize": 219125475 + }, + { + "role": "notice", + "path": "talking-head-1080p24-final/notices/talking-head-1080p24-final.txt", + "downloadName": "talking-head-1080p24-final--talking-head-1080p24-final.txt", + "sha256": "63fb7ce61b5ebcdc4aabd02c5075094ef2ebd9ecec6dff6d4a1d12987dc8c143", + "byteSize": 1430 + }, + { + "role": "license", + "path": "talking-head-1080p24-final/notices/CC-BY-3.0.txt", + "downloadName": "talking-head-1080p24-final--CC-BY-3.0.txt", + "sha256": "e6bc9e9c474700b708f568bac9e5a8a9bcb2b1dad53442f5ba449fcb848b8e76", + "byteSize": 19467 + } + ] + } + } +} diff --git a/client/runtime_lock.py b/client/runtime_lock.py index aef66014..ca1bff30 100644 --- a/client/runtime_lock.py +++ b/client/runtime_lock.py @@ -3,7 +3,8 @@ import os import shutil import subprocess -from .campaign import check_preparation_cancelled, run_measurement_process +from .campaign import (check_preparation_cancelled, preparation_heartbeat, + preparation_stage, run_measurement_process) import sys from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence @@ -18,6 +19,10 @@ "runtime-lock.json", ) REQUIRED_FILTERS: Sequence[str] = ("libvmaf", "xpsnr") +RUNTIME_PROBE_TIMEOUT_SECONDS = 30.0 +RUNTIME_BINARY_HASH_BUDGET_SECONDS = 300.0 +RUNTIME_DEPENDENCY_HASH_BUDGET_SECONDS = 300.0 +RUNTIME_PROBE_STAGE_BUDGET_SECONDS = 240.0 PLATFORM_REQUIRED_ENCODERS: Mapping[str, Sequence[str]] = { "linux": ("libaom-av1", "libsvtav1", "libvpx-vp9", "libx264", "libx265"), "mac": ("libaom-av1", "libsvtav1", "libvpx-vp9", "libx264", "libx265"), @@ -76,10 +81,13 @@ def _canonical_json(value: Any) -> str: def _sha256_path(path: str) -> str: digest = hashlib.sha256() + completed = 0 with open(path, "rb") as handle: for chunk in iter(lambda: handle.read(1024 * 1024), b""): check_preparation_cancelled() digest.update(chunk) + completed += len(chunk) + preparation_heartbeat(path=path, completedBytes=completed) return digest.hexdigest() @@ -123,13 +131,22 @@ def runtime_capability_requirements(platform_key: Optional[str] = None) -> Dict[ def _run_text(command: Sequence[str]) -> str: - proc = run_measurement_process( - list(command), - check=False, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - ) + # Every runtime probe owns a finite wall-clock budget even outside a + # preparation scope (release tooling, bare CLI): a hung ffmpeg/ffprobe + # must fail the check with the command named, never pin the caller. + try: + proc = run_measurement_process( + list(command), + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + timeout=RUNTIME_PROBE_TIMEOUT_SECONDS, + ) + except subprocess.TimeoutExpired: + raise RuntimeLockError( + f"{command[0]} exceeded its {RUNTIME_PROBE_TIMEOUT_SECONDS:g}s probe budget; " + "the runtime binary did not respond and was terminated") if proc.returncode != 0: detail = (proc.stderr or proc.stdout or "").strip() raise RuntimeLockError(f"{command[0]} failed with exit code {proc.returncode}: {detail}") @@ -187,6 +204,7 @@ def runtime_dependency_records(ffmpeg_path: str) -> List[Dict[str, Any]]: path = os.path.join(current, name) if os.path.commonpath([os.path.realpath(root), os.path.realpath(path)]) != os.path.realpath(root): raise RuntimeLockError("runtime dependency escapes bundled binary directory") + preparation_heartbeat(path=path, substage="dependencies") records.append({"relativePath": os.path.relpath(path, root).replace(os.sep, "/"), "sha256": _sha256_path(path), "byteSize": os.path.getsize(path)}) return records @@ -211,6 +229,30 @@ def probe_runtime_identity( required_encoders: Optional[Iterable[str]] = None, optional_encoders: Optional[Iterable[str]] = None, smoke_test_encoders: Optional[Iterable[str]] = None, +) -> Dict[str, Any]: + # The whole identity check is one bounded preparation stage; each probe + # inside also owns its own finite probe timeout. + with preparation_stage("runtime-probe", RUNTIME_PROBE_STAGE_BUDGET_SECONDS): + return _probe_runtime_identity( + ffmpeg_path=ffmpeg_path, + ffprobe_path=ffprobe_path, + platform_key=platform_key, + required_filters=required_filters, + required_encoders=required_encoders, + optional_encoders=optional_encoders, + smoke_test_encoders=smoke_test_encoders, + ) + + +def _probe_runtime_identity( + *, + ffmpeg_path: str, + ffprobe_path: str, + platform_key: Optional[str] = None, + required_filters: Optional[Iterable[str]] = None, + required_encoders: Optional[Iterable[str]] = None, + optional_encoders: Optional[Iterable[str]] = None, + smoke_test_encoders: Optional[Iterable[str]] = None, ) -> Dict[str, Any]: requirements = runtime_capability_requirements(platform_key) ffmpeg_version_output = _run_text([ffmpeg_path, "-version"]) @@ -462,9 +504,11 @@ def verify_runtime_lock( resolved_ffmpeg = _resolve_binary_path(lock_path=resolved_lock_path, explicit_path=ffmpeg_path, entry=expected_ffmpeg) resolved_ffprobe = _resolve_binary_path(lock_path=resolved_lock_path, explicit_path=ffprobe_path, entry=expected_ffprobe) - for label, path, expected in (("ffmpeg", resolved_ffmpeg, expected_ffmpeg), ("ffprobe", resolved_ffprobe, expected_ffprobe)): - _assert_binary_bytes(label, path, expected, {"sha256": _sha256_path(path), "byteSize": os.path.getsize(path)}) - _verify_dependencies(resolved_ffmpeg, platform_entry.get("runtimeDependencies")) + with preparation_stage("runtime-binary-hash", RUNTIME_BINARY_HASH_BUDGET_SECONDS): + for label, path, expected in (("ffmpeg", resolved_ffmpeg, expected_ffmpeg), ("ffprobe", resolved_ffprobe, expected_ffprobe)): + _assert_binary_bytes(label, path, expected, {"sha256": _sha256_path(path), "byteSize": os.path.getsize(path)}) + with preparation_stage("runtime-dependency-hash", RUNTIME_DEPENDENCY_HASH_BUDGET_SECONDS): + _verify_dependencies(resolved_ffmpeg, platform_entry.get("runtimeDependencies")) default_requirements = runtime_capability_requirements(selected_platform) observed = probe_runtime_identity( ffmpeg_path=resolved_ffmpeg, diff --git a/client/spool.py b/client/spool.py index 0e551dfb..f5cdde15 100644 --- a/client/spool.py +++ b/client/spool.py @@ -5,18 +5,226 @@ import shutil import time import random +import tempfile from pathlib import Path +import contextvars from contextlib import contextmanager from dataclasses import dataclass from typing import Any, Dict, List, Optional, Tuple +from . import network from .artifacts import AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND, submit_artifact_submission from .campaign import directory_bytes -from .network import SubmitError, submit +from .network import SubmissionCancelled, SubmitError, submit SPOOL_VERSION = 1 MANAGED_ARTIFACT_DIRNAME = "artifacts" SPOOL_METADATA_RESERVE_BYTES = 64 * 1024 +REPLAY_WINDOW = 25 +REPLAY_SECONDS = 60.0 + +# True only inside a collector's own batch: its checkpoint uploads legitimately run +# while it holds this queue's measurement lock. External publishers must defer. +_COLLECTOR_PUBLICATION_SCOPE: contextvars.ContextVar = contextvars.ContextVar( + "encodingdb_collector_publication", default=False) + +# True while this process holds the host phase lock; nested publication passes +# (collector checkpoint uploads, publish wrapper around replay) reuse it. +_HOST_PHASE_HELD: contextvars.ContextVar = contextvars.ContextVar( + "encodingdb_host_phase_held", default=False) + + +@contextmanager +def collector_publication_scope(): + """Declare that the calling process owns the live collection for this queue.""" + token = _COLLECTOR_PUBLICATION_SCOPE.set(True) + try: + yield + finally: + _COLLECTOR_PUBLICATION_SCOPE.reset(token) + + +def _refuse_publication_during_measurement(queue_dir: str) -> None: + """Publication defers to a collector measuring on this queue (C11). + + The probe opens measurement.lock non-blockingly from a fresh descriptor; a + held lock - even one owned by this same process through another descriptor - + means authoritative collection is timing-sensitive right now. Only the + collector's own in-batch uploads (``collector_publication_scope``) skip it.""" + if _COLLECTOR_PUBLICATION_SCOPE.get(): + return + from .campaign import active_collection + try: + active = active_collection(queue_dir) + except OSError as exc: + raise SpoolCapacityError(f"Cannot verify publication exclusion: {exc}") from exc + if active is not None: + raise SpoolCapacityError( + "A collection is measuring in this queue; publication defers to its next checkpoint") + + +def publication_lock_busy(queue_dir: str) -> bool: + """True when another publisher currently owns this queue's publication lock.""" + try: + with _spool_write_lock(queue_dir): + return False + except SpoolCapacityError: + return True + + +def _host_phase_lock_path() -> str: + """Host-scoped lock file shared by every queue of this user on this machine. + + C11: two different queue directories on the same host still share one disk + and network path, so measurement timing and publication cannot overlap even + across queues. The kernel releases flock/msvcrt ownership on process death, + so a crash needs no stale-lock cleanup - and none is ever performed.""" + root = os.environ.get("ENCODINGDB_HOST_PHASE_DIR") or os.path.join( + tempfile.gettempdir(), "encodingdb-host-phase-{}".format(getattr(os, "getuid", lambda: "shared")())) + os.makedirs(root, exist_ok=True) + return os.path.join(root, "phase.lock") + + +def _host_phase_probe_held() -> bool: + """True when some process currently owns the host phase lock (advisory). + + Fail-closed: an uninspectable lock file raises OSError; correctness never + depends on this probe - both phases acquire the lock non-blockingly before + doing timing-sensitive work, and the loser defers.""" + path = _host_phase_lock_path() + try: + with open(path, "a+b") as handle: + if os.fstat(handle.fileno()).st_size == 0: + handle.write(b"0") + handle.flush() + if os.name == "nt": + import msvcrt + handle.seek(0) + try: + msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1) + except OSError: + return True + msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1) + return False + import fcntl + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError: + return True + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + return False + except OSError: + raise + + +def host_phase_busy() -> bool: + """True when another phase owner currently holds the host phase lock (C11). + + Advisory probe for start guards; the authoritative arbiter remains the + non-blocking acquire inside ``host_phase_hold``. Re-entry by the current + holder reports free - a collector guard runs inside its own measurement + hold and must not refuse itself. Fails closed: an uninspectable lock file + raises SpoolCapacityError.""" + if _HOST_PHASE_HELD.get(): + return False + try: + return _host_phase_probe_held() + except OSError as exc: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + + +@contextmanager +def host_phase_hold(role: str = "publication"): + """Kernel-backed host phase lock for one collector batch or one publication + pass (C11). + + Measurement takes the lock exclusively and publication shared, so a held + collector defers publishers and a held publisher defers the next collector, + in either start order: the non-blocking acquire is the atomic arbiter, so + neither side can slip between the other's probe and acquisition. Two + publishers coexist (uploads are idempotent and time-insensitive); only + measurement timing needs the exclusive phase. (msvcrt has no shared lock, + so on Windows publishers serialize - deferral is always safe.) A busy lock + raises SpoolCapacityError - publishers defer, collectors exit 6 - while any + other inspection failure also fails closed. The kernel releases flock + ownership on process death; no stale-lock deletion exists. Re-entry inside + the same process/thread (collector checkpoint upload, publish wrapper + around replay) reuses the held lock. Only acquisition errors are wrapped: + an OSError raised by the body propagates unchanged.""" + if _HOST_PHASE_HELD.get(): + yield False + return + path = _host_phase_lock_path() + try: + handle = open(path, "a+b") + except OSError as exc: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + try: + if os.fstat(handle.fileno()).st_size == 0: + handle.write(b"0") + handle.flush() + if os.name == "nt": + import msvcrt + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1) + else: + import fcntl + mode = fcntl.LOCK_EX if role == "measurement" else fcntl.LOCK_SH + fcntl.flock(handle.fileno(), mode | fcntl.LOCK_NB) + except OSError as exc: + busy = getattr(exc, "errno", None) in (errno.EAGAIN, errno.EACCES, errno.EWOULDBLOCK, + errno.EDEADLK) or type(exc).__name__ == "BlockingIOError" + try: + handle.close() + except OSError: + pass + if not busy: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + if role == "measurement": + raise SpoolCapacityError( + "A publication or measurement pass owns this host right now; " + "wait for it to finish before starting collection") from exc + raise SpoolCapacityError( + "A collector is measuring on this host; publication defers to its next checkpoint") from exc + token = _HOST_PHASE_HELD.set(True) + # R05: every transport worker started inside this hold is owned by it. On + # release the workers are reaped (bounded sync join); if one is still + # mid-I/O, the kernel lock stays held (escalated to exclusive, deferred + # releaser) until the worker is quiescent — a collector can never start + # while a cancelled/timed-out worker still owns live transport I/O. + worker_phase = network.begin_owned_worker_phase() + + def _release_kernel_lock() -> None: + try: + if os.name == "nt": + import msvcrt + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1) + else: + import fcntl + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + finally: + handle.close() + + try: + yield True + finally: + _HOST_PHASE_HELD.reset(token) + if not network.end_owned_worker_phase(worker_phase, _release_kernel_lock): + _escalate_deferred_phase_lock(handle) + + +def _escalate_deferred_phase_lock(handle) -> None: + """Best-effort SH→EX escalation while a deferred releaser still owns the + phase (POSIX flock conversion). Other publishers may keep coexisting when + escalation is impossible; deferral of this hold still blocks collectors.""" + if os.name == "nt": + return + try: + import fcntl + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError: + pass class SpoolCapacityError(OSError): @@ -115,6 +323,8 @@ class ReplayStats: retained: int = 0 dead_lettered: int = 0 corrupt: int = 0 + cancelled: int = 0 + deferred: int = 0 @dataclass @@ -174,6 +384,46 @@ def count_pending_entries(queue_dir: str) -> int: return 0 +def due_first_queue_paths(queue_dir: str, *, limit: int = REPLAY_WINDOW) -> Tuple[List[str], int]: + """Fair bounded due-first selection (C08). + + Returns (paths, deferred). Entries whose Retry-After has arrived - or whose + retry deadline has expired, so the verdict can be finalized - are due; the + window admits only due entries, oldest-scheduled first, so a failing prefix + can never starve healthy entries behind it. Delayed entries are counted as + deferred, never attempted early.""" + now = time.time() + due: List[Tuple[float, int, str, str]] = [] + deferred = 0 + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + return [], 0 + for name in names: + if not name.endswith(".json"): + continue + path = os.path.join(queue_dir, name) + if not os.path.isfile(path): + continue + try: + entry = load_spool_entry(path) + next_at = float(entry.get("nextAttemptAt") or 0) + deadline = float(entry.get("retryDeadlineAt") or 0) + queued = int(entry.get("queuedAt") or 0) + except Exception: + # Corrupt files are due immediately: the cheap terminal verdict + # retires them before healthy traffic waits behind them. + due.append((float("-inf"), 0, name, path)) + continue + if next_at <= now or deadline <= now: + due.append((next_at, queued, name, path)) + else: + deferred += 1 + due.sort(key=lambda item: (item[0], item[1], item[2])) + selected = [item[3] for item in due[:max(0, limit)]] + return selected, deferred + max(0, len(due) - len(selected)) + + def _iter_files(root: str) -> List[str]: files: List[str] = [] if not os.path.isdir(root): @@ -310,7 +560,35 @@ def _managed_artifact_path(queue_dir: str, artifact_sha256: str, source_path: st return os.path.join(_managed_artifact_dir(queue_dir), f"{artifact_sha256}{ext.lower()}") -def _preserve_artifact_for_spool(queue_dir: str, payload: Dict[str, Any]) -> Dict[str, Any]: +def _admission_guard(cancel_event: Optional[Any], deadline: Optional[float], + phase: str) -> None: + """R03: cooperative cancel + real caller deadline between blocking local + steps (capacity scan, artifact staging copy, replay hashing).""" + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + if deadline is not None and time.monotonic() >= deadline: + raise SubmitError(f"{phase} exceeded caller deadline", retryable=True) + + +def spool_payload(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int = 2048, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> Tuple[str, Dict[str, Any]]: + _admission_guard(cancel_event, deadline, "queue admission") + try: + with host_phase_hold("publication"): + _refuse_publication_during_measurement(queue_dir) + with _spool_write_lock(queue_dir): + return _spool_payload_locked(queue_dir, payload, max_storage_mb=max_storage_mb, + cancel_event=cancel_event, deadline=deadline) + except OSError as exc: + if exc.errno in (errno.ENOSPC, errno.EDQUOT): + raise SpoolCapacityError("Publication ran out of disk space; free space and resume the retained campaign") from exc + raise + + +def _preserve_artifact_for_spool(queue_dir: str, payload: Dict[str, Any], *, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> Dict[str, Any]: if payload.get("submissionKind") != AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: return dict(payload) artifact_path = str(payload.get("artifactPath") or "").strip() @@ -330,7 +608,11 @@ def _preserve_artifact_for_spool(queue_dir: str, payload: Dict[str, Any]) -> Dic raise ValueError("Retained artifact size differs from immutable upload metadata") copied = 0 with open(artifact_path, "rb") as source, open(tmp_path, "xb") as target: - for chunk in iter(lambda: source.read(1024 * 1024), b""): + while True: + _admission_guard(cancel_event, deadline, "artifact staging") + chunk = source.read(1024 * 1024) + if not chunk: + break copied += len(chunk) if copied > expected_size: raise ValueError("Retained artifact grew during bounded upload staging") @@ -376,17 +658,11 @@ def _terminal_spool_entry_locked(queue_dir: str, local_hash: str) -> Optional[Tu return None -def spool_payload(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int = 2048) -> Tuple[str, Dict[str, Any]]: - try: - with _spool_write_lock(queue_dir): - return _spool_payload_locked(queue_dir, payload, max_storage_mb=max_storage_mb) - except OSError as exc: - if exc.errno in (errno.ENOSPC, errno.EDQUOT): - raise SpoolCapacityError("Publication ran out of disk space; free space and resume the retained campaign") from exc - raise -def _spool_payload_locked(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int) -> Tuple[str, Dict[str, Any]]: +def _spool_payload_locked(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> Tuple[str, Dict[str, Any]]: receipt_path = os.path.join(queue_dir, "receipts", f"{local_hash_for_payload(payload)}.json") if os.path.isfile(receipt_path): return receipt_path, _envelope_for_payload(payload) @@ -405,7 +681,10 @@ def _spool_payload_locked(queue_dir: str, payload: Dict[str, Any], *, max_storag raise ValueError("Corrupt upload could not retain terminal identity") return terminal _check_spool_capacity(queue_dir, payload, max_storage_mb) - spool_payload_value = _preserve_artifact_for_spool(queue_dir, payload) + _admission_guard(cancel_event, deadline, "queue admission") + spool_payload_value = _preserve_artifact_for_spool(queue_dir, payload, + cancel_event=cancel_event, + deadline=deadline) envelope = _envelope_for_payload(spool_payload_value) _write_json_atomic(path, envelope) return path, envelope @@ -524,7 +803,9 @@ def _is_managed_artifact_path(queue_dir: str, artifact_path: str) -> bool: return common == managed_root -def _validate_managed_artifact_for_replay(queue_dir: str, payload: Dict[str, Any]) -> None: +def _validate_managed_artifact_for_replay(queue_dir: str, payload: Dict[str, Any], *, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> None: artifact_path = str(payload.get("artifactPath") or "").strip() if not artifact_path or not _is_managed_artifact_path(queue_dir, artifact_path): raise SubmitError("spooled artifact path is outside the managed queue", retryable=False) @@ -535,7 +816,14 @@ def _validate_managed_artifact_for_replay(queue_dir: str, payload: Dict[str, Any digest = hashlib.sha256() observed_size = 0 with open(artifact_path, "rb") as handle: - for chunk in iter(lambda: handle.read(1024 * 1024), b""): + while True: + # R03: replay hashing can run over multi-GB artifacts; check + # cancel/deadline between chunks so a Stop does not wait for the + # full hash. Raised errors are retryable -> entry is retained. + _admission_guard(cancel_event, deadline, "replay artifact validation") + chunk = handle.read(1024 * 1024) + if not chunk: + break observed_size += len(chunk) digest.update(chunk) if observed_size != expected_size or digest.hexdigest() != expected_hash: @@ -621,6 +909,225 @@ def _submission_success_message(payload: Dict[str, Any], response: Any) -> str: return "" +def _journal_self_publish(queue_dir: str, payload: Dict[str, Any], response: Any) -> None: + """Commit journal acceptance evidence for a completed queue upload (C11). + + A checkpoint upload that finishes outside a live batch (standalone replay, + --upload-only, GUI retry) must still write the campaign journal's own receipt + and retire the owned artifact, so a crash between server acceptance and + journal commit cannot resurrect an uploaded group for re-encode or re-upload. + Without a server run id nothing is treated as accepted; an existing faithful + receipt simply wins (idempotent replay).""" + if payload.get("submissionKind") != AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: + return + run_id = _submission_success_message(payload, response) + if not run_id: + return + local_hash = local_hash_for_payload(payload) + campaigns = Path(queue_dir) / "campaigns" + run_create = payload.get("runCreate") if isinstance(payload.get("runCreate"), dict) else {} + scoped = str(run_create.get("campaignId") or "") + try: + # The payload carries its campaign identity; scope the faithful-identity scan + # to that journal instead of parsing every campaign's submissions. + if scoped and (campaigns / scoped).is_dir(): + candidates = sorted((campaigns / scoped).glob("submission-*.json")) + else: + candidates = sorted(campaigns.glob("*/submission-*.json")) + except OSError: + return + for candidate in candidates: + if candidate.name.endswith(".accepted.json"): + continue + try: + journal_payload = json.loads(candidate.read_text()) + except (OSError, ValueError): + continue + if not isinstance(journal_payload, dict) or local_hash_for_payload(journal_payload) != local_hash: + continue + run_create = journal_payload.get("runCreate") + repetition_group = str((run_create or {}).get("repetitionGroupId") or "") if isinstance(run_create, dict) else "" + artifact_path = str(journal_payload.get("artifactPath") or "").strip() + artifact_sha = str(journal_payload.get("artifactSha256") or "").strip() + try: + order_value = int(candidate.stem.split("-", 1)[1]) + except (ValueError, IndexError): + continue + if (not artifact_path or not artifact_sha or ":" not in repetition_group + or str((run_create or {}).get("campaignId") or "") != candidate.parent.name): + continue + _write_json_atomic(str(candidate.with_name(f"submission-{order_value:06d}.accepted.json")), + {"schemaVersion": 1, + "executionOrder": order_value, + "recipeId": repetition_group.split(":", 1)[1], + "artifactPath": artifact_path, + "artifactSha256": artifact_sha, + "benchmarkRunId": run_id, + "acceptedAt": time.time()}) + owned = Path(artifact_path).resolve() + if candidate.parent.resolve() in owned.parents and owned.is_file(): + try: + owned.unlink() + except OSError: + pass + return + + +def drain_committed_receipts(queue_dir: str) -> int: + """Retire pending entries whose acceptance receipt is already committed (C10). + + A crash between receipt commit and entry unlink leaves both files; the + receipt is terminal evidence, so the entry (and its managed staging copy) + must drain BEFORE new staging consumes the storage budget. Journal + self-publish runs here too, covering a crash inside the other process.""" + drained = 0 + with _spool_write_lock(queue_dir): + try: + names = os.listdir(queue_dir) + except OSError: + return 0 + for name in names: + if not name.endswith(".json") or not os.path.isfile(os.path.join(queue_dir, name)): + continue + receipt_path = os.path.join(queue_dir, "receipts", name) + if not os.path.isfile(receipt_path): + continue + path = os.path.join(queue_dir, name) + try: + entry = load_spool_entry(path) + except Exception: + entry = None + if entry is not None: + try: + receipt = json.loads(Path(receipt_path).read_text()) + except (OSError, ValueError): + receipt = {} + _journal_self_publish(queue_dir, entry.get("payload") or {}, receipt.get("response")) + _cleanup_managed_artifact_if_unreferenced(queue_dir, entry, excluding_entry_path=path) + try: + os.remove(path) + drained += 1 + except OSError: + pass + return drained + + +def campaign_queue_summary(queue_dir: str, campaign_id: str) -> Dict[str, Any]: + """Reconciled publication counters for ONE campaign across queue+receipts+terminal. + + Counts entries by durable identity: pending staging, accepted receipts, + terminal dead letters. Measurement counters live in the journal; this view + shows what publication actually holds for the campaign right now.""" + summary = {"pendingEntries": 0, "pendingBytes": 0, "dueEntries": 0, "acceptedReceipts": 0, + "terminalEntries": 0, "nextAttemptAt": None} + now = time.time() + def belongs(payload: Any) -> bool: + run_create = payload.get("runCreate") if isinstance(payload, dict) else None + return isinstance(run_create, dict) and str(run_create.get("campaignId") or "") == campaign_id + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + return summary + for name in names: + path = os.path.join(queue_dir, name) + if not name.endswith(".json") or not os.path.isfile(path): + continue + try: + entry = load_spool_entry(path) + except Exception: + continue + if not belongs(entry.get("payload")): + continue + summary["pendingEntries"] += 1 + summary["pendingBytes"] += os.path.getsize(path) + if float(entry.get("nextAttemptAt") or 0) <= now: + summary["dueEntries"] += 1 + else: + next_at = float(entry.get("nextAttemptAt") or 0) + if summary["nextAttemptAt"] is None or next_at < summary["nextAttemptAt"]: + summary["nextAttemptAt"] = next_at + receipts = Path(queue_dir) / "receipts" + try: + for receipt_file in sorted(receipts.glob("*.json")): + try: + receipt = json.loads(receipt_file.read_text()) + except (OSError, ValueError): + continue + response = receipt.get("response") if isinstance(receipt, dict) else None + run_id = _submission_success_message({"submissionKind": AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND}, response) + if run_id and _receipt_matches_campaign(queue_dir, receipt_file.stem, campaign_id): + summary["acceptedReceipts"] += 1 + except OSError: + pass + terminal_dir = Path(queue_dir) / "terminal" + try: + for terminal_file in sorted(terminal_dir.glob("*.json")): + try: + terminal = json.loads(terminal_file.read_text()) + except (OSError, ValueError): + continue + if belongs((terminal or {}).get("payload")): + summary["terminalEntries"] += 1 + except OSError: + pass + return summary + + +def _receipt_matches_campaign(queue_dir: str, local_hash: str, campaign_id: str) -> bool: + """A queue receipt names only its hash; the journal holds the campaign link.""" + root = Path(queue_dir) / "campaigns" / campaign_id + if not root.is_dir(): + return False + for submission in root.glob("submission-*.json"): + if submission.name.endswith(".accepted.json"): + continue + try: + payload = json.loads(submission.read_text()) + except (OSError, ValueError): + continue + if isinstance(payload, dict) and local_hash_for_payload(payload) == local_hash: + return True + return False + + +def queue_recovery_summary(queue_dir: str) -> Dict[str, Any]: + """Read-only queue-wide publication view for status/recovery projection (C06).""" + summary = {"pendingEntries": 0, "pendingBytes": 0, "dueEntries": 0, "acceptedReceipts": 0, + "terminalEntries": 0, "nextAttemptAt": None} + now = time.time() + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + names = [] + for name in names: + path = os.path.join(queue_dir, name) + if not name.endswith(".json") or not os.path.isfile(path): + continue + summary["pendingEntries"] += 1 + try: + summary["pendingBytes"] += os.path.getsize(path) + except OSError: + pass + try: + entry = load_spool_entry(path) + next_at = float(entry.get("nextAttemptAt") or 0) + except Exception: + continue + if next_at <= now: + summary["dueEntries"] += 1 + elif summary["nextAttemptAt"] is None or next_at < summary["nextAttemptAt"]: + summary["nextAttemptAt"] = next_at + try: + summary["acceptedReceipts"] = sum(1 for f in (Path(queue_dir) / "receipts").glob("*.json") if f.is_file()) + except OSError: + pass + try: + summary["terminalEntries"] = sum(1 for f in (Path(queue_dir) / "terminal").glob("*.json") if f.is_file()) + except OSError: + pass + return summary + + def _current_spooled_entry_locked(path: str, queue_dir: str): # A different replay may have finished while our network transaction was in # progress. Its receipt/terminal verdict wins; never recreate stale entries. @@ -634,6 +1141,10 @@ def _current_spooled_entry_locked(path: str, queue_dir: str): except ValueError: pending = None if pending is not None: + # Crash recovery (C11): the other process committed the receipt but may + # have died before publishing journal acceptance. Replay the journal + # receipt from the still-pending payload before releasing its bytes. + _journal_self_publish(queue_dir, pending.get("payload") or {}, receipt.get("response")) _cleanup_managed_artifact_if_unreferenced(queue_dir, pending, excluding_entry_path=pending_path) os.remove(pending_path) return None, ("submitted", _submission_success_message( @@ -665,6 +1176,33 @@ def submit_spooled_path( api_key: str, retries: int, use_token: bool, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, +) -> Tuple[str, str]: + # C11: the whole network transaction must live inside the host publication + # phase; otherwise a collector could start mid-upload through the very disk + # and network path the upload is saturating. In-process re-entry (a + # collector's checkpoint upload, or replay_spool's pass-level hold) is free. + # R03: `deadline` (monotonic) is a real caller deadline honoured before any + # I/O and through every transport phase; R05: the hold reaps (or defers + # around) any worker this transaction leaves behind. + with host_phase_hold("publication"): + return _submit_spooled_path_unheld( + path, queue_dir=queue_dir, base_url=base_url, api_key=api_key, + retries=retries, use_token=use_token, cancel_event=cancel_event, + deadline=deadline) + + +def _submit_spooled_path_unheld( + path: str, + *, + queue_dir: str, + base_url: str, + api_key: str, + retries: int, + use_token: bool, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, ) -> Tuple[str, str]: try: with _spool_write_lock(queue_dir): @@ -683,8 +1221,15 @@ def submit_spooled_path( _move_to_dead_letter_locked(queue_dir, path, entry, "missing_spooled_artifact") return "dead_lettered", "missing_spooled_artifact" try: - _validate_managed_artifact_for_replay(queue_dir, payload) + _validate_managed_artifact_for_replay(queue_dir, payload, + cancel_event=cancel_event, + deadline=deadline) except SubmitError as exc: + if exc.retryable: + # R03: cancel/deadline interrupted replay hashing — + # retain with identity intact, perform no network I/O. + _retain_entry(path, entry, str(exc)) + return "retained", str(exc) _move_to_dead_letter_locked(queue_dir, path, entry, str(exc)) return "dead_lettered", str(exc) except Exception as exc: @@ -695,10 +1240,18 @@ def submit_spooled_path( response: Any = None error: Optional[Exception] = None try: + if deadline is not None and time.monotonic() >= deadline: + raise SubmitError("spool transport deadline reached before network", + retryable=True) + if _event_cancelled(cancel_event): + raise SubmissionCancelled("spool transport") if payload.get("submissionKind") == AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: - response = submit_artifact_submission(base_url, payload, retries=retries) + response = submit_artifact_submission(base_url, payload, retries=retries, + cancel_event=cancel_event, + deadline=deadline) else: - submit(base_url, payload, api_key=api_key, retries=retries, use_token=use_token) + submit(base_url, payload, api_key=api_key, retries=retries, use_token=use_token, + cancel_event=cancel_event, deadline=deadline) except Exception as exc: error = exc with _spool_write_lock(queue_dir): @@ -714,6 +1267,10 @@ def submit_spooled_path( _write_json_atomic(os.path.join(queue_dir, "receipts", os.path.basename(path)), {"localHash": entry["localHash"], "uploadedAt": time.time(), "response": response, "status": "uploaded_analysis_pending"}) + # Publish journal acceptance while the pending entry still exists: a crash + # here leaves replayable state (receipt + pending), never a journal that + # lost its artifact without a receipt. + _journal_self_publish(queue_dir, payload, response) _cleanup_managed_artifact_if_unreferenced(queue_dir, entry, excluding_entry_path=path) try: os.remove(path) @@ -733,20 +1290,61 @@ def replay_spool( api_key: str, retries: int, use_token: bool, + limit: int = REPLAY_WINDOW, + time_budget: float = REPLAY_SECONDS, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, +) -> ReplayStats: + """Bounded, cancellable, due-first replay (C08/C11/C12). + + - The whole pass holds the host publication phase atomically: the probe and + every upload are one phase, so a collector starting on ANY queue of this + host cannot slip between them (kernel lock; crash releases). + - Only entries whose Retry-After has arrived (or whose deadline lapsed) enter + the window; a failing prefix consumes slots but never blocks later due work. + - A cancel request stops admission between entries; an in-flight network + transaction keeps its durable entry, so the ambiguous outcome is retried + idempotently rather than lost or double-submitted. + - A measuring collector on this queue defers publication (SpoolCapacityError). + """ + with host_phase_hold("publication"): + return _replay_spool_unheld( + queue_dir, base_url=base_url, api_key=api_key, retries=retries, + use_token=use_token, limit=limit, time_budget=time_budget, + cancel_event=cancel_event, deadline=deadline) + + +def _replay_spool_unheld( + queue_dir: str, + *, + base_url: str, + api_key: str, + retries: int, + use_token: bool, + limit: int = REPLAY_WINDOW, + time_budget: float = REPLAY_SECONDS, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None, ) -> ReplayStats: stats = ReplayStats() - try: - files: List[str] = sorted([ - os.path.join(queue_dir, name) - for name in os.listdir(queue_dir) - if name.endswith(".json") and os.path.isfile(os.path.join(queue_dir, name)) - ]) - except Exception: - return stats - + _refuse_publication_during_measurement(queue_dir) + files, deferred = due_first_queue_paths(queue_dir, limit=limit) + stats.deferred += deferred started = time.monotonic() - for path in files[:25]: - if time.monotonic() - started >= 60: + for index, path in enumerate(files): + if _event_cancelled(cancel_event): + # Reconciled counters (C05): this entry counts once, as cancelled; + # only the entries behind it count as deferred. + stats.cancelled += 1 + stats.deferred += len(files) - index - 1 + break + if deadline is not None and time.monotonic() >= deadline: + # R03: real caller deadline (monotonic wall-clock), distinct from + # the per-pass throughput budget below. + stats.deferred += len(files) - index + break + if time.monotonic() - started >= time_budget: + stats.deferred += len(files) - index break status, _message = submit_spooled_path( path, @@ -755,6 +1353,8 @@ def replay_spool( api_key=api_key, retries=retries, use_token=use_token, + cancel_event=cancel_event, + deadline=deadline, ) if status == "submitted": stats.submitted += 1 @@ -765,3 +1365,12 @@ def replay_spool( elif status == "corrupt": stats.corrupt += 1 return stats + + +def _event_cancelled(cancel_event: Optional[Any]) -> bool: + if cancel_event is None: + return False + try: + return bool(cancel_event.is_set()) + except Exception: + return False diff --git a/client/suite.py b/client/suite.py index c4ddfa69..cf413c64 100644 --- a/client/suite.py +++ b/client/suite.py @@ -1,6 +1,8 @@ +import errno import hashlib import json import math +import re import os import shutil import queue @@ -18,7 +20,7 @@ import struct from dataclasses import dataclass from fractions import Fraction -from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple +from typing import Any, Callable, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple from . import config @@ -30,6 +32,11 @@ SUITE_PACK_METADATA_RELATIVE_PATH = os.path.join("resources", "test_suite_v1", "suite-pack.json") SUITE_PACK_SCHEMA_VERSION = 1 DEFAULT_SUITE_PACK_FILE_NAME = f"{SUITE_VERSION}.tar.gz" +CLIP_DISTRIBUTION_RELATIVE_PATH = os.path.join("resources", "test_suite_v1", "clip-distribution.json") +CLIP_DISTRIBUTION_SCHEMA_VERSION = 1 +SUITE_CLIP_BASE_URL_ENV = "ENCODINGDB_SUITE_CLIP_BASE_URL" +SUITE_ALLOW_FULL_PACK_ENV = "ENCODINGDB_ALLOW_FULL_PACK" +SUITE_MIN_FREE_MB_ENV = "ENCODINGDB_SUITE_MIN_FREE_MB" REQUIRED_CONTENT_CLASSES: Tuple[str, ...] = ( "high-motion-sports", @@ -228,6 +235,210 @@ def load_suite_pack_metadata(path: Optional[str] = None) -> Dict[str, Any]: return payload +def get_clip_distribution_path() -> Optional[str]: + for manifest_path in _manifest_resource_candidates(): + candidate = os.path.join(os.path.dirname(manifest_path), os.path.basename(CLIP_DISTRIBUTION_RELATIVE_PATH)) + if candidate and os.path.exists(candidate): + return candidate + return None + + +def _clip_notice_asset_names(manifest: SuiteManifest, suite_root: str) -> Dict[str, List[str]]: + """Map each clip to its attribution notice plus the license texts it names. + + Bindings come from the frozen notice bodies themselves: every clip notice + declares "License: ;" and a font notice beside that clip's license is + included only when the clip notice mentions the font. + """ + notices_root = os.path.join(suite_root, "notices") + available = {name for name in os.listdir(notices_root) if name.endswith(".txt")} if os.path.isdir(notices_root) else set() + bindings: Dict[str, List[str]] = {} + for clip in manifest.clips: + own = f"{clip.clip_id}.txt" + if own not in available: + raise RuntimeError(f"missing attribution notice for {clip.clip_id}") + with open(os.path.join(notices_root, own), "r", encoding="utf-8") as handle: + text = handle.read() + names = [own] + license_match = re.search(r"License:\s*([^;\n]+);", text) + declared = license_match.group(1).strip() if license_match else "" + if declared: + license_file = f"{declared}.txt" + if license_file not in available: + raise RuntimeError(f"missing license text {license_file} for {clip.clip_id}") + names.append(license_file) + if re.search(r"plex|font", text, re.IGNORECASE): + for name in sorted(available): + if name in names or name == own: + continue + with open(os.path.join(notices_root, name), "r", encoding="utf-8") as other_handle: + other = other_handle.read(400) + if re.search(r"font software is licensed", other, re.IGNORECASE): + names.append(name) + bindings[clip.clip_id] = names + return bindings + + +def build_clip_distribution_metadata( + suite_root: str, + *, + manifest: Optional[SuiteManifest] = None, + base_url: str = "", + release_tag: Optional[str] = None, + published: bool = False, +) -> Dict[str, Any]: + """Derive the per-clip distribution manifest from frozen suite resources. + + Every clip/notice hash and size comes from manifest.json and notices/; the + canonical media bytes themselves are not required. ``path`` records the + logical source layout; ``downloadName`` is unique and flat for releases. + """ + suite_root_abs = os.path.abspath(suite_root) + manifest_value = manifest + if manifest_value is None: + with open(os.path.join(suite_root_abs, "manifest.json"), "r", encoding="utf-8") as handle: + manifest_value = manifest_from_payload(json.load(handle)) + manifest_path = os.path.join(suite_root_abs, "manifest.json") + clips: Dict[str, Any] = {} + for clip_id, names in _clip_notice_asset_names(manifest_value, suite_root_abs).items(): + clip = get_clip(manifest_value, clip_id) + assets: List[Dict[str, Any]] = [ + { + "role": "clip", + "path": f"{clip_id}/{clip.file_name}", + "downloadName": f"{clip_id}--{clip.file_name}", + "sha256": clip.sha256, + "byteSize": clip.byte_size, + } + ] + for name in names: + notice_path = os.path.join(suite_root_abs, "notices", name) + assets.append( + { + "role": "notice" if name == f"{clip_id}.txt" else "license", + "path": f"{clip_id}/notices/{name}", + "downloadName": f"{clip_id}--{name}", + "sha256": _sha256_of_file(notice_path), + "byteSize": os.path.getsize(notice_path), + } + ) + clips[clip_id] = { + "fileName": clip.file_name, + "license": str(clip.provenance.get("license") or ""), + "assets": assets, + } + return { + "schemaVersion": CLIP_DISTRIBUTION_SCHEMA_VERSION, + "suiteId": "encodingdb-test-suite", + "suiteVersion": manifest_value.suite_version, + "manifestVersion": manifest_value.manifest_version, + "source": "github-release" if published else "staged-unpublished", + "releaseTag": release_tag, + "distribution": { + "baseUrl": base_url, + "clipUrlOverrideEnv": SUITE_CLIP_BASE_URL_ENV, + "published": bool(published), + }, + "manifest": { + "sha256": _sha256_of_file(manifest_path), + "byteSize": os.path.getsize(manifest_path), + }, + "clips": clips, + } + + +def write_clip_distribution_metadata(suite_root: str, output_path: Optional[str] = None, **kwargs: Any) -> str: + payload = build_clip_distribution_metadata(suite_root, **kwargs) + destination = os.path.abspath(output_path or os.path.join(suite_root, "clip-distribution.json")) + os.makedirs(os.path.dirname(destination), exist_ok=True) + with open(destination, "w", encoding="utf-8") as handle: + json.dump(payload, handle, indent=2, sort_keys=False) + handle.write("\n") + return destination + + +def load_clip_distribution_metadata(path: Optional[str] = None, + manifest: Optional[SuiteManifest] = None) -> Optional[Dict[str, Any]]: + """Load the clip distribution manifest; None when the file is absent. + + Absence only disables the per-clip route (development fixtures). A present + file that contradicts the frozen suite identity fails closed — it would + otherwise redirect acquisition to unreviewed bytes. + """ + resolved = path if path is not None else get_clip_distribution_path() + if resolved is None: + return None + with open(resolved, "r", encoding="utf-8") as handle: + payload = json.load(handle) + if not isinstance(payload, dict): + raise RuntimeError("clip distribution metadata is invalid") + if int(payload.get("schemaVersion") or 0) != CLIP_DISTRIBUTION_SCHEMA_VERSION: + raise RuntimeError("clip distribution metadata has an unsupported schemaVersion") + suite_manifest = manifest if manifest is not None else load_default_suite_manifest() + if str(payload.get("suiteVersion") or "") != suite_manifest.suite_version: + raise RuntimeError("clip distribution metadata suite version mismatch") + manifest_path = os.path.join(os.path.dirname(resolved), "manifest.json") + record = dict(payload.get("manifest") or {}) + if int(record.get("byteSize") or 0) != os.path.getsize(manifest_path) \ + or str(record.get("sha256") or "").lower() != _sha256_of_file(manifest_path): + raise RuntimeError("clip distribution metadata manifest identity mismatch") + clips = dict(payload.get("clips") or {}) + manifest_ids = {clip.clip_id for clip in suite_manifest.clips} + if set(clips) != manifest_ids: + raise RuntimeError("clip distribution metadata clip inventory mismatch") + bindings = _clip_notice_asset_names(suite_manifest, os.path.dirname(resolved)) + for clip in suite_manifest.clips: + entry = dict(clips.get(clip.clip_id) or {}) + if str(entry.get("fileName") or "") != clip.file_name: + raise RuntimeError(f"clip distribution metadata file name mismatch for {clip.clip_id}") + assets = list(entry.get("assets") or []) + roles = [str(asset.get("role") or "") for asset in assets] + if roles.count("clip") != 1: + raise RuntimeError(f"clip distribution metadata needs exactly one clip asset for {clip.clip_id}") + expected_names = bindings[clip.clip_id] + actual_names = [ + str(asset.get("path") or "").rsplit("notices/", 1)[-1] + for asset in assets if str(asset.get("role") or "") in ("notice", "license") + ] + if sorted(actual_names) != sorted(expected_names): + raise RuntimeError(f"clip distribution metadata notice binding mismatch for {clip.clip_id}") + for asset in assets: + relative = str(asset.get("path") or "") + if not relative.startswith(f"{clip.clip_id}/") or ".." in relative.split("/"): + raise RuntimeError(f"clip distribution metadata has an unsafe asset path: {relative}") + expected_download_name = f"{clip.clip_id}--{relative.rsplit('/', 1)[-1]}" + if str(asset.get("downloadName") or "") != expected_download_name: + raise RuntimeError(f"clip distribution metadata download name mismatch: {relative}") + expected_sha = str(asset.get("sha256") or "").lower() + expected_bytes = int(asset.get("byteSize") or 0) + if str(asset.get("role") or "") == "clip": + if expected_bytes != clip.byte_size or expected_sha != clip.sha256.lower(): + raise RuntimeError(f"clip distribution metadata identity mismatch for {clip.clip_id}") + else: + notice_file = relative.rsplit("notices/", 1)[-1] + notice_path = os.path.join(os.path.dirname(resolved), "notices", notice_file) + if not os.path.isfile(notice_path) or os.path.getsize(notice_path) != expected_bytes \ + or _sha256_of_file(notice_path).lower() != expected_sha: + raise RuntimeError(f"clip distribution metadata notice mismatch: {relative}") + return payload + + +def clip_distribution_available(metadata: Optional[Mapping[str, Any]]) -> bool: + if not metadata: + return False + override = os.environ.get(SUITE_CLIP_BASE_URL_ENV, "").strip() + base = str(dict(metadata.get("distribution") or {}).get("baseUrl") or "").strip() + return bool(override or base) + + +def _clip_asset_url(metadata: Mapping[str, Any], asset_path: str) -> str: + override = os.environ.get(SUITE_CLIP_BASE_URL_ENV, "").strip() + base = override or str(dict(metadata.get("distribution") or {}).get("baseUrl") or "").strip() + if not base: + raise RuntimeError("clip distribution has no base URL") + return urljoin(base.rstrip("/") + "/", asset_path.lstrip("/")) + + def _suite_cache_root() -> str: custom = os.environ.get("ENCODINGDB_SUITE_CACHE_DIR", "").strip() if custom: @@ -1186,9 +1397,15 @@ def read_network(): # The reader closes it on return/timeout and cannot mutate retained data. -def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any]) -> None: - distribution = dict(metadata.get("distribution") or {}) - expected_size = int(distribution.get("byteSize") or 0) +def _download_verified_file(url: str, destination: str, *, expected_size: int, + verify: "Callable[[str], ClipVerificationResult]", + exceeds_message: str) -> None: + """Resumable, size-bounded, hash-verified download into `destination`. + + Shared by the frozen suite pack and per-clip assets: one owned reader + thread, `.part` resume via Range, hard stop at the declared size, and a + caller-supplied final verification before the atomic install. + """ temp_path = f"{destination}.part" os.makedirs(os.path.dirname(destination), exist_ok=True) resume_from = 0 @@ -1214,25 +1431,346 @@ def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any] os.remove(temp_path) continue completed = resume_from if mode == "ab" else 0 - with open(temp_path, mode) as handle: - for kind, chunk in events: - if kind != "chunk": - raise RuntimeError("Unexpected suite download event") - check_preparation_cancelled() - if completed + len(chunk) > expected_size: - raise RuntimeError("Suite download exceeds declared pack size") - handle.write(chunk) - completed += len(chunk) - preparation_progress("download", path=destination, completedBytes=completed, totalBytes=expected_size) + try: + with open(temp_path, mode) as handle: + for kind, chunk in events: + if kind != "chunk": + raise RuntimeError("Unexpected suite download event") + check_preparation_cancelled() + if completed + len(chunk) > expected_size: + raise RuntimeError(exceeds_message) + handle.write(chunk) + completed += len(chunk) + preparation_progress("download", path=destination, completedBytes=completed, totalBytes=expected_size) + except OSError as exc: + if getattr(exc, "errno", None) in (errno.ENOSPC, errno.EDQUOT): + raise RuntimeError( + f"the suite cache volume ran out of free space while downloading " + f"{destination}: {exc}; free space on that volume and start the run again" + ) from exc + if getattr(exc, "errno", None) in (errno.EACCES, errno.EPERM, errno.EROFS): + raise RuntimeError( + f"suite content could not be written to {destination}: the folder is " + f"write-protected or unwritable for this account ({exc})" + ) from exc + raise break finally: events.close() - result = _verify_suite_pack_file(temp_path, metadata) + result = verify(temp_path) if not result.ok: + # The bytes are untrusted; never keep a corrupt partial for resume. + try: + os.remove(temp_path) + except FileNotFoundError: + pass raise RuntimeError(result.message) os.replace(temp_path, destination) +def _suite_free_bytes(path: str) -> Optional[int]: + try: + probe_path = path + while not os.path.exists(probe_path): + parent = os.path.dirname(probe_path) + if parent == probe_path: + break + probe_path = parent + return int(shutil.disk_usage(probe_path).free) + except OSError: + return None + + +def _suite_cache_write_error(cache_root: str) -> Optional[str]: + """Return a clear cause when the cache root cannot accept new files.""" + try: + os.makedirs(cache_root, exist_ok=True) + probe_dir = os.path.join(cache_root, "canonical") + os.makedirs(probe_dir, exist_ok=True) + fd, probe_path = tempfile.mkstemp(prefix=".encodingdb-write-probe-", dir=probe_dir) + os.close(fd) + os.remove(probe_path) + except OSError as exc: + return f"the suite cache {cache_root} is not writable by this account ({exc})" + return None + + +def _check_acquisition_storage(cache_root: str, required_bytes: int) -> None: + """Gate a costly fetch on free space + writability before the first byte. + + Requires room for the transfer plus the ENCODINGDB_SUITE_MIN_FREE_MB floor + (default 64 MiB; 0 disables). A declared-but-unwritable cache is reported + as such instead of failing mid-download. + """ + floor_mb = max(0, int(os.environ.get(SUITE_MIN_FREE_MB_ENV, "64") or "0")) + required = int(required_bytes) + floor_mb * 1024 * 1024 + write_error = _suite_cache_write_error(cache_root) + if write_error is not None: + raise RuntimeError(write_error) + free = _suite_free_bytes(cache_root) + if free is not None and free < required: + raise RuntimeError( + f"acquiring suite content needs {required_bytes:,} bytes plus a " + f"{floor_mb * 1024 * 1024:,} byte free-space floor, but the volume holding " + f"{cache_root} has only {free:,} bytes free; free space or set " + f"{SUITE_MIN_FREE_MB_ENV} to 0 to override the floor" + ) + + +def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any]) -> None: + distribution = dict(metadata.get("distribution") or {}) + expected_size = int(distribution.get("byteSize") or 0) + _download_verified_file( + url, + destination, + expected_size=expected_size, + verify=lambda path: _verify_suite_pack_file(path, metadata), + exceeds_message="Suite download exceeds declared pack size", + ) + + +def _full_pack_fallback_allowed(allow_full_pack: bool) -> bool: + if not allow_full_pack: + return False + setting = os.environ.get(SUITE_ALLOW_FULL_PACK_ENV, "1").strip().lower() + return setting not in ("0", "false", "no", "off") + + +def _disclose_large_download(clip: SuiteClip, pack_bytes: int, clip_error: Optional[str]) -> None: + """Make an unexpected full-pack transfer visible before it starts.""" + reason = f" ({clip_error})" if clip_error else " (no per-clip assets are published for this suite yet)" + message = ( + f"The small per-clip asset for {clip.clip_id} could not be acquired{reason}. " + f"Falling back to the full frozen suite pack: {pack_bytes:,} bytes " + f"(about {pack_bytes / (2 ** 30):.1f} GiB) will be transferred into the local suite cache. " + f"Set {SUITE_CLIP_BASE_URL_ENV} to a host with published per-clip assets to avoid this." + ) + preparation_progress("large-download", clipId=clip.clip_id, totalBytes=pack_bytes, message=message) + print(f"EncodingDB: {message}", file=sys.stderr) + + +def _clip_asset_target(cache_base: str, clip: SuiteClip, asset: Mapping[str, Any]) -> str: + if str(asset.get("role") or "") == "clip": + return clip_cache_path(clip, cache_base) + name = str(asset.get("path") or "").rsplit("notices/", 1)[-1] + return os.path.join(cache_base, "notices", name) + + +def _verified_asset_result(path: str, asset: Mapping[str, Any], clip: SuiteClip) -> ClipVerificationResult: + if str(asset.get("role") or "") == "clip": + return _verify_suite_clip_bytes(path, clip) + if not os.path.exists(path): + return ClipVerificationResult(False, f"{os.path.basename(path)} not found", {"path": path}) + actual_size = os.path.getsize(path) + if actual_size != int(asset.get("byteSize") or 0): + return ClipVerificationResult(False, f"{os.path.basename(path)} size mismatch", {"path": path}) + if _sha256_of_file(path).lower() != str(asset.get("sha256") or "").lower(): + return ClipVerificationResult(False, f"{os.path.basename(path)} checksum mismatch", {"path": path}) + return ClipVerificationResult(True, "ok", {"path": path}) + + +def _pending_clip_assets(clip: SuiteClip, cache_base: str, + assets: Sequence[Mapping[str, Any]]) -> Tuple[List[Tuple[Mapping[str, Any], str]], int]: + pending: List[Tuple[Mapping[str, Any], str]] = [] + cached_bytes = 0 + for asset in assets: + target = _clip_asset_target(cache_base, clip, asset) + if _verified_asset_result(target, asset, clip).ok: + cached_bytes += int(asset.get("byteSize") or 0) + else: + pending.append((asset, target)) + return pending, cached_bytes + + +def _materialize_clip_from_distribution(clip: SuiteClip, distribution: Mapping[str, Any], + cache_base: str) -> str: + """Fetch only this clip's canonical file plus its license notices. + + Each asset resumes via its own .part, stops at the declared size, and is + hash-verified against the frozen manifest before the atomic install; a + corrupt response never installs. + """ + entry = dict(distribution.get("clips") or {}).get(clip.clip_id) + if not entry: + raise RuntimeError(f"clip distribution metadata has no entry for {clip.clip_id}") + assets = list(dict(entry).get("assets") or []) + pending, _ = _pending_clip_assets(clip, cache_base, assets) + if not pending: + return clip_cache_path(clip, cache_base) + transfer_bytes = sum(int(asset.get("byteSize") or 0) for asset, _ in pending) + headroom_bytes = max(64 * 1024 * 1024, transfer_bytes // 20) + _check_acquisition_storage(cache_base, transfer_bytes + headroom_bytes) + for _, target in pending: + try: + os.makedirs(os.path.dirname(target), exist_ok=True) + except OSError as exc: + raise RuntimeError( + f"the suite cache folder {os.path.dirname(target)} is not writable by this account ({exc})" + ) from exc + label = f"Suite clip download for {clip.clip_id}" + for asset, target in pending: + url = _clip_asset_url(distribution, str(asset.get("downloadName") or "")) + _download_verified_file( + url, + target, + expected_size=int(asset.get("byteSize") or 0), + verify=lambda path, current=asset: _verified_asset_result(path, current, clip), + exceeds_message=f"{label} exceeds declared file size", + ) + result = _verify_suite_clip_bytes(clip_cache_path(clip, cache_base), clip) + if not result.ok: + raise RuntimeError(result.message) + return clip_cache_path(clip, cache_base) + + +def _packaged_canonical_path(clip: SuiteClip) -> Optional[str]: + for manifest_path in _manifest_resource_candidates(): + candidate = os.path.join(os.path.dirname(manifest_path), "canonical", clip.file_name) + if os.path.exists(candidate): + return candidate + return None + + +def _pack_state(pack_metadata: Mapping[str, Any], cache_base: str) -> Dict[str, bool]: + pack_cached = _verify_suite_pack_file(_cache_suite_pack_path(pack_metadata, cache_base), pack_metadata).ok + fingerprint = str(pack_metadata.get("suiteFingerprint") or "").strip() + # Without a fingerprint the extraction directory is unnameable; assume absent. + extracted = bool(fingerprint) and os.path.isdir( + os.path.join(_suite_pack_extract_root(pack_metadata, cache_base), "canonical")) + return {"packCached": pack_cached, "extracted": extracted} + + +def _plan_clip_acquisition(clip: SuiteClip, cache_base: str, *, clip_route_available: bool, + pack_metadata: Mapping[str, Any], pack_state: Mapping[str, bool], + allow_full_pack: bool) -> Dict[str, Any]: + """Decide one clip's acquisition without side effects beyond hash reads.""" + packaged = _packaged_canonical_path(clip) + if packaged is not None and _verify_suite_clip_bytes(packaged, clip).ok: + return {"source": "packaged", "transferBytes": 0, "cachedBytes": clip.byte_size, "peakBytes": 0, + "note": "verified canonical asset is bundled with the client"} + cached_path = clip_cache_path(clip, cache_base) + if _verify_suite_clip_bytes(cached_path, clip).ok: + return {"source": "cache", "transferBytes": 0, "cachedBytes": clip.byte_size, "peakBytes": 0, "note": "hash-verified cache hit"} + pack_bytes = int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0) + if clip_route_available: + distribution = load_clip_distribution_metadata(manifest=None) + entry = dict((distribution or {}).get("clips") or {}).get(clip.clip_id) or {} + assets = list(dict(entry).get("assets") or []) + pending, cached_bytes = _pending_clip_assets(clip, cache_base, assets) + transfer = sum(int(asset.get("byteSize") or 0) for asset, _ in pending) + headroom = max(64 * 1024 * 1024, transfer // 20) + return {"source": "clip", "transferBytes": transfer, "cachedBytes": cached_bytes, + "peakBytes": transfer + headroom, "note": "per-clip assets are published for this suite"} + if not _full_pack_fallback_allowed(allow_full_pack): + return {"source": "unavailable", "transferBytes": 0, "cachedBytes": 0, + "peakBytes": clip.byte_size, + "note": (f"clip is not cached, no per-clip assets are available, and the full pack " + f"({pack_bytes:,} bytes) is disabled via {SUITE_ALLOW_FULL_PACK_ENV}/allow_full_pack")} + transfer = 0 if pack_state.get("packCached") else pack_bytes + extract_bytes = 0 if pack_state.get("extracted") else pack_bytes + return {"source": "pack", "transferBytes": transfer, "cachedBytes": 0, + "peakBytes": transfer + extract_bytes + clip.byte_size, + "note": f"only the full frozen suite pack ({pack_bytes:,} bytes) is published for this suite"} + + +def acquisition_estimate(clip_ids: Optional[Sequence[str]] = None, *, + cache_root: Optional[str] = None, + allow_full_pack: bool = True, + manifest: Optional[SuiteManifest] = None) -> Dict[str, Any]: + """Truthful, side-effect-free transfer/peak-storage estimate for a fetch. + + Stable read-only API for UIs: reports exactly what `ensure_suite_clip`/ + `ensure_suite` would transfer per clip (cache hit, per-clip assets, or the + full pack) before any byte is downloaded. + """ + suite_manifest = manifest or load_default_suite_manifest() + resolved_cache_root = cache_root or _suite_cache_root() + target_ids = list(clip_ids) if clip_ids else [clip.clip_id for clip in suite_manifest.clips] + distribution = load_clip_distribution_metadata(manifest=suite_manifest) + route_available = clip_distribution_available(distribution) + pack_metadata = load_suite_pack_metadata() + pack_state = _pack_state(pack_metadata, resolved_cache_root) + plans: Dict[str, Any] = {} + transfer_total = 0 + cached_total = 0 + peak_total = 0 + warnings: List[str] = [] + worst = "cache" + pack_counted = False + severity = {"cache": 0, "packaged": 0, "pack": 1, "clip": 1, "unavailable": 2} + for clip_id in target_ids: + clip = get_clip(suite_manifest, clip_id) + plan = _plan_clip_acquisition(clip, resolved_cache_root, clip_route_available=route_available, + pack_metadata=pack_metadata, pack_state=pack_state, + allow_full_pack=allow_full_pack) + if plan["source"] == "pack": + if pack_counted: + plan["transferBytes"] = 0 + plan["peakBytes"] = clip.byte_size + plan["note"] = "shared full suite pack counted with the first missing clip" + pack_counted = True + plans[clip_id] = {"clipId": clip_id, "fileName": clip.file_name, **plan} + transfer_total += int(plan["transferBytes"]) + cached_total += int(plan["cachedBytes"]) + peak_total += int(plan["peakBytes"]) + if severity.get(plan["source"], 0) > severity.get(worst, 0): + worst = plan["source"] + if plan["source"] in ("unavailable", "pack"): + warnings.append(f"{clip_id}: {plan['note']}") + free = _suite_free_bytes(resolved_cache_root) + floor_mb = max(0, int(os.environ.get(SUITE_MIN_FREE_MB_ENV, "64") or "0")) + storage_ok = free is None or free >= peak_total + floor_mb * 1024 * 1024 + return { + "schemaVersion": 1, + "suiteVersion": suite_manifest.suite_version, + "cacheRoot": resolved_cache_root, + "strategy": worst, + "clipRouteAvailable": bool(route_available), + "allowFullPack": bool(_full_pack_fallback_allowed(allow_full_pack)), + "fullPackBytes": int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0), + "bytesToTransfer": transfer_total, + "bytesAlreadyVerified": cached_total, + "peakStorageBytes": peak_total, + "freeBytes": free, + "freeFloorBytes": floor_mb * 1024 * 1024, + "storageOk": bool(storage_ok), + "clips": plans, + "warnings": warnings, + } + + +def _acquire_missing_clip(clip: SuiteClip, cache_base: str, *, allow_full_pack: bool) -> str: + distribution = load_clip_distribution_metadata(manifest=None) + clip_error: Optional[str] = None + if clip_distribution_available(distribution): + try: + return _materialize_clip_from_distribution(clip, distribution, cache_base) + except RuntimeError as exc: + clip_error = str(exc) + preparation_progress("recovery", clipId=clip.clip_id, + message=f"per-clip acquisition failed ({clip_error}); considering the full pack") + pack_metadata = load_suite_pack_metadata() + pack_bytes = int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0) + if not _full_pack_fallback_allowed(allow_full_pack): + detail = f" Per-clip download failed: {clip_error}." if clip_error else "" + raise RuntimeError( + f"suite clip {clip.clip_id} is not cached and the full frozen suite pack " + f"({pack_bytes:,} bytes) download is disabled. Set {SUITE_CLIP_BASE_URL_ENV} to a host with " + "the published per-clip assets, provide the pack via ENCODINGDB_SUITE_PACK_PATH, or allow " + f"{SUITE_ALLOW_FULL_PACK_ENV}." + detail + ) + state = _pack_state(pack_metadata, cache_base) + needed = clip.byte_size + if not state["packCached"]: + needed += pack_bytes + if not state["extracted"]: + needed += pack_bytes + _check_acquisition_storage(cache_base, needed) + if not state["packCached"] or not state["extracted"]: + _disclose_large_download(clip, pack_bytes, clip_error) + return _materialize_clip_from_suite_pack(clip, pack_metadata, cache_base) + + def _local_suite_pack_candidates(file_name: str) -> List[str]: candidates: List[str] = [] explicit = os.environ.get("ENCODINGDB_SUITE_PACK_PATH", "").strip() @@ -1412,6 +1950,7 @@ def ensure_suite_clip( *, cache_root: Optional[str] = None, regenerate_on_mismatch: bool = True, + allow_full_pack: bool = True, ) -> PreparedSuiteClip: wait_for_owned_acquisition() resolved_cache_root = cache_root or _suite_cache_root() @@ -1430,8 +1969,8 @@ def ensure_suite_clip( path = clip_cache_path(clip, resolved_cache_root) result = verify_suite_clip(path, clip) if not result.ok and regenerate_on_mismatch: - path = _materialize_clip_from_suite_pack(clip, load_suite_pack_metadata(), resolved_cache_root) - # That exact staged stream passed this clip's complete media contract. + path = _acquire_missing_clip(clip, resolved_cache_root, allow_full_pack=allow_full_pack) + # The acquired stream passed this clip's complete media contract. # Recheck the renamed bytes, including SHA, before reusing that result. result = _verify_suite_clip_bytes(path, clip) if not result.ok: @@ -1474,6 +2013,7 @@ def ensure_suite( *, clip_ids: Optional[Sequence[str]] = None, cache_root: Optional[str] = None, + allow_full_pack: bool = True, ) -> List[PreparedSuiteClip]: suite = manifest or load_default_suite_manifest() target_ids = set(clip_ids or [clip.clip_id for clip in suite.clips]) @@ -1481,7 +2021,7 @@ def ensure_suite( for index, clip in enumerate(suite.clips, 1): preparation_progress("clip", clipId=clip.clip_id, completedClips=index - 1, totalClips=len(suite.clips)) if clip.clip_id in target_ids: - prepared.append(ensure_suite_clip(clip, cache_root=cache_root)) + prepared.append(ensure_suite_clip(clip, cache_root=cache_root, allow_full_pack=allow_full_pack)) return prepared diff --git a/client/tests/test_artifact_cancellation.py b/client/tests/test_artifact_cancellation.py new file mode 100644 index 00000000..9d8a91d8 --- /dev/null +++ b/client/tests/test_artifact_cancellation.py @@ -0,0 +1,407 @@ +"""C12 revision: phase-aware cooperative cancellation and bounded transport. + +Covers: stalled create/auth/PUT, cancel during each phase, lost response after +server acceptance, idempotent replay with the same identity, redacted public +errors (no raw bodies/tokens), and GUI-safe progress reporting. +""" +import hashlib +import json +import socket +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import ClassVar, List, Optional + +import pytest + +from client.artifacts import SubmitError, SubmissionCancelled, build_payload_hash, submit_artifact_submission +from client.network import submit as legacy_submit + + +def _artifact_bytes() -> bytes: + return b"C" * (1 << 20) # 1 MiB + + +def _submission(tmp_path, run_create=None): + path = tmp_path / "artifact.mp4" + path.write_bytes(_artifact_bytes()) + create = run_create if run_create is not None else { + "campaignId": "campaign-c12", + "repetitionGroupId": "campaign-c12:recipe-1", + "repetitionIndex": 1, + "artifact": {"role": "ENCODED", "sha256": hashlib.sha256(_artifact_bytes()).hexdigest(), + "byteSize": len(_artifact_bytes()), "mediaContainer": "mp4"}, + } + create = dict(create) + create.setdefault("payloadHash", build_payload_hash(create)) + return {"submissionKind": "authoritative-artifact-run-v1", + "artifactPath": str(path), "contentType": "video/mp4", "runCreate": create} + + +class _Server(ThreadingHTTPServer): + daemon_threads = True + allow_reuse_address = True + + +class _StallHandler(BaseHTTPRequestHandler): + """Stalls each phase on a class-level control event.""" + phase: ClassVar[str] = "create" # where to stall: create|auth|put + release: ClassVar[Optional[threading.Event]] = None + first_body_chunk: ClassVar[Optional[threading.Event]] = None + seen: ClassVar[List[str]] = [] + + def _read_body(self) -> bytes: + n = int(self.headers.get("Content-Length", "0")) + return self.rfile.read(n) if n else b"" + + def _stall(self, phase: str) -> None: + if type(self).phase == phase and type(self).release is not None: + type(self).release.wait(10) + + def do_POST(self) -> None: + body = self._read_body() + type(self).seen.append(self.path) + if self.path == "/v7/benchmark-runs": + self._stall("create") + payload = json.loads(body or b"{}") + self.send_response(201) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({ + "benchmarkRun": {"id": "run-c12"}, + "artifact": {"storageState": "PENDING"}, + "analyses": [], + }).encode()) + return + if self.path.endswith("/upload-authorizations"): + self._stall("auth") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": True, "token": "tok-c12"}).encode()) + return + self.send_response(404) + self.end_headers() + + def do_PUT(self) -> None: + # Read the body slowly: one chunk, then wait, so the client generator + # keeps pulling chunks and can observe cancellation mid-upload. + n = int(self.headers.get("Content-Length", "0")) + remaining = n + first = True + while remaining > 0: + chunk = self.rfile.read(min(65536, remaining)) + if not chunk: + break + remaining -= len(chunk) + if first: + first = False + if type(self).first_body_chunk is not None: + type(self).first_body_chunk.set() + self._stall("put") + if remaining > 0: + time.sleep(0.05) # throttle so upload stays in-flight + type(self).seen.append("PUT") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"benchmarkRun": {"id": "run-c12"}}).encode()) + + def log_message(self, format: str, *args) -> None: # noqa: A003 + return + + +def _start(handler): + server = _Server(("127.0.0.1", 0), handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + return server, f"http://127.0.0.1:{server.server_port}" + + +@pytest.fixture(autouse=True) +def _reset_handler_state(): + _StallHandler.phase = "create" + _StallHandler.release = None + _StallHandler.first_body_chunk = None + _StallHandler.seen = [] + yield + + +def test_cancel_before_call_touches_no_network(tmp_path): + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + cancel.set() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + assert caught.value.retryable + assert _StallHandler.seen == [] + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_stalled_create_is_bounded(tmp_path): + _StallHandler.phase = "create" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + threading.Timer(0.3, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + elapsed = time.monotonic() - started + assert caught.value.phase == "run create" + assert caught.value.retryable + # Bounded shutdown: cancel observed far under the old 300 s PUT window. + assert elapsed < 3.0 + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_stalled_create_hits_phase_budget_without_cancel(tmp_path): + _StallHandler.phase = "create" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=0.5) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 3.0 # hard bound, never the 60 s legacy timeout + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_stalled_auth(tmp_path): + _StallHandler.phase = "auth" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + threading.Timer(0.3, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + assert caught.value.phase == "upload authorization" + assert time.monotonic() - started < 3.0 + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_upload_and_progress_reporting(tmp_path): + _StallHandler.phase = "put" + _StallHandler.release = threading.Event() + _StallHandler.first_body_chunk = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + progress: List[tuple] = [] + + def on_progress(phase, sent, total): + progress.append((phase, sent, total)) + if sent >= 262144: # second chunk already accepted + cancel.set() + + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), + cancel_event=cancel, progress=on_progress, + upload_seconds=5.0) + elapsed = time.monotonic() - started + assert caught.value.phase == "artifact upload" + assert caught.value.retryable + # The caller returned while the worker was still socket-blocked: the + # transaction was genuinely in-flight (server receives bytes after the + # cancellation returns), and shutdown was fast. + assert _StallHandler.first_body_chunk.wait(5) + assert elapsed < 3.0 + # GUI-safe progress: upload phase, monotonic sent, correct total. + assert progress and all(p[0] == "artifact upload" for p in progress) + assert [p[1] for p in progress] == sorted(p[1] for p in progress) + assert progress[-1][2] == len(_artifact_bytes()) + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_upload_stall_is_hard_bounded_below_legacy_300s(tmp_path): + """No cancel at all: the upload phase budget alone terminates the stall.""" + class _NeverResponds(_StallHandler): + def do_PUT(self): + n = int(self.headers.get("Content-Length", "0")) + self.rfile.read(min(1024, n)) # read a little, then hang + time.sleep(10) + + server, base_url = _start(_NeverResponds) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), upload_seconds=1.0) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 8.0 # bounded ~1 s budget, not 300 s + finally: + server.shutdown() + server.server_close() + + +def test_lost_response_after_acceptance_replays_same_identity(tmp_path): + """Server accepts the PUT, then the response is lost. Retry with the same + submission must reuse the existing run and skip re-upload — never create a + second run, never discard the artifact.""" + class _IdempotentFlow(_StallHandler): + runs: ClassVar[dict] = {} + uploaded: ClassVar[bool] = False + lost_once: ClassVar[bool] = False + + def do_POST(self): + body = self._read_body() + if self.path == "/v7/benchmark-runs": + payload = json.loads(body or b"{}") + key = str(payload.get("payloadHash")) + run = _IdempotentFlow.runs.get(key) + if run is None: + run = {"id": f"run-{len(_IdempotentFlow.runs) + 1}"} + _IdempotentFlow.runs[key] = run + state = "RETAINED" if _IdempotentFlow.uploaded else "PENDING" + analyses = [{"id": "analysis-1", "status": "COMPLETE"}] if _IdempotentFlow.uploaded else [] + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({ + "benchmarkRun": {"id": run["id"]}, + "artifact": {"storageState": state}, + "analyses": analyses, + }).encode()) + return + if self.path.endswith("/upload-authorizations"): + if _IdempotentFlow.uploaded: + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": False}).encode()) + return + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": True, "token": "t1"}).encode()) + return + self.send_response(404) + self.end_headers() + + def do_PUT(self): + n = int(self.headers.get("Content-Length", "0")) + self.rfile.read(n) + _IdempotentFlow.uploaded = True + if _IdempotentFlow.lost_once: + _IdempotentFlow.lost_once = False + # Accept bytes, then drop the connection without a response. + self.close_connection = True + try: + self.connection.shutdown(socket.SHUT_RDWR) + self.connection.close() + except OSError: + pass + return + super().do_PUT() + + server, base_url = _start(_IdempotentFlow) + try: + submission = _submission(tmp_path) + _IdempotentFlow.lost_once = True + with pytest.raises(SubmitError) as lost: + submit_artifact_submission(base_url, submission) + assert lost.value.retryable # ambiguous: spool retains the entry + + # Replay after restart with the SAME submission identity. + result = submit_artifact_submission(base_url, submission) + assert result["benchmarkRun"]["id"] == "run-1" + assert len(_IdempotentFlow.runs) == 1 # no duplicate run created + finally: + server.shutdown() + server.server_close() + + +def test_public_errors_never_leak_server_bodies_or_tokens(tmp_path): + class _LeakyRejection(_StallHandler): + def do_POST(self): + self._read_body() + self.send_response(400) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(b'{"error":"bad","token":"raw-secret-token","body":"raw-secret-body"}') + + server, base_url = _start(_LeakyRejection) + try: + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path)) + exc = caught.value + assert exc.retryable is False + assert exc.status_code == 400 + text = str(exc) + assert "raw-secret-token" not in text + assert "raw-secret-body" not in text + assert "raw-secret-token" not in repr(exc) + # Raw body is retained privately for diagnostics only. + assert "raw-secret-body" in exc._server_body + finally: + server.shutdown() + server.server_close() + + +def test_submit_error_message_is_bounded(): + exc = SubmitError("x" * 5000, retryable=True) + assert len(str(exc)) <= SubmitError._MAX_MESSAGE_CHARS + + +def test_legacy_submit_cancel_during_retry_wait(): + """network.submit honours cancel_event while waiting out a 429 backoff.""" + class _RateLimited(_StallHandler): + def do_POST(self): + self._read_body() + self.send_response(429) + self.send_header("Retry-After", "30") + self.end_headers() + + server, base_url = _start(_RateLimited) + try: + cancel = threading.Event() + threading.Timer(0.2, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled): + legacy_submit(base_url, {"cpuModel": "T"}, retries=3, + backoff_seconds=1, use_token=False, cancel_event=cancel) + assert time.monotonic() - started < 3.0 + finally: + server.shutdown() + server.server_close() + + +def test_legacy_submit_transaction_bound_stops_stalled_attempt(): + """No cancel: the wall-clock transaction bound ends a stalled POST chain.""" + class _Silent(_StallHandler): + def do_POST(self): + time.sleep(10) + + server, base_url = _start(_Silent) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + legacy_submit(base_url, {"cpuModel": "T"}, retries=3, backoff_seconds=0.1, + use_token=False, transaction_seconds=1.0) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 8.0 # far under 30 s × retries legacy behavior + finally: + server.shutdown() + server.server_close() diff --git a/client/tests/test_assemble_client_release.py b/client/tests/test_assemble_client_release.py new file mode 100644 index 00000000..f5a5af7c --- /dev/null +++ b/client/tests/test_assemble_client_release.py @@ -0,0 +1,148 @@ +import hashlib +import json +import tempfile +import unittest +from pathlib import Path + +from scripts.assemble_client_release import assemble + + +def digest(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +class AssembleClientReleaseTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.entries = [] + for role, platform, name in ( + ("macos-dmg", "mac", "EncodingDB-macOS-arm64.dmg"), + ("windows-gui", "win", "encodingdb-client-windows.exe"), + ("windows-console", "win", "encodingdb-client-windows-console.exe"), + ("linux-archive", "linux", "encodingdb-client-linux.tar.gz"), + ): + data = role.encode() + artifact = self.root / name + artifact.write_bytes(data) + binary_sha = digest((role + "-binary").encode()) if role in {"macos-dmg", "linux-archive"} else digest(data) + manifest = { + "schemaVersion": 1, "source": {"revision": "a" * 40, "trackedChanges": False}, + "projectVersion": "1.3.0-rc.8", "platform": platform, + "protocol": {"clientVersion": "client/0.3.8", "benchmarkProtocolVersion": "7.1", + "minimumClientVersion": "client/0.3.0"}, + "suite": {"suiteVersion": "encodingdb-test-suite-v1", "manifestVersion": 2, + "suiteFingerprint": "f" * 64, "isFrozen": True}, + "runtime": {"fingerprint": platform + "-runtime"}, + "artifact": {"fileName": name, "sha256": binary_sha, "byteSize": len(data), + "executableIdentity": { + "format": {"mac": "Mach-O", "win": "PE", "linux": "ELF"}[platform], + "architecture": "arm64" if platform == "mac" else "x86_64", + }}, + } + manifest_path = self.root / f"{name}.release-manifest.json" + manifest_path.write_text(json.dumps(manifest)) + entry = {"role": role, "artifact": name, "releaseManifest": manifest_path.name} + if role in {"macos-dmg", "linux-archive"}: + wrapper = {"fileName": name, "sha256": digest(data), "byteSize": len(data)} + if role == "macos-dmg": + wrapper["readOnlyMountVerification"] = {"mounted": True, "innerSha256": binary_sha} + package = {"provisional": False, "source": manifest["source"], "dmg": wrapper, + "cliEmbeddedSha256": binary_sha, "cliBytesChangedByPackaging": False, + "minimumSystemVersion": "27.0"} + else: + wrapper["members"] = [{"sha256": binary_sha}] + package = {"provisional": False, "source": manifest["source"], "archive": wrapper, + "verification": {"verified": True}} + package_path = self.root / f"{name}.package-info.json" + package_path.write_text(json.dumps(package)) + entry["packageInfo"] = package_path.name + self.entries.append(entry) + + def spec(self): + return {"expectedSourceRevision": "a" * 40, "expectedProjectVersion": "1.3.0-rc.8", + "assets": self.entries} + + def test_assembles_four_verified_assets_from_one_clean_identity(self): + release = assemble(self.spec(), self.root) + self.assertEqual(release["sourceRevision"], "a" * 40) + self.assertEqual(release["clientVersion"], "client/0.3.8") + self.assertEqual(len(release["assets"]), 4) + self.assertEqual(release["lifecycle"], { + "builtFromReviewedSource": True, + "nativeAcceptance": "not_certified", + "published": False, + "independentRedownloadVerified": False, + }) + self.assertEqual({asset["role"] for asset in release["assets"]}, + {"macos-dmg", "windows-gui", "windows-console", "linux-archive"}) + support = {asset["role"]: asset["support"] for asset in release["assets"]} + self.assertEqual(support["macos-dmg"], { + "operatingSystem": "macOS", "architecture": "arm64", "minimumVersion": "27.0"}) + self.assertEqual(support["windows-gui"], { + "operatingSystem": "Windows", "architecture": "x86_64", "supportedVersion": "11", + "otherVersions": "unverified"}) + self.assertEqual(support["linux-archive"], { + "operatingSystem": "Ubuntu Linux", "architecture": "x86_64", + "supportedVersion": "24.04", "otherDistributions": "unverified"}) + + def test_rejects_tampered_asset_and_mixed_revision(self): + (self.root / "encodingdb-client-windows.exe").write_bytes(b"different") + with self.assertRaisesRegex(ValueError, "file bytes differ"): + assemble(self.spec(), self.root) + (self.root / "encodingdb-client-windows.exe").write_bytes(b"windows-gui") + path = self.root / "encodingdb-client-windows.exe.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["source"]["revision"] = "b" * 40 + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "differs from the reviewed source/version"): + assemble(self.spec(), self.root) + + def test_rejects_missing_package_proof_and_dirty_source(self): + path = self.root / "EncodingDB-macOS-arm64.dmg.package-info.json" + package = json.loads(path.read_text()) + package["dmg"]["readOnlyMountVerification"] = None + path.write_text(json.dumps(package)) + with self.assertRaisesRegex(ValueError, "read-only mount verification"): + assemble(self.spec(), self.root) + package["dmg"]["readOnlyMountVerification"] = {"mounted": True, "innerSha256": package["cliEmbeddedSha256"]} + path.write_text(json.dumps(package)) + path = self.root / "encodingdb-client-linux.tar.gz.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["source"]["trackedChanges"] = True + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "clean committed source"): + assemble(self.spec(), self.root) + + def test_rejects_different_runtime_for_windows_pair(self): + path = self.root / "encodingdb-client-windows-console.exe.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["runtime"]["fingerprint"] = "other-runtime" + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "different win runtime identities"): + assemble(self.spec(), self.root) + + def test_rejects_architecture_or_macos_floor_missing_from_native_receipts(self): + path = self.root / "encodingdb-client-windows.exe.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["artifact"]["executableIdentity"]["architecture"] = "arm64" + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "supported architecture"): + assemble(self.spec(), self.root) + receipt["artifact"]["executableIdentity"]["architecture"] = "x86_64" + path.write_text(json.dumps(receipt)) + package_path = self.root / "EncodingDB-macOS-arm64.dmg.package-info.json" + package = json.loads(package_path.read_text()) + package.pop("minimumSystemVersion") + package_path.write_text(json.dumps(package)) + with self.assertRaisesRegex(ValueError, "minimum OS version"): + assemble(self.spec(), self.root) + package["minimumSystemVersion"] = "26.0" + package_path.write_text(json.dumps(package)) + with self.assertRaisesRegex(ValueError, "qualified runtime floor"): + assemble(self.spec(), self.root) + + +if __name__ == "__main__": + unittest.main() diff --git a/client/tests/test_auto_saved_continuation.py b/client/tests/test_auto_saved_continuation.py new file mode 100644 index 00000000..c81cd30d --- /dev/null +++ b/client/tests/test_auto_saved_continuation.py @@ -0,0 +1,50 @@ +"""Normal Windows idle work must restage explicitly consented saved campaigns.""" + +from unittest import mock + +from client import main, windows_gui as gui +from test_progress_contract import _build_gui_app +from test_windows_gui import FakeThread + + +def _state(campaign_id, fingerprint): + return { + "publicationConsent": True, + "activeCollection": None, + "publicationLockBusy": False, + "publication": {"pendingEntries": 0, "dueEntries": 0, + "acceptedReceipts": 0, "terminalEntries": 0}, + "campaigns": [{ + "campaignId": campaign_id, "complete": True, + "pendingUploads": 1, "queueDue": 0, "queueTerminal": 0, + "unavailableSources": 0, + "actions": [{"action": "publish_saved"}], + "publicationIntent": {"active": True, + "baseUrlFingerprint": fingerprint, + "nextAttemptAt": 0}, + }], + } + + +def test_idle_windows_restages_saved_suffix_without_another_click(): + app = _build_gui_app() + app.no_submit_var.set(False) + base_url = app.base_url_var.get().strip() or str(app.base_args.base_url) + campaign_id = "campaign-0123456789abcdef" + app._load_recovery_state = lambda: _state( + campaign_id, main.publication_endpoint_fingerprint(base_url)) + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._idle_retry() + assert app.upload_thread is not None + assert app.upload_thread.target.__name__ == "_publish_saved_worker" + assert app.upload_thread.args[0] == campaign_id + + +def test_idle_windows_does_not_publish_to_changed_endpoint(): + app = _build_gui_app() + app.no_submit_var.set(False) + app._load_recovery_state = lambda: _state( + "campaign-0123456789abcdef", "different-endpoint-fingerprint") + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._idle_retry() + assert app.upload_thread is None diff --git a/client/tests/test_campaign_durability.py b/client/tests/test_campaign_durability.py index b8c45609..44e49a87 100644 --- a/client/tests/test_campaign_durability.py +++ b/client/tests/test_campaign_durability.py @@ -116,7 +116,7 @@ def test_retry_after_survives_restart_and_deadline_expires(tmp_path): def test_compatibility_refuses_old_epoch_before_campaign(tmp_path): - response = SimpleNamespace(status_code=200,json=lambda:{'protocolVersion':'7.0','minimumClientVersion':'client/0.2.0','encodeTimerBoundary':'old'}) + response = SimpleNamespace(status_code=200,json=lambda:{'protocolVersion':'7.0','minimumClientVersion':'client/0.2.0','encodeTimerBoundary':'old'},close=lambda:None) with mock.patch('client.network._load_requests',return_value=SimpleNamespace(get=lambda *a,**k:response)): with pytest.raises(SubmitError): check_compatibility('http://example.invalid','client/0.3.0') @@ -158,15 +158,161 @@ def test_completed_local_campaign_publishes_without_source_or_encoder(tmp_path): campaign_id='campaign-0123456789abcdef' root=journal_path(str(tmp_path),campaign_id) atomic_json(root/'campaign-complete.json',{'skipped':0,'failed':0}) - payload={'retained':'immutable'} + artifact=root/'saved.mp4' + artifact.write_bytes(b'retained original bytes') + payload={'artifactPath':str(artifact),'runCreate':{'campaignId':campaign_id}} atomic_json(root/'submission-000001.json',payload) - with mock.patch.object(main,'check_compatibility'), mock.patch.object(main,'spool_payload',return_value=('retained.json',{})) as save, mock.patch.object(main,'replay_spool',return_value=spool.ReplayStats(submitted=1)), mock.patch.object(main,'_prepare_named_suite_clip') as source, mock.patch.object(main,'run_benchmark_batch') as encode: + def accepted_replay(*_args, **_kwargs): + local_hash=spool.local_hash_for_payload(payload) + atomic_json(tmp_path/'receipts'/f'{local_hash}.json',{ + 'localHash':local_hash,'response':{'benchmarkRun':{'id':'run-saved'}}}) + return spool.ReplayStats(submitted=1) + with mock.patch.object(main,'check_compatibility'), mock.patch.object(main,'spool_payload',return_value=('retained.json',{})) as save, mock.patch.object(main,'replay_spool',side_effect=accepted_replay), mock.patch.object(main,'_prepare_named_suite_clip') as source, mock.patch.object(main,'run_benchmark_batch') as encode: assert main.main(['prog','--resume-campaign',campaign_id,'--submit','--queue-dir',str(tmp_path)]) == 0 - save.assert_called_once_with(str(tmp_path),payload,max_storage_mb=2048) + save.assert_called_once() + assert save.call_args.args == (str(tmp_path), payload) + assert save.call_args.kwargs['max_storage_mb'] == 2048 + assert not save.call_args.kwargs['cancel_event'].is_set() + assert isinstance(save.call_args.kwargs['deadline'], float) source.assert_not_called() encode.assert_not_called() +def test_reconstruction_refuses_changed_runtime_identity(tmp_path): + from client.campaign import journal_path + campaign_id = 'campaign-0123456789abcdef' + root = journal_path(str(tmp_path), campaign_id) + root.mkdir(parents=True) + atomic_json(root / 'manifest.json', { + 'protocolVersion': '7.1', 'clientVersion': main.CLIENT_VERSION, + 'runtime': {'ffmpeg': {'sha256': 'old-runtime'}}, + }) + with mock.patch('client.identity.runtime_identity', return_value={'ffmpeg': {'sha256': 'new-runtime'}}): + outcome = main._reconstruct_saved_submissions( + queue_dir=str(tmp_path), campaign_id=campaign_id, max_storage_mb=2048, + ) + assert 'saved runtime identity differs' in outcome['failure'] + assert outcome['reconstructed'] == 0 + + +def test_reconstruction_refuses_changed_client_version(tmp_path): + from client.campaign import journal_path + campaign_id = 'campaign-0123456789abcdef' + root = journal_path(str(tmp_path), campaign_id) + root.mkdir(parents=True) + atomic_json(root / 'manifest.json', { + 'protocolVersion': '7.1', 'clientVersion': 'client/0.0.0', + }) + outcome = main._reconstruct_saved_submissions( + queue_dir=str(tmp_path), campaign_id=campaign_id, max_storage_mb=2048, + ) + assert 'saved client identity differs' in outcome['failure'] + assert outcome['reconstructed'] == 0 + + +def test_publish_saved_rebuilds_envelopes_for_complete_group_after_controlled_stop(tmp_path): + # C09: a controlled stop after a group's measured attempts are durable but + # before submission-*.json envelopes exist must NOT publish zero groups. + # Restarting with the original source unavailable, Publish saved rebuilds + # the exact envelope (same group ID) from the retained journal with zero + # encodes, while the still-extendable group stays unfinished. + import threading + from test_main_routing import MainRoutingTests, _DummyDashboard + fixture = MainRoutingTests() + clip_a = fixture._quick_clip() + clip_b = dataclasses.replace(clip_a, clip_id="film-grain-1080p24-final", + workload_id="film-grain-1080p24-final") + args = fixture._batch_args(str(tmp_path), no_submit=True) + args.local_metrics = False + args.campaign_seed = 41 + args.max_duration_minutes = 1.0 + calls = [] + cancel = threading.Event() + def encode(**kwargs): + calls.append(kwargs['artifact_name']) + if len(calls) == 6: # stop during the first group's SECOND measured attempt + cancel.set() # (its outcome is discarded; five attempts stay journaled) + main.check_measurement_budget() + artifact = Path(kwargs['out_dir']) / kwargs['artifact_name'] + artifact.write_bytes(b'encoded') + return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', + 'fileSizeBytes': 7, 'encodeStartMonotonicNs': 1_000_000_000, + 'encodeEndMonotonicNs': 2_000_000_000, 'elapsedMs': 1000, 'error': None} + hardware = main.HardwareInfo('CPU', None, 16, 'OS') + with mock.patch.object(main, 'detect_hardware', return_value=hardware), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, '_build_protocol_config', + return_value=protocol.ProtocolConfig.for_version('7.1', max_adaptive_repeats=0)), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(main, '_probe_artifact_contract', side_effect=lambda path: fixture._artifact_contract()), \ + mock.patch.object(main, '_capture_protocol_environment_snapshot', + return_value=protocol.EnvironmentSnapshot(selected_accelerator='software')), \ + mock.patch.object(main, 'encode_to_artifact', side_effect=encode), \ + mock.patch.object(main, 'BatchRunDashboard', _DummyDashboard): + # Controlled stop: every attempt up to the pause is journalized, but the + # submit loop refuses before materializing any envelope. + assert main.run_benchmark_batch(hardware=hardware, base_url='https://example.invalid', + args=args, cancel_event=cancel, + tasks=[{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_a}, + {'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_b}]) == 130 + root = next((tmp_path / 'campaigns').iterdir()) + measured = {} + for path in sorted(root.glob('attempt-*.json')): + data = json.loads(path.read_text()) + schedule = data.get('schedule', {}) + if schedule.get('phase') == 'measured': + measured.setdefault(schedule['recipe_id'], []).append(schedule['execution_order']) + complete = [rid for rid, orders in measured.items() if len(orders) >= 2] + partial = [rid for rid, orders in measured.items() if len(orders) == 1] + assert complete and partial, f"fixture must yield one complete and one partial group: {measured}" + assert not list(root.glob('submission-*.json')), "stop happened before envelope creation" + # A damaged unrelated sibling cannot hide or block the independently + # complete group that still has faithful measured records and bytes. + (root / 'attempt-999999.json').write_text('{damaged sibling') + sent = [] + def transport(base_url, submission, **kwargs): + sent.append(submission) + return {'benchmarkRun': {'id': f'run-{len(sent)}'}} + source = mock.Mock(side_effect=AssertionError('original source must never be re-fetched')) + encode_after = mock.Mock(side_effect=AssertionError('publish must never encode')) + with mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, '_prepare_named_suite_clip', source), \ + mock.patch.object(main, 'encode_to_artifact', encode_after), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=transport): + assert main.main(['prog', '--publish-saved', root.name, + '--queue-dir', str(tmp_path), '--base-url', 'https://example.invalid']) == 0 + # Rebuilt payloads carry the identical group identity the live path emits. + assert len(sent) == len(measured[complete[0]]), \ + "the complete group publishes every counted attempt, once" + for submission in sent: + run_create = submission['runCreate'] + assert run_create['measurementGroup']['repetitionGroupId'] == f"{root.name}:{complete[0]}" + assert run_create['measurementGroup']['completed'] is True + source.assert_not_called() + encode_after.assert_not_called() + accepted_orders = {path.stem.replace('submission-', '').replace('.accepted', '') + for path in root.glob('submission-*.accepted.json')} + for order in measured[complete[0]]: + assert f"{order:06d}" in accepted_orders # honest server receipts + for order in measured[partial[0]]: + assert not (root / f'submission-{order:06d}.json').exists() + assert not (root / f'submission-{order:06d}.accepted.json').exists() + assert spool.count_pending_entries(str(tmp_path)) == 0 + # Accepted bytes retired; the unfinished group's bytes remain for resume. + assert (root / 'campaign-complete.json').exists() is False + remaining = [path for path in root.glob('*.mp4') if path.read_bytes() == b'encoded'] + assert len(remaining) == 1, measured + # Second Publish saved is a no-op: accepted receipts authorize, nothing re-uploads. + with mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=AssertionError('replay must not re-upload')): + assert main.main(['prog', '--publish-saved', root.name, + '--queue-dir', str(tmp_path), '--base-url', 'https://example.invalid']) == 0 + def test_cancellation_stops_only_owned_process(): import subprocess import sys diff --git a/client/tests/test_close_deadline.py b/client/tests/test_close_deadline.py new file mode 100644 index 00000000..ab78b565 --- /dev/null +++ b/client/tests/test_close_deadline.py @@ -0,0 +1,27 @@ +"""A confirmed Windows Close has one finite grace period.""" + +import time +from unittest import mock + +import pytest + +from client import windows_gui as gui +from test_progress_contract import _build_gui_app +from test_windows_gui import FakeThread + + +def test_close_grace_does_not_renew_forever_with_live_owned_worker(): + app = _build_gui_app() + root = app.root + uploader = FakeThread() + uploader.start() + app.upload_thread = uploader + app._close_deadline = time.monotonic() - 1 + with mock.patch.object(gui, "_terminate_owned_children", create=True) as terminate, \ + mock.patch.object(gui.os, "_exit", side_effect=SystemExit(130)) as exit_process: + with pytest.raises(SystemExit): + app._close_when_stopped() + terminate.assert_called_once() + exit_process.assert_called_once_with(130) + root.destroy.assert_called_once() + assert app._close_deadline < time.monotonic() diff --git a/client/tests/test_encoding_regressions.py b/client/tests/test_encoding_regressions.py index 6ac58cd9..2f1d2f81 100644 --- a/client/tests/test_encoding_regressions.py +++ b/client/tests/test_encoding_regressions.py @@ -18,7 +18,7 @@ def setUp(self) -> None: def test_corrected_metrics_use_distinguishable_client_version(self) -> None: from client import main as client_main - self.assertEqual(client_main.CLIENT_VERSION, "client/0.3.3") + self.assertEqual(client_main.CLIENT_VERSION, "client/0.3.8") self.assertEqual(client_main.PROTOCOL_MINIMUM_CLIENT_VERSION, "client/0.3.0") def test_vmaf_passes_distorted_input_before_reference(self) -> None: diff --git a/client/tests/test_live_upload_cancellation.py b/client/tests/test_live_upload_cancellation.py new file mode 100644 index 00000000..4eee66eb --- /dev/null +++ b/client/tests/test_live_upload_cancellation.py @@ -0,0 +1,53 @@ +"""Normal collection must use the same cancellation as manual Retry.""" + +import threading +from unittest import mock + +from client import main, spool +from client.network import SubmissionCancelled + + +def test_live_batch_submission_passes_its_stop_event_to_network_replay(tmp_path): + stop = threading.Event() + with mock.patch.object(main, "spool_payload", return_value=("saved", {})), \ + mock.patch.object(main, "submit_spooled_path", return_value=("retained", "cancelled")) as send, \ + mock.patch.object(main, "count_pending_entries", return_value=1): + main._submit_payload_with_spool( + queue_dir=str(tmp_path), base_url="http://127.0.0.1:9", + payload={"runCreate": {"payloadHash": "a" * 64}}, api_key="", + retries=1, use_token=False, cancel_event=stop, + ) + assert send.call_args.kwargs["cancel_event"] is stop + + +def test_checkpoint_queue_replay_passes_the_collection_stop_event(tmp_path): + stop = threading.Event() + with mock.patch.object(main, "replay_spool", return_value=spool.ReplayStats()) as replay, \ + mock.patch.object(main, "count_pending_entries", return_value=0): + main._replay_pending_uploads( + queue_dir=str(tmp_path), base_url="http://127.0.0.1:9", + api_key="", retries=1, use_token=False, cancel_event=stop, + ) + assert replay.call_args.kwargs["cancel_event"] is stop + + +def test_real_batch_upload_uses_the_gui_stop_event_and_retains_queue(tmp_path): + from test_progress_contract import _run_batch + + stop = threading.Event() + observed = [] + + def cancelled_create(_base_url, _payload, **kwargs): + observed.append(kwargs.get("cancel_event") is stop) + observed.append(isinstance(kwargs.get("deadline"), float)) + stop.set() + raise SubmissionCancelled("run create") + + events = [] + rc = _run_batch(tmp_path, seed=712, events=events, + transport=cancelled_create, no_submit=False, + cancel_event=stop) + assert rc == 130 + assert observed == [True, True] + assert spool.count_pending_entries(str(tmp_path)) >= 1 + assert any(event.get("type") == "run_interrupted" for event in events) diff --git a/client/tests/test_main_routing.py b/client/tests/test_main_routing.py index 42d63c44..b50c91c4 100644 --- a/client/tests/test_main_routing.py +++ b/client/tests/test_main_routing.py @@ -2,6 +2,7 @@ import itertools import json import os +import signal import tempfile import unittest from contextlib import ExitStack @@ -38,6 +39,8 @@ def update_machine_metrics(self, *args, **kwargs) -> None: def update_counters(self, *args, **kwargs) -> None: self.calls.append(("update_counters", args, kwargs)) + def set_progress(self, *args, **kwargs) -> None: + self.calls.append(("set_progress", args, kwargs)) def advance_phase(self, *args, **kwargs) -> None: self.calls.append(("advance_phase", args, kwargs)) @@ -219,6 +222,60 @@ def test_queue_status_command_exits_without_running_benchmark(self) -> None: self.assertEqual(rc, 0) run_mock.assert_not_called() + def test_saved_publish_sigint_cancels_owned_upload_and_returns_130(self) -> None: + with tempfile.TemporaryDirectory() as queue_dir: + previous = signal.getsignal(signal.SIGINT) + + def publish(**kwargs): + cancel = kwargs.get("cancel_event") + self.assertIsNotNone(cancel) + handler = signal.getsignal(signal.SIGINT) + self.assertIsNot(handler, previous) + handler(signal.SIGINT, None) + self.assertTrue(cancel.is_set()) + return 10, {"status": "cancelled", "pending": 1} + + with mock.patch.object(client_main, "check_compatibility"), \ + mock.patch.object(client_main, "publish_saved_campaign", side_effect=publish): + rc = client_main.main(["prog", "--cli", "--publish-saved", "campaign-saved", + "--queue-dir", queue_dir]) + + self.assertEqual(rc, 130) + self.assertIs(signal.getsignal(signal.SIGINT), previous) + + def test_retry_sigint_cancels_owned_upload_and_returns_130(self) -> None: + with tempfile.TemporaryDirectory() as queue_dir: + previous = signal.getsignal(signal.SIGINT) + + def retry(**kwargs): + cancel = kwargs.get("cancel_event") + self.assertIsNotNone(cancel) + handler = signal.getsignal(signal.SIGINT) + self.assertIsNot(handler, previous) + handler(signal.SIGINT, None) + self.assertTrue(cancel.is_set()) + return 10, {"status": "pending", "cancelled": 1} + + with mock.patch.object(client_main, "check_compatibility"), \ + mock.patch.object(client_main, "retry_due_uploads", side_effect=retry): + rc = client_main.main(["prog", "--cli", "--upload-only", + "--queue-dir", queue_dir]) + + self.assertEqual(rc, 130) + self.assertIs(signal.getsignal(signal.SIGINT), previous) + + def test_upload_interrupt_during_compatibility_returns_130(self) -> None: + with tempfile.TemporaryDirectory() as queue_dir: + previous = signal.getsignal(signal.SIGINT) + with mock.patch.object(client_main, "check_compatibility", + side_effect=KeyboardInterrupt), \ + mock.patch.object(client_main, "retry_due_uploads") as retry: + rc = client_main.main(["prog", "--cli", "--upload-only", + "--queue-dir", queue_dir]) + self.assertEqual(rc, 130) + self.assertIs(signal.getsignal(signal.SIGINT), previous) + retry.assert_not_called() + def test_authoritative_run_create_carries_canonical_energy_and_decode_evidence(self) -> None: record = BenchmarkRunRecord( schedule=ScheduledRun("campaign-a", "recipe-a", "measured", 0, 0), @@ -621,7 +678,10 @@ def test_run_benchmark_batch_submits_authoritative_artifact_bundle(self) -> None "recipeFingerprint": "f" * 64, } - def fake_submit(*, queue_dir, base_url, payload, api_key, retries, use_token, max_storage_mb): + def fake_submit(*, queue_dir, base_url, payload, api_key, retries, use_token, + max_storage_mb, cancel_event=None, deadline=None): + self.assertIsNone(cancel_event) + self.assertIsInstance(deadline, float) captured_submission.update(payload) return "submitted", "", 0 @@ -988,7 +1048,7 @@ def fake_batch(**kwargs): def test_sweep_checkpoint_without_progress_stops(self) -> None: calls = [] - client_main.config._BATCH_COMPLETED_COUNT = 0 + client_main.config._BATCH_LEDGER = None def fake_batch(**kwargs): calls.append(kwargs) @@ -1003,7 +1063,7 @@ def fake_batch(**kwargs): rc = client_main.run_sweep_mode(mode="small", base_args=self.args(queue_dir=queue_dir), show_end_screen=False, interactive=False, presets_cfg={}) finally: - client_main.config._BATCH_COMPLETED_COUNT = 0 + client_main.config._BATCH_LEDGER = None self.assertEqual(rc, 11) self.assertEqual(len(calls), 1, "no-progress checkpoint must not spin") diff --git a/client/tests/test_measurement_budget.py b/client/tests/test_measurement_budget.py index d3fa49bb..4bd9eb29 100644 --- a/client/tests/test_measurement_budget.py +++ b/client/tests/test_measurement_budget.py @@ -134,7 +134,10 @@ def encode(**kwargs): return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', 'fileSizeBytes': 7, 'encodeStartMonotonicNs': 1_000_000_000, 'encodeEndMonotonicNs': 2_000_000_000, 'elapsedMs': 1000, 'error': None} - def submit(queue_dir, *, base_url, payload, max_storage_mb, api_key, retries, use_token): + def submit(queue_dir, *, base_url, payload, max_storage_mb, api_key, retries, + use_token, cancel_event=None, deadline=None): + assert cancel_event is None + assert isinstance(deadline, float) submissions.append(payload) return "submitted", "run-checkpoint-test", 0 hardware = main.HardwareInfo('CPU', None, 16, 'OS') @@ -222,3 +225,87 @@ def submit(queue_dir, *, base_url, payload, max_storage_mb, api_key, retries, us assert entry['artifactSha256'] != '' and len(entry['artifactSha256']) == 64 assert not Path(entry['artifactPath']).exists() # all accepted bytes retired assert (root / 'campaign-complete.json').exists() + + +def test_batch_counters_reconcile_with_spool_and_journal(tmp_path): + # C04/C05: the end-screen counters (submitted/queued/failed) and the global + # task_complete progress must reconcile against real durable evidence: + # accepted receipts in the journal, one retryable pending queue entry, and + # one permanently rejected dead-letter - with safe, distinct errorCategory + # values and never a raw server body. + from dataclasses import replace + from test_main_routing import MainRoutingTests, _DummyDashboard + from client import spool + from client.network import SubmitError + fixture = MainRoutingTests() + clip_a = fixture._quick_clip() + clip_b = replace(clip_a, clip_id="film-grain-1080p24-final", workload_id="film-grain-1080p24-final") + args = fixture._batch_args(str(tmp_path), no_submit=False) + args.local_metrics = False + args.campaign_seed = 29 + args.max_duration_minutes = 1.0 + def encode(**kwargs): + artifact = Path(kwargs['out_dir']) / kwargs['artifact_name'] + artifact.write_bytes(b'encoded') + return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', 'fileSizeBytes': 7, + 'encodeStartMonotonicNs': 1_000_000_000, 'encodeEndMonotonicNs': 2_000_000_000, + 'elapsedMs': 1000, 'error': None} + def transport(base_url, submission, **kwargs): + run_create = submission['runCreate'] + group = str(run_create['repetitionGroupId']) + index = int(run_create['repetitionIndex']) + if 'athletic' in group: # first group: both attempts accepted + return {'benchmarkRun': {'id': 'run-counters-ok'}} + if index == 1: # transient upstream outage: durable queue must defer, not fail + raise SubmitError('submit failed (503)', retryable=True) + raise SubmitError('server rejected the evidence (400)', retryable=False) + events = [] + hardware = main.HardwareInfo('CPU', None, 16, 'OS') + with mock.patch.object(main, 'detect_hardware', return_value=hardware), \ + mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, 'fetch_baseline_rows', return_value=[]), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=transport), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, '_build_protocol_config', + return_value=protocol.ProtocolConfig.for_version('7.1', max_adaptive_repeats=0)), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(main, '_probe_artifact_contract', side_effect=lambda path: fixture._artifact_contract()), \ + mock.patch.object(main, '_capture_protocol_environment_snapshot', + return_value=protocol.EnvironmentSnapshot(selected_accelerator='software')), \ + mock.patch.object(main, 'encode_to_artifact', side_effect=encode), \ + mock.patch.object(main, 'BatchRunDashboard', _DummyDashboard): + # One accepted pair, one 503-deferred upload, one terminal 400. + assert main.run_benchmark_batch(hardware=hardware, base_url='https://example.invalid', args=args, + event_sink=events.append, + tasks=[{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_a}, + {'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_b}]) == 1 + counters = next(e for e in reversed(events) if e.get('type') == 'counters') + assert (counters['submitted'], counters['skipped'], counters['queued'], counters['failed']) == (2, 0, 1, 1) + complete = next(e for e in events if e.get('type') == 'run_complete') + assert (complete['submitted'], complete['queued'], complete['failed']) == (2, 1, 1) + assert complete['locallyComplete'] == 0 + # Global progress: every measured unit of work advanced the batch, uploads included. + assert [e['processed'] for e in events if e.get('type') == 'task_complete' and e.get('scope') == 'batch'] == [1, 2, 3, 4] + outcomes = {e['status']: e for e in events if e.get('type') == 'submit_result'} + assert outcomes['queued']['errorCategory'] == 'server_error' + assert outcomes['failed']['errorCategory'] == 'protocol_rejected' + assert 'benchmarkRunId' in outcomes['submitted'] + # Durable truth: journal accepted receipts, queue pending entry, dead-letter. + root = next((tmp_path / 'campaigns').iterdir()) + accepted = [json.loads(p.read_text()) for p in sorted(root.glob('submission-*.accepted.json'))] + assert len(accepted) == 2 + assert all(entry['benchmarkRunId'] == 'run-counters-ok' for entry in accepted) + assert all(not Path(entry['artifactPath']).exists() for entry in accepted) + pending = [json.loads(p.read_text()) for p in tmp_path.glob('*.json') + if p.name != 'queue' and json.loads(p.read_text()).get('payload')] + assert len(pending) == 1 + assert 'film-grain' in pending[0]['payload']['runCreate']['repetitionGroupId'] + assert int(pending[0]['payload']['runCreate']['repetitionIndex']) == 1 + dead = [json.loads(p.read_text()) for p in (tmp_path / 'dead-letter').glob('*.json')] + assert len(dead) == 1 + assert int(dead[0]['payload']['runCreate']['repetitionIndex']) == 2 + receipts = [json.loads(p.read_text()) for p in (tmp_path / 'receipts').glob('*.json')] + assert len(receipts) == 2 + assert {r['status'] for r in receipts} == {'uploaded_analysis_pending'} + assert 'run-counters-ok' in json.dumps(receipts) diff --git a/client/tests/test_preparation.py b/client/tests/test_preparation.py index 822fc084..e5a258dd 100644 --- a/client/tests/test_preparation.py +++ b/client/tests/test_preparation.py @@ -76,7 +76,7 @@ def test_explicit_upload_only_with_resume_and_submit_never_falls_through_to_enco mock.patch.object(main, '_prepare_quick_suite_clip') as prepare, \ mock.patch.object(main, '_preparation_runtime_integrity') as runtime: self.assertEqual(main.main(['prog', '--upload-only', '--resume-campaign', 'campaign-0123456789abcdef', - '--submit', '--queue-dir', directory]), 0) + '--submit', '--queue-dir', directory]), 1) batch.assert_not_called() prepare.assert_not_called() runtime.assert_not_called() @@ -86,7 +86,7 @@ def test_menu_exit_does_not_prepare_sources_or_probe_hardware(self) -> None: with mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ mock.patch.object(main, 'detect_hardware', return_value=hardware), \ mock.patch.object(main, 'list_all_available_encoders', return_value=['libx264']), \ - mock.patch.object(main, 'prompt_choice', return_value=5), \ + mock.patch.object(main, 'prompt_choice', return_value=6), \ mock.patch.object(main.sys, 'stdin', None), \ mock.patch.object(main, '_prepare_quick_suite_clip') as prepare, \ mock.patch.object(main, 'is_hardware_encoder_usable') as probe: diff --git a/client/tests/test_preparation_integration.py b/client/tests/test_preparation_integration.py new file mode 100644 index 00000000..7449ada9 --- /dev/null +++ b/client/tests/test_preparation_integration.py @@ -0,0 +1,89 @@ +"""Real preparation entry points expose bounded, actionable stages.""" + +import time +import subprocess +import io +from types import SimpleNamespace +from unittest import mock + +import pytest + +from client import campaign, encoders, main + + +def test_source_acquisition_has_its_own_finite_stage_budget(): + clip = SimpleNamespace(clip_id="frozen-clip") + manifest = SimpleNamespace(clips=[clip]) + with mock.patch.object(main, "load_default_suite_manifest", return_value=manifest), \ + mock.patch.object(main, "get_default_quick_clip", return_value=clip), \ + mock.patch.object(main, "ensure_suite_clip", side_effect=lambda _clip: time.sleep(0.08)), \ + mock.patch.object(main, "SOURCE_CLIP_BUDGET_SECONDS", 0.05, create=True): + with campaign.PreparationScope(heartbeat_path=None).activate(): + with pytest.raises(campaign.PreparationBudgetExceeded): + main._prepare_quick_suite_clip() + + +def test_preparation_budget_reports_visible_error_and_nonzero_exit(): + events = [] + + @main._preparation_operation + def prepare(*, event_sink=None): + with campaign.preparation_stage("test-runtime", 0.05): + time.sleep(0.08) + + assert prepare(event_sink=events.append) == 2 + assert any(event.get("type") == "run_error" and + "test-runtime" in str(event.get("message")) for event in events) + + +def test_source_contract_has_independent_finite_stage_budget(): + from test_main_routing import MainRoutingTests + + clip = MainRoutingTests()._quick_clip() + main._SOURCE_CONTRACT_CACHE.clear() + with mock.patch.object(main, "SOURCE_CONTRACT_BUDGET_SECONDS", 0.05), \ + mock.patch.object(main, "_probe_artifact_contract", side_effect=lambda _path: time.sleep(0.08)): + with campaign.PreparationScope(heartbeat_path=None).activate(): + with pytest.raises(campaign.PreparationBudgetExceeded, match="source-contract"): + main._build_protocol_recipe_specs( + [{"encoder": "libx264", "preset": "fast", "crf": 24, + "suiteClip": clip}], + default_input_path=clip.path, default_input_hash=clip.input_hash) + + +@pytest.mark.parametrize("operation,stage", [ + (lambda: encoders.ensure_ffmpeg_and_ffprobe(), "encoder-version-probe"), + (lambda: encoders._get_encoder_set(), "encoder-list-probe"), + (lambda: encoders.has_libvmaf(), "filter-list-probe"), +]) +def test_encoder_discovery_timeout_is_actionable(operation, stage): + encoders._ENCODER_LIST_CACHE = None + with mock.patch.object(encoders, "run_measurement_process", + side_effect=subprocess.TimeoutExpired("ffmpeg", 30)): + with campaign.PreparationScope(heartbeat_path=None).activate(): + with pytest.raises(campaign.PreparationBudgetExceeded, match=stage): + operation() + + +def test_cli_queue_and_local_policy_keep_guided_menu(tmp_path): + with mock.patch.object(main, "interactive_menu_flow", return_value=37) as guided, \ + mock.patch.object(main, "run_with_args", side_effect=AssertionError("single flow")): + assert main.main(["encodingdb", "--cli", "--no-submit", + "--queue-dir", str(tmp_path)]) == 37 + guided.assert_called_once() + assert main._has_direct_single_run_intent(["--codec", "libx264", "--no-submit"]) + + +def test_cli_entrypoint_survives_non_ascii_output_on_windows_code_page(tmp_path): + bytes_out = io.BytesIO() + output = io.TextIOWrapper(bytes_out, encoding="cp1252", errors="strict") + + def guided(_parser, _args): + main.print_info("about 123 MB; source ≈ estimate") + return 0 + + with mock.patch.object(main.sys, "stdout", output), \ + mock.patch.object(main, "interactive_menu_flow", side_effect=guided): + assert main.main(["encodingdb", "--menu", "--queue-dir", str(tmp_path)]) == 0 + output.flush() + assert b"source \\u2248 estimate" in bytes_out.getvalue() diff --git a/client/tests/test_progress_contract.py b/client/tests/test_progress_contract.py new file mode 100644 index 00000000..53a3a643 --- /dev/null +++ b/client/tests/test_progress_contract.py @@ -0,0 +1,322 @@ +"""R06/R07: the real run_benchmark_batch producer and the real GUI consumer. + +Every assertion here compares the GUI's observable bar state against durable +attempts written by the producer, while publication and receipt state stay in +the separate ledger. A mock that agreed with itself could not catch a unit +mismatch between producer and consumer. +""" +import argparse +import json +import sys +import threading +from dataclasses import replace +from pathlib import Path +from unittest import mock + +from client import main, protocol, spool +from client import windows_gui as gui +from client.network import SubmitError + + +def _build_gui_app(): + from test_windows_gui import Variable, Widget, cached_estimate, cached_recovery_state + root = mock.MagicMock() + bindings = {} + root.bind.side_effect = lambda key, callback: bindings.__setitem__(key, callback) + tk = mock.MagicMock() + tk.Tk.return_value = root + tk.StringVar = tk.IntVar = tk.BooleanVar = Variable + for name in ("Frame", "LabelFrame", "Label", "Combobox", "Checkbutton", "Spinbox", + "Entry", "Button", "Progressbar"): + setattr(tk.ttk, name, Widget) + tk.scrolledtext.ScrolledText = Widget + base_args = argparse.Namespace(codec="libx264", presets="fast", crf=0, no_submit=True, + queue_dir="test-only", base_url="http://127.0.0.1:9") + with mock.patch.dict(sys.modules, {"tkinter": tk}), mock.patch.object(gui.os, "name", "nt"), \ + mock.patch.object(gui, "list_all_available_encoders", return_value=["libx264"]), \ + mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state), \ + mock.patch.object(gui.threading, "Thread"): + gui.launch_windows_gui(base_args) + app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state + return app + + +def _drive_gui(app, events): + """Feed every producer event through the real consumer handler.""" + for event in events: + app._handle_event(event) + return app + + +def _encode(**kwargs): + artifact = Path(kwargs['out_dir']) / kwargs['artifact_name'] + artifact.write_bytes(b'encoded') + return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', + 'fileSizeBytes': 7, 'encodeStartMonotonicNs': 1_000_000_000, + 'encodeEndMonotonicNs': 2_000_000_000, 'elapsedMs': 1000, 'error': None} + + +def _producer_patches(fixture, tmp_path, *, transport=None): + from test_main_routing import _DummyDashboard + stack = [ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), + mock.patch.object(main, '_build_protocol_config', + return_value=protocol.ProtocolConfig.for_version('7.1', max_adaptive_repeats=0)), + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), + mock.patch.object(main, '_probe_artifact_contract', + side_effect=lambda path: fixture._artifact_contract()), + mock.patch.object(main, '_capture_protocol_environment_snapshot', + return_value=protocol.EnvironmentSnapshot(selected_accelerator='software')), + mock.patch.object(main, 'encode_to_artifact', side_effect=_encode), + mock.patch.object(main, 'BatchRunDashboard', _DummyDashboard), + ] + if transport is not None: + stack.append(mock.patch.object(spool, 'submit_artifact_submission', side_effect=transport)) + stack.append(mock.patch.object(main, 'fetch_baseline_rows', return_value=[])) + return stack + + +def _run_batch(tmp_path, *, seed, events, transport=None, tasks=None, no_submit=True, + cancel_event=None, encode_side_effect=None): + from test_main_routing import MainRoutingTests + fixture = MainRoutingTests() + clip = fixture._quick_clip() + args = fixture._batch_args(str(tmp_path), no_submit=no_submit) + args.local_metrics = False + args.campaign_seed = seed + args.max_duration_minutes = 60 + hardware = main.HardwareInfo('CPU', None, 16, 'OS') + if tasks is None: + tasks = [{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip}] + with mock.patch.object(main, 'detect_hardware', return_value=hardware), \ + mock.patch.object(main, 'check_compatibility', return_value={}): + stack = _producer_patches(fixture, tmp_path, transport=transport) + if encode_side_effect is not None: + stack = [p for p in stack if p.attribute != 'encode_to_artifact'] + stack.append(mock.patch.object(main, 'encode_to_artifact', side_effect=encode_side_effect)) + for patcher in stack: + patcher.start() + try: + rc = main.run_benchmark_batch( + hardware=hardware, base_url='https://example.invalid', args=args, + event_sink=events.append, cancel_event=cancel_event, tasks=tasks) + finally: + for patcher in stack: + patcher.stop() + return rc + + +def _campaign_root(tmp_path): + return next((tmp_path / 'campaigns').iterdir()) + + +def _journal_for(tmp_path, root): + return main.CampaignJournal(str(tmp_path), root.name, + json.loads((root / 'manifest.json').read_text()), 2048) + + +def test_local_only_success_bars_match_durable_saves(tmp_path): + events = [] + rc = _run_batch(tmp_path, seed=41, events=events) + assert rc == 0 + root = _campaign_root(tmp_path) + saved = list(root.glob('submission-*.json')) + attempts = list(root.glob('attempt-*.json')) + progress = [e for e in events if e.get('type') == 'campaign_progress'] + assert progress, "producer must emit campaign_progress" + final = progress[-1] + # Progress counts every durable encode attempt. Publication is a separate + # ledger, so the two measured envelopes must not drive the bars. + assert len(attempts) == 3 # 1 warmup + 2 measured + assert len(saved) == 2 + assert final['total'] == len(attempts) == 3 + assert final['done'] == final['total'] + assert final['batchDone'] == final['batchTotal'] == 3 + assert final['unit'] == 'durable-attempt' + # Monotonic: both fractions never decrease across the whole event stream. + overall = [(e['done'], e['total']) for e in progress] + batch = [(e['batchDone'], e['batchTotal']) for e in progress] + assert all(a[0] / a[1] <= b[0] / b[1] + 1e-9 for a, b in zip(overall, overall[1:])) + assert all(a[0] / a[1] <= b[0] / b[1] + 1e-9 for a, b in zip(batch, batch[1:])) + # The real GUI consumer ends at 100% with matching options. + app = _drive_gui(_build_gui_app(), events) + assert app.overall_pb.options['maximum'] == 3 + assert app.overall_pb.options['value'] == 3 + assert app.batch_pb.options['value'] == 3 + # task_complete after the final progress must not push past the maximum. + app._handle_event({'type': 'task_complete', 'scope': 'batch', 'processed': 99, 'total': 9}) + assert app.overall_pb.options['value'] == 3 + # Durable ledger for the end screen: saved, nothing uploaded. + ledger = main._durable_campaign_ledger(str(tmp_path), root.name, _journal_for(tmp_path, root), True) + assert ledger['savedLocal'] == 2 + assert ledger['uploaded'] == 0 + assert ledger['queued'] == 0 + + +def test_warmup_and_measured_attempts_advance_bars_before_publication(tmp_path, monkeypatch): + monkeypatch.setenv('ENCODINGDB_HOST_PHASE_DIR', str(tmp_path / 'host-phase')) + events = [] + assert _run_batch(tmp_path, seed=405, events=events) == 0 + app = _build_gui_app() + progress_before_next_encode = {} + for event in events: + app._handle_event(event) + if event.get('type') == 'encode_start': + progress_before_next_encode[event['index']] = ( + app.overall_pb.options['value'], app.batch_pb.options['value']) + # The first warmup and first measured attempt are durably recorded before + # the following encode starts; a publication-only bar stayed at (0, 0). + assert progress_before_next_encode[2] == (1, 1) + assert progress_before_next_encode[3] == (2, 2) + assert app.overall_pb.options['maximum'] == 3 + assert app.overall_pb.options['value'] == 3 + assert app.batch_pb.options['maximum'] == 3 + assert app.batch_pb.options['value'] == 3 + + +def test_local_only_resume_baseline_never_rewinds_or_double_counts(tmp_path): + first = [] + assert _run_batch(tmp_path, seed=42, events=first) == 0 + root = _campaign_root(tmp_path) + resume_events = [] + assert _run_batch(tmp_path, seed=42, events=resume_events) == 0 + recorded = len(list(root.glob('attempt-*.json'))) + progress = [e for e in resume_events if e.get('type') == 'campaign_progress'] + assert progress[0]['done'] == 3, "resume must start from the durable baseline" + assert progress[-1]['done'] == progress[-1]['total'] == recorded == 3 + # GUI consumer: overall starts at the durable baseline and never drops. + values = [] + app = _build_gui_app() + for event in resume_events: + app._handle_event(event) + if event.get('type') in ('run_start', 'campaign_progress'): + values.append(app.overall_pb.options['value']) + assert values and values[0] == 3 + assert all(a <= b for a, b in zip(values, values[1:])) + + +def test_interrupted_resume_keeps_overall_baseline_and_restarts_batch(tmp_path, monkeypatch): + monkeypatch.setenv('ENCODINGDB_HOST_PHASE_DIR', str(tmp_path / 'host-phase')) + cancel = threading.Event() + + class StopAfterFirstSave(list): + def append(self, event): + super().append(event) + if event.get('type') == 'campaign_progress' and event.get('done') == 1: + cancel.set() + + first = StopAfterFirstSave() + assert _run_batch(tmp_path, seed=404, events=first, cancel_event=cancel) == 130 + root = _campaign_root(tmp_path) + assert len(list(root.glob('attempt-*.json'))) == 1 + + resumed = [] + assert _run_batch(tmp_path, seed=404, events=resumed) == 0 + progress = [event for event in resumed if event.get('type') == 'campaign_progress'] + assert (progress[0]['done'], progress[0]['batchDone']) == (1, 0) + assert any((event['done'], event['batchDone']) == (2, 1) for event in progress) + assert (progress[-1]['done'], progress[-1]['total']) == (3, 3) + assert (progress[-1]['batchDone'], progress[-1]['batchTotal']) == (2, 2) + app = _drive_gui(_build_gui_app(), resumed) + assert app.overall_pb.options['value'] == 3 + assert app.batch_pb.options['value'] == 2 + + +def test_partial_upload_run_measurement_bars_finish_but_publication_does_not(tmp_path): + from test_main_routing import MainRoutingTests + fixture = MainRoutingTests() + clip_a = fixture._quick_clip() + clip_b = replace(clip_a, clip_id="film-grain-1080p24-final", + workload_id="film-grain-1080p24-final") + + def transport(base_url, submission, **kwargs): + run_create = submission['runCreate'] + group = str(run_create['repetitionGroupId']) + index = int(run_create['repetitionIndex']) + if 'athletic' in group: + return {'benchmarkRun': {'id': 'run-progress-ok'}} + if index == 1: + raise SubmitError('submit failed (503)', retryable=True) + raise SubmitError('server rejected the evidence (400)', retryable=False) + + events = [] + rc = _run_batch(tmp_path, seed=29, events=events, transport=transport, no_submit=False, + tasks=[{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_a}, + {'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_b}]) + assert rc == 1 + root = _campaign_root(tmp_path) + accepted = list(root.glob('submission-*.accepted.json')) + assert len(accepted) == 2 # only the athletic group confirmed + progress = [e for e in events if e.get('type') == 'campaign_progress'] + final = progress[-1] + # Six warmup/measured attempts are recorded even though only two uploads + # are acknowledged. The GUI separates measurement progress from delivery. + assert final['total'] == final['done'] == 6 + app = _drive_gui(_build_gui_app(), events) + assert app.overall_pb.options['maximum'] == 6 + assert app.overall_pb.options['value'] == 6 + assert app.batch_pb.options['value'] == app.batch_pb.options['maximum'] == 6 + assert app.stage_var.get() == 'Finished with issues' + overall = [(e['done'], e['total']) for e in progress] + assert all(a[0] / a[1] <= b[0] / b[1] + 1e-9 for a, b in zip(overall, overall[1:])) + ledger = main._durable_campaign_ledger(str(tmp_path), root.name, _journal_for(tmp_path, root), False) + assert ledger['uploaded'] == 2 + assert ledger['queued'] == 1 + assert ledger['terminalFailures'] == 1 + + +def test_terminal_cancellation_bars_match_durable_attempts(tmp_path): + # The GUI Stop path: cancel_event set mid-run; the producer exits 130 and + # the bars must reflect only durably saved work. + calls = [] + + def encode_then_cancel(**kwargs): + result = _encode(**kwargs) + calls.append(kwargs['artifact_name']) + if kwargs['artifact_name'].endswith('measured-r1.mp4'): + cancel.set() + return result + + cancel = threading.Event() + events = [] + rc = _run_batch(tmp_path, seed=43, events=events, cancel_event=cancel, + encode_side_effect=encode_then_cancel) + assert rc == 130 + assert events[-1]['type'] == 'run_interrupted' + progress = [e for e in events if e.get('type') == 'campaign_progress'] + final = progress[-1] + root = _campaign_root(tmp_path) + recorded = len(list(root.glob('attempt-*.json'))) + # The first warmup is durable; the interrupted measured output is not. + assert final['done'] == recorded == 1 + assert final['warmupsDone'] == 1 + assert final['measuredDone'] == 0 + app = _drive_gui(_build_gui_app(), events) + assert app.overall_pb.options['value'] == recorded + if recorded < final['total']: + assert app.overall_pb.options['value'] < app.overall_pb.options['maximum'] + + +def test_end_screen_ledger_separates_saved_from_confirmed(tmp_path): + # R06 end screen: the rendered text must separate saved / uploaded / + # queued / confirmed and never call saved work "submitted". + events = [] + assert _run_batch(tmp_path, seed=44, events=events) == 0 + root = _campaign_root(tmp_path) + ledger = main._durable_campaign_ledger(str(tmp_path), root.name, _journal_for(tmp_path, root), True) + assert ledger['savedLocal'] == 2 + assert ledger['uploaded'] == 0 + from client import ui + lines = [] + with mock.patch.object(ui, '_rich_tty', return_value=False), \ + mock.patch('builtins.print', side_effect=lambda *a, **k: lines.append(' '.join(str(x) for x in a))): + ui.print_end_screen(0, 12.0, status='complete', ledger=ledger) + text = '\n'.join(lines) + assert 'Saved locally: 2' in text + assert 'Server-confirmed: 0' in text + assert 'analysis pending' in text + assert 'Submitted' not in text diff --git a/client/tests/test_progress_review.py b/client/tests/test_progress_review.py new file mode 100644 index 00000000..071808d3 --- /dev/null +++ b/client/tests/test_progress_review.py @@ -0,0 +1,127 @@ +"""Independent GUI final-state and durable receipt checks.""" + +import dataclasses +import hashlib +import json +from unittest import mock + +from client import main, protocol, spool +from client.campaign import CampaignJournal, atomic_json +from test_progress_contract import (_build_gui_app, _campaign_root, _drive_gui, + _encode, _journal_for, _run_batch) + + +def _finish(app, rc): + app.event_queue.put(("done", rc)) + app._poll_events() + + +def test_final_exit_failure_cannot_leave_gui_stage_complete(): + app = _build_gui_app() + app._handle_event({"type": "run_complete", "scope": "batch", + "completed": 0, "failed": 0, "skipped": 0, "queued": 0}) + assert app.stage_var.get() == "Complete" + _finish(app, 1) + assert app.stage_var.get() == "Error" + assert "failed" in app.summary_var.get().lower() + + +def test_final_queue_exit_sets_pending_stage(): + app = _build_gui_app() + app._handle_event({"type": "run_complete", "scope": "batch", + "completed": 1, "failed": 0, "skipped": 0, "queued": 1}) + _finish(app, 10) + assert app.stage_var.get() == "Pending uploads" + assert "queued" in app.summary_var.get().lower() + + +def test_checkpoint_exit_sets_paused_stage(): + app = _build_gui_app() + app._handle_event({"type": "run_budget_exhausted", "scope": "batch", + "status": "budget_exhausted"}) + _finish(app, 11) + assert app.stage_var.get() == "Paused" + assert "campaign saved" in app.summary_var.get().lower() + + +def test_end_ledger_counts_queue_receipt_when_old_journal_marker_is_missing(tmp_path): + events = [] + assert _run_batch(tmp_path, seed=501, events=events) == 0 + root = _campaign_root(tmp_path) + for index, path in enumerate(sorted(root.glob("submission-*.json")), start=1): + payload = json.loads(path.read_text()) + local_hash = spool.local_hash_for_payload(payload) + atomic_json(tmp_path / "receipts" / f"{local_hash}.json", { + "localHash": local_hash, + "response": {"benchmarkRun": {"id": f"run-older-{index}"}}, + }) + ledger = main._durable_campaign_ledger( + str(tmp_path), root.name, _journal_for(tmp_path, root), False, + ) + assert ledger["uploaded"] == 2 + assert ledger["groupsConfirmed"] == 1 + assert ledger["queued"] == 0 + + +def test_adaptive_group_bars_wait_for_actual_terminal_attempt(tmp_path): + original_factory = protocol.ProtocolConfig.for_version + calls = [0] + + def variable_encode(**kwargs): + calls[0] += 1 + result = _encode(**kwargs) + result["encodeEndMonotonicNs"] = 1_000_000_000 + calls[0] * 1_000_000_000 + return result + + events = [] + with mock.patch.object(protocol.ProtocolConfig, "for_version", + side_effect=lambda version, **_kwargs: original_factory( + version, max_adaptive_repeats=2, + stability_threshold_ratio=0.03)): + assert _run_batch(tmp_path, seed=502, events=events, + encode_side_effect=variable_encode) == 0 + root = _campaign_root(tmp_path) + assert len(list(root.glob("attempt-*.json"))) == 5 # warmup + four measured + progress = [event for event in events if event.get("type") == "campaign_progress"] + assert progress[-1]["done"] == progress[-1]["total"] == 5 + assert any(0 < event["done"] < event["total"] for event in progress) + app = _drive_gui(_build_gui_app(), events) + assert app.overall_pb.options["maximum"] == 5 + assert app.overall_pb.options["value"] == 5 + assert app.batch_pb.options["maximum"] == 5 + assert app.batch_pb.options["value"] == 5 + + +def test_durable_ledger_never_calls_one_repetition_a_finished_group(tmp_path): + campaign_id = "campaign-0123456789abcdef" + root = tmp_path / "campaigns" / campaign_id + root.mkdir(parents=True) + config = protocol.ProtocolConfig.for_version("7.1", max_adaptive_repeats=2) + manifest = {"protocolVersion": "7.1", "seed": 1, + "protocolConfig": dataclasses.asdict(config), + "tasks": [{"clipId": "clip", "encoder": "libx264", "preset": "fast", "crf": 24}]} + artifact = root / "measured.mp4" + artifact.write_bytes(b"durable original") + digest = hashlib.sha256(artifact.read_bytes()).hexdigest() + record = protocol.BenchmarkRunRecord( + schedule=protocol.ScheduledRun(campaign_id, "clip|libx264|fast|24", "measured", 1, 1), + timing=protocol.EncodeTiming.from_measurement( + start_monotonic_ns=1, end_monotonic_ns=1_000_000_001, + source_frame_count=24, encoded_frame_count=24, source_fps=24), + metadata={"info": {"artifactPath": str(artifact), "artifactSha256": digest}}, + counted_for_stability=True, + ) + atomic_json(root / "manifest.json", manifest) + atomic_json(root / "attempt-000001.json", record.to_dict()) + atomic_json(root / "submission-000001.accepted.json", { + "schemaVersion": 1, "executionOrder": 1, + "recipeId": record.schedule.recipe_id, + "artifactPath": str(artifact), "artifactSha256": digest, + "benchmarkRunId": "run-confirmed", + }) + journal = CampaignJournal(str(tmp_path), campaign_id, manifest, 2048) + ledger = main._durable_campaign_ledger(str(tmp_path), campaign_id, journal, False) + assert ledger["uploaded"] == 1 + assert ledger["groupsTotal"] == 1 + assert ledger["groupsConfirmed"] == 0 + assert ledger["requiredMeasured"] == 2 diff --git a/client/tests/test_publication_result.py b/client/tests/test_publication_result.py new file mode 100644 index 00000000..6bf54e6a --- /dev/null +++ b/client/tests/test_publication_result.py @@ -0,0 +1,44 @@ +from client.network import SubmitError +from client.publication_result import failure_info, failure_text + + +def test_structured_failure_is_bounded_and_redacts_credentials(): + failure = failure_info( + {"category": "network", "retryable": True, + "reason": "POST https://private.example/upload?token=secret Authorization:Bearer abc123 failed"}, + operation="upload_put", campaign_id="campaign-0123456789abcdef", + ) + assert failure["category"] == "network" + assert failure["operation"] == "upload_put" + assert failure["campaignId"] == "campaign-0123456789abcdef" + assert "private.example" not in failure["reason"] + assert "secret" not in failure["reason"] + assert "abc123" not in failure["reason"] + assert len(failure["reason"]) <= 200 + + +def test_transport_status_and_legacy_string_have_actionable_causes(): + rate_limit = failure_info( + SubmitError("rate limited", retryable=True, status_code=429), + operation="create_run", + ) + assert rate_limit["category"] == "rate_limited" + assert rate_limit["retryable"] is True + assert rate_limit["statusCode"] == 429 + mismatch = failure_info( + "saved runtime identity differs from this client", operation="reconstruct", + ) + assert mismatch["category"] == "incompatible" + assert mismatch["retryable"] is False + assert "original compatible client" in failure_text(mismatch) + disk = failure_info( + FileNotFoundError("/Users/example/private/campaign.json is missing"), + operation="journal_reopen", + ) + assert "/Users/example" not in disk["reason"] + assert disk["operation"] == "journal_reopen" + assert disk["category"] == "corrupt_evidence" + expired = failure_info("retry_deadline_expired", operation="replay") + assert expired["category"] == "expired" + assert expired["retryable"] is False + assert "do not restart" in expired["nextAction"] diff --git a/client/tests/test_publication_storage.py b/client/tests/test_publication_storage.py index 99521e04..f2ea829f 100644 --- a/client/tests/test_publication_storage.py +++ b/client/tests/test_publication_storage.py @@ -12,6 +12,24 @@ MIB = 1024 * 1024 +def test_saved_publish_never_reports_success_with_unadmitted_envelopes(tmp_path): + campaign_id = 'campaign-0123456789abcdef' + root = tmp_path / 'campaigns' / campaign_id + root.mkdir(parents=True) + atomic_json(root / 'submission-000001.json', {'saved': 'immutable'}) + with mock.patch.object(main, 'drain_committed_receipts', return_value=0), \ + mock.patch.object(main, 'spool_payload', side_effect=spool.SpoolCapacityError('volume full')), \ + mock.patch.object(main, 'replay_spool', return_value=spool.ReplayStats()): + rc, info = main.publish_saved_campaign( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url='http://127.0.0.1:9', api_key='', max_storage_mb=2048, + ) + assert rc == 10 + assert info['status'] == 'deferred' + assert info['unadmitted'] == 1 + assert info['pending'] == 0 + + def payload_at(path, size=400 * 1024): from test_spool import SpoolTests path.parent.mkdir(parents=True, exist_ok=True) diff --git a/client/tests/test_release_packaging.py b/client/tests/test_release_packaging.py index 75b83c02..a463cd75 100644 --- a/client/tests/test_release_packaging.py +++ b/client/tests/test_release_packaging.py @@ -98,6 +98,7 @@ def test_release_version_is_assigned_and_missing_version_is_rejected(self) -> No def test_read_client_minimum_version_is_coherent(self) -> None: self.assertEqual(release_manifest_lib.read_client_minimum_version(), "client/0.3.0") + self.assertEqual(release_manifest_lib.read_client_implementation_version(), "client/0.3.8") def test_client_patch_version_does_not_change_protocol_minimum(self) -> None: with mock.patch.object(release_manifest_lib, "read_text", side_effect=[ @@ -231,6 +232,7 @@ def test_primary_artifact_contracts_and_launch_wiring(self) -> None: self.assertIn('"--gui"', gui_entry) console_entry = (root / "client/_pyinstaller_entry.py").read_text(encoding="utf-8") self.assertIn("from client.main import main", console_entry) + for relative_path in ("packaging/macos/launcher.sh", "packaging/macos/EncodingDB.command", "packaging/linux/start.sh"): @@ -252,6 +254,15 @@ def test_primary_artifact_contracts_and_launch_wiring(self) -> None: self.assertEqual(macos_client_package.format_version( macos_client_package.DOCUMENTED_MACOS_FLOOR), "27.0") + def test_console_onefile_builds_handle_process_group_interrupt_once(self) -> None: + root = release_manifest_lib.ROOT_DIR + for name in ("build_macos_client.sh", "build_linux_client.sh"): + script = (root / "scripts" / name).read_text(encoding="utf-8") + self.assertIn("--bootloader-ignore-signals", script) + windows = (root / "scripts/build_windows_client.ps1").read_text(encoding="utf-8") + self.assertIn('if ($Windowed) {', windows) + self.assertIn('$buildArgs += "--bootloader-ignore-signals"', windows) + if __name__ == "__main__": unittest.main() diff --git a/client/tests/test_release_preflight.py b/client/tests/test_release_preflight.py index b7b20448..9cd7ff92 100644 --- a/client/tests/test_release_preflight.py +++ b/client/tests/test_release_preflight.py @@ -58,8 +58,8 @@ def test_release_json_declares_coherent_frozen_release(self) -> None: payload = json.loads((release_manifest_lib.ROOT_DIR / "release.json").read_text(encoding="utf-8")) self.assertEqual(payload["suiteVersion"], "encodingdb-test-suite-v1") - self.assertEqual(payload["projectVersion"], "1.3.0-rc.5") - self.assertEqual(payload["releaseDate"], "2026-09-22") + self.assertEqual(payload["projectVersion"], "1.3.0-rc.8") + self.assertEqual(payload["releaseDate"], "2026-09-29") for tree in ("client", "server"): root = release_manifest_lib.ROOT_DIR / tree / "resources/test_suite_v1" status = json.loads((root / "finalization-status.json").read_text()) @@ -90,13 +90,14 @@ def test_preflight_reports_every_requested_gate_and_retains_logs(self) -> None: (repo / "README.md").write_text("EncodingDB release metadata test fixture\n", encoding="utf-8") (repo / "frontend" / "DEPLOYMENT.md").write_text("No deprecated Next.js references here.\n", encoding="utf-8") (repo / "CHANGELOG.md").write_text("## [Unreleased]\n\n- Pending freeze.\n", encoding="utf-8") + (repo / "client" / "main.py").write_text('CLIENT_VERSION = "client/0.2.0"\n', encoding="utf-8") (repo / "release.json").write_text( json.dumps( { "schemaVersion": 1, "projectVersion": None, "releaseDate": None, - "benchmarkProtocolVersion": "7.0", + "benchmarkProtocolVersion": "7.1", "plFormulaVersion": "7.0", "suiteVersion": "encodingdb-test-suite-v1", "clientImplementationVersion": "client/0.2.0", diff --git a/client/tests/test_reliability_recovery.py b/client/tests/test_reliability_recovery.py new file mode 100644 index 00000000..ef311fc8 --- /dev/null +++ b/client/tests/test_reliability_recovery.py @@ -0,0 +1,341 @@ +"""Regressions for saved work that must remain visible and publishable.""" + +import dataclasses +import hashlib +import json +from collections import Counter +from pathlib import Path +from unittest import mock + +from client import main, protocol, recovery_projection, spool +from client.campaign import atomic_json + + +def _root(tmp_path: Path, suffix: int = 1) -> tuple[str, Path]: + campaign_id = f"campaign-{suffix:016x}" + root = tmp_path / "campaigns" / campaign_id + root.mkdir(parents=True) + return campaign_id, root + + +def test_intact_saved_envelope_publishes_with_old_runtime_and_client(tmp_path): + campaign_id, root = _root(tmp_path) + atomic_json(root / "manifest.json", { + "protocolVersion": "7.1", "clientVersion": "client/0.0.0", + "runtime": {"ffmpeg": {"sha256": "original-runtime"}}, + }) + artifact = root / "measured.mp4" + artifact.write_bytes(b"retained original bytes") + payload = {"artifactPath": str(artifact), "artifactSha256": "original-hash"} + atomic_json(root / "submission-000001.json", payload) + (root / "attempt-000002.json").write_text("{corrupt unrelated sibling") + + def accepted_replay(*_args, **_kwargs): + local_hash = spool.local_hash_for_payload(payload) + atomic_json(tmp_path / "receipts" / f"{local_hash}.json", { + "localHash": local_hash, + "response": {"benchmarkRun": {"id": "run-original"}}, + }) + return spool.ReplayStats(submitted=1) + + with mock.patch.object(main, "spool_payload", return_value=("queued", {})) as stage, \ + mock.patch.object(main, "replay_spool", side_effect=accepted_replay), \ + mock.patch.object(main, "count_pending_entries", return_value=1), \ + mock.patch.object(main, "ensure_ffmpeg_and_ffprobe", side_effect=AssertionError("runtime not needed")): + rc, info = main._publish_saved_campaign_gated( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url="http://127.0.0.1:9", api_key="", max_storage_mb=2048, + ) + assert rc == 0, info + assert info["selectedPending"] == 0 + assert info["pending"] == 1 # An unrelated campaign remains queued. + stage.assert_called_once_with(str(tmp_path), payload, max_storage_mb=2048) + assert artifact.read_bytes() == b"retained original bytes" + + +def test_one_saved_publish_restages_after_queue_retirement(tmp_path): + campaign_id, root = _root(tmp_path) + for order in (1, 2): + artifact = root / f"measured-{order}.mp4" + artifact.write_bytes(f"saved-{order}".encode()) + atomic_json(root / f"submission-{order:06d}.json", { + "order": order, "artifactPath": str(artifact), + "runCreate": {"campaignId": campaign_id}, + }) + drained = False + admissions = [] + receipted = set() + + def stage(_queue, payload, **_kwargs): + if payload["order"] == 2 and not drained: + raise spool.SpoolCapacityError("one-artifact headroom") + admissions.append(payload) + return f"queued-{payload['order']}", {} + + def replay(*_args, **_kwargs): + nonlocal drained + drained = True + for payload in admissions: + order = payload["order"] + if order in receipted: + continue + local_hash = spool.local_hash_for_payload(payload) + atomic_json(tmp_path / "receipts" / f"{local_hash}.json", { + "localHash": local_hash, + "response": {"benchmarkRun": {"id": f"run-{order}"}}, + }) + receipted.add(order) + return spool.ReplayStats(submitted=len(admissions)) + + with mock.patch.object(main, "spool_payload", side_effect=stage), \ + mock.patch.object(main, "replay_spool", side_effect=replay), \ + mock.patch.object(main, "count_pending_entries", return_value=0): + rc, info = main._publish_saved_campaign_gated( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url="http://127.0.0.1:9", api_key="", max_storage_mb=2048, + ) + assert rc == 0, info + assert Counter(payload["order"] for payload in admissions) == Counter({1: 1, 2: 1}) + assert info["unadmitted"] == 0 + + +def test_one_saved_publish_drains_real_staged_bytes_to_admit_suffix(tmp_path): + from test_spool import SpoolTests + + campaign_id, root = _root(tmp_path) + for order, byte in ((1, b"A"), (2, b"B")): + artifact = root / f"measured-{order}.mp4" + artifact.write_bytes(byte * (600 * 1024)) + payload = SpoolTests()._authoritative_payload(str(artifact)) + payload["runCreate"]["campaignId"] = campaign_id + payload["runCreate"]["payloadHash"] = f"{order:064x}" + atomic_json(root / f"submission-{order:06d}.json", payload) + + run_ids = iter(("run-1", "run-2")) + with mock.patch.object(spool, "submit_artifact_submission", + side_effect=lambda *_args, **_kwargs: {"benchmarkRun": {"id": next(run_ids)}}) as send: + rc, info = main.publish_saved_campaign( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url="http://127.0.0.1:9", api_key="", max_storage_mb=2, + ) + assert rc == 0, info + assert send.call_count == 2 + assert len(list((tmp_path / "receipts").glob("*.json"))) == 2 + assert info["unadmitted"] == 0 + + +def test_blocked_publication_reports_its_cause_without_crashing(capsys): + main._report_recovery_result({ + "status": "blocked", "campaignId": "campaign-0000000000000001", + "failure": "saved runtime identity differs from this client", + }) + output = capsys.readouterr() + assert "saved runtime identity differs" in output.out + output.err + + +def test_windows_saved_publication_shows_typed_blocker_cause(): + from test_progress_contract import _build_gui_app + + app = _build_gui_app() + failure = main.failure_info( + "saved runtime identity differs from this client", + operation="reconstruct", campaign_id="campaign-0123456789abcdef", + ) + message = app._publication_result_text(1, { + "status": "blocked", "terminal": 0, "failure": failure, + }) + assert "runtime identity differs" in message + assert "original compatible client" in message + assert "0 terminal" not in message + + +def test_journal_only_finished_group_is_discoverable_for_publish(tmp_path): + campaign_id, root = _root(tmp_path) + cfg = protocol.ProtocolConfig.for_version("7.1", max_adaptive_repeats=0) + atomic_json(root / "manifest.json", { + "protocolVersion": "7.1", "protocolConfig": dataclasses.asdict(cfg), + "tasks": [{"clipId": "clip", "encoder": "libx264", "preset": "fast", "crf": 24}], + }) + timing = protocol.EncodeTiming.from_measurement( + start_monotonic_ns=1, end_monotonic_ns=1_000_000_001, + source_frame_count=24, encoded_frame_count=24, source_fps=24, + ) + for order, phase, repetition in ((1, "warmup", 1), (2, "measured", 1), (3, "measured", 2)): + metadata = {} + if phase == "measured": + artifact = root / f"measured-{order}.mp4" + artifact.write_bytes(f"measured-{order}".encode()) + metadata = {"info": { + "artifactPath": str(artifact), + "artifactSha256": hashlib.sha256(artifact.read_bytes()).hexdigest(), + "fileSizeBytes": artifact.stat().st_size, + "error": None, + }} + record = protocol.BenchmarkRunRecord( + schedule=protocol.ScheduledRun(campaign_id, "clip|libx264|fast|24", phase, repetition, order), + timing=timing, + metadata=metadata, + counted_for_stability=phase == "measured", + ) + atomic_json(root / f"attempt-{order:06d}.json", record.to_dict()) + (root / "attempt-000004.json").write_text("{corrupt sibling") + projected = recovery_projection.project_attempt_groups(root, campaign_id) + assert projected["finishedGroupIds"] == ["clip|libx264|fast|24"] + assert projected["candidateOrders"] == [2, 3] + assert [item["path"] for item in projected["corruptEntries"]] == ["attempt-000004.json"] + state = main.campaign_recovery_state(str(tmp_path), campaign_id) + assert state is not None + assert state["completedGroups"] == 1 + assert any(action["action"] == "publish_saved" for action in state["actions"]) + + +def test_terminal_menu_does_not_hide_older_or_interrupted_publishable_work(tmp_path): + ids = [] + for suffix in range(1, 8): + campaign_id, root = _root(tmp_path, suffix) + artifact = root / "measured.mp4" + artifact.write_bytes(b"saved") + atomic_json(root / "submission-000001.json", {"artifactPath": str(artifact)}) + if suffix != 7: + atomic_json(root / "campaign-complete.json", {"failed": 0, "skipped": 0}) + ids.append(campaign_id) + publishable = {item[0] for item in main._publishable_campaigns(str(tmp_path))} + assert publishable == set(ids) + + +def test_terminal_menu_lists_every_incomplete_campaign(tmp_path): + ids = [] + for suffix in range(1, 9): + campaign_id, _ = _root(tmp_path, suffix) + ids.append(campaign_id) + assert {item[0] for item in main._incomplete_campaigns(str(tmp_path))} == set(ids) + + +def test_terminal_saved_publish_is_available_before_broken_runtime_setup(tmp_path): + campaign_id, root = _root(tmp_path) + artifact = root / "measured.mp4" + artifact.write_bytes(b"saved") + atomic_json(root / "submission-000001.json", {"artifactPath": str(artifact)}) + args = main.build_arg_parser().parse_args(["--queue-dir", str(tmp_path)]) + + def choose(_title, labels, **_kwargs): + return next(index for index, label in enumerate(labels) if "Publish saved" in label) + + with mock.patch.object(main, "ensure_ffmpeg_and_ffprobe", side_effect=AssertionError("runtime is broken")), \ + mock.patch.object(main, "prompt_choice", side_effect=choose), \ + mock.patch.object(main, "publish_saved_campaign", return_value=(0, {"status": "published", "campaignId": campaign_id})) as publish: + assert main.interactive_menu_flow(main.build_arg_parser(), args) == 0 + assert publish.call_args.kwargs["campaign_id"] == campaign_id + + +def test_recovery_counts_envelope_and_spool_copy_once(tmp_path): + campaign_id, root = _root(tmp_path) + artifact = root / "measured.mp4" + artifact.write_bytes(b"saved") + payload = {"artifactPath": str(artifact), "runCreate": {"campaignId": campaign_id}} + atomic_json(root / "submission-000001.json", payload) + spool.spool_payload(str(tmp_path), payload) + state = main.campaign_recovery_state(str(tmp_path), campaign_id) + assert state is not None + assert state["pendingUploads"] == 1 + assert state["queuePending"] == 1 + assert state["logicalPendingUploads"] == 1 + assert main._publishable_campaigns(str(tmp_path))[0][2] == 1 + + +def test_queue_receipt_wins_over_unmarked_saved_envelope(tmp_path): + campaign_id, root = _root(tmp_path) + artifact = root / "measured.mp4" + artifact.write_bytes(b"already accepted") + payload = {"artifactPath": str(artifact), "runCreate": {"campaignId": campaign_id}} + atomic_json(root / "submission-000001.json", payload) + receipt_hash = spool.local_hash_for_payload(payload) + atomic_json(tmp_path / "receipts" / f"{receipt_hash}.json", { + "localHash": receipt_hash, + "response": {"benchmarkRun": {"id": "run-already-accepted"}}, + }) + state = main.campaign_recovery_state(str(tmp_path), campaign_id) + assert state is not None + assert state["acceptedUploads"] == 1 + assert state["logicalPendingUploads"] == 0 + assert not main._publishable_campaigns(str(tmp_path)) + + +def test_failure_actions_preserve_evidence_without_generic_reencode_or_cleanup(): + for cause in ("retry_deadline_expired", "missing_spooled_artifact", "corrupt_existing_spool"): + action = main._submit_failure_fields("failed", cause)["recoveryAction"].lower() + assert "queue-cleanup" not in action + assert "re-encode" not in action + assert "resume or publish" not in action + + +def test_missing_retained_artifact_is_visible_blocker_not_published(tmp_path): + campaign_id, root = _root(tmp_path) + atomic_json(root / "submission-000001.json", { + "artifactPath": str(root / "missing.mp4"), + "runCreate": {"campaignId": campaign_id}, + }) + with mock.patch.object(main, "spool_payload", return_value=("queued", {})), \ + mock.patch.object(main, "replay_spool", return_value=spool.ReplayStats()): + rc, info = main.publish_saved_campaign( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url="http://127.0.0.1:9", api_key="", max_storage_mb=2048, + ) + assert rc == 1 + assert info["status"] == "blocked" + assert info["unavailableSources"] == 1 + assert "artifact" in info["failure"]["reason"].lower() + + +def test_missing_legacy_client_identity_cannot_be_relabelled_during_reconstruction(tmp_path): + campaign_id, root = _root(tmp_path) + atomic_json(root / "manifest.json", {"protocolVersion": "7.1"}) + outcome = main._reconstruct_saved_submissions( + queue_dir=str(tmp_path), campaign_id=campaign_id, max_storage_mb=2048, + ) + assert outcome["reconstructed"] == 0 + assert "saved client identity is missing" in outcome["failure"] + assert not list(root.glob("submission-*.json")) + + +def test_consented_saved_publish_persists_continuation_until_confirmed(tmp_path): + campaign_id, _root_dir = _root(tmp_path) + arguments = dict(queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url="http://127.0.0.1:9", api_key="", + max_storage_mb=2048, continue_when_open=True) + with mock.patch.object(main, "_has_publication_consent", return_value=True), \ + mock.patch.object(main, "_publish_saved_campaign_gated", side_effect=[ + (10, {"campaignId": campaign_id, "status": "pending", "pending": 1}), + (0, {"campaignId": campaign_id, "status": "published", "pending": 0}), + ]): + assert main.publish_saved_campaign(**arguments)[0] == 10 + pending = main.publication_intent_state(str(tmp_path), campaign_id) + assert pending["active"] is True + assert pending["nextAttemptAt"] >= pending["createdAt"] + assert main.publish_saved_campaign(**arguments)[0] == 0 + confirmed = main.publication_intent_state(str(tmp_path), campaign_id) + assert confirmed["active"] is False + assert confirmed["status"] == "published" + assert confirmed["baseUrlFingerprint"] == pending["baseUrlFingerprint"] + + +def test_legacy_attempts_without_client_identity_cannot_resume_as_new_client(tmp_path): + from test_progress_contract import _run_batch, _campaign_root + + events = [] + assert _run_batch(tmp_path, seed=713, events=events) == 0 + root = _campaign_root(tmp_path) + (root / "campaign-complete.json").unlink() + manifest_path = root / "manifest.json" + manifest = json.loads(manifest_path.read_text()) + manifest.pop("clientVersion") + atomic_json(manifest_path, manifest) + before = {path.name for path in root.glob("attempt-*.json")} + resumed_events = [] + assert _run_batch(tmp_path, seed=713, events=resumed_events) == 6 + assert {path.name for path in root.glob("attempt-*.json")} == before + assert any("saved client identity is missing" in str(event.get("message")).lower() + for event in resumed_events if event.get("type") == "run_error") + view = main.campaign_recovery_state(str(tmp_path), root.name) + assert "original compatible client" in view["measurementBlocked"] + assert not any(action["action"] == "resume" for action in view["actions"]) diff --git a/client/tests/test_runtime_preparation_budgets.py b/client/tests/test_runtime_preparation_budgets.py new file mode 100644 index 00000000..76b44ad3 --- /dev/null +++ b/client/tests/test_runtime_preparation_budgets.py @@ -0,0 +1,221 @@ +"""R02: every runtime probe and preparation stage owns a finite wall-clock budget, +a persistent substage/heartbeat trail, responsive cancellation, and bounded +owned-process termination. No media runtime or network required.""" +import hashlib +import itertools +import json +import os +import subprocess +import sys +import threading +import time +from unittest import mock + +import pytest + +from client import campaign, protocol, runtime_lock + + +def _journal_with_artifact(tmp_path): + campaign_id = "campaign-0123456789abcdef" + root = campaign.journal_path(str(tmp_path), campaign_id) + root.mkdir(parents=True) + manifest = {"seed": 3, "tasks": [{"encoder": "libx264", "preset": "fast", "crf": 24}]} + campaign.atomic_json(root / "manifest.json", manifest) + artifact = root / "attempt-000000.mkv" + artifact.write_bytes(b"a" * (1024 * 1024) + b"b" * (1024 * 1024)) + record = protocol.BenchmarkRunRecord( + schedule=protocol.ScheduledRun(campaign_id, "clip|libx264|fast|24", "measured", 0, 0), + metadata={"info": {"artifactPath": str(artifact), + "artifactSha256": hashlib.sha256(artifact.read_bytes()).hexdigest()}}, + counted_for_stability=True) + campaign.atomic_json(root / "attempt-000000.json", record.to_dict()) + return campaign_id, root, manifest, artifact + + +def _run_in_thread(action): + outcome = [] + def body(): + try: + outcome.append(("returned", action())) + except BaseException as exc: # noqa: BLE001 - surfaced to the assertion + outcome.append(("raised", exc)) + thread = threading.Thread(target=body, daemon=True) + started = time.monotonic() + thread.start() + thread.join(10) + return thread, outcome, time.monotonic() - started + + +def test_hanging_runtime_probe_is_killed_with_budget_context(): + spawned = [] + real_popen = subprocess.Popen + def launch(command, **kwargs): + process = real_popen(command, **kwargs) + spawned.append(process) + return process + with mock.patch.object(runtime_lock, "RUNTIME_PROBE_TIMEOUT_SECONDS", 1.0), \ + mock.patch.object(campaign.subprocess, "Popen", side_effect=launch): + thread, outcome, elapsed = _run_in_thread( + lambda: runtime_lock._run_text([sys.executable, "-c", "import time; time.sleep(30)"])) + try: + assert not thread.is_alive(), "a hung runtime probe outran its finite probe budget" + assert elapsed < 10 + kind, value = outcome[0] + assert kind == "raised" and isinstance(value, runtime_lock.RuntimeLockError) + assert "probe budget" in str(value) + assert len(spawned) == 1 and spawned[0].poll() is not None + finally: + for process in spawned: + if process.poll() is None: + process.kill() + process.wait(timeout=5) + + +def test_scoped_probe_without_explicit_timeout_is_bounded_and_reaped(): + spawned = [] + real_popen = subprocess.Popen + def launch(command, **kwargs): + process = real_popen(command, **kwargs) + # The patch is process-global; ignore the interpreter's own lazy + # platform probes (uname/file) that land inside the window. + if list(command)[0] == sys.executable: + spawned.append(process) + return process + def action(): + with campaign.PreparationScope().activate(): + return campaign.run_measurement_process( + [sys.executable, "-c", "import time; time.sleep(30)"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + with mock.patch.object(campaign, "PROBE_PROCESS_TIMEOUT_SECONDS", 1.0), \ + mock.patch.object(campaign.subprocess, "Popen", side_effect=launch): + thread, outcome, elapsed = _run_in_thread(action) + try: + assert not thread.is_alive(), "a scoped probe with no explicit timeout hung forever" + assert elapsed < 10 + kind, value = outcome[0] + assert kind == "raised" and isinstance(value, subprocess.TimeoutExpired) + assert campaign._PREPARATION.get() is None + assert len(spawned) == 1 and spawned[0].poll() is not None, \ + "timed-out owned probe process was not terminated" + finally: + for process in spawned: + if process.poll() is None: + process.kill() + process.wait(timeout=5) + + +def test_preparation_stage_budget_fails_closed_on_wall_clock_expiry(): + scope = campaign.PreparationScope() + with mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=20)): + with scope.activate(): # activate opens the acquisition fence; body stages nest inside + with pytest.raises(campaign.PreparationBudgetExceeded) as raised: + with scope.stage("runtime-verify", 30): + campaign.check_preparation_cancelled() + campaign.check_preparation_cancelled() + assert raised.value.stage == "runtime-verify" + assert raised.value.budget_seconds == 30 + assert scope.stage_stack == [] + + outer = campaign.PreparationScope() + with mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=10)): + with outer.activate(): + with pytest.raises(campaign.PreparationBudgetExceeded) as raised: + with outer.stage("outer", 1000): + with outer.stage("inner", 20): + campaign.check_preparation_cancelled() + campaign.check_preparation_cancelled() + assert raised.value.stage == "inner" + assert outer.stage_stack == [] + + +def test_stage_stall_watchdog_tolerates_live_progress_but_fails_on_silence(): + events = [] + scope = campaign.PreparationScope(progress=lambda stage, **details: events.append(stage)) + with mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=10)): + with scope.activate(): + with scope.stage("runtime-hash", 1000, stall_seconds=50): + for count in range(10): + campaign.preparation_progress("runtime-hash", path="bundle", completedBytes=count) + assert events == ["runtime-hash"] * 10 + with pytest.raises(campaign.PreparationStalled) as raised: + with scope.stage("runtime-dependency-hash", 10000, stall_seconds=50): + for _ in range(5): + campaign.check_preparation_cancelled() + assert raised.value.stage == "runtime-dependency-hash" + + +def test_watchdogs_never_fire_outside_an_open_stage(): + scope = campaign.PreparationScope() + token = campaign._PREPARATION.set(scope) + try: + with mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=10000)): + for _ in range(5): + campaign.check_preparation_cancelled() + campaign.preparation_progress("upload", path="x") + finally: + campaign._PREPARATION.reset(token) + + +def test_journal_reopen_is_cancellable_between_hash_chunks(tmp_path): + campaign_id, root, manifest, artifact = _journal_with_artifact(tmp_path) + campaign.atomic_json(root / "budget.json", {"schemaVersion": 1, "maxStorageMb": 256}) + stop = threading.Event() + events = [] + def progress(stage, **details): + events.append((stage, details)) + stop.set() + def snapshot(): + # budget.json is a policy file reopen rewrites (byte-identical); the + # evidence files a failed reopen must never touch are manifest + attempt. + return {p.name: p.read_bytes() for p in root.iterdir() if p.name != "budget.json"} + before = snapshot() + with mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=1.0)): + with pytest.raises(KeyboardInterrupt), campaign.PreparationScope(stop, progress).activate(): + campaign.CampaignJournal(str(tmp_path), campaign_id, manifest, 256) + assert [stage for stage, _ in events] == ["journal-reopen"] + assert events[0][1]["path"] == str(artifact) + assert events[0][1]["completedBytes"] == 1024 * 1024 + assert snapshot() == before + + +def test_journal_reopen_heartbeat_persists_throttled_substage(tmp_path): + campaign_id, root, manifest, artifact = _journal_with_artifact(tmp_path) + heartbeat = tmp_path / "state" / "preparation.json" + observed = [] + def progress(stage, **details): + observed.append(json.loads(heartbeat.read_text())) + scope = campaign.PreparationScope(progress=progress, heartbeat_path=str(heartbeat)) + with mock.patch.object(campaign.time, "monotonic", return_value=10): + with scope.activate(): + campaign.CampaignJournal(str(tmp_path), campaign_id, manifest, 256) + assert len(observed) == 1 + mid = observed[0] + assert mid["stage"] == "journal-reopen" and mid["substage"] == "journal-reopen" + assert mid["completedBytes"] == 1024 * 1024 and mid["pid"] == os.getpid() + final = json.loads(heartbeat.read_text()) + assert final["stage"] == "journal-reopen" and final["status"] == "completed" + assert final["stageBudgetSeconds"] == campaign.JOURNAL_REOPEN_BUDGET_SECONDS + + +def test_acquisition_fence_fails_closed_when_a_reader_never_drains(): + release = threading.Event() + reader = threading.Thread(target=release.wait, daemon=True) + reader.start() + try: + with mock.patch.object(campaign, "_ACQUISITION_READER", reader), \ + mock.patch.object(campaign, "_ACQUISITION_FENCE_BUDGET_SECONDS", 0.5), \ + mock.patch.object(campaign.time, "monotonic", side_effect=itertools.count(step=0.1)): + def action(): + with campaign.PreparationScope().activate(): + return "entered" + thread, outcome, elapsed = _run_in_thread(action) + assert not thread.is_alive(), "the acquisition fence waited forever on a stuck reader" + assert elapsed < 10 + kind, value = outcome[0] + assert kind == "raised" and isinstance(value, campaign.PreparationTimeout) + assert value.stage == "acquisition-fence" + assert campaign._PREPARATION.get() is None + finally: + release.set() + reader.join(5) \ No newline at end of file diff --git a/client/tests/test_runtime_provisioning.py b/client/tests/test_runtime_provisioning.py new file mode 100644 index 00000000..9be48b73 --- /dev/null +++ b/client/tests/test_runtime_provisioning.py @@ -0,0 +1,14 @@ +"""The CI runtime recovery path must reject unreviewed archive bytes.""" + +import pytest + +from scripts.provision_pinned_runtime import provision + + +@pytest.mark.parametrize("platform", ["linux", "win"]) +def test_candidate_archive_tamper_fails_before_extract(tmp_path, platform): + archive = tmp_path / "candidate" + archive.write_bytes(b"unreviewed") + with pytest.raises(RuntimeError, match="reviewed SHA-256/size"): + provision(platform, tmp_path / "out", archive_path=archive) + assert not list((tmp_path / "out").rglob("ffmpeg*")) diff --git a/client/tests/test_runtime_review.py b/client/tests/test_runtime_review.py new file mode 100644 index 00000000..b7bac65c --- /dev/null +++ b/client/tests/test_runtime_review.py @@ -0,0 +1,24 @@ +"""Independent review of preparation stage deadline accounting.""" + +import time + +import pytest + +from client.campaign import PreparationBudgetExceeded, PreparationScope + + +def test_stage_reports_overrun_when_blocking_step_returns_after_budget(): + scope = PreparationScope(heartbeat_path=None, stall_seconds=10) + with scope.activate(): + with pytest.raises(PreparationBudgetExceeded): + with scope.stage("review-blocking-step", 0.05): + time.sleep(0.08) + + +def test_nested_stage_does_not_consume_parent_independent_budget(): + scope = PreparationScope(heartbeat_path=None, stall_seconds=10) + with scope.activate(): + with scope.stage("review-parent", 0.07): + with scope.stage("review-child", 0.2): + time.sleep(0.09) + scope.check() diff --git a/client/tests/test_spool.py b/client/tests/test_spool.py index ecc60258..bd75d084 100644 --- a/client/tests/test_spool.py +++ b/client/tests/test_spool.py @@ -1,6 +1,7 @@ import json import hashlib import os +import sys import tempfile import threading import unittest @@ -9,7 +10,7 @@ from unittest import mock from client.artifacts import AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND -from client.spool import cleanup_spool, count_pending_entries, inspect_spool, load_spool_entry, replay_spool, spool_payload +from client.spool import cleanup_spool, collector_publication_scope, count_pending_entries, due_first_queue_paths, inspect_spool, load_spool_entry, replay_spool, spool_payload class _SpoolHandler(BaseHTTPRequestHandler): @@ -339,6 +340,236 @@ def test_spool_capacity_charges_own_campaign_and_honors_free_floor(self) -> None with self.assertRaisesRegex(OSError, "safety"): spool_payload(queue_dir, payload, max_storage_mb=2048) + def test_publication_defers_to_a_measuring_collector(self) -> None: + # C11, collector-first order: while measurement.lock is held, both new + # staging and replay must defer (recoverable pause), never upload. + import fcntl + with tempfile.TemporaryDirectory() as queue_dir: + source_path = os.path.join(queue_dir, "artifact.mp4") + with open(source_path, "wb") as handle: + handle.write(b"test") + payload = self._authoritative_payload(source_path) + with open(os.path.join(queue_dir, "measurement.lock"), "a+b") as held: + fcntl.flock(held.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + with self.assertRaisesRegex(OSError, "publication defers"): + spool_payload(queue_dir, payload, max_storage_mb=2048) + with self.assertRaisesRegex(OSError, "publication defers"): + replay_spool(queue_dir, base_url="http://127.0.0.1:1", api_key="", retries=1, use_token=False) + with collector_publication_scope(): + # The collector's own checkpoint uploads legitimately proceed. + path, entry = spool_payload(queue_dir, payload, max_storage_mb=2048) + self.assertTrue(os.path.exists(path)) + self.assertFalse(entry.get("terminal")) + + def test_collector_entry_refuses_while_publisher_owns_queue(self) -> None: + # C11 reverse order: a publisher holding publication.lock blocks the next + # collection; releasing it lets the collector proceed. + from client import main as client_main + import fcntl + with tempfile.TemporaryDirectory() as queue_dir: + with open(os.path.join(queue_dir, "publication.lock"), "a+b") as held: + fcntl.flock(held.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + events = [] + rc = client_main.active_collection_guard(queue_dir, lambda event: events.append(event)) + self.assertEqual(rc, 6) + self.assertTrue(any("publication pass" in str(event.get("message")) for event in events), + "the refusal must name the publisher, not a phantom collector") + self.assertIsNone(client_main.active_collection_guard(queue_dir)) + + def test_due_first_window_skips_delayed_and_stays_fair(self) -> None: + # C08: delayed (Retry-After in the future) entries never enter the + # window; due entries are admitted oldest-scheduled first and the + # window stays bounded when more than `limit` are due. + import time as _time + with tempfile.TemporaryDirectory() as queue_dir: + for index in range(25): + path1, entry1 = spool_payload(queue_dir, {"cpuModel": f"delayed-{index}", "fps": 1}) + entry1["nextAttemptAt"] = _time.time() + 3600 + entry1["retryDeadlineAt"] = _time.time() + 86400 + with open(path1, "w") as handle: + json.dump(entry1, handle) + due_paths = [] + for index in range(3): + path2, entry2 = spool_payload(queue_dir, {"cpuModel": f"due-{index}", "fps": 1}) + entry2["nextAttemptAt"] = _time.time() - (10 - index) # due, staggered + with open(path2, "w") as handle: + json.dump(entry2, handle) + due_paths.append(path2) + selected, deferred = due_first_queue_paths(queue_dir, limit=25) + self.assertEqual(sorted(selected), sorted(due_paths), + "only due entries may be attempted; delayed must wait") + self.assertEqual(deferred, 25) + # Oldest-scheduled first (fairness) + self.assertEqual([os.path.basename(p) for p in selected], + [os.path.basename(p) for p in sorted(due_paths, key=lambda p: load_spool_entry(p)["nextAttemptAt"])]) + # Bound: with limit=2 the two earliest win; the third is deferred, not dropped silently + capped, deferred2 = due_first_queue_paths(queue_dir, limit=2) + self.assertEqual(len(capped), 2) + self.assertEqual(deferred2, 26) + + def test_replay_cancellation_is_bounded_and_keeps_entries_durable(self) -> None: + # C12: a cancel request stops admission between entries; every entry + # keeps its durable file so the next pass retries idempotently, and a + # pre-set cancel performs zero network attempts. + with tempfile.TemporaryDirectory() as queue_dir: + for index in range(3): + spool_payload(queue_dir, {"cpuModel": f"p{index}", "fps": 1}) + cancelled = type("AlwaysCancelled", (), {"is_set": lambda self: True})() + stats = replay_spool(queue_dir, base_url="http://127.0.0.1:1", api_key="", + retries=1, use_token=False, cancel_event=cancelled) + self.assertEqual(stats.submitted, 0) + self.assertEqual(stats.cancelled, 1) + self.assertEqual(stats.deferred, 2) + self.assertEqual(count_pending_entries(queue_dir), 3, + "cancellation must never lose or duplicate durable work") + + def test_lost_response_stays_durable_and_retries_idempotently(self) -> None: + # C12: a dropped/ambiguous network outcome must keep the exact pending + # entry (same localHash identity) and the next pass completes it once. + _SpoolHandler.mode = "ok" + server, thread, base_url = self._start_server() + try: + with tempfile.TemporaryDirectory() as queue_dir: + path, entry = spool_payload(queue_dir, {"cpuModel": "ambiguous", "fps": 1}) + hash_before = entry["localHash"] + from client.spool import submit as real_submit + calls = {"n": 0} + def flaky(*args, **kwargs): + calls["n"] += 1 + if calls["n"] == 1: + raise ConnectionError("response lost mid-transaction") + return real_submit(*args, **kwargs) + with mock.patch("client.spool.submit", side_effect=flaky): + stats = replay_spool(queue_dir, base_url=base_url, api_key="", retries=1, use_token=False) + self.assertEqual((stats.submitted, stats.retained), (0, 1)) + retained = load_spool_entry(path) + self.assertEqual(retained["localHash"], hash_before, + "ambiguous outcome must not replace retry identity") + retained["nextAttemptAt"] = 0 # due now (test clock shortcut) + with open(path, "w") as handle: + json.dump(retained, handle) + stats2 = replay_spool(queue_dir, base_url=base_url, api_key="", retries=1, use_token=False) + self.assertEqual(stats2.submitted, 1) + self.assertEqual(count_pending_entries(queue_dir), 0) + finally: + server.shutdown() + thread.join(timeout=5) + + +class HostPhaseExclusionTests(unittest.TestCase): + """C11: kernel-backed host phase exclusion across distinct queue dirs.""" + + def _phase_dir_like(self, tmp: str) -> str: + root = os.path.join(tmp, "host-phase") + os.makedirs(root, exist_ok=True) + return root + + + def test_collector_first_defers_publisher_on_other_queue(self) -> None: + # The publisher runs on its own thread: same-thread re-entry is the + # collector's own checkpoint-upload path and legitimately bypasses the + # probe; a *different* publisher (thread/process) must be refused. + from client.spool import SpoolCapacityError, host_phase_hold + errors: list = [] + queue_b: list = [] + def publish(): + try: + spool_payload(queue_b[0], {"cpuModel": "other-queue", "fps": 1}) + except BaseException as exc: # noqa: BLE001 - recorded for assertion + errors.append(exc) + with tempfile.TemporaryDirectory() as tmp, \ + mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": self._phase_dir_like(tmp)}): + queue_b.append(os.path.join(tmp, "qb")) + collector = host_phase_hold("measurement") + collector.__enter__() + try: + thread = threading.Thread(target=publish) + thread.start() + thread.join(5) + self.assertFalse(thread.is_alive()) + self.assertEqual(len(errors), 1) + self.assertIsInstance(errors[0], SpoolCapacityError) + self.assertIn("measuring on this host", str(errors[0])) + finally: + collector.__exit__(None, None, None) + queue_b[0] = os.path.join(tmp, "qb2") + with mock.patch("client.spool.submit_artifact_submission", return_value={}): + spool_payload(queue_b[0], {"cpuModel": "after-release", "fps": 1}) + + def test_publisher_first_defers_collector_and_allows_second_publisher(self) -> None: + from client.spool import SpoolCapacityError, host_phase_hold + errors: list = [] + with tempfile.TemporaryDirectory() as tmp, \ + mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": self._phase_dir_like(tmp)}): + publisher = host_phase_hold("publication") + publisher.__enter__() + def collector_start(): + try: + hold = host_phase_hold("measurement") + hold.__enter__() + except BaseException as exc: # noqa: BLE001 - recorded for assertion + errors.append(exc) + return + errors.append(None) + hold.__exit__(None, None, None) + try: + thread = threading.Thread(target=collector_start) + thread.start() + thread.join(5) + self.assertFalse(thread.is_alive()) + self.assertEqual(len(errors), 1) + self.assertIsInstance(errors[0], SpoolCapacityError) + self.assertIn("owns this host right now", str(errors[0])) + # A second publisher coexists: idempotent uploads are not a + # timing hazard and must not deadlock each other. + with host_phase_hold("publication"): + pass + finally: + publisher.__exit__(None, None, None) + + def test_crashed_owner_releases_host_phase_for_reopen(self) -> None: + import signal + import subprocess + from client.spool import host_phase_busy, host_phase_hold + with tempfile.TemporaryDirectory() as tmp: + phase_dir = self._phase_dir_like(tmp) + env = dict(os.environ, ENCODINGDB_HOST_PHASE_DIR=phase_dir) + script = ( + "import os, fcntl, time\n" + "path = os.path.join(os.environ['ENCODINGDB_HOST_PHASE_DIR'], 'phase.lock')\n" + "handle = open(path, 'a+b')\n" + "fcntl.flock(handle.fileno(), fcntl.LOCK_EX)\n" + "print('HELD', flush=True)\n" + "time.sleep(30)\n") + proc = subprocess.Popen([sys.executable, "-c", script], env=env, + stdout=subprocess.PIPE, text=True) + try: + self.assertEqual(proc.stdout.readline().strip(), "HELD") + with mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": phase_dir}): + self.assertTrue(host_phase_busy()) + os.kill(proc.pid, signal.SIGKILL) + proc.wait(5) + self.assertFalse(host_phase_busy(), + "kernel must release a crashed owner's phase lock") + with host_phase_hold("measurement") as acquired: + self.assertTrue(acquired) + finally: + if proc.poll() is None: + proc.kill() + proc.wait(5) + + def test_uninspectable_host_phase_lock_fails_closed(self) -> None: + from client.spool import SpoolCapacityError, host_phase_busy + with tempfile.TemporaryDirectory() as tmp: + phase_dir = self._phase_dir_like(tmp) + os.chmod(phase_dir, 0o000) + try: + with mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": phase_dir}): + with self.assertRaises(SpoolCapacityError): + host_phase_busy() + finally: + os.chmod(phase_dir, 0o755) + if __name__ == "__main__": unittest.main() diff --git a/client/tests/test_suite_clip_acquisition.py b/client/tests/test_suite_clip_acquisition.py new file mode 100644 index 00000000..f15f033c --- /dev/null +++ b/client/tests/test_suite_clip_acquisition.py @@ -0,0 +1,397 @@ +"""PLA-546 C02: cold Small contributions fetch only their frozen quick clip. + +Synthetic media only (never production suite resources). A loopback HTTP server +stands in for the published per-clip host; request logs prove exactly which +assets each run transferred. +""" +import hashlib +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import json +import os +from pathlib import Path +import posixpath +import shutil +import stat +import threading +from unittest import mock + +import pytest +from client import campaign, suite +from scripts.prepare_client_suite_distribution import build_clip_bundle + +from test_suite_v1 import small_media_fixture + + +def add_notices(root: Path, manifest) -> None: + notices = root / "notices" + notices.mkdir(exist_ok=True) + for clip in manifest.clips: + license_id = str(clip.provenance.get("license") or "CC-BY-4.0") + (notices / f"{license_id}.txt").write_text(f"license text for {license_id}\n") + (notices / f"{clip.clip_id}.txt").write_text( + f"Attribution notice for {clip.clip_id}\nLicense: {license_id}; see {license_id}.txt\n" + ) + + +class ClipServer: + """Serves the staged per-clip bundle; supports Range and tampered assets.""" + + def __init__(self, root: Path, distribution, *, corrupt_paths=(), short_paths=()): + manifest = json.loads((root / "manifest.json").read_text()) + by_id = {clip["id"]: clip for clip in manifest["clips"]} + self.files = {} + for entry in distribution["clips"].values(): + for asset in entry["assets"]: + if str(asset["role"]) == "clip": + clip_id = str(asset["path"]).split("/")[0] + data = (root / "canonical" / by_id[clip_id]["fileName"]).read_bytes() + else: + data = (root / "notices" / posixpath.basename(str(asset["path"]))).read_bytes() + if str(asset["path"]) in corrupt_paths: + data = bytes([data[0] ^ 255]) + data[1:] + self.files["/" + str(asset["downloadName"])] = data + self.short_paths = set(short_paths) + self.requests = [] + self.range_headers = [] + server = self + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_GET(self): + server.requests.append(self.path) + data = server.files.get(self.path) + range_header = self.headers.get("Range") + if range_header: + server.range_headers.append((self.path, range_header)) + if data is None: + self.send_error(404) + return + if self.path in server.short_paths: + # A truncated complete response: declared short, body short. + truncated = data[: max(1, len(data) // 2)] + self.send_response(200) + self.send_header("Content-Length", str(len(truncated))) + self.end_headers() + self.wfile.write(truncated) + return + start = 0 + if range_header and range_header.startswith("bytes=") and range_header.endswith("-"): + start = int(range_header[6:-1]) + if start: + self.send_response(206) + self.send_header("Content-Range", f"bytes {start}-{len(data) - 1}/{len(data)}") + self.send_header("Content-Length", str(len(data) - start)) + self.end_headers() + self.wfile.write(data[start:]) + else: + self.send_response(200) + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + self.server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + self.serving = threading.Thread(target=self.server.serve_forever, daemon=True) + self.serving.start() + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self.server.server_port}/" + + def close(self): + self.server.shutdown() + self.server.server_close() + self.serving.join(timeout=2) + campaign.wait_for_owned_acquisition() + + +def staged_fixture(tmp_path, **server_kwargs): + """Staged suite tree: metadata + notices + served media, no local canonical bytes.""" + outer = small_media_fixture() + root, manifest = outer.__enter__() + server = None + try: + add_notices(root, manifest) + distribution_path = suite.write_clip_distribution_metadata(str(root)) + metadata = json.loads(Path(distribution_path).read_text()) + clip = manifest.clips[0] + clip_bytes = (root / "canonical" / clip.file_name).read_bytes() + server = ClipServer(root, metadata, **server_kwargs) + # A distributed client tree carries no canonical bytes; only the pack or + # the published per-clip host can supply the media. + shutil.rmtree(root / "canonical") + cache = str(tmp_path / "cache") + with mock.patch.dict(os.environ, {suite.SUITE_CLIP_BASE_URL_ENV: server.base_url}): + yield server, cache, clip, manifest, clip_bytes + finally: + if server is not None: + server.close() + outer.__exit__(None, None, None) + + +@pytest.fixture +def clip_env(tmp_path): + yield from staged_fixture(tmp_path) + + +def distribution_entry(clip_id): + payload = json.loads(Path(suite.get_clip_distribution_path()).read_text()) + return payload["clips"][clip_id] + + +def test_cold_quick_clip_downloads_only_selected_clip_assets(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + served_ids = {path.lstrip("/").split("--", 1)[0] for path in server.requests} + assert served_ids == {clip.clip_id} + assert not any("pack" in path or path.endswith(".tar.gz") for path in server.requests) + installed = Path(cache) / "canonical" / clip.file_name + assert installed.read_bytes() == clip_bytes + assert Path(prepared.path) == installed + assert prepared.input_hash == clip.sha256 + # License notices install with matching frozen hashes. + for asset in distribution_entry(clip.clip_id)["assets"]: + if str(asset["role"]) == "clip": + continue + installed_notice = Path(suite._clip_asset_target(cache, clip, asset)) + assert hashlib.sha256(installed_notice.read_bytes()).hexdigest() == asset["sha256"] + + +def test_repeat_uses_hash_verified_cache_without_requests(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + suite.ensure_suite_clip(clip, cache_root=cache) + before = len(server.requests) + suite.ensure_suite_clip(clip, cache_root=cache) + assert len(server.requests) == before + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert estimate["strategy"] == "cache" + assert estimate["bytesToTransfer"] == 0 + + +def test_estimate_is_truthful_before_any_download(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + entry = distribution_entry(clip.clip_id) + expected = sum(int(asset["byteSize"]) for asset in entry["assets"]) + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert server.requests == [] # the estimate itself transfers nothing + assert estimate["strategy"] == "clip" + assert estimate["bytesToTransfer"] == expected + assert estimate["peakStorageBytes"] >= expected + assert estimate["clipRouteAvailable"] is True + assert estimate["storageOk"] is True + suite.ensure_suite_clip(clip, cache_root=cache) + assert len(server.requests) == len(entry["assets"]) # one GET per declared asset + + +def test_full_suite_only_downloads_missing_clips(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + suite.ensure_suite_clip(clip, cache_root=cache) + server.requests.clear() + server.range_headers.clear() + prepared = suite.ensure_suite(manifest, cache_root=cache) + assert len(prepared) == len(manifest.clips) + served_ids = {path.lstrip("/").split("--", 1)[0] for path in server.requests} + assert clip.clip_id not in served_ids + assert served_ids == {other.clip_id for other in manifest.clips if other.clip_id != clip.clip_id} + + +def test_truncated_response_never_installs_and_resumes_with_range(clip_env, monkeypatch): + server, cache, clip, manifest, clip_bytes = clip_env + monkeypatch.setenv(suite.SUITE_ALLOW_FULL_PACK_ENV, "0") # surface the clip-route error + entry = distribution_entry(clip.clip_id) + clip_asset = next(asset for asset in entry["assets"] if asset["role"] == "clip") + server.short_paths.add("/" + clip_asset["downloadName"]) + with pytest.raises(RuntimeError, match="size mismatch"): + suite.ensure_suite_clip(clip, cache_root=cache) + target = Path(cache) / "canonical" / clip.file_name + part = Path(str(target) + ".part") + assert not target.exists() + assert not part.exists() # corrupt/incomplete bytes are never retained + # A retained partial from an interrupted run must resume via Range. + target.parent.mkdir(parents=True, exist_ok=True) + part.write_bytes(clip_bytes[: len(clip_bytes) // 2]) + server.short_paths.clear() + server.requests.clear() + server.range_headers.clear() + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + assert Path(prepared.path).read_bytes() == clip_bytes + resumed = [path for path, header in server.range_headers if header.startswith("bytes=")] + assert any(clip.file_name in path for path in resumed) + + +def test_corrupt_clip_response_never_installs(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + server = None + try: + add_notices(root, manifest) + distribution_path = suite.write_clip_distribution_metadata(str(root)) + metadata = json.loads(Path(distribution_path).read_text()) + clip = manifest.clips[0] + corrupt = [asset["path"] for asset in metadata["clips"][clip.clip_id]["assets"] if asset["role"] == "clip"] + server = ClipServer(root, metadata, corrupt_paths=corrupt) + shutil.rmtree(root / "canonical") + cache = str(tmp_path / "cache") + with mock.patch.dict(os.environ, {suite.SUITE_CLIP_BASE_URL_ENV: server.base_url, + suite.SUITE_ALLOW_FULL_PACK_ENV: "0"}): + with pytest.raises(RuntimeError, match="checksum mismatch"): + suite.ensure_suite_clip(clip, cache_root=cache) + target = Path(cache) / "canonical" / clip.file_name + assert not target.exists() + assert not Path(str(target) + ".part").exists() + finally: + if server is not None: + server.close() + outer.__exit__(None, None, None) + + +def test_full_pack_fallback_requires_explicit_size_disclosure(clip_env, capsys): + server, cache, clip, manifest, clip_bytes = clip_env + clip_error = "forced per-clip failure" + with mock.patch.dict(os.environ, {suite.SUITE_ALLOW_FULL_PACK_ENV: "0"}), \ + mock.patch.object(suite, "_materialize_clip_from_distribution", side_effect=RuntimeError(clip_error)): + with pytest.raises(RuntimeError, match="disabled") as blocked: + suite.ensure_suite_clip(clip, cache_root=cache) + assert suite.SUITE_ALLOW_FULL_PACK_ENV in str(blocked.value) + + pack_bytes = int(suite.load_suite_pack_metadata()["distribution"]["byteSize"]) + + def fake_pack(clip_arg, pack_metadata, cache_root=None): + target = Path(suite.clip_cache_path(clip_arg, cache_root)) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(clip_bytes) + return str(target) + + with mock.patch.object(suite, "_materialize_clip_from_distribution", side_effect=RuntimeError(clip_error)), \ + mock.patch.object(suite, "_materialize_clip_from_suite_pack", side_effect=fake_pack) as pack: + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + pack.assert_called_once() + assert Path(prepared.path).read_bytes() == clip_bytes + err = capsys.readouterr().err + assert f"{pack_bytes:,} bytes" in err # size disclosed before the pack transfer + assert clip_error in err + + +def test_full_pack_compatibility_is_preserved_by_default(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + clip = manifest.clips[0] + clip_bytes = (root / "canonical" / clip.file_name).read_bytes() + shutil.rmtree(root / "canonical") # distributed tree: no packaged media + cache = str(tmp_path / "cache") + + def fake_pack(clip_arg, pack_metadata, cache_root=None): + target = Path(suite.clip_cache_path(clip_arg, cache_root)) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(clip_bytes) + return str(target) + + with mock.patch.object(suite, "_materialize_clip_from_suite_pack", side_effect=fake_pack) as pack: + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + pack.assert_called_once() # no clip-distribution.json -> pack route unchanged + assert Path(prepared.path).read_bytes() == clip_bytes + finally: + outer.__exit__(None, None, None) + + +def test_protected_cache_reports_clear_error_without_partial_install(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + if os.name == "nt": + pytest.skip("POSIX permission semantics") + locked = Path(cache) / "canonical" + locked.mkdir(parents=True) + locked.chmod(stat.S_IRUSR | stat.S_IXUSR) + try: + with pytest.raises(RuntimeError, match="not writable|write-protected"): + suite.ensure_suite_clip(clip, cache_root=cache) + finally: + locked.chmod(stat.S_IRWXU) + assert not list(locked.glob("*.part")) + assert not list(locked.glob(clip.file_name)) + assert server.requests == [] # blocked before any transfer + + +def test_disk_exhaustion_gate_stops_before_any_transfer(clip_env, monkeypatch): + server, cache, clip, manifest, clip_bytes = clip_env + monkeypatch.setenv(suite.SUITE_MIN_FREE_MB_ENV, "4194304") # 4 TiB floor + with pytest.raises(RuntimeError, match="bytes free"): + suite.ensure_suite_clip(clip, cache_root=cache) + assert server.requests == [] + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert estimate["storageOk"] is False + + +def test_tampered_clip_distribution_fails_closed(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + add_notices(root, manifest) + path = suite.write_clip_distribution_metadata(str(root)) + payload = json.loads(Path(path).read_text()) + clip = manifest.clips[0] + payload["clips"][clip.clip_id]["assets"][0]["sha256"] = "0" * 64 + Path(path).write_text(json.dumps(payload)) + with pytest.raises(RuntimeError, match="identity mismatch"): + suite.load_clip_distribution_metadata() + finally: + outer.__exit__(None, None, None) + + +def test_frozen_clip_distribution_matches_repo_suite_locks(): + metadata = suite.load_clip_distribution_metadata() + assert metadata is not None + assert metadata["source"] == "staged-unpublished" + assert metadata["distribution"]["published"] is False + assert metadata["distribution"]["baseUrl"] == "" + frozen = suite.load_default_suite_manifest() + lock = json.loads(Path(suite.get_suite_lock_path()).read_text()) + # The lock binds the manifest's canonical JSON, not its file bytes. + suite_root = Path(suite.get_suite_lock_path()).parent + manifest_payload = json.loads((suite_root / "manifest.json").read_text()) + assert suite._sha256_text(suite._canonical_json(manifest_payload)) == lock["manifestSha256"] + lock_clips = {entry["id"]: entry for entry in lock["clips"]} + for clip in frozen.clips: + entry = metadata["clips"][clip.clip_id] + clip_asset = next(asset for asset in entry["assets"] if asset["role"] == "clip") + assert clip_asset["sha256"] == clip.sha256 == lock_clips[clip.clip_id]["sha256"] + assert clip_asset["byteSize"] == clip.byte_size == lock_clips[clip.clip_id]["byteSize"] + assert clip_asset["downloadName"] == f"{clip.clip_id}--{clip.file_name}" + roles = [asset["role"] for asset in entry["assets"]] + assert roles.count("clip") == 1 + assert "notice" in roles and "license" in roles + + +def test_full_pack_estimate_counts_one_shared_download(tmp_path): + manifest = suite.load_default_suite_manifest() + clips = manifest.clips[:2] + with mock.patch.object(suite, "_packaged_canonical_path", return_value=None), \ + mock.patch.object(suite, "_pack_state", return_value={"packCached": False, "extracted": False}): + estimate = suite.acquisition_estimate( + [clip.clip_id for clip in clips], cache_root=str(tmp_path), manifest=manifest, + ) + pack_bytes = estimate["fullPackBytes"] + assert estimate["strategy"] == "pack" + assert estimate["bytesToTransfer"] == pack_bytes + assert estimate["peakStorageBytes"] == 2 * pack_bytes + sum(clip.byte_size for clip in clips) + assert estimate["clips"][clips[0].clip_id]["transferBytes"] == pack_bytes + assert estimate["clips"][clips[1].clip_id]["transferBytes"] == 0 + + +def test_release_clip_bundle_uses_flat_unique_asset_names(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + add_notices(root, manifest) + bundle = tmp_path / "bundle" + with mock.patch.object(suite, "load_default_suite_manifest", return_value=manifest): + build_clip_bundle(source_suite_dir=root, bundle_out=bundle) + metadata = json.loads((bundle / "clip-distribution.json").read_text()) + names = [asset["downloadName"] for entry in metadata["clips"].values() + for asset in entry["assets"]] + assert len(names) == len(set(names)) + assert {path.name for path in bundle.iterdir()} == set(names) | {"clip-distribution.json"} + assert all(path.is_file() for path in bundle.iterdir()) + finally: + outer.__exit__(None, None, None) diff --git a/client/tests/test_transport_lifecycle.py b/client/tests/test_transport_lifecycle.py new file mode 100644 index 00000000..db32a1fb --- /dev/null +++ b/client/tests/test_transport_lifecycle.py @@ -0,0 +1,621 @@ +"""R03/R04/R05 (September 28 audit): real cancellation/deadline propagation through +transport admission, bounded error-body consumption, and owned-worker lifecycle. + +Failing-before regressions against the September 28 code: +- R03: spool admission (hash/copy/lock) ignores cancel/deadline; artifact + transport has no caller deadline; legacy submit has no caller deadline. +- R04: artifacts.py reads error bodies with deadline=None; the byte cap only + limits retained text while consumption continues; blocked/drip-fed bodies + ignore cancellation; responses are never closed. +- R05: _run_cancellable abandons daemon workers while host-phase exclusion + ownership is released immediately, so a cancelled worker can still perform + I/O after a collector could start. Release must reap owned workers or + retain exclusion ownership until quiescent. +""" +import hashlib +import json +import os +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import ClassVar, List, Optional + +import pytest + +from client import network, spool +from client.artifacts import (AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND, + submit_artifact_submission) +from client.network import SubmissionCancelled, SubmitError + + +# -------------------------------------------------------------------------- +# fault servers +# -------------------------------------------------------------------------- + +class _FaultServer(ThreadingHTTPServer): + daemon_threads = True + allow_reuse_address = True + + +class _FaultHandler(BaseHTTPRequestHandler): + """Configurable fault server: stalls, blocked/drip/oversized error bodies.""" + protocol_version = "HTTP/1.1" + + mode: ClassVar[str] = "ok" + stall: ClassVar[Optional[threading.Event]] = None + first_body_byte: ClassVar[Optional[threading.Event]] = None + seen: ClassVar[List[str]] = [] + body_bytes_written: ClassVar[int] = 0 + client_closed_early: ClassVar[bool] = False + drip_seconds: ClassVar[float] = 5.0 + + def _read_body(self) -> bytes: + n = int(self.headers.get("Content-Length", "0")) + return self.rfile.read(n) if n else b"" + + def _status(self, code: int, headers: Optional[dict] = None) -> None: + self.send_response(code) + for key, value in (headers or {}).items(): + self.send_header(key, str(value)) + + def _end(self) -> None: + self.end_headers() + self.wfile.flush() + + def _blocked_body(self, code: int) -> None: + """Headers + a few bytes, then hold the socket silent for 10 s.""" + self._status(code, {"Content-Type": "text/plain", "Content-Length": "100000"}) + self._end() + try: + self.wfile.write(b"0123456789") + self.wfile.flush() + except OSError: + self.close_connection = True + return + if type(self).first_body_byte is not None: + type(self).first_body_byte.set() + time.sleep(10.0) + self.close_connection = True + + def _drip_body(self, code: int, extra_headers: Optional[dict] = None) -> None: + """Chunked drip: 4 KiB every 50 ms for drip_seconds.""" + self._status(code, {"Content-Type": "text/plain", + "Transfer-Encoding": "chunked", **(extra_headers or {})}) + self._end() + end = time.time() + type(self).drip_seconds + try: + while time.time() < end: + chunk = b"x" * 4096 + self.wfile.write(f"{len(chunk):x}\r\n".encode() + chunk + b"\r\n") + self.wfile.flush() + type(self).body_bytes_written += len(chunk) + time.sleep(0.05) + self.wfile.write(b"0\r\n\r\n") + self.wfile.flush() + except (BrokenPipeError, ConnectionResetError, OSError): + type(self).client_closed_early = True + self.close_connection = True + + def _oversized_body(self, code: int, total: int = 8 << 20) -> None: + """Fixed-length 8 MiB body written only as fast as the client reads.""" + self._status(code, {"Content-Type": "text/plain", "Content-Length": str(total)}) + self._end() + written = 0 + try: + while written < total: + chunk = b"y" * 65536 + self.wfile.write(chunk) + self.wfile.flush() + written += len(chunk) + type(self).body_bytes_written = written + except (BrokenPipeError, ConnectionResetError, OSError): + type(self).client_closed_early = True + self.close_connection = True + + def _json(self, code: int, value: dict) -> None: + body = json.dumps(value).encode() + self._status(code, {"Content-Type": "application/json", + "Content-Length": str(len(body))}) + self._end() + try: + self.wfile.write(body) + self.wfile.flush() + except OSError: + self.close_connection = True + + def _stall(self) -> None: + if type(self).stall is not None: + type(self).stall.wait(30) + + def do_GET(self) -> None: + type(self).seen.append(f"GET {self.path}") + self._json(404, {"error": "unused"}) + + def do_POST(self) -> None: + body = self._read_body() + type(self).seen.append(f"POST {self.path}") + mode = type(self).mode + if self.path == "/submit": + if mode == "stall-submit": + self._stall() + self._json(200, {"ok": True}) + else: + self._json(200, {"ok": True}) + return + if self.path == "/v7/benchmark-runs": + if mode == "stall-create": + self._stall() + elif mode == "blocked-create-body": + self._blocked_body(500) + return + elif mode == "drip-create-5xx": + self._drip_body(503) + return + elif mode == "drip-create-429": + self._drip_body(429, {"Retry-After": "7"}) + return + elif mode == "drip-create-400": + self._drip_body(400) + return + elif mode == "oversized-create": + self._oversized_body(500) + return + self._json(201, {"benchmarkRun": {"id": "run-lifecycle"}, + "artifact": {"storageState": "PENDING"}, "analyses": []}) + return + if self.path.endswith("/upload-authorizations"): + if mode == "stall-auth": + self._stall() + self._json(200, {"uploadRequired": True, "token": "tok-lifecycle"}) + return + self._json(404, {"error": "unused"}) + + def do_PUT(self) -> None: + remaining = int(self.headers.get("Content-Length", "0")) + while remaining > 0: + chunk = self.rfile.read(min(65536, remaining)) + if not chunk: + break + remaining -= len(chunk) + type(self).seen.append("PUT") + self._json(200, {"benchmarkRun": {"id": "run-lifecycle"}}) + + def log_message(self, format: str, *args) -> None: # noqa: A003 + return + + +def _start(handler): + server = _FaultServer(("127.0.0.1", 0), handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + return server, f"http://127.0.0.1:{server.server_port}" + + +@pytest.fixture(autouse=True) +def _reset_fault_state(): + _FaultHandler.mode = "ok" + _FaultHandler.stall = None + _FaultHandler.first_body_byte = None + _FaultHandler.seen = [] + _FaultHandler.body_bytes_written = 0 + _FaultHandler.client_closed_early = False + _FaultHandler.drip_seconds = 5.0 + yield + + +def _submission(tmp_path): + path = tmp_path / "artifact.mp4" + path.write_bytes(b"C" * (1 << 16)) + digest = hashlib.sha256(path.read_bytes()).hexdigest() + create = {"campaignId": "c-lifecycle", "payloadHash": "h-1", + "artifact": {"sha256": digest, "byteSize": path.stat().st_size}} + return {"artifactPath": str(path), "contentType": "video/mp4", + "submissionKind": AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND, + "artifactSha256": digest, "artifactByteSize": path.stat().st_size, + "runCreate": create} + + +def _wait_until(predicate, timeout=5.0): + end = time.monotonic() + timeout + while time.monotonic() < end: + if predicate(): + return True + time.sleep(0.02) + return False + + +# -------------------------------------------------------------------------- +# R04 — error-body bounds: byte cap, real deadline, cancellation, close +# -------------------------------------------------------------------------- + +def test_oversized_error_body_stops_at_byte_cap_and_closes(tmp_path): + """An 8 MiB 5xx body must be consumed only up to the hard read cap; the + current code keeps draining the whole body after the retained-text cap.""" + _FaultHandler.mode = "oversized-create" + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=20.0) + assert time.monotonic() - started < 15.0 + assert raised.value.status_code == 500 + assert raised.value.retryable + # Consumption is bounded: the client stopped reading at the cap and + # closed, so the server could not write the full 8 MiB. + assert _wait_until(lambda: _FaultHandler.client_closed_early, timeout=5.0), \ + "client never closed the oversized response" + assert _FaultHandler.body_bytes_written <= (65536 + (3 << 20)) + finally: + server.server_close() + + +def test_drip_fed_error_body_stops_at_phase_deadline(tmp_path): + """A 5 s drip must still stop at the phase deadline. The current code + passes deadline=None and waits out the whole drip.""" + _FaultHandler.mode = "drip-create-5xx" + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=1.0) + elapsed = time.monotonic() - started + assert elapsed < 3.0, f"drip-fed body was consumed for {elapsed:.1f}s" + assert raised.value.retryable + finally: + server.server_close() + + +def test_permanent_rejection_with_endless_body_is_bounded(tmp_path): + """400 with an endless chunked body: still a bounded, permanent verdict.""" + _FaultHandler.mode = "drip-create-400" + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=1.0) + elapsed = time.monotonic() - started + assert elapsed < 3.0, f"permanent-rejection body ran for {elapsed:.1f}s" + assert raised.value.retryable is False + assert raised.value.status_code == 400 + finally: + server.server_close() + + +def test_429_retry_after_survives_hostile_body(tmp_path): + """429 + Retry-After must be preserved even when the body is hostile.""" + _FaultHandler.mode = "drip-create-429" + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=1.0) + elapsed = time.monotonic() - started + assert elapsed < 3.0, f"429 body ran for {elapsed:.1f}s" + assert raised.value.retryable + assert raised.value.retry_after == pytest.approx(7.0) + finally: + server.server_close() + + +def test_cancel_during_blocked_error_body_is_bounded(tmp_path): + """Cancel must interrupt a blocked error-body read within ~poll, not wait + for the socket timeout; the old code only checked cancel between chunks.""" + _FaultHandler.mode = "blocked-create-body" + arrived = threading.Event() + _FaultHandler.first_body_byte = arrived + server, base_url = _start(_FaultHandler) + try: + cancel = threading.Event() + outcome: List[BaseException] = [] + + def _call() -> None: + try: + submit_artifact_submission(base_url, _submission(tmp_path), + create_seconds=8.0, cancel_event=cancel) + except BaseException as exc: # noqa: BLE001 - relayed below + outcome.append(exc) + + caller = threading.Thread(target=_call, daemon=True) + started = time.monotonic() + caller.start() + assert arrived.wait(5), "server never sent the first blocked-body byte" + cancel.set() + caller.join(5) + elapsed = time.monotonic() - started + assert not caller.is_alive(), "cancel never interrupted the blocked read" + assert elapsed < 2.5, f"cancel took {elapsed:.1f}s" + assert len(outcome) == 1 and isinstance(outcome[0], SubmissionCancelled), outcome + finally: + server.server_close() + + +def test_legacy_submit_stall_honours_caller_deadline(tmp_path): + """network.submit must accept a real caller deadline (monotonic) that + bounds a stalled POST; today there is no such parameter.""" + _FaultHandler.mode = "stall-submit" + stall = threading.Event() + _FaultHandler.stall = stall + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + network.submit(base_url, {"fps": 1}, retries=1, use_token=False, + deadline=time.monotonic() + 0.7) + assert raised.value.retryable + assert time.monotonic() - started < 3.0 + finally: + stall.set() + server.server_close() + + +# -------------------------------------------------------------------------- +# R03 — cancellation/deadline through admission, hashing, create/auth/PUT +# -------------------------------------------------------------------------- + +def test_spool_admission_honours_cancel(tmp_path): + """A cancelled Stop must not stage new work: no queue entry, no copy.""" + queue = str(tmp_path / "queue") + cancel = threading.Event() + cancel.set() + with pytest.raises(SubmissionCancelled): + spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}, cancel_event=cancel) + assert spool.count_pending_entries(queue) == 0 + + +def test_spool_admission_honours_deadline(tmp_path): + queue = str(tmp_path / "queue") + with pytest.raises(SubmitError) as raised: + spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}, + deadline=time.monotonic() - 0.001) + assert raised.value.retryable + assert spool.count_pending_entries(queue) == 0 + + +class _ReadWatchProxy: + """File proxy: invokes a callback on each read (io objects reject attrs).""" + + def __init__(self, handle, on_read) -> None: + self._handle = handle + self._on_read = on_read + + def read(self, *args, **kwargs): + self._on_read() + return self._handle.read(*args, **kwargs) + + def __getattr__(self, name): + return getattr(self._handle, name) + + def __enter__(self): + return self + + def __exit__(self, *exc): + return self._handle.__exit__(*exc) + + +def test_spool_admission_cancel_during_staging_aborts_copy(tmp_path): + """Cancellation observed mid-copy must abort staging with no residue.""" + queue_dir = tmp_path / "queue" + queue = str(queue_dir) + source = tmp_path / "artifact.bin" + source.write_bytes(b"Z" * (4 << 20)) + digest = hashlib.sha256(source.read_bytes()).hexdigest() + payload = {"submissionKind": AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND, + "artifactPath": str(source), "artifactSha256": digest, + "artifactByteSize": source.stat().st_size, + "runCreate": {"artifact": {"sha256": digest, + "byteSize": source.stat().st_size}}} + cancel = threading.Event() + real_open = open + state = {"n": 0} + + def on_read(): + state["n"] += 1 + if state["n"] == 2: + cancel.set() + + def watched_open(file, *args, **kwargs): + handle = real_open(file, *args, **kwargs) + if str(file) == str(source): + return _ReadWatchProxy(handle, on_read) + return handle + + with pytest.MonkeyPatch.context() as patch: + # `open` lives in builtins; injecting it into module globals shadows it + # for the staging copy loop's name resolution. + patch.setattr(spool, "open", watched_open, raising=False) + with pytest.raises(SubmissionCancelled): + spool.spool_payload(queue, payload, cancel_event=cancel) + assert spool.count_pending_entries(queue) == 0 + staged = list((queue_dir / "artifacts").glob("*")) if (queue_dir / "artifacts").exists() else [] + assert staged == [] + + +def test_submit_spooled_path_deadline_prevents_network(tmp_path): + """A timed-out call must not perform I/O: no request leaves the host.""" + server, base_url = _start(_FaultHandler) + queue = str(tmp_path / "queue") + path, _entry = spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}) + try: + status, _message = spool.submit_spooled_path( + path, queue_dir=queue, base_url=base_url, api_key="", retries=1, + use_token=False, deadline=time.monotonic() - 0.001) + assert status == "retained" + assert _FaultHandler.seen == [] + assert spool.count_pending_entries(queue) == 1 + finally: + server.server_close() + + +def test_replay_spool_deadline_admits_nothing(tmp_path): + server, base_url = _start(_FaultHandler) + queue = str(tmp_path / "queue") + spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}) + try: + stats = spool.replay_spool(queue, base_url=base_url, api_key="", retries=1, + use_token=False, deadline=time.monotonic() - 0.001) + assert stats.submitted == 0 + assert _FaultHandler.seen == [] + assert spool.count_pending_entries(queue) == 1 + finally: + server.server_close() + + +def test_artifact_transport_external_deadline_during_auth(tmp_path): + """A caller deadline must stop a stalled auth phase (and never reach PUT).""" + _FaultHandler.mode = "stall-auth" + stall = threading.Event() + _FaultHandler.stall = stall + server, base_url = _start(_FaultHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as raised: + submit_artifact_submission(base_url, _submission(tmp_path), + deadline=time.monotonic() + 0.8) + assert time.monotonic() - started < 3.0 + assert raised.value.retryable + assert "PUT" not in _FaultHandler.seen + finally: + stall.set() + server.server_close() + + +def test_checkpoint_replay_cancel_keeps_entry(tmp_path): + """Stop during a checkpoint upload: entry retained, pass ends promptly.""" + _FaultHandler.mode = "stall-submit" + stall = threading.Event() + _FaultHandler.stall = stall + server, base_url = _start(_FaultHandler) + queue = str(tmp_path / "queue") + path, _entry = spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}) + try: + cancel = threading.Event() + outcome = {} + + def result(): + try: + outcome["stats"] = spool.replay_spool( + queue, base_url=base_url, api_key="", retries=1, + use_token=False, cancel_event=cancel) + except BaseException as exc: # noqa: BLE001 - recorded for assertion + outcome["error"] = exc + runner = threading.Thread(target=result, daemon=True) + runner.start() + assert _wait_until(lambda: any("POST" in s for s in _FaultHandler.seen)) + cancel.set() + runner.join(8) + assert not runner.is_alive(), "replay did not stop within 8s of cancel" + assert outcome.get("error") is None + assert spool.count_pending_entries(queue) == 1 + assert os.path.isfile(path) + finally: + stall.set() + server.server_close() + + +# -------------------------------------------------------------------------- +# R05 — owned lifecycle: reap on release, or retain exclusion until quiescent +# -------------------------------------------------------------------------- + +def test_hold_release_reaps_owned_worker_before_returning(tmp_path, monkeypatch): + """Ownership release must not return while an owned worker is still doing + I/O: the releasing call reaps it (bounded sync join) before finishing.""" + monkeypatch.setattr(network, "WORKER_QUIESCE_SYNC_JOIN_SECONDS", 10.0, raising=False) + monkeypatch.setenv("ENCODINGDB_HOST_PHASE_DIR", str(tmp_path / "phase")) + _FaultHandler.mode = "stall-submit" + stall = threading.Event() + _FaultHandler.stall = stall + server, base_url = _start(_FaultHandler) + queue = str(tmp_path / "queue") + path, _entry = spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}) + results = {} + cancel = threading.Event() + try: + def body(): + try: + results["value"] = spool.submit_spooled_path( + path, queue_dir=queue, base_url=base_url, api_key="", + retries=1, use_token=False, cancel_event=cancel) + except BaseException as exc: # noqa: BLE001 - recorded for assertion + results["error"] = exc + runner = threading.Thread(target=body, daemon=True) + runner.start() + assert _wait_until(lambda: any("POST" in s for s in _FaultHandler.seen)) + cancel.set() + # The caller has raised, but the owned worker is still stalled inside + # its request: the releasing call must NOT have returned yet. + time.sleep(0.5) + assert runner.is_alive(), "release returned while an owned worker was still doing I/O" + stall.set() + runner.join(10) + assert not runner.is_alive() + assert results.get("error") is None + assert results["value"][0] == "retained" + assert network.owned_worker_census() == 0 + assert spool.count_pending_entries(queue) == 1 # cancelled -> retained + finally: + stall.set() + server.server_close() + + +def test_nonquiescent_release_retains_exclusion(tmp_path, monkeypatch): + """If reaping is not yet complete, the host phase exclusion must stay held + until the owned worker is quiescent — a collector cannot start while a + cancelled worker still owns live transport I/O.""" + monkeypatch.setattr(network, "WORKER_QUIESCE_SYNC_JOIN_SECONDS", 0.1, raising=False) + phase_dir = str(tmp_path / "phase") + monkeypatch.setenv("ENCODINGDB_HOST_PHASE_DIR", phase_dir) + _FaultHandler.mode = "stall-submit" + stall = threading.Event() + _FaultHandler.stall = stall + server, base_url = _start(_FaultHandler) + queue = str(tmp_path / "queue") + path, _entry = spool.spool_payload(queue, {"cpuModel": "A", "fps": 1}) + results = {} + cancel = threading.Event() + try: + def body(): + try: + results["value"] = spool.submit_spooled_path( + path, queue_dir=queue, base_url=base_url, api_key="", + retries=1, use_token=False, cancel_event=cancel) + except BaseException as exc: # noqa: BLE001 - recorded for assertion + results["error"] = exc + runner = threading.Thread(target=body, daemon=True) + runner.start() + assert _wait_until(lambda: any("POST" in s for s in _FaultHandler.seen)) + cancel.set() + # Drain grace (0.1 s) expires while the worker is still stalled; the + # releasing call returns, but exclusion must NOT be released yet. + runner.join(5) + assert not runner.is_alive(), "release should not block past the drain grace" + assert results.get("error") is None + assert network.owned_worker_census() >= 1 + + probe = {} + + def probe_phase(): + try: + with spool.host_phase_hold("measurement"): + probe["acquired"] = True + except spool.SpoolCapacityError: + probe["acquired"] = False + watcher = threading.Thread(target=probe_phase, daemon=True) + watcher.start() + watcher.join(5) + assert probe.get("acquired") is False, ( + "host phase exclusion was released while an owned worker still " + "performed transport I/O") + + stall.set() + assert _wait_until(lambda: network.owned_worker_census() == 0, timeout=10) + assert _wait_until(lambda: not spool.host_phase_busy(), timeout=10) + # The cancelled worker finished its single in-flight request; it must + # not start new I/O after release. + assert _FaultHandler.seen.count("POST /submit") == 1 + finally: + stall.set() + server.server_close() + + +def test_census_reports_no_owned_workers_without_activity(): + assert network.owned_worker_census() == 0 \ No newline at end of file diff --git a/client/tests/test_transport_review.py b/client/tests/test_transport_review.py new file mode 100644 index 00000000..3b6872c8 --- /dev/null +++ b/client/tests/test_transport_review.py @@ -0,0 +1,153 @@ +"""Independent review regressions for transport worker ownership.""" + +import threading +import time + +import pytest + +from client import network, spool + + +def test_body_reader_never_consumes_past_its_hard_cap(): + class Response: + def __init__(self): + self.consumed = 0 + self.closed = False + + def iter_content(self, chunk_size): + while True: + self.consumed += chunk_size + yield b"x" * chunk_size + + def close(self): + self.closed = True + + response = Response() + network._read_response_body(None, response, None, time.monotonic() + 2, + max_bytes=65536) + assert response.consumed <= 65536 + assert response.closed + + +def test_second_publication_retains_host_lock_for_its_straggler(tmp_path, monkeypatch): + monkeypatch.setenv("ENCODINGDB_HOST_PHASE_DIR", str(tmp_path)) + with spool.host_phase_hold("publication"): + pass # The first phase must not own workers started by a later phase. + + stop = threading.Event() + started = threading.Event() + finish = threading.Event() + + def blocked(): + started.set() + finish.wait(5) + return None + + def cancel_after_start(): + assert started.wait(1) + stop.set() + + canceller = threading.Thread(target=cancel_after_start) + canceller.start() + try: + with pytest.raises(network.SubmissionCancelled): + with spool.host_phase_hold("publication"): + network._run_cancellable( + blocked, phase="review-blocked", cancel_event=stop, + deadline=time.monotonic() + 5, bound_seconds=5, + ) + assert network.owned_worker_census() >= 1 + with pytest.raises(spool.SpoolCapacityError): + with spool.host_phase_hold("measurement"): + pass + finally: + finish.set() + canceller.join(1) + deadline = time.monotonic() + 2 + while network.owned_worker_census() and time.monotonic() < deadline: + time.sleep(0.01) + assert network.owned_worker_census() == 0 + + +def test_abandoned_response_is_closed_when_worker_finishes(tmp_path, monkeypatch): + monkeypatch.setenv("ENCODINGDB_HOST_PHASE_DIR", str(tmp_path)) + started = threading.Event() + finish = threading.Event() + stop = threading.Event() + + class Response: + closed = False + + def close(self): + self.closed = True + + response = Response() + + def late_response(): + started.set() + finish.wait(5) + return response + + def cancel_after_start(): + assert started.wait(1) + stop.set() + + canceller = threading.Thread(target=cancel_after_start) + canceller.start() + try: + with pytest.raises(network.SubmissionCancelled): + with spool.host_phase_hold("publication"): + network._run_cancellable( + late_response, phase="review-response", cancel_event=stop, + deadline=time.monotonic() + 5, bound_seconds=5, + ) + finally: + finish.set() + canceller.join(1) + deadline = time.monotonic() + 2 + while not response.closed and time.monotonic() < deadline: + time.sleep(0.01) + assert response.closed + + +def test_host_lock_covers_late_response_close_io(tmp_path, monkeypatch): + monkeypatch.setenv("ENCODINGDB_HOST_PHASE_DIR", str(tmp_path)) + started = threading.Event() + stop = threading.Event() + return_response = threading.Event() + closing = threading.Event() + allow_close = threading.Event() + + class Response: + def close(self): + closing.set() + allow_close.wait(5) + + def late_response(): + started.set() + return_response.wait(5) + return Response() + + def release_after_cancel(): + assert started.wait(1) + stop.set() + time.sleep(0.1) + return_response.set() + + releaser = threading.Thread(target=release_after_cancel) + releaser.start() + try: + with pytest.raises(network.SubmissionCancelled): + with spool.host_phase_hold("publication"): + network._run_cancellable( + late_response, phase="review-close-io", cancel_event=stop, + deadline=time.monotonic() + 5, bound_seconds=5, + ) + assert closing.wait(2) + with pytest.raises(spool.SpoolCapacityError): + with spool.host_phase_hold("measurement"): + pass + finally: + allow_close.set() + return_response.set() + releaser.join(1) diff --git a/client/tests/test_windows_gui.py b/client/tests/test_windows_gui.py index 5178a33b..607b6183 100644 --- a/client/tests/test_windows_gui.py +++ b/client/tests/test_windows_gui.py @@ -37,12 +37,24 @@ def set(self, value): self.value = value +def cached_estimate(_mode): + return {"strategy": "cache", "storageOk": True, "bytesToTransfer": 0, + "peakStorageBytes": 0, "warnings": []} + + +def cached_recovery_state(_queue=None): + return {"publicationConsent": False, "campaigns": [], + "publication": {"pendingEntries": 0, "dueEntries": 0, + "acceptedReceipts": 0, "terminalEntries": 0}} + + class Widget: def __init__(self, *args, **kwargs): self.variable = kwargs.get("textvariable") self.values = kwargs.get("values", []) self.index = -1 self.options = kwargs + self.visible = False def __setitem__(self, key, value): if key == "values": @@ -61,7 +73,10 @@ def configure(self, **kwargs): self.options.update(kwargs) def pack(self, **kwargs): - pass + self.visible = True + + def pack_forget(self): + self.visible = False def grid(self, **kwargs): pass @@ -113,9 +128,12 @@ def test_initialized_controls_and_keyboard_handlers_use_real_running_guards(self with mock.patch.dict(sys.modules, {"tkinter": tk}), mock.patch.object(gui.os, "name", "nt"), \ mock.patch.object(gui, "list_all_available_encoders", return_value=["libx265", "libx264"]), \ mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state), \ mock.patch.object(gui.threading, "Thread") as thread: self.assertEqual(gui.launch_windows_gui(self.args()), 0) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state self.assertEqual((app._selected_encoder(), app._selected_preset(), app.crf_var.get()), ("libx264", "fast", 0)) app._handle_event({"type": "preparation_progress", "stage": "probe", "path": "test.mkv"}) self.assertEqual(app.stage_var.get(), "Preparing: probe") @@ -142,7 +160,7 @@ def test_initialized_controls_and_keyboard_handlers_use_real_running_guards(self self.assertEqual(app.start_btn.options["state"], "disabled") bindings[""](None) self.assertTrue(app.cancel_event.is_set()) - self.assertEqual(app.summary_var.get(), "Stopping owned work; retaining downloads and campaign...") + self.assertEqual(app.summary_var.get(), "Stopping owned work; retaining saved results...") def _build_app(self, mode: str = "Small"): """Instantiate the Tk app against the mocked widget harness.""" @@ -160,9 +178,12 @@ def _build_app(self, mode: str = "Small"): mock.patch.object(gui, "list_all_available_encoders", return_value=["libx264", "h264_videotoolbox"]), \ mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state), \ mock.patch.object(gui.threading, "Thread"): gui.launch_windows_gui(self.args()) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state app.mode_var.set(mode) return app @@ -172,11 +193,13 @@ def test_mode_choices_expose_shared_sweeps_with_single_as_advanced(self): self.assertEqual(app.mode_var.get(), "Small") self.assertIn("native recipes", app.summary_var.get()) self.assertIn("quick clip", app.summary_var.get()) + self.assertFalse(app.advanced_frame.visible) app.mode_var.set("Full") app._update_single_fields_state() self.assertIn("all seven frozen clips", app.summary_var.get()) app.mode_var.set("Single (advanced)") app._update_single_fields_state() + self.assertTrue(app.advanced_frame.visible) self.assertEqual(app.encoder_combo.options["state"], "readonly") def test_sweep_worker_dispatches_to_shared_planner_run(self): @@ -208,8 +231,9 @@ def test_upload_retry_never_encodes(self): from client import main as client_main with mock.patch.object(client_main, "count_pending_entries", side_effect=[2, 0]), \ - mock.patch.object(client_main, "replay_spool", - return_value=mock.Mock(dead_lettered=0, corrupt=0)) as replay_mock, \ + mock.patch.object(client_main, "retry_due_uploads", + return_value=(0, {"status": "published", "pending": 0, + "deadLettered": 0, "corrupt": 0})) as replay_mock, \ mock.patch.object(client_main, "run_sweep_mode") as sweep_mock, \ mock.patch.object(client_main, "run_with_args") as single_mock: app._retry_uploads_worker("queue-dir", "https://example.invalid", "", 2) @@ -266,11 +290,14 @@ def build(self, mode: str = "Small"): setattr(tk.ttk, name, Widget) tk.scrolledtext.ScrolledText = Widget with mock.patch.dict(sys.modules, {"tkinter": tk}), mock.patch.object(gui.os, "name", "nt"), \ - mock.patch.object(gui, "list_all_available_encoders", + mock.patch.object(gui, "list_all_available_encoders", return_value=["libx264", "h264_videotoolbox"]), \ - mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)): + mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state): gui.launch_windows_gui(self.args()) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state app.mode_var.set(mode) return app, root, bindings, tk @@ -280,7 +307,7 @@ def states(self, app): def test_idle_states_then_active_benchmark_locks_retry_and_start(self): app, _root, bindings, _tk = self.build() - self.assertEqual(self.states(app), ("normal", "disabled", "normal")) + self.assertEqual(self.states(app), ("normal", "disabled", "disabled")) with mock.patch.object(gui.threading, "Thread", FakeThread): bindings[""](None) self.assertEqual(app.worker_thread.target.__name__, "_run_worker") @@ -293,27 +320,286 @@ def test_idle_states_then_active_benchmark_locks_retry_and_start(self): app.event_queue.put(("done", 0)) app._poll_events() self.assertFalse(app.running) - self.assertEqual(self.states(app), ("normal", "disabled", "normal"), - "Retry must be available again after the run finishes") + self.assertEqual(self.states(app), ("normal", "disabled", "disabled"), + "Local-only policy keeps upload actions disabled after the run") def test_finished_run_surfaces_pending_uploads(self): app, _root, _bindings, _tk = self.build() app.no_submit_var.set(False) + app._active_no_submit = False + app.event_queue.put(("event", {"type": "submit_result", "status": "submitted"})) with mock.patch.object(gui.client_main, "count_pending_entries", return_value=2): app.event_queue.put(("done", 0)) app._poll_events() self.assertIn("2 upload(s) queued", app.summary_var.get()) - self.assertIn("Uploaded; analysis pending", app.summary_var.get()) + self.assertIn("some uploads are queued", app.summary_var.get()) self.assertEqual(app.upload_btn.options["state"], "normal") + def test_campaign_progress_owns_both_bars_on_declared_attempt_units(self): + app, _root, _bindings, _tk = self.build() + app._handle_event({"type": "run_start", "scope": "batch", "totalTasks": 30, + "declaredAttempts": 12, "totalBatches": 1, + "progressUnit": "measured-attempt", "doneTotal": 0}) + self.assertEqual(app.overall_pb.options["maximum"], 12) + app._handle_event({"type": "batch_start", "batchNo": 1, "totalBatches": 1, + "batchDeclaredAttempts": 8}) + self.assertEqual(app.batch_pb.options["maximum"], 8) + self.assertEqual(app.batch_pb.options["value"], 0) + # A per-record completion must never move batch bars toward 100%. + app._handle_event({"type": "task_complete", "scope": "batch", "processed": 5, + "total": 30}) + self.assertEqual(app.overall_pb.options["value"], 0) + self.assertEqual(app.batch_pb.options["value"], 0) + app._handle_event({"type": "campaign_progress", "scope": "batch", "unit": "measured-attempt", + "done": 4, "total": 12, "batchDone": 4, "batchTotal": 8, + "groupsTotal": 3}) + self.assertEqual(app.overall_pb.options["maximum"], 12) + self.assertEqual(app.overall_pb.options["value"], 4) + self.assertEqual(app.batch_pb.options["value"], 4) + self.assertIn("4/12", app.summary_var.get()) + + def test_resume_baseline_and_monotonic_bounds_survive_segments(self): + app, _root, _bindings, _tk = self.build() + # Segment 2 of a resume: run_start carries the durable baseline and a + # refined (smaller) declared bound; bars must not rewind below done. + app._handle_event({"type": "campaign_progress", "scope": "batch", + "done": 6, "total": 12, "batchDone": 6, "batchTotal": 6, + "groupsTotal": 3}) + app._handle_event({"type": "run_start", "scope": "batch", "totalTasks": 30, + "declaredAttempts": 9, "totalBatches": 1, + "progressUnit": "measured-attempt", "doneTotal": 6}) + self.assertEqual(app.overall_pb.options["maximum"], 9) + self.assertEqual(app.overall_pb.options["value"], 6) + # A new segment resets only the batch view. + app._handle_event({"type": "batch_start", "batchNo": 1, "totalBatches": 1, + "batchDeclaredAttempts": 3}) + self.assertEqual(app.batch_pb.options["value"], 0) + self.assertEqual(app.overall_pb.options["value"], 6) + + def test_invalid_typed_settings_leave_idle_without_worker(self): + for field, value, mode in ( + ("retries_var", "abc", "Small"), ("retries_var", "", "Small"), + ("retries_var", "11", "Small"), ("batch_size_var", " ", "Small"), + ("batch_size_var", "65", "Small"), ("crf_var", "oops", "Single (advanced)"), + ("bitrate_var", "-3", "Single (advanced)"), + ): + with self.subTest(field=field, value=value): + app, _root, _bindings, tk = self.build(mode=mode) + getattr(app, field).set(value) + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.start_btn.options["state"], "normal") + tk.messagebox.showerror.assert_called_once() + + def test_real_tcl_invalid_intvar_becomes_field_error(self): + try: + import tkinter as tkinter_real + interpreter = tkinter_real.Tcl() + except Exception as exc: + self.skipTest(f"Tcl unavailable: {exc}") + value = tkinter_real.IntVar(master=interpreter, value=3) + interpreter.setvar(value._name, "abc") + with self.assertRaises(tkinter_real.TclError): + value.get() + with self.assertRaisesRegex(ValueError, "Retries must be a whole number"): + gui._validated_integer(value, "Retries", 1, 10) + + def test_local_only_run_does_not_require_a_live_server_url(self): + app, _root, _bindings, tk = self.build() + app.base_url_var.set("unconfigured") + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertTrue(app.running) + self.assertTrue(app.worker_thread.args[0].no_submit) + tk.messagebox.showerror.assert_not_called() + + def test_download_cost_is_confirmed_before_worker_starts(self): + app, _root, _bindings, tk = self.build() + app._estimate_acquisition = lambda _mode: { + "strategy": "pack", "storageOk": True, + "bytesToTransfer": 1506890018, "peakStorageBytes": 3100000000, + } + tk.messagebox.askyesno.return_value = False + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertIn("1.4 GB", tk.messagebox.askyesno.call_args.args[1]) + self.assertIn("declined", app.summary_var.get()) + + def test_storage_estimate_refuses_before_worker(self): + app, _root, _bindings, tk = self.build() + app._estimate_acquisition = lambda _mode: { + "strategy": "clip", "storageOk": False, + "bytesToTransfer": 100, "peakStorageBytes": 5000000000, + } + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertIn("disk space", app.summary_var.get()) + tk.messagebox.showerror.assert_called_once() + + def test_acquisition_preview_uses_quick_or_complete_frozen_clip_set(self): + manifest = gui.suite.load_default_suite_manifest() + with mock.patch.object(gui.suite, "acquisition_estimate", return_value={}) as estimate: + gui._acquisition_preview("small") + quick_ids = estimate.call_args.args[0] + gui._acquisition_preview("full") + full_ids = estimate.call_args.args[0] + self.assertEqual(quick_ids, [gui.suite.get_default_quick_clip(manifest).clip_id]) + self.assertEqual(set(full_ids), {clip.clip_id for clip in manifest.clips}) + + def test_consent_save_and_thread_start_failures_restore_idle(self): + app, _root, _bindings, tk = self.build() + app.no_submit_var.set(False) + with mock.patch.object(gui.client_main, "_ensure_interactive_publication_consent", + side_effect=OSError("consent storage unavailable")): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.no_submit_var.get(), False) + tk.messagebox.showerror.assert_called_once() + + app.no_submit_var.set(True) + class FailingThread(FakeThread): + def start(self): + raise RuntimeError("thread unavailable") + with mock.patch.object(gui.threading, "Thread", FailingThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.stage_var.get(), "Idle") + self.assertEqual(app.start_btn.options["state"], "normal") + + app.base_args.max_duration_minutes_explicit = True + app.base_args.max_duration_minutes = "invalid" + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.start_btn.options["state"], "normal") + + def test_single_worker_uses_snapshot_even_if_widgets_change(self): + app, _root, _bindings, _tk = self.build(mode="Single (advanced)") + app._update_single_fields_state() + app.crf_var.set(19) + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertEqual(app.no_submit_check.options["state"], "disabled") + app.crf_var.set("broken later") + app.no_submit_var.set(False) + with mock.patch.object(gui.client_main, "run_with_args", return_value=0) as run: + app.worker_thread.target(*app.worker_thread.args) + args = run.call_args.args[0] + self.assertEqual(args.crf, 19) + self.assertTrue(args.no_submit) + app._poll_events() + self.assertIn("Saved locally", app.summary_var.get()) + + def test_local_completion_reports_saved_groups_without_upload_claim(self): + app, _root, _bindings, _tk = self.build() + app._active_no_submit = True + app.event_queue.put(("event", {"type": "submit_result", "status": "locally_complete"})) + app.event_queue.put(("event", {"type": "submit_result", "status": "locally_complete"})) + app.event_queue.put(("done", 0)) + app._poll_events() + self.assertIn("2 measured attempt(s) locally", app.summary_var.get()) + self.assertNotIn("Uploaded", app.summary_var.get()) + + def test_saved_work_is_visible_and_publish_uses_zero_encode_api(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["publicationConsent"] = True + state["publication"] = {"pendingEntries": 2, "dueEntries": 1, + "acceptedReceipts": 3, "terminalEntries": 1} + state["campaigns"] = [{"campaignId": "campaign-test", "complete": True, + "pendingUploads": 2, "unavailableSources": 0, + "actions": [{"action": "publish_saved"}]}] + app._load_recovery_state = lambda: state + app._refresh_saved_work() + app.no_submit_var.set(False) + app._refresh_controls() + self.assertIn("1 due / 1 delayed", app.saved_summary_var.get()) + self.assertIn("terminal", app.saved_summary_var.get()) + self.assertEqual(app._selected_saved_campaign(), "campaign-test") + self.assertEqual(app.publish_btn.options["state"], "normal") + with mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ + mock.patch.object(gui.threading, "Thread", FakeThread): + app._publish_saved() + self.assertEqual(app.publish_btn.options["state"], "disabled") + with mock.patch.object(gui.client_main, "publish_saved_campaign", + return_value=(0, {"submitted": 2, "pending": 0})) as publish, \ + mock.patch.object(gui.client_main, "run_with_args") as encode: + app.upload_thread.target(*app.upload_thread.args) + publish.assert_called_once() + self.assertEqual(publish.call_args.kwargs["campaign_id"], "campaign-test") + encode.assert_not_called() + self.assertIn("analysis pending", app.event_queue.get_nowait()[1]) + self.assertIn("1 not yet staged", app._publication_result_text( + 10, {"pending": 0, "unadmitted": 1, "deferredReason": "storage_or_exclusion"})) + + def test_resume_selected_campaign_keeps_its_identity(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["campaigns"] = [{"campaignId": "campaign-saved", "complete": False, + "pendingUploads": 0, "unavailableSources": 0, + "actions": [{"action": "resume"}]}] + app._load_recovery_state = lambda: state + app._refresh_saved_work() + with mock.patch.object(app, "_start_run") as start: + app._resume_saved() + start.assert_called_once_with(resume_id="campaign-saved") + args = self.args(resume_campaign="campaign-saved") + with mock.patch.object(gui.client_main, "_resume_campaign", return_value=0) as resume, \ + mock.patch.object(gui.client_main, "run_with_args") as new_run: + app._run_worker(args, "Small", "small") + resume.assert_called_once() + self.assertEqual(resume.call_args.args[0].resume_campaign, "campaign-saved") + new_run.assert_not_called() + + def test_idle_retry_requires_due_work_and_saved_consent(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["publicationConsent"] = True + state["publication"]["pendingEntries"] = 2 + state["publication"]["dueEntries"] = 1 + app._load_recovery_state = lambda: state + app.no_submit_var.set(False) + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_called_once_with(automatic=True) + state["publicationConsent"] = False + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_not_called() + state["publicationConsent"] = True + app.no_submit_var.set(True) + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_not_called() + + def test_stop_cancels_active_upload_replay(self): + app, _root, _bindings, _tk = self.build() + app.upload_thread = FakeThread() + app.upload_thread.start() + app._refresh_controls() + self.assertEqual(app.stop_btn.options["state"], "normal") + app._stop_run() + self.assertTrue(app.upload_cancel_event.is_set()) + def test_active_replay_blocks_benchmark_start_and_restores_on_status(self): app, _root, _bindings, _tk = self.build() + app.no_submit_var.set(False) with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ mock.patch.object(gui.client_main, "count_pending_entries", return_value=3), \ mock.patch.object(gui.client_main, "replay_spool"): app._retry_uploads() self.assertIsNotNone(app.upload_thread) - self.assertEqual(self.states(app), ("disabled", "disabled", "disabled")) + self.assertEqual(self.states(app), ("disabled", "normal", "disabled")) app._start_run() self.assertIsNone(app.worker_thread, "benchmark must not race a live replay") # A second click while replaying must not spawn another uploader. @@ -331,35 +617,40 @@ def test_active_replay_blocks_benchmark_start_and_restores_on_status(self): def test_close_during_replay_waits_then_closes_bounded(self): app, root, _bindings, tk = self.build() + app.no_submit_var.set(False) tk.messagebox.askyesno.return_value = True with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ mock.patch.object(gui.client_main, "count_pending_entries", return_value=1), \ mock.patch.object(gui.client_main, "replay_spool"): app._retry_uploads() uploader = app.upload_thread app._on_close() + self.assertTrue(app.upload_cancel_event.is_set()) root.destroy.assert_not_called() app._close_when_stopped() root.destroy.assert_not_called() # A slow request must not outlive an apparently closed application. - app._close_deadline = gui.time.monotonic() - 1 - log_lines = [] - with mock.patch.object(app, "_append_log", log_lines.append): - app._close_when_stopped() - root.destroy.assert_not_called() - self.assertIn("Waiting for the current operation", app.summary_var.get()) self.assertTrue(uploader.is_alive(), "uploader object unaffected by close bookkeeping") # Declining the confirmation leaves the window open. root.reset_mock() tk.messagebox.askyesno.return_value = False app._on_close() root.destroy.assert_not_called() - uploader.finish() - app._close_when_stopped() + # Once the original grace expires, do not renew it forever. A single + # confirmed Close ends the owned process and helper tree. + app._close_deadline = gui.time.monotonic() - 1 + with mock.patch.object(gui, "_terminate_owned_children") as terminate, \ + mock.patch.object(gui.os, "_exit", side_effect=SystemExit(130)) as exit_process: + with self.assertRaises(SystemExit): + app._close_when_stopped() + terminate.assert_called_once() + exit_process.assert_called_once_with(130) root.destroy.assert_called_once() def test_retry_exception_surfaces_and_restores_controls(self): app, _root, _bindings, _tk = self.build() + app.no_submit_var.set(False) with mock.patch.object(gui.client_main, "count_pending_entries", side_effect=OSError("disk gone")): app._retry_uploads_worker("queue-dir", "http://127.0.0.1:9", "", 2) app.upload_thread = FakeThread() @@ -370,7 +661,9 @@ def test_retry_exception_surfaces_and_restores_controls(self): def test_upload_status_does_not_release_a_still_running_worker(self): app, _root, _bindings, _tk = self.build() - with mock.patch.object(gui.threading, "Thread", FakeThread): + app.no_submit_var.set(False) + with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True): app._retry_uploads() uploader = app.upload_thread app.event_queue.put(("upload_status", "Upload complete")) @@ -398,6 +691,21 @@ def test_submit_result_events_get_browse_link_not_fake_run_url(self): self.assertNotIn("browse", lines[1], "browse hint appears once per run") self.assertNotIn("browse", lines[2], "queued is not a browsable success claim") + def test_submission_failure_shows_safe_cause_and_run_id(self): + app, _root, _bindings, _tk = self.build() + app._active_no_submit = False + app._handle_event({"type": "submit_result", "status": "failed", + "errorCategory": "protocol_rejected", "safeReason": "Suite identity differs", + "recoveryAction": "Update the client", "reasonCodes": ["SUITE_MISMATCH"], + "benchmarkRunId": "run-123", "error": "token=SECRET"}) + self.assertIn("Suite identity differs", app.summary_var.get()) + self.assertIn("Update the client", app.summary_var.get()) + self.assertIn("run-123", app.summary_var.get()) + self.assertNotIn("SECRET", app.summary_var.get()) + app.event_queue.put(("done", 1)) + app._poll_events() + self.assertIn("Suite identity differs", app.summary_var.get()) + def test_explicit_allowance_reported_when_set(self): app, _root, bindings, _tk = self.build() log_lines = [] diff --git a/client/ui.py b/client/ui.py index 5335c75d..51527213 100644 --- a/client/ui.py +++ b/client/ui.py @@ -288,8 +288,15 @@ def _format_duration(seconds: float) -> str: def print_end_screen(completed_count: int, elapsed_seconds: float, status: str = "complete", - recovery: Optional[str] = None) -> None: - """Render the terminal state honestly; only rc-0 work may claim completion.""" + recovery: Optional[str] = None, + ledger: Optional[Dict[str, Any]] = None) -> None: + """Render the terminal state honestly; only rc-0 work may claim completion. + + `ledger` is the durable campaign view (journal + queue). When present it + replaces the single "Submitted data points" line with distinct counters so + measured, locally-saved, queued and server-confirmed work are never + conflated; an accepted upload is analysis-pending, not an accepted result. + """ time_str = _format_duration(elapsed_seconds) headline, title, plain = { "complete": ("[ok] Benchmark run complete [/ok]", "Thank You", "Benchmark complete."), @@ -301,14 +308,49 @@ def print_end_screen(completed_count: int, elapsed_seconds: float, status: str = "Run did not complete; nothing was marked finished."), }.get(status, ("[accent] Run ended with an unknown state [/accent]", "Failed", "Run ended with an unknown state.")) + counts_line: Optional[str] = None + coverage_line: Optional[str] = None + if isinstance(ledger, dict): + uploaded = int(ledger.get("uploaded") or 0) + saved_local = int(ledger.get("savedLocal") or 0) + queued = int(ledger.get("queued") or 0) + terminal = int(ledger.get("terminalFailures") or 0) + measured = int(ledger.get("measuredAttempts") or 0) + if status == "failed" and uploaded + saved_local > 0: + headline = "[accent] Run finished with failures — retained evidence remains actionable [/accent]" + plain = "Run finished with failures; retained evidence remains actionable." + counts_line = ( + f"Measured attempts: {measured} · Saved locally: {saved_local} · " + f"Queued for upload: {queued} · Server-confirmed: {uploaded} (analysis pending) · " + f"Terminal failures: {terminal}" + ) + planned = int(ledger.get("groupsTotal") or 0) + finished = int(ledger.get("groupsFinished") or 0) + required = int(ledger.get("requiredMeasured") or 0) + optional = int(ledger.get("optionalMeasured") or 0) + if planned: + coverage_line = (f"Frozen groups: {finished}/{planned} finished · " + f"Required measured floor: {required} · " + f"Optional adaptive attempts used: {optional}") recovery_line = f"\n[muted]{recovery}[/muted]" if recovery else "" if _rich_tty(): - body = ( - f"{headline}\n\n" - f"[muted]Submitted data points:[/muted] [accent]{completed_count}[/accent]\n" - f"[muted]Time donated:[/muted] [accent2]{time_str}[/accent2]{recovery_line}" - ) + if counts_line is not None: + body = f"{headline}\n\n[muted]{counts_line}[/muted]\n" + if coverage_line: + body += f"[muted]{coverage_line}[/muted]\n" + body += f"[muted]Time donated:[/muted] [accent2]{time_str}[/accent2]{recovery_line}" + else: + body = ( + f"{headline}\n\n" + f"[muted]Submitted data points:[/muted] [accent]{completed_count}[/accent]\n" + f"[muted]Time donated:[/muted] [accent2]{time_str}[/accent2]{recovery_line}" + ) _console.print(_Panel(body, title=f"[title] {title} [/title]", border_style="accent2")) + elif counts_line is not None: + print(f"{plain} {counts_line}" + + (f" {coverage_line}." if coverage_line else "") + + f" in {time_str}." + + (f" {recovery}" if recovery else "")) else: print(f"{plain} Submitted {completed_count} data points in {time_str}." + (f" {recovery}" if recovery else "")) @@ -406,40 +448,36 @@ def print_batch_summary(summary: Dict[str, Any]) -> None: total_batches = int(summary.get("totalBatches") or 0) completed = int(summary.get("completed") or 0) submitted = int(summary.get("submitted") or 0) + locally_complete = int(summary.get("locallyComplete") or 0) skipped = int(summary.get("skipped") or 0) queued = int(summary.get("queued") or 0) failed = int(summary.get("failed") or 0) elapsed_seconds = float(summary.get("elapsedSeconds") or 0.0) throughput_per_hour = float(summary.get("throughputPerHour") or 0.0) + rows = [ + ("Planned attempts", str(total)), + *( [("Batches", str(total_batches))] if total_batches > 0 else [] ), + ("Measured processed", str(completed)), + ("Saved locally", str(locally_complete)), + ("Server-confirmed", f"{submitted} (analysis pending)"), + ("Protocol invalid", str(skipped)), + ("Queued for upload", str(queued)), + ("Terminal failures", str(failed)), + ] + if elapsed_seconds > 0: + rows.append(("Elapsed", _format_duration(elapsed_seconds))) + rows.append(("Throughput", f"{throughput_per_hour:.1f} attempts/hour")) if _rich_tty(): table = _Table(show_header=False, box=None, pad_edge=False) table.add_column(style="muted", width=22) table.add_column(style="accent", justify="right") - table.add_row("Planned Tasks", str(total)) - if total_batches > 0: - table.add_row("Batches", str(total_batches)) - table.add_row("Completed Encodes", str(completed)) - table.add_row("Submitted", str(submitted)) - table.add_row("Skipped", str(skipped)) - table.add_row("Queued", str(queued)) - table.add_row("Failures", str(failed)) - if elapsed_seconds > 0: - table.add_row("Elapsed", _format_duration(elapsed_seconds)) - table.add_row("Throughput", f"{throughput_per_hour:.1f} encodes/hour") + for label, value in rows: + table.add_row(label, value) _console.print(_Panel(table, title="[title] Batch Summary [/title]", border_style="accent2")) return print("Batch Summary") - print(f" Planned Tasks: {total}") - if total_batches > 0: - print(f" Batches: {total_batches}") - print(f" Completed Encodes: {completed}") - print(f" Submitted: {submitted}") - print(f" Skipped: {skipped}") - print(f" Queued: {queued}") - print(f" Failures: {failed}") - if elapsed_seconds > 0: - print(f" Elapsed: {_format_duration(elapsed_seconds)}") - print(f" Throughput: {throughput_per_hour:.1f} encodes/hour") + for label, value in rows: + print(f" {label}: {value}") class BenchmarkProgress: @@ -495,7 +533,6 @@ def __init__(self, total_tasks: int, total_batches: int, hardware: Optional[Any] self.total_tasks = max(1, int(total_tasks)) self.total_batches = max(1, int(total_batches)) self.hardware = hardware - self._phase_steps_per_task = 3 # encode + metrics + submit self._live = None self._overall_progress = None @@ -503,16 +540,20 @@ def __init__(self, total_tasks: int, total_batches: int, hardware: Optional[Any] self._overall_task_id = None self._batch_task_id = None - self._overall_phase_steps = 0 - self._batch_phase_steps = 0 - self._overall_count = 0 - self._batch_count = 0 + # Bar bounds are durable measurement attempts, set only + # via set_progress. Legacy counters below drive descriptions only. + self._display_done = 0 + self._display_total = self.total_tasks + self._display_batch_done = 0 + self._display_batch_total = 1 self._batch_no = 1 self._batch_size = 1 self._description = "Preparing batch..." self._task_info: Dict[str, Any] = {} self._metrics: Dict[str, Any] = {} - self._counters: Dict[str, int] = {"submitted": 0, "skipped": 0, "queued": 0, "failed": 0} + self._counters: Dict[str, int] = { + "submitted": 0, "locally": 0, "skipped": 0, "queued": 0, "failed": 0, + } self._prev_sigwinch: Any = None self._sigwinch_installed = False @@ -545,16 +586,18 @@ def __enter__(self) -> "BatchRunDashboard": # teardown/interrupt on some terminals. self._overall_task_id = self._overall_progress.add_task( "Overall progress", - total=self.total_tasks * self._phase_steps_per_task, - display_done=0, - display_total=self.total_tasks, + total=self._display_total, + completed=self._display_done, + display_done=self._display_done, + display_total=self._display_total, label="Overall", ) self._batch_task_id = self._batch_progress.add_task( "Batch progress", - total=max(1, self._batch_size * self._phase_steps_per_task), - display_done=0, - display_total=self._batch_size, + total=self._display_batch_total, + completed=self._display_batch_done, + display_done=self._display_batch_done, + display_total=self._display_batch_total, label="Batch", ) self._live = _Live( @@ -592,20 +635,60 @@ def __exit__(self, exc_type, exc, tb) -> None: def start_batch(self, batch_no: int, batch_size: int) -> None: self._batch_no = max(1, int(batch_no)) self._batch_size = max(1, int(batch_size)) - self._batch_count = 0 - self._batch_phase_steps = 0 + self._display_batch_done = 0 + self._display_batch_total = self._batch_size if self._batch_progress is not None and self._batch_task_id is not None: self._batch_progress.reset( self._batch_task_id, - total=max(1, self._batch_size * self._phase_steps_per_task), + total=self._display_batch_total, completed=0, description=f"Batch {self._batch_no}/{self.total_batches}", display_done=0, - display_total=self._batch_size, + display_total=self._display_batch_total, label="Batch", ) self._refresh() + def set_progress( + self, + *, + done: int, + total: int, + batch_done: int, + batch_total: int, + ) -> None: + """Move both bars using producer-declared durable-truth bounds. + + ``done`` counts journaled warmup and measured attempts; ``total`` is + the declared attempt bound. The producer keeps + both monotonic per campaign; this clamps display only. + """ + total_n = max(1, int(total)) + done_n = max(0, min(int(done), total_n)) + batch_total_n = max(1, int(batch_total)) + batch_done_n = max(0, min(int(batch_done), batch_total_n)) + self._display_total = total_n + self._display_done = done_n + self._display_batch_total = batch_total_n + self._display_batch_done = batch_done_n + if self._overall_progress is not None and self._overall_task_id is not None: + self._overall_progress.update( + self._overall_task_id, + total=total_n, + completed=done_n, + display_done=done_n, + display_total=total_n, + ) + if self._batch_progress is not None and self._batch_task_id is not None: + self._batch_progress.update( + self._batch_task_id, + total=batch_total_n, + completed=batch_done_n, + display_done=batch_done_n, + display_total=batch_total_n, + ) + self._refresh() + def set_description(self, description: str) -> None: self._description = description if self._overall_progress is not None and self._overall_task_id is not None: @@ -625,9 +708,11 @@ def update_machine_metrics(self, metrics: Dict[str, Any]) -> None: self._metrics = dict(metrics or {}) self._refresh() - def update_counters(self, *, submitted: int, skipped: int, queued: int, failed: int) -> None: + def update_counters(self, *, submitted: int, skipped: int, queued: int, failed: int, + locally: int = 0) -> None: self._counters = { "submitted": max(0, int(submitted)), + "locally": max(0, int(locally)), "skipped": max(0, int(skipped)), "queued": max(0, int(queued)), "failed": max(0, int(failed)), @@ -635,66 +720,20 @@ def update_counters(self, *, submitted: int, skipped: int, queued: int, failed: self._refresh() def advance_phase(self, description: Optional[str] = None, step: int = 1) -> None: - step_n = max(1, int(step)) - self._overall_phase_steps += step_n - self._batch_phase_steps += step_n + """Phase chatter: updates the description only. Bars follow set_progress.""" if description: self._description = description - display_overall = min(self.total_tasks, max(self._overall_count, self._overall_phase_steps)) - display_batch = min(self._batch_size, max(self._batch_count, self._batch_phase_steps)) - - if self._overall_progress is not None and self._overall_task_id is not None: - kwargs: Dict[str, Any] = { - "advance": step_n, - "display_done": display_overall, - "display_total": self.total_tasks, - } - if description: - kwargs["description"] = description - self._overall_progress.update(self._overall_task_id, **kwargs) - - if self._batch_progress is not None and self._batch_task_id is not None: - self._batch_progress.update( - self._batch_task_id, - advance=step_n, - display_done=display_batch, - display_total=self._batch_size, - ) - self._refresh() + self.set_description(description) def advance(self, description: Optional[str] = None, step: int = 1) -> None: - step_n = max(1, int(step)) - self._overall_count += step_n - self._batch_count += step_n - self._overall_phase_steps += step_n - self._batch_phase_steps += step_n + """Legacy task step: description only; bars follow set_progress.""" if description: self._description = description - display_overall = min(self.total_tasks, max(self._overall_count, self._overall_phase_steps)) - display_batch = min(self._batch_size, max(self._batch_count, self._batch_phase_steps)) - - if self._overall_progress is not None and self._overall_task_id is not None: - kwargs: Dict[str, Any] = { - "advance": step_n, - "display_done": display_overall, - "display_total": self.total_tasks, - } - if description: - kwargs["description"] = description - self._overall_progress.update(self._overall_task_id, **kwargs) - if self._batch_progress is not None and self._batch_task_id is not None: - self._batch_progress.update( - self._batch_task_id, - advance=step_n, - display_done=display_batch, - display_total=self._batch_size, - ) - if not _rich_tty(): if description: - print(f"Progress: {self._overall_count}/{self.total_tasks} - {description}") + print(f"Progress: {self._display_done}/{self._display_total} - {description}") else: - print(f"Progress: {self._overall_count}/{self.total_tasks}") + print(f"Progress: {self._display_done}/{self._display_total}") self._refresh() def _install_resize_handler(self) -> None: @@ -825,7 +864,8 @@ def _test_info_lines(self, compact: bool = False) -> List[str]: f"[muted]Encoder:[/muted] [vanilla]{enc}[/vanilla]", f"[muted]Preset:[/muted] {preset} [muted]CRF:[/muted] {crf if crf is not None else '-'}", f"[muted]Mode:[/muted] CRF (1-pass)", - f"[muted]Queue:[/muted] ok={self._counters['submitted']} skip={self._counters['skipped']} queue={self._counters['queued']} fail={self._counters['failed']}", + f"[muted]Queue:[/muted] confirmed={self._counters['submitted']} local={self._counters['locally']} " + f"invalid={self._counters['skipped']} queued={self._counters['queued']} fail={self._counters['failed']}", f"[muted]Now:[/muted] {self._description}", ] if compact: diff --git a/client/windows_gui.py b/client/windows_gui.py index 2a2c431a..62cc302c 100644 --- a/client/windows_gui.py +++ b/client/windows_gui.py @@ -1,13 +1,17 @@ import argparse import os import queue +import re import threading import time import traceback from typing import Any, Dict, Optional +import psutil + from . import main as client_main from . import sweep_plan +from . import suite from .encoders import ( enumerate_supported_presets_for_encoder, get_encoder_friendly_label, @@ -28,10 +32,89 @@ "Single (advanced)": None, } # Replay checks its time budget between entries; an in-flight request can take -# longer. Keep the window visible until both owned workers have actually exited. +# longer. A confirmed Close gives owned workers one finite shutdown grace. GUI_CLOSE_GRACE_SECONDS = 70.0 +def _terminate_owned_children() -> None: + """Reap FFmpeg and helper descendants before a forced process exit.""" + try: + children = psutil.Process(os.getpid()).children(recursive=True) + except (psutil.Error, OSError): + return + for child in children: + try: + child.terminate() + except (psutil.Error, OSError): + pass + try: + _gone, alive = psutil.wait_procs(children, timeout=2) + except (psutil.Error, OSError): + alive = children + for child in alive: + try: + child.kill() + except (psutil.Error, OSError): + pass + try: + psutil.wait_procs(alive, timeout=1) + except (psutil.Error, OSError): + pass + + +def _validated_integer(value: Any, label: str, minimum: int, maximum: int) -> int: + """Read a Tk variable without leaving the window running on TclError.""" + try: + raw = str(value.get()).strip() + if not re.fullmatch(r"[0-9]+", raw): + raise ValueError + number = int(raw) + except Exception as exc: + raise ValueError(f"{label} must be a whole number from {minimum} to {maximum}.") from exc + if not minimum <= number <= maximum: + raise ValueError(f"{label} must be a whole number from {minimum} to {maximum}.") + return number + + +def _submission_line(event: Dict[str, Any]) -> str: + def display(value: Any, limit: int) -> str: + text = " ".join(str(value or "").split()) + text = re.sub(r"(?i)\b(bearer)\s+\S+", r"\1 [redacted]", text) + text = re.sub(r"(?i)\b(token|api[_-]?key|secret|authorization)\s*[:=]\s*\S+", + r"\1=[redacted]", text) + return text[:limit] + + status = display(event.get("status") or "unknown", 24) + category = display(event.get("errorCategory"), 40) + reason = display(event.get("safeReason"), 240) + action = display(event.get("recoveryAction"), 160) + codes = event.get("reasonCodes") + if isinstance(codes, (list, tuple)): + valid_codes = [str(code) for code in codes if re.fullmatch(r"[A-Za-z0-9_-]{1,48}", str(code))] + else: + valid_codes = [] + run_id = str(event.get("benchmarkRunId") or "") + if not re.fullmatch(r"[A-Za-z0-9_-]{1,80}", run_id): + run_id = "" + details = [item for item in (category, reason, ", ".join(valid_codes[:4]), action) if item] + label = display(event.get("preset") or event.get("codec") or event.get("campaignId"), 80) + line = f"Submission {status}" + (f" ({label})" if label else "") + if details: + line += ": " + "; ".join(details) + if run_id: + line += f" [run {run_id}]" + return line + + +def _acquisition_preview(mode_key: Optional[str]) -> Dict[str, Any]: + manifest = suite.load_default_suite_manifest() + if mode_key in (None, "small"): + clip_ids = [suite.get_default_quick_clip(manifest).clip_id] + else: + clip_ids = [clip.clip_id for clip in manifest.clips] + return suite.acquisition_estimate(clip_ids, manifest=manifest) + + def plan_summary_text(mode: str, encoders: list[str], presets_cfg: dict[str, Any]) -> str: """One-line honest preview of a sweep plan: finite work counts, no wall-clock claims.""" plan = sweep_plan.plan_sweep(mode, encoders, presets_cfg=presets_cfg) @@ -136,15 +219,24 @@ def __init__(self) -> None: self.event_queue: queue.Queue = queue.Queue() self.worker_thread: Optional[threading.Thread] = None self.upload_thread: Optional[threading.Thread] = None + self.upload_cancel_event = threading.Event() self.cancel_event = threading.Event() self.running = False + self._active_no_submit: Optional[bool] = None + self._last_submission_failure = "" + self._run_counts = {"submitted": 0, "locally_complete": 0, "queued": 0, "failed": 0} self._browse_shown = False self._close_deadline = 0.0 + self._estimate_acquisition = _acquisition_preview + self._load_recovery_state = lambda: client_main.recovery_state(str(self.base_args.queue_dir)) + self.saved_campaign_ids: list[str] = [] + self.saved_state: Dict[str, Any] = {} # Causal message from the most recent run_error/unhandled failure in this # run; the done handler must not replace it with a bare exit code. self.last_failure: Optional[str] = None self.mode_var = tk.StringVar(value="Small") + self.advanced_var = tk.BooleanVar(value=False) self.no_submit_var = tk.BooleanVar(value=bool(getattr(base_args, "no_submit", False))) self.base_url_var = tk.StringVar(value=str(getattr(base_args, "base_url", ""))) self.retries_var = tk.IntVar(value=max(1, int(getattr(base_args, "retries", 3)))) @@ -159,7 +251,9 @@ def __init__(self) -> None: self.current_var = tk.StringVar(value="-") self.summary_var = tk.StringVar(value="Ready") self.telemetry_var = tk.StringVar(value="-") - self.counter_var = tk.StringVar(value="ok=0 skip=0 queue=0 fail=0") + self.counter_var = tk.StringVar(value="local=0 uploaded=0 queued=0 failed=0") + self.saved_summary_var = tk.StringVar(value="Checking saved work...") + self.selected_saved_var = tk.StringVar(value="") self.overall_total = 1 self.overall_done = 0 @@ -170,10 +264,13 @@ def __init__(self) -> None: self.preset_values = [] self._build_ui(ttk, tk, scrolledtext) + self._toggle_advanced() self._refresh_encoders() self._update_single_fields_state() + self._refresh_saved_work() self._refresh_controls() self._poll_events() + self.root.after(30_000, self._idle_retry) self.root.protocol("WM_DELETE_WINDOW", self._on_close) self.root.bind("", self._start_shortcut) self.root.bind("", self._stop_shortcut) @@ -198,21 +295,30 @@ def _build_ui(self, ttk: Any, tk: Any, scrolledtext: Any) -> None: self.mode_combo.pack(side="left", padx=(8, 16)) self.mode_combo.bind("<>", lambda _evt: self._update_single_fields_state()) - ttk.Checkbutton(row1, text="No submit (local dry run only)", variable=self.no_submit_var).pack(side="left", padx=(0, 12)) - ttk.Label(row1, text="Retries").pack(side="left") - self.retries_spin = ttk.Spinbox(row1, from_=1, to=10, textvariable=self.retries_var, width=6) + self.no_submit_check = ttk.Checkbutton(row1, text="Save locally; publish later", + variable=self.no_submit_var, command=self._refresh_controls) + self.no_submit_check.pack(side="left", padx=(0, 12)) + + self.advanced_toggle = ttk.Checkbutton(config_frame, text="Advanced settings", variable=self.advanced_var, + command=self._toggle_advanced) + self.advanced_toggle.pack(anchor="w") + self.advanced_frame = ttk.LabelFrame(config_frame, text="Advanced settings", padding=10) + advanced_row = ttk.Frame(self.advanced_frame) + advanced_row.pack(fill="x", pady=(0, 8)) + ttk.Label(advanced_row, text="Retries").pack(side="left") + self.retries_spin = ttk.Spinbox(advanced_row, from_=1, to=10, textvariable=self.retries_var, width=6) self.retries_spin.pack(side="left", padx=(6, 12)) - ttk.Label(row1, text="Batch size").pack(side="left") - self.batch_spin = ttk.Spinbox(row1, from_=0, to=64, textvariable=self.batch_size_var, width=6) + ttk.Label(advanced_row, text="Batch size").pack(side="left") + self.batch_spin = ttk.Spinbox(advanced_row, from_=0, to=64, textvariable=self.batch_size_var, width=6) self.batch_spin.pack(side="left", padx=(6, 0)) - row2 = ttk.Frame(config_frame) + row2 = ttk.Frame(self.advanced_frame) row2.pack(fill="x", pady=(0, 8)) ttk.Label(row2, text="Base URL").pack(side="left") self.base_url_entry = ttk.Entry(row2, textvariable=self.base_url_var) self.base_url_entry.pack(side="left", fill="x", expand=True, padx=(8, 0)) - row3 = ttk.Frame(config_frame) + row3 = ttk.Frame(self.advanced_frame) row3.pack(fill="x") ttk.Label(row3, text="Encoder").pack(side="left") self.encoder_combo = ttk.Combobox(row3, textvariable=self.selected_encoder_var, state="readonly", width=34) @@ -232,14 +338,30 @@ def _build_ui(self, ttk: Any, tk: Any, scrolledtext: Any) -> None: self.bitrate_entry.pack(side="left", padx=(6, 0)) buttons = ttk.Frame(config_frame) + self.buttons_frame = buttons buttons.pack(fill="x", pady=(10, 0)) self.start_btn = ttk.Button(buttons, text="Start benchmark (Alt+B)", underline=6, command=self._start_run) self.start_btn.pack(side="left") self.stop_btn = ttk.Button(buttons, text="Stop (Alt+S)", underline=0, command=self._stop_run, state="disabled") self.stop_btn.pack(side="left", padx=(8, 0)) - self.upload_btn = ttk.Button(buttons, text="Retry Queued Uploads", command=self._retry_uploads) + self.upload_btn = ttk.Button(buttons, text="Retry due uploads", command=self._retry_uploads) self.upload_btn.pack(side="left", padx=(16, 0)) + saved_frame = ttk.LabelFrame(outer, text="Saved work", padding=10) + saved_frame.pack(fill="x", pady=(12, 0)) + ttk.Label(saved_frame, textvariable=self.saved_summary_var).pack(anchor="w") + saved_row = ttk.Frame(saved_frame) + saved_row.pack(fill="x", pady=(6, 0)) + self.saved_combo = ttk.Combobox(saved_row, textvariable=self.selected_saved_var, + values=[], state="readonly", width=64) + self.saved_combo.pack(side="left", fill="x", expand=True) + self.saved_combo.bind("<>", lambda _evt: self._refresh_controls()) + self.resume_btn = ttk.Button(saved_row, text="Resume", command=self._resume_saved) + self.resume_btn.pack(side="left", padx=(8, 0)) + self.publish_btn = ttk.Button(saved_row, text="Publish saved results", command=self._publish_saved) + self.publish_btn.pack(side="left", padx=(8, 0)) + ttk.Label(saved_frame, text="To publish, turn off Save locally and approve uploads.").pack(anchor="w", pady=(6, 0)) + progress_frame = ttk.LabelFrame(outer, text="Live Progress", padding=10) progress_frame.pack(fill="x", pady=(12, 12)) @@ -287,8 +409,73 @@ def _site_root(self) -> str: def _refresh_controls(self) -> None: idle = not self.running and not self._upload_active() self.start_btn.configure(state="normal" if idle else "disabled") - self.stop_btn.configure(state="normal" if self.running else "disabled") - self.upload_btn.configure(state="normal" if idle else "disabled") + self.stop_btn.configure(state="normal" if self.running or self._upload_active() else "disabled") + self.upload_btn.configure(state="normal" if idle and not self.no_submit_var.get() else "disabled") + self.no_submit_check.configure(state="normal" if idle else "disabled") + selected = self._selected_saved_state() + actions = {str(action.get("action") or "") for action in (selected or {}).get("actions", [])} + self.resume_btn.configure(state="normal" if idle and "resume" in actions else "disabled") + self.publish_btn.configure(state="normal" if idle and not self.no_submit_var.get() + and "publish_saved" in actions else "disabled") + + def _selected_saved_campaign(self) -> str: + index = self.saved_combo.current() + if index is None or index < 0 or index >= len(self.saved_campaign_ids): + return "" + return self.saved_campaign_ids[index] + + def _selected_saved_state(self) -> Optional[Dict[str, Any]]: + campaign_id = self._selected_saved_campaign() + return next((item for item in self.saved_state.get("campaigns", []) + if item.get("campaignId") == campaign_id), None) + + def _confirm_publication_consent(self) -> bool: + return client_main._ensure_interactive_publication_consent( + queue_dir=str(self.base_args.queue_dir), + prompt_callback=lambda disclosure: bool(messagebox.askyesno( + "Allow Benchmark Publication", disclosure, icon="warning", + )), + ) + + def _refresh_saved_work(self) -> None: + previous = self._selected_saved_campaign() if self.saved_campaign_ids else "" + try: + state = self._load_recovery_state() + except Exception as exc: + self.saved_state = {} + self.saved_summary_var.set(f"Saved work unavailable: {exc}") + return + self.saved_state = state + publication = state.get("publication") or {} + pending = int(publication.get("pendingEntries") or 0) + due = int(publication.get("dueEntries") or 0) + terminal = int(publication.get("terminalEntries") or 0) + accepted = int(publication.get("acceptedReceipts") or 0) + campaigns = list(state.get("campaigns") or []) + self.saved_summary_var.set( + f"{len(campaigns)} campaign(s) · {due} due / {max(0, pending - due)} delayed uploads · " + f"{accepted} uploaded (analysis pending) · {terminal} terminal" + ) + self.saved_campaign_ids = [str(item.get("campaignId") or "") for item in campaigns] + labels = [] + for item in campaigns: + finished = int(item.get("completedGroups") or 0) + planned = int(item.get("plannedGroups") or 0) + coverage = (f"{finished}/{planned} finished groups" if planned + else f"{finished} finished groups") + labels.append( + f"{item.get('campaignId')} — {coverage}, " + f"{int(item.get('pendingUploads') or 0)} unpublished, " + f"{int(item.get('queueDue') or 0)} due, " + f"{int(item.get('queueTerminal') or 0)} terminal, " + f"{int(item.get('unavailableSources') or 0)} unavailable" + + (" · needs original client" if item.get("measurementBlocked") else "")) + self.saved_combo["values"] = labels + if labels: + self.saved_combo.current(self.saved_campaign_ids.index(previous) if previous in self.saved_campaign_ids else 0) + else: + self.selected_saved_var.set("") + self._refresh_controls() def _set_running(self, running: bool) -> None: self.running = running @@ -302,12 +489,22 @@ def _set_running(self, running: bool) -> None: self.batch_spin.configure(state="normal" if not self.running and not self._upload_active() else "disabled") self.base_url_entry.configure(state="normal" if not self.running and not self._upload_active() else "disabled") self.crf_spin.configure(state=enabled) + self._update_single_fields_state(preview=False) def _selected_mode_key(self) -> Optional[str]: return GUI_MODE_BY_LABEL.get(self.mode_var.get().strip()) + def _toggle_advanced(self) -> None: + if self.advanced_var.get(): + self.advanced_frame.pack(fill="x", pady=(8, 0), before=self.buttons_frame) + else: + self.advanced_frame.pack_forget() + def _update_single_fields_state(self, preview: bool = True) -> None: single = self._selected_mode_key() is None + if single and not self.advanced_var.get(): + self.advanced_var.set(True) + self._toggle_advanced() state = "readonly" if single and not self.running else "disabled" spin_state = "normal" if single and not self.running else "disabled" self.encoder_combo.configure(state=state) @@ -366,7 +563,12 @@ def _stop_shortcut(self, _event: Any = None) -> str: self._stop_run() return "break" - def _start_run(self) -> None: + def _resume_saved(self) -> None: + campaign_id = self._selected_saved_campaign() + if campaign_id: + self._start_run(resume_id=campaign_id) + + def _start_run(self, *, resume_id: str = "") -> None: if self.running or self._upload_active(): return try: # Advisory only; run_benchmark_batch refuses authoritatively before any preparation. @@ -377,73 +579,109 @@ def _start_run(self) -> None: who = f" (campaign {active['campaignId']}, PID {active['pid']})" if active.get("campaignId") else "" self.summary_var.set(f"Another collection is actively running in this queue{who}. " f"Its checkpoints continue automatically - let it finish, or " - f"stop/cancel that run first. Retry Queued Uploads stays available.") + f"stop/cancel that run first. Due uploads resume after measurement.") self._append_log("Start refused: active collection detected") return - mode = self.mode_var.get().strip() - mode_key = GUI_MODE_BY_LABEL.get(mode) - if mode_key is None and ( - not self._selected_encoder() or self._selected_preset() not in self.preset_values - ): - messagebox.showerror("Unsupported configuration", "Select an available encoder and supported preset before starting.") - return - self.cancel_event.clear() - self.last_failure = None - self._set_running(True) - self.summary_var.set("Run started...") - self.stage_var.set("Starting") - self.current_var.set("-") - self.telemetry_var.set("-") - self.counter_var.set("ok=0 skip=0 queue=0 fail=0") - self.overall_total = 1 - self.overall_done = 0 - self.batch_total = 1 - self.batch_done = 0 - self.overall_pb.configure(maximum=1, value=0) - self.batch_pb.configure(maximum=1, value=0) - self._append_log(f"Starting {mode} run") - - run_args = argparse.Namespace(**vars(self.base_args)) - run_args.base_url = self.base_url_var.get().strip() or self.base_args.base_url - run_args.no_submit = bool(self.no_submit_var.get()) - run_args.retries = max(1, int(self.retries_var.get() or 1)) - run_args.batch_size = max(0, int(self.batch_size_var.get() or 0)) - run_args.pause_on_exit = False - run_args.menu = False - self._browse_shown = False - if getattr(run_args, "max_duration_minutes_explicit", False): - self._append_log( - f"Explicit measurement allowance: {float(run_args.max_duration_minutes):g} minutes; " - "the run stops there with the campaign saved for a later continuation." - ) - else: - self._append_log( - f"Checkpoint segments: {getattr(run_args, 'max_duration_minutes', 60):g} minutes each; " - "the run continues automatically until the plan completes. Acquisition and uploads are separate." - ) - bitrate = self.bitrate_var.get().strip() try: - run_args.target_bitrate_kbps = int(bitrate) if bitrate else None - except ValueError: - messagebox.showerror("Bitrate", "Enter a positive integer bitrate in kbps") - self._set_running(False) + mode = self.mode_var.get().strip() + if mode not in GUI_MODE_BY_LABEL: + raise ValueError("Choose a listed contribution mode.") + mode_key = GUI_MODE_BY_LABEL[mode] + run_args = argparse.Namespace(**vars(self.base_args)) + run_args.base_url = self.base_url_var.get().strip() or str(self.base_args.base_url) + run_args.no_submit = bool(self.no_submit_var.get()) + if resume_id: + run_args.resume_campaign = resume_id + run_args.submit = not run_args.no_submit + if not run_args.no_submit and not re.match(r"^https?://[^/\s]+", run_args.base_url): + raise ValueError("Base URL must be an HTTP or HTTPS address.") + run_args.retries = _validated_integer(self.retries_var, "Retries", 1, 10) + run_args.batch_size = _validated_integer(self.batch_size_var, "Batch size", 0, 64) + run_args.pause_on_exit = False + run_args.menu = False + bitrate = str(self.bitrate_var.get()).strip() + if mode_key is None and not resume_id: + encoder, preset = self._selected_encoder(), self._selected_preset() + if not encoder or preset not in self.preset_values: + raise ValueError("Select an available encoder and supported preset before starting.") + quality = _validated_integer(self.crf_var, "Native quality value", 0, 40) + if bitrate and not re.fullmatch(r"[0-9]+", bitrate): + raise ValueError("Bitrate must be a positive whole number in kbps.") + run_args.target_bitrate_kbps = int(bitrate) if bitrate else None + if run_args.target_bitrate_kbps is not None and run_args.target_bitrate_kbps <= 0: + raise ValueError("Bitrate must be a positive whole number in kbps.") + run_args = client_main.build_single_effective_args( + base_args=run_args, encoder=encoder, preset=preset, crf=quality, + ) + if not resume_id: + estimate = self._estimate_acquisition(mode_key) + if not estimate.get("storageOk", True): + raise ValueError( + "Not enough writable disk space for this contribution. " + f"Estimated peak: {client_main._format_byte_count(int(estimate['peakStorageBytes']))}." + ) + if estimate.get("strategy") == "unavailable": + raise ValueError("The selected frozen clips are unavailable. " + "; ".join(estimate.get("warnings") or [])) + transfer = int(estimate.get("bytesToTransfer") or 0) + if transfer: + approved = messagebox.askyesno( + "Download and storage estimate", + f"This run may download {client_main._format_byte_count(transfer)} of frozen reference media. " + f"Estimated peak extra storage: {client_main._format_byte_count(int(estimate.get('peakStorageBytes') or 0))}. " + "Continue?", + ) + if not approved: + self.summary_var.set("Run not started; download estimate declined") + return + if not run_args.no_submit: + consent_ok = self._confirm_publication_consent() + if not consent_ok: + run_args.no_submit = True + self.no_submit_var.set(True) + self._append_log("Publication consent not granted; saving locally.") + except Exception as exc: + messagebox.showerror("Check settings", str(exc)) + self.summary_var.set(f"Check settings: {exc}") return - if not run_args.no_submit: - consent_ok = client_main._ensure_interactive_publication_consent( - queue_dir=str(run_args.queue_dir), - prompt_callback=lambda disclosure: bool(messagebox.askyesno( - "Allow Benchmark Publication", - disclosure, - icon="warning", - )), - ) - if not consent_ok: - run_args.no_submit = True - self.no_submit_var.set(True) - self._append_log("Publication consent not granted; switching to local dry-run mode.") - self.worker_thread = threading.Thread(target=self._run_worker, args=(run_args, mode, mode_key), daemon=False) - self.worker_thread.start() + try: + self._active_no_submit = run_args.no_submit + self._last_submission_failure = "" + self._run_counts = {"submitted": 0, "locally_complete": 0, "queued": 0, "failed": 0} + self.cancel_event.clear() + self.last_failure = None + self._set_running(True) + self.summary_var.set("Run started...") + self.stage_var.set("Starting") + self.current_var.set("-") + self.telemetry_var.set("-") + self.counter_var.set("local=0 uploaded=0 queued=0 failed=0") + self.overall_total = self.batch_total = 1 + self.overall_done = self.batch_done = 0 + self.overall_pb.configure(maximum=1, value=0) + self.batch_pb.configure(maximum=1, value=0) + self._append_log(f"Resuming {resume_id}" if resume_id else f"Starting {mode} run") + self._browse_shown = False + if getattr(run_args, "max_duration_minutes_explicit", False): + self._append_log( + f"Explicit measurement allowance: {float(run_args.max_duration_minutes):g} minutes; " + "the run stops there with the campaign saved for a later continuation." + ) + else: + self._append_log( + f"Checkpoint segments: {getattr(run_args, 'max_duration_minutes', 60):g} minutes each; " + "the run continues automatically until the plan completes. Acquisition and uploads are separate." + ) + self.worker_thread = threading.Thread(target=self._run_worker, args=(run_args, mode, mode_key), daemon=False) + self.worker_thread.start() + except Exception as exc: + self.worker_thread = None + self._active_no_submit = None + self._set_running(False) + self._update_single_fields_state(preview=False) + self.stage_var.set("Idle") + self.summary_var.set(f"Could not start run: {exc}") + messagebox.showerror("Could not start", str(exc)) def _run_worker(self, run_args: argparse.Namespace, mode: str, mode_key: Optional[str]) -> None: def sink(event: Dict[str, Any]) -> None: @@ -451,7 +689,10 @@ def sink(event: Dict[str, Any]) -> None: rc = 1 try: - if mode_key is not None: + if getattr(run_args, "resume_campaign", ""): + rc = client_main._resume_campaign(run_args, event_sink=sink, + cancel_event=self.cancel_event, interactive=False) + elif mode_key is not None: rc = client_main.run_sweep_mode( mode=mode_key, base_args=run_args, @@ -462,11 +703,7 @@ def sink(event: Dict[str, Any]) -> None: presets_cfg=dict(presets_cfg), ) else: - effective_args = client_main.build_single_effective_args( - base_args=run_args, encoder=self._selected_encoder(), - preset=self._selected_preset(), crf=int(self.crf_var.get()), - ) - rc = client_main.run_with_args(effective_args, event_sink=sink, + rc = client_main.run_with_args(run_args, event_sink=sink, cancel_event=self.cancel_event, show_end_screen=False) except Exception as e: self.event_queue.put(("error", f"{e}\n{traceback.format_exc()}")) @@ -474,22 +711,101 @@ def sink(event: Dict[str, Any]) -> None: finally: self.event_queue.put(("done", rc)) - def _retry_uploads(self) -> None: + def _publish_saved(self) -> None: if self.running or self._upload_active(): return + if self.no_submit_var.get(): + self.summary_var.set("Turn off Save locally before publishing saved results") + return + campaign_id = self._selected_saved_campaign() + if not campaign_id: + return + try: + retries = _validated_integer(self.retries_var, "Retries", 1, 10) + if not self._confirm_publication_consent(): + self.summary_var.set("Saved work remains local; publication consent was not granted") + return + self.upload_cancel_event.clear() + self.upload_thread = threading.Thread( + target=self._publish_saved_worker, + args=(campaign_id, self.base_url_var.get().strip() or str(self.base_args.base_url), + str(getattr(self.base_args, "api_key", "") or ""), retries), + daemon=False, + ) + self.upload_thread.start() + self.summary_var.set(f"Publishing saved results from {campaign_id}; no encoding") + self._refresh_controls() + except Exception as exc: + self.upload_thread = None + self.summary_var.set(f"Could not publish saved work: {exc}") + messagebox.showerror("Publish saved results", str(exc)) + self._refresh_controls() + + def _publish_saved_worker(self, campaign_id: str, base_url: str, api_key: str, retries: int) -> None: + try: + rc, info = client_main.publish_saved_campaign( + queue_dir=str(self.base_args.queue_dir), campaign_id=campaign_id, + base_url=base_url, api_key=api_key, retries=retries, + interactive=False, continue_when_open=True, + cancel_event=self.upload_cancel_event, + event_sink=lambda event: self.event_queue.put(("event", event)), + ) + self.event_queue.put(("upload_status", self._publication_result_text(rc, info))) + except Exception as exc: + self.event_queue.put(("upload_status", f"Saved publication failed: {exc}")) + + @staticmethod + def _publication_result_text(rc: int, info: Dict[str, Any]) -> str: + submitted = int(info.get("submitted") or 0) + pending = int(info.get("selectedPending") or 0) + unadmitted = int(info.get("unadmitted") or 0) + terminal = int(info.get("terminal") or 0) + int(info.get("deadLettered") or 0) + if rc == 0: + return f"Uploaded {submitted} saved result(s); analysis pending" + if rc == 10: + reason = str(info.get("deferredReason") or "uploads_pending") + if info.get("failure"): + reason = client_main.failure_text(info["failure"]) + return (f"Saved work retained: {pending} queued, {unadmitted} not yet staged " + f"({reason}); no encoding") + if info.get("failure"): + return f"Saved publication blocked: {client_main.failure_text(info['failure'])}" + return f"Saved publication has {terminal} terminal failure(s); review saved work" + + def _retry_uploads(self, *, automatic: bool = False) -> None: + if self.running or self._upload_active(): + return + if self.no_submit_var.get(): + self.summary_var.set("Turn off Save locally before retrying uploads") + return + try: + retries = _validated_integer(self.retries_var, "Retries", 1, 10) + if not automatic and not self._confirm_publication_consent(): + self.summary_var.set("Queued uploads remain local; publication consent was not granted") + return + except Exception as exc: + messagebox.showerror("Check settings", str(exc)) + self.summary_var.set(f"Check settings: {exc}") + return base_url = self.base_url_var.get().strip() or str(self.base_args.base_url) api_key = str(getattr(self.base_args, "api_key", "") or "") queue_dir = str(self.base_args.queue_dir) - retries = max(1, int(self.retries_var.get() or 1)) self._append_log("Retrying queued uploads (never encodes)...") self.summary_var.set("Retrying queued uploads...") - self.upload_thread = threading.Thread( - target=self._retry_uploads_worker, - args=(queue_dir, base_url, api_key, retries), - daemon=False, - ) - self.upload_thread.start() - self._refresh_controls() + self.upload_cancel_event.clear() + try: + self.upload_thread = threading.Thread( + target=self._retry_uploads_worker, + args=(queue_dir, base_url, api_key, retries), + daemon=False, + ) + self.upload_thread.start() + self._refresh_controls() + except Exception as exc: + self.upload_thread = None + self.summary_var.set(f"Could not start upload retry: {exc}") + messagebox.showerror("Retry uploads", str(exc)) + self._refresh_controls() def _retry_uploads_worker(self, queue_dir: str, base_url: str, api_key: str, retries: int) -> None: try: @@ -497,22 +813,71 @@ def _retry_uploads_worker(self, queue_dir: str, base_url: str, api_key: str, ret if not pending_before: self.event_queue.put(("upload_status", "Upload queue is empty; nothing to retry.")) return - stats = client_main.replay_spool(queue_dir, base_url=base_url, api_key=api_key, - retries=retries, use_token=False) - remaining = client_main.count_pending_entries(queue_dir) + rc, info = client_main.retry_due_uploads( + queue_dir=queue_dir, base_url=base_url, api_key=api_key, + retries=retries, use_token=False, cancel_event=self.upload_cancel_event, + ) + remaining = int(info.get("pending") or client_main.count_pending_entries(queue_dir)) self.event_queue.put(( "upload_status", f"Upload retry: {pending_before} pending before, {remaining} still pending, " - f"dead-lettered={stats.dead_lettered}, corrupt={stats.corrupt}.", + f"dead-lettered={int(info.get('deadLettered') or 0)}, " + f"corrupt={int(info.get('corrupt') or 0)}, status={info.get('status') or rc}.", )) except Exception as e: self.event_queue.put(("upload_status", f"Upload retry failed: {e}")) + def _idle_retry(self) -> None: + try: + if not self.running and not self._upload_active(): + self._refresh_saved_work() + publication = self.saved_state.get("publication") or {} + permitted = (self.saved_state.get("publicationConsent") + and not self.no_submit_var.get() + and not self.saved_state.get("activeCollection") + and not self.saved_state.get("publicationLockBusy")) + if permitted: + base_url = self.base_url_var.get().strip() or str(self.base_args.base_url) + fingerprint = client_main.publication_endpoint_fingerprint(base_url) + for campaign in self.saved_state.get("campaigns") or []: + intent = campaign.get("publicationIntent") or {} + if (not intent.get("active") + or intent.get("baseUrlFingerprint") != fingerprint + or float(intent.get("nextAttemptAt") or 0) > time.time()): + continue + campaign_id = str(campaign.get("campaignId") or "") + if not campaign_id: + continue + try: + retries = _validated_integer(self.retries_var, "Retries", 1, 10) + except ValueError as exc: + self.summary_var.set(f"Saved continuation paused: {exc}") + break + self.upload_cancel_event.clear() + self.upload_thread = threading.Thread( + target=self._publish_saved_worker, + args=(campaign_id, base_url, + str(getattr(self.base_args, "api_key", "") or ""), retries), + daemon=False, + ) + self.upload_thread.start() + self.summary_var.set( + f"Continuing saved publication for {campaign_id}; no encoding") + self._refresh_controls() + return + if int(publication.get("dueEntries") or 0): + self._retry_uploads(automatic=True) + finally: + self.root.after(30_000, self._idle_retry) + def _stop_run(self) -> None: - if not self.running: + if not self.running and not self._upload_active(): return - self.cancel_event.set() - self.summary_var.set("Stopping owned work; retaining downloads and campaign...") + if self.running: + self.cancel_event.set() + if self._upload_active(): + self.upload_cancel_event.set() + self.summary_var.set("Stopping owned work; retaining saved results...") self._append_log("Cancellation requested") def _handle_event(self, event: Dict[str, Any]) -> None: @@ -539,22 +904,57 @@ def _handle_event(self, event: Dict[str, Any]) -> None: return if event_type == "run_start": - total = max(1, int(event.get("totalTasks") or 1)) + # Overall counts durably recorded warmup and measured attempts against the + # producer-declared bound; doneTotal carries the durable + # baseline so a resume segment never rewinds the bar. + if event.get("scope") == "batch": + total = max(1, int(event.get("declaredAttempts") or event.get("totalTasks") or 1)) + done = max(0, int(event.get("doneTotal") or 0)) + self.overall_total = total + self.overall_done = min(done, total) + self.overall_pb.configure(maximum=self.overall_total, value=self.overall_done) + unit = str(event.get("progressUnit") or "durable-attempt") + self.summary_var.set(f"Running {event.get('scope', 'benchmark')} ({unit}s)") + self._append_log(f"Run start: unit={unit} baseline={self.overall_done}/{self.overall_total}") + else: + total = max(1, int(event.get("totalTasks") or 1)) + self.overall_total = total + self.overall_done = 0 + self.overall_pb.configure(maximum=total, value=0) + self.summary_var.set(f"Running {event.get('scope', 'benchmark')} tasks") + self._append_log(f"Run start: tasks={total}") + return + + if event_type == "campaign_progress": + # Sole authority for both bars: producer counts durable + # measurement attempts; publication has separate counters. + total = max(1, int(event.get("total") or 1)) + done = max(0, min(int(event.get("done") or 0), total)) + batch_total = max(1, int(event.get("batchTotal") or 1)) + batch_done = max(0, min(int(event.get("batchDone") or 0), batch_total)) self.overall_total = total - self.overall_done = 0 - self.overall_pb.configure(maximum=total, value=0) - self.batch_pb.configure(maximum=total, value=0) - self.summary_var.set(f"Running {event.get('scope', 'benchmark')} tasks") - self._append_log(f"Run start: total={total}") + self.overall_done = done + self.overall_pb.configure(maximum=total, value=done) + self.batch_total = batch_total + self.batch_done = batch_done + self.batch_pb.configure(maximum=batch_total, value=batch_done) + self.summary_var.set( + f"Recorded {done}/{total} attempts " + f"({int(event.get('warmupsDone') or 0)} warmups, " + f"{int(event.get('measuredDone') or 0)} measured)" + ) return if event_type == "batch_start": - batch_size = max(1, int(event.get("batchSize") or 1)) - self.batch_total = batch_size + # A new segment restarts the batch view; Overall keeps its + # durable baseline (reset here would be non-monotonic). + batch_total = max(1, int(event.get("batchDeclaredAttempts") or event.get("batchSize") or 1)) + self.batch_total = batch_total self.batch_done = 0 - self.batch_pb.configure(maximum=batch_size, value=0) + self.batch_pb.configure(maximum=batch_total, value=0) self._append_log( - f"Batch {event.get('batchNo')}/{event.get('totalBatches')} start ({batch_size} tasks)" + f"Batch {event.get('batchNo')}/{event.get('totalBatches')} start " + f"({batch_total} attempts)" ) return @@ -598,7 +998,19 @@ def _handle_event(self, event: Dict[str, Any]) -> None: return if event_type == "submit_result": - line = f"Submit result: {event.get('status')} ({event.get('preset') or event.get('codec')})" + line = _submission_line(event) + status = str(event.get("status") or "") + if status in self._run_counts: + self._run_counts[status] += 1 + if status == "locally_complete": + self.counter_var.set( + f"local={self._run_counts['locally_complete']} " + f"uploaded={self._run_counts['submitted']} " + f"queued={self._run_counts['queued']} failed={self._run_counts['failed']}" + ) + if status in {"failed", "rejected", "queued"}: + self._last_submission_failure = line + self.summary_var.set(line) # The ingest response carries only the BenchmarkRun id, which the site does # not resolve; never fabricate a per-run URL. Point at the corpus browse page. if event.get("status") == "submitted" and not self._browse_shown: @@ -609,30 +1021,43 @@ def _handle_event(self, event: Dict[str, Any]) -> None: if event_type == "counters": self.counter_var.set( - f"ok={int(event.get('submitted') or 0)} " - f"skip={int(event.get('skipped') or 0)} " - f"queue={int(event.get('queued') or 0)} " - f"fail={int(event.get('failed') or 0)}" + f"local={self._run_counts['locally_complete']} " + f"uploaded={int(event.get('submitted') or 0)} (analysis pending) " + f"skipped={int(event.get('skipped') or 0)} " + f"queued={int(event.get('queued') or 0)} " + f"failed={int(event.get('failed') or 0)}" ) return if event_type == "task_complete": processed = max(0, int(event.get("processed") or 0)) - self.overall_done = processed - self.overall_pb.configure(value=min(self.overall_total, processed)) - if event.get("scope") == "batch" and self.batch_total > 0: - self.batch_done = min(self.batch_total, self.batch_done + 1) - self.batch_pb.configure(value=self.batch_done) - elif event.get("scope") == "single": - self.batch_pb.configure(maximum=max(1, int(event.get("total") or 1)), value=processed) - self.summary_var.set(f"Completed {processed}/{self.overall_total}") + if event.get("scope") == "single": + self.overall_pb.configure(maximum=max(1, int(event.get("total") or 1)), + value=min(max(1, int(event.get("total") or 1)), processed)) + self.batch_pb.configure(maximum=max(1, int(event.get("total") or 1)), + value=min(max(1, int(event.get("total") or 1)), processed)) + self.summary_var.set(f"Completed {processed}/{max(1, int(event.get('total') or 1))}") + # Batch progress is owned by campaign_progress (durable + # attempts); a per-record completion must never move the + # bars toward 100% on its own. return if event_type == "run_complete": - completed = event.get("completed") elapsed = event.get("elapsedSeconds") - self.stage_var.set("Complete") - self.summary_var.set(f"Completed {completed} task(s) in {elapsed:.1f}s" if isinstance(elapsed, (int, float)) else "Run complete") + failed = int(event.get("failed") or 0) + int(event.get("skipped") or 0) + queued = int(event.get("queued") or 0) + # The loop finishing is not success: stage follows the counts, + # and the done payload's exit code gives the final word. + self.stage_var.set("Complete" if not (failed or queued) else "Finished with issues") + if isinstance(elapsed, (int, float)): + text = f"Run loop finished in {elapsed:.1f}s" + else: + text = "Run loop finished" + if failed: + text += f"; {failed} attempt(s) failed or were invalid" + elif queued: + text += f"; {queued} upload(s) still queued" + self.summary_var.set(text) self._append_log(self.summary_var.get()) return @@ -670,30 +1095,54 @@ def _poll_events(self) -> None: elif kind == "done": self._set_running(False) rc = int(payload) + active_no_submit = self._active_no_submit + self._active_no_submit = None pending_note = "" - if rc in (0, 10, 11) and not self.no_submit_var.get(): + if rc in (0, 10, 11) and not active_no_submit: try: pending = client_main.count_pending_entries(str(self.base_args.queue_dir)) except Exception: pending = 0 if pending: - pending_note = f" — {pending} upload(s) queued; use Retry Queued Uploads" + pending_note = f" — {pending} upload(s) queued; due work retries while this window is open" + # Final stage is owned by the exit code: whatever the + # event stream showed mid-run, the terminal state must + # agree with summary, buttons and exit semantics. if rc == 0: - self.summary_var.set(("Locally complete" if self.no_submit_var.get() else "Uploaded; analysis pending") + pending_note) + if active_no_submit: + saved = self._run_counts["locally_complete"] + count = f"{saved} measured attempt(s) " if saved else "" + self.stage_var.set("Complete") + self.summary_var.set(f"Saved {count}locally; use Publish saved results when ready") + elif self._run_counts["failed"]: + self.stage_var.set("Error") + self.summary_var.set(self._last_submission_failure + pending_note) + elif self._run_counts["queued"] or pending_note: + self.stage_var.set("Pending uploads") + self.summary_var.set("Measurements saved; some uploads are queued" + pending_note) + elif self._run_counts["submitted"]: + self.stage_var.set("Complete") + self.summary_var.set("Uploaded; analysis pending") + else: + self.stage_var.set("Complete") + self.summary_var.set("Run finished; review saved work") elif rc == 11: + self.stage_var.set("Paused") self.summary_var.set("Measurement allowance reached; campaign saved — starting this mode again continues it" + pending_note) elif rc == 10: - self.summary_var.set(f"Saved locally; upload queued{pending_note}") + self.stage_var.set("Pending uploads") + self.summary_var.set("Measurements retained; uploads queued for retry" + pending_note) elif rc == 130: + self.stage_var.set("Cancelled") self.summary_var.set("Run cancelled") else: - if self.last_failure: - failure = f"Run failed (exit code {rc}): {self.last_failure}" - else: - failure = f"Run failed (exit code {rc}); see event log for details" + self.stage_var.set("Error") + failure = self._last_submission_failure or self.last_failure + failure = f"Run failed (exit code {rc}): {failure}" if failure else f"Run failed (exit code {rc}); see event log for details" self.summary_var.set(failure) self._append_log(failure) self._update_single_fields_state(preview=False) + self._refresh_saved_work() elif kind == "upload_status": self._append_log(payload) if not self.running: @@ -702,6 +1151,7 @@ def _poll_events(self) -> None: pass if self.upload_thread is not None and not self.upload_thread.is_alive(): self.upload_thread = None + self._refresh_saved_work() self._refresh_controls() self.root.after(120, self._poll_events) @@ -711,6 +1161,8 @@ def _on_close(self) -> None: return if self.running: self.cancel_event.set() + if self._upload_active(): + self.upload_cancel_event.set() self.summary_var.set("Stopping owned work before close...") self._close_deadline = time.monotonic() + GUI_CLOSE_GRACE_SECONDS self.root.after(100, self._close_when_stopped) @@ -722,8 +1174,13 @@ def _close_when_stopped(self): uploader = self._upload_active() if benchmark or uploader: if time.monotonic() >= self._close_deadline: - self.summary_var.set("Waiting for the current operation to finish safely before closing...") - self._close_deadline = time.monotonic() + GUI_CLOSE_GRACE_SECONDS + self.summary_var.set("Stop deadline reached; saved work remains available after restart") + self._append_log("Owned work did not quiesce in the Close grace period; ending this process and its helpers.") + try: + _terminate_owned_children() + self.root.destroy() + finally: + os._exit(130) self.root.after(100, self._close_when_stopped) return self.root.destroy() diff --git a/docs/CLIENT_BUILD_RUNTIME.md b/docs/CLIENT_BUILD_RUNTIME.md index de9c96f9..82444ccd 100644 --- a/docs/CLIENT_BUILD_RUNTIME.md +++ b/docs/CLIENT_BUILD_RUNTIME.md @@ -45,9 +45,16 @@ Supported builder overrides: Runtime-lock-sensitive CI does not rely on ambient `apt`, `brew`, or `choco` FFmpeg packages, because those runner packages do not consistently expose the required `xpsnr` filter. -- Linux and Windows use the immutable BtbN `autobuild-2026-09-09-14-51` - FFmpeg `n8.1.2-51-g7ba069f4f1` GPL archives, checked against pinned upstream - SHA256 digests. Platform identities were generated on native runners in +- Linux and Windows use the reviewed BtbN `autobuild-2026-09-09-14-51` + FFmpeg `n8.1.2-51-g7ba069f4f1` GPL runtime bytes. The original upstream + release tag now returns HTTP 404. CI recovers those *same* FFmpeg/FFprobe + bytes from the already-published [Linux candidate archive](https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909/encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167.tar.gz) + (SHA256 `cb2712b94b705cf447eb4b550e8f3b2be1bfcc6120db1f361a6534234bfb325c`) + or [Windows candidate archive](https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909/encodingdb-windows-candidate-ci34430919675-2d3ed7d4d167.zip) + (SHA256 `9e00b681397044ced552ec9539d10af42c731fe5a8b583eb9efb72de917fbe05`). + `scripts/provision_pinned_runtime.py` hashes each archive and extracted binary + against the committed lock before any build or smoke test. Platform identities + were generated on native runners in [CI run 34419226010](https://github.com/oliverdougherC/Encoding_Database/actions/runs/34419226010), then reviewed into the committed lock. A proposed lock is not a packaged build. - macOS uses Evermeet `126386-gc27482a18d7`, containing libvmaf `3.2.0-13`. diff --git a/docs/client-release-parity.md b/docs/client-release-parity.md new file mode 100644 index 00000000..9b2f92ff --- /dev/null +++ b/docs/client-release-parity.md @@ -0,0 +1,36 @@ +# Client release parity gate + +For the next candidate, build the macOS app/DMG, Windows GUI and console, and +Linux launcher/archive from the **same clean commit**. Keep the native +`*.release-manifest.json` sidecars and the macOS/Linux `*.package-info.json` +beside the downloaded build artifacts. The native build helpers already emit +these receipts; new sidecars also record `protocol.clientVersion`. + +Create a spec on the machine holding all four artifacts: + +```json +{ + "expectedSourceRevision": "", + "expectedProjectVersion": "", + "assets": [ + {"role": "macos-dmg", "artifact": "EncodingDB-macOS-arm64.dmg", "releaseManifest": "encodingdb-client-macos.release-manifest.json", "packageInfo": "EncodingDB-macOS-arm64.dmg.package-info.json"}, + {"role": "windows-gui", "artifact": "encodingdb-client-windows.exe", "releaseManifest": "encodingdb-client-windows.exe.release-manifest.json"}, + {"role": "windows-console", "artifact": "encodingdb-client-windows-console.exe", "releaseManifest": "encodingdb-client-windows-console.exe.release-manifest.json"}, + {"role": "linux-archive", "artifact": "encodingdb-client-linux.tar.gz", "releaseManifest": "encodingdb-client-linux.release-manifest.json", "packageInfo": "encodingdb-client-linux.tar.gz.package-info.json"} + ] +} +``` + +Paths are relative to the spec. Use the actual sidecar file names emitted by +each build. Then run: + +```bash +python scripts/assemble_client_release.py --spec candidate-spec.json --output client-release-manifest.json +``` + +The command rehashes each artifact, checks wrapper/embedded executable +identity, and rejects mixed source revisions, dirty builds, protocol/client +versions, frozen-suite fingerprints and runtime identities. The output gives +per-asset sizes and hashes for the release page. It does not publish an asset +or replace the platform-specific G01 checks in Linear PLA-90. Promotion also +requires rehashing the public download and matching its bytes to the manifest. diff --git a/docs/collection-readiness/reliability-20260928/ORIGINAL-INCIDENTS.md b/docs/collection-readiness/reliability-20260928/ORIGINAL-INCIDENTS.md new file mode 100644 index 00000000..d0964a11 --- /dev/null +++ b/docs/collection-readiness/reliability-20260928/ORIGINAL-INCIDENTS.md @@ -0,0 +1,143 @@ +# September 28 original-run evidence, before repair + +This is a read-only inventory of the owner's retained runs. It records what the +original files and current backend prove. The original state on each machine +was not cleared or resumed during this inventory. Private byte-for-byte copies +and SHA-256 file inventories are retained locally under +`.test-reports/reliability-20260928/original-{mac,windows,linux}-state/`. +Every copied file was compared by SHA-256 with its original: macOS 5,247/5,247, +Windows 815/815 and Linux 240/240 matched. +The independent read-only `scripts/campaign_conservation.py` reported zero +accounting violations for each preserved campaign; its three JSON outputs are +in `.test-reports/reliability-20260928/`. + +| Campaign | Frozen groups | Finished | Unfinished | Unstarted | Eligible measured uploads | Confirmed | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| macOS Medium | 98 | 98 | 0 | 0 | 198 | 198 | +| Windows Medium | 126 | 0 | 126 | 0 | 0 | 0 | +| Linux retained Large | 462 | 0 | 42 | 420 | 0 | 0 | + +| Machine | Original asset SHA-256 | Preserved state | Matched published asset | +| --- | --- | --- | --- | +| macOS | `cc6c6503293c8f01b223e1fac9a5da56888352d39c5e777cf90d23acec8aa8c6` | 5,247 files, 2,534,742,857 bytes | `1.3.0-rc.5` macOS DMG | +| Windows GUI | `2ec0a4bcb7d94af340e61a1837bfc9380b51b2dd60bbd1f48eb023dfa24c0379` | 815 files, 7,636,096,174 bytes | `1.3.0-rc.5` Windows GUI executable | +| Linux | `b1a68a039ce78a6bc9718adbae99409865326ea8a8b3fc29e47aaa090ef0f4d8` | 240 files, 458,284,784 bytes | `1.3.0-rc.5` Linux archive | + +Read-only machine inventory: macOS 27.0 arm64, M4 Pro; Windows 11 Pro build +26200 x64, Ryzen 9 7950X and RTX 5090; Linux 6.8.0-142 x86-64, Xeon +E5-2699 v4 and GTX 1070. Hardware presence does not certify a supported +encoder path in a new package. + +The release notes explicitly say rc.5 rebuilt macOS from `b0607f0`, carried +Windows binaries from `5d379c7`, and carried a Linux rc.2-era binary. Matching +the rc.5 release asset is therefore not evidence of shared source parity. +The original campaign manifests bind protocol and pinned FFmpeg/FFprobe SHA-256 +values, but these older manifests do not record `clientVersion`. The macOS +submission envelopes independently identify `client/0.3.3`; the Windows and +Linux incomplete runs have no envelopes from which to prove their client +version. Package hashes, local download times and release notes support the +package association but do not replace missing launch provenance. +The original macOS DMG mounted read-only and verified its image checksum. Its +app declares version `1.3.0-rc.5` and macOS minimum `27.0`; the embedded CLI is +Mach-O arm64 with SHA-256 +`82c4075927ca62bc18ac8be6cc284ba8454a7b7701f1bf7aa405d4f0a980cde4`. +The app has an ad-hoc code signature and no team ID. It was ejected after +inspection. +The original Windows GUI asset has a PE x86-64 header. The Linux archive's +embedded ELF x86-64 executable has SHA-256 +`6793d0892f95dd5334d0faae9d252376a1484cd32d8de33515c72d58116d2e8b`, +matching the already extracted native launcher directory. +Windows `Get-AuthenticodeSignature` reports `NotSigned` for the original GUI +asset; no signature assurance is inferred from its release checksum. + +| Original campaign runtime | FFmpeg SHA-256 | FFprobe SHA-256 | +| --- | --- | --- | +| macOS Medium | `ef1da73e929cb11aae5f4770ba0fcab617641d9feb03c5d00bb7c0702ebdf5d2` | `12da1d9d189806ba249f3c44b703f20b804e309a6150b3bef1e0ab4b569e9b66` | +| Windows Medium | `b74bd2209fe38026ae9f73ad676e626599bdc1efd950fe32d40b23184fb1a060` | `70be4ea32723b313df89603ba7cfbd818862e8785c20eb9008c27ef161ae84c0` | +| Linux retained Large | `d223345af606df9ab0c99e4222375376e218976c8e1e4ccaf8292d35d504f782` | `fb1364aa21beb8f514fb1b8cc5a9c5d557f9347c9433541c45bd5f3146ccb719` | + +## macOS Medium: `campaign-7e6d70761d73929e` + +The frozen manifest names 98 recipe by clip groups with protocol 7.1: one +warmup, two required measured repetitions, and at most two optional adaptive +repetitions per group. Its 296 durable attempts are 98 warmups plus 198 measured +records. Ninety-seven groups have exactly two measured records; one used four. +All 296 recorded attempts have valid client validity. The 198 immutable +submission envelopes each have an accepted marker and a unique server run ID. +The recorded attempts span 05:04:13–05:51:08 UTC; accepted markers span +05:51:12–06:02:16 UTC. These are observed intervals, not a promised duration. + +A read-only production database query found exactly those 198 run IDs, with +198/198 artifact hashes matching the local accepted markers. Server disposition: +98 `ACCEPTED` runs with `RETAINED` artifacts and `COMPLETE` analyses; 100 +`SUSPECT` runs with `VERIFIED` artifacts and `SUSPECT` analyses. All 100 suspect +runs have the server reason `Metric disagreement diagnostics flagged the run +for review`. Grouping by immutable repetition group gives 47 groups with two +accepted runs, one group with four accepted runs, and 50 groups with two +suspect runs. There is no missing required repetition or pending upload in +this campaign. The UI's +reported 198 is the number of measured uploads, not the number of encodes or +the number of scientifically accepted runs. The reported initial display of +"over 600" has no retained screenshot or log here; this manifest's protocol +range is 294–490 encodes, so the exact origin of that displayed figure remains +unresolved. The terminal implementation also mislabeled an in-memory count as +"submitted" and requires repair independently of this particular campaign's +successful delivery. + +## Windows Medium: `campaign-c820031ccf718637` + +The original GUI state has a 126-group frozen plan. It durably recorded 126 +valid warmups and 40 valid first measured repetitions. No group has its second +required measured repetition, and there are no submission envelopes or +accepted markers. `budget-exhausted-1790573546663034900.json` records a +60-minute measurement checkpoint at 2026-09-28 05:32:26 UTC with attempt 167 +in flight. Attempt 167 left output bytes but no durable attempt record; those +bytes are preserved as interrupted evidence, not counted as a valid measured +repetition. The backend has zero runs for this campaign. The first proven +reason for zero submissions at that checkpoint is incomplete measurement +groups; the code publishes only finalizable groups. Why automatic continuation +did not subsequently finish the campaign is unresolved without a retained GUI +event log or operator action record. The original executable and all 166 +attempts remain available for resume. No backend create, authorization or PUT +failure for this campaign can be inferred from zero publication attempts. + +## Linux reported Medium stall + +The retained Linux journals do not include a Medium campaign. The incomplete +overnight journal is **Large** `campaign-125d142f511fadad`, with 462 planned +groups, 42 valid warmups, no measured repetition and no submission envelope. +Its budget record shows a 60-minute measurement checkpoint at 2026-09-28 +08:57:17 UTC, attempt 43 in flight. Attempt 43 likewise has output bytes but +no durable attempt record. The other retained Linux journal is a +completed Small campaign. The backend has zero runs for the Large campaign. + +At inventory time the Linux host had no EncodingDB client, FFmpeg or FFprobe +process owned by this run, so the process tree and blocked operation from the +reported separate Medium "preparing runtime" stall cannot be reconstructed. +No Medium journal or matching retained terminal log was found. The unbounded +runtime-probe path in source is a confirmed defect, but it is not proof of the +owner's exact blocked substage. The original archive, runtime state, journals, +in-flight marker and retained bytes are preserved for follow-up. + +## Adjacent saved-work reconciliation + +The same original Windows state holds two completed Small campaigns +(`campaign-52eef614f3c5bd00` and `campaign-be40e3c9e4abe756`), and Linux +holds completed Small `campaign-7f05346afacc4b2e`. Each has 14 submission +envelopes and 14 durable queue receipts, but **zero journal accepted markers**. +The backend contains 14 `ACCEPTED`/`RETAINED` runs with `COMPLETE` analysis for +each campaign; original artifact +paths still exist on their native machines. Recovery must reconcile the queue +receipts with these envelopes by immutable payload identity. Treating each +envelope as pending and adding the queue receipt count would double-count work +already accepted by the server. +All 42 queue receipt run IDs and artifact hashes were matched independently +against read-only backend rows, with zero mismatches. + +## Evidence limits + +This document is incident diagnosis only. It does not certify a fixed client, +native Medium reruns, campaign recovery, or a new release. Backend status is +the current authoritative analysis disposition, not an assertion that all +uploaded evidence is accepted. Production was queried read-only; no synthetic +data was sent there. diff --git a/docs/collection-readiness/reliability-20260928/STATE-MODEL.md b/docs/collection-readiness/reliability-20260928/STATE-MODEL.md new file mode 100644 index 00000000..f318044b --- /dev/null +++ b/docs/collection-readiness/reliability-20260928/STATE-MODEL.md @@ -0,0 +1,41 @@ +# Durable campaign state and recovery contract + +The contributor UI must project persisted evidence rather than treating the +last callback or process exit code as the source of truth. Protocol 7.1 fixes +one warmup and two required measured repetitions per recipe by clip group; +up to two adaptive repetitions are optional and run only while needed for +stability. A maximum encode estimate is a storage/time bound, not a count of +required uploads. +The Windows Overall bar declares durable warmup and measured attempts against +the frozen maximum; the Batch bar shows durable attempts in the current segment. +The producer declares these units in each event. Unused optional adaptive slots +leave the maximum only when the measurement plan settles, and a checkpoint +cannot reset already durable campaign progress. Finished recipe by clip groups +are reported separately from both bars, as are publication and analysis status. + +| Stage | Durable evidence | Allowed recovery | +| --- | --- | --- | +| Frozen plan | `campaigns//manifest.json`, saved budget and exact suite/runtime/recipe identities | Resume the same plan only with compatible measurement provenance. A new plan gets a new campaign ID. | +| Source acquisition | frozen suite lock, hash-checked cache and license notices | Resume verified partial downloads; never substitute media or suppress hash checks. | +| Runtime readiness | pinned runtime lock and persistent preparation substage/heartbeat | Retry a bounded failed probe or preparation stage without touching campaign attempts. | +| Active measurement | `in-flight.json`, process checkpoint and attempt budget | An in-flight output without `attempt-*.json` is interrupted evidence; resume the same attempt, not a completed measurement. | +| Durable attempt | fsynced `attempt-*.json`, environment and exact artifact hash | Reuse every faithful completed attempt. Warmups may be released only with verified release evidence. | +| Finalized group | stability or measured-attempt cap proved from the frozen plan; immutable `submission-*.json` for eligible measured records | Publish saved complete records with zero encoding. An unfinished group remains resumable; optional unused repeats are not missing work. | +| Queued/delayed upload | spool entry, payload hash, artifact copy, original retry deadline and next due time | Replay the same logical upload after faults. A saved envelope and its spool copy count once. Rejections and expiry stay terminal. | +| Acknowledgment | queue receipt and/or faithful journal `submission-*.accepted.json`, bound to the original run and artifact hash | Reconcile a receipt before retiring only owned accepted bytes. A lost local acknowledgment must not produce a new server identity. | +| Analysis disposition | authoritative backend run, artifact and analysis rows | Show `ACCEPTED`/`RETAINED`/`COMPLETE` separately from `SUSPECT`, `INVALID`, `REJECTED` and analysis pending. An HTTP upload receipt alone is not scientific acceptance. | + +Two conservation equations must hold at every supported persistence boundary: + +1. Frozen groups = finished groups + unfinished started groups + unstarted groups. +2. Eligible finalized measured records = confirmed + durably queued + unstaged envelope + journal-only candidate + explicitly terminal or blocked records. An envelope and its queued copy are one record. + +`scripts/campaign_conservation.py` checks these equations independently of +the client implementation. Its seeded transition test moves one identity +through journal-only, envelope, queue, queue receipt and journal marker states. +The oracle counts recorded identities; publication separately verifies that +each retained artifact still has its promised bytes and SHA-256. +The original September 28 Mac, Windows and Linux campaign JSON reports are +retained under `.test-reports/reliability-20260928/`; they all conserve at the +observed boundaries. The repaired client still requires fault-driven native +acceptance before these invariants can be certified for its new packages. diff --git a/frontend/app/run/page.test.tsx b/frontend/app/run/page.test.tsx index c02ff2db..b6d35f1c 100644 --- a/frontend/app/run/page.test.tsx +++ b/frontend/app/run/page.test.tsx @@ -82,6 +82,13 @@ describe("RunPage", () => { expect(screen.getByRole("link", { name: "Download for macOS (Apple Silicon)" })).toHaveAttribute("href", `${plannedBase}/EncodingDB-macOS-arm64.dmg`); expect(screen.getByRole("link", { name: "Download for Windows (GUI)" })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-windows.exe`); expect(screen.getByRole("link", { name: "Download for Linux (x86-64)" })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-linux.tar.gz`); + expect(screen.getByRole("link", { name: `Windows console (${projectTag})` })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-windows-console.exe`); + const macCard = screen.getByRole("link", { name: "Download for macOS (Apple Silicon)" }).closest("article"); + expect(macCard?.textContent?.indexOf("requires macOS 27 or later")).toBeLessThan(macCard?.textContent?.indexOf("Download for macOS") ?? 0); + const windowsCard = screen.getByRole("link", { name: "Download for Windows (GUI)" }).closest("article"); + expect(windowsCard?.textContent?.indexOf("Windows 11 x86-64")).toBeLessThan(windowsCard?.textContent?.indexOf("Download for Windows") ?? 0); + const linuxCard = screen.getByRole("link", { name: "Download for Linux (x86-64)" }).closest("article"); + expect(linuxCard?.textContent?.indexOf("Ubuntu 24.04 x86-64")).toBeLessThan(linuxCard?.textContent?.indexOf("Download for Linux") ?? 0); expect(screen.getByRole("link", { name: new RegExp(`Release notes, checksums, and build evidence`) })).toHaveAttribute("href", `${repoReleases}/tag/${projectTag}`); expect(screen.queryByText(/pending publication/)).toBeNull(); }); diff --git a/frontend/app/run/page.tsx b/frontend/app/run/page.tsx index 62caf52b..401c3dd9 100644 --- a/frontend/app/run/page.tsx +++ b/frontend/app/run/page.tsx @@ -1,5 +1,5 @@ import styles from "./page.module.css"; -import { cliTag, downloadModel, historicalTag, projectTag, repoReleases, supersededAssets, supersededTag } from "./releaseAssets"; +import { cliTag, currentWindowsConsole, downloadModel, historicalTag, projectTag, repoReleases, supersededAssets, supersededTag } from "./releaseAssets"; export const dynamic = "force-dynamic"; @@ -14,6 +14,7 @@ export default function RunPage() { const historical = `${repoReleases}/download/${historicalTag}`; const superseded = `${repoReleases}/download/${supersededTag}`; const cliBase = `${repoReleases}/download/${cliTag}`; + const currentBase = `${repoReleases}/download/${projectTag}`; return

Contribute results

@@ -28,12 +29,12 @@ export default function RunPage() {
{downloads.items.map((asset) =>

{asset.label}

+

{asset.support}

{asset.href ? Download for {asset.label} : Download for {asset.label} (pending publication)}

{asset.file}

    {(platformNotes[asset.file] ?? []).map((line) =>
  1. {line}
  2. )}
-

{asset.support}

{asset.sha256 ? <>SHA-256 {asset.sha256} : "SHA-256 published with the release."}

)}
@@ -75,8 +76,14 @@ export default function RunPage() { )} -

Plain command-line builds ({cliTag})

-

Protocol-compatible executables that predate the packaged apps; the macOS file is a bare extensionless executable.

+

Current command-line access ({projectTag})

+

For Windows scripts, use the console executable from the current release. The current macOS and Linux command-line entry points are inside their packages above.

+

{downloads.published + ? Windows console ({projectTag}) + : Windows console ({projectTag}) available when current downloads are enabled} · {currentWindowsConsole.file} · SHA-256 {currentWindowsConsole.sha256}

+ +

Historical plain executables ({cliTag})

+

These protocol-compatible files predate the packaged apps and current shared-core repairs. The macOS file is a bare extensionless executable. Use the current packages above for contribution.

  • Windows GUI (rc.1) · encodingdb-client-windows.exe
  • Windows console (rc.1) · encodingdb-client-windows-console.exe
  • diff --git a/frontend/app/run/releaseAssets.ts b/frontend/app/run/releaseAssets.ts index d65add79..24096df3 100644 --- a/frontend/app/run/releaseAssets.ts +++ b/frontend/app/run/releaseAssets.ts @@ -36,16 +36,23 @@ export const primaryAssets: ReleaseAsset[] = [ file: "encodingdb-client-windows.exe", label: "Windows (GUI)", sha256: "2ec0a4bcb7d94af340e61a1837bfc9380b51b2dd60bbd1f48eb023dfa24c0379", - support: "No Authenticode signature, so SmartScreen may warn at first launch. The window exposes the same Small/Medium/Large/Full sweeps as the guided interface and recovers automatically from a cache folder protected against the current user.", + support: "Tested on Windows 11 x86-64; other Windows versions are unverified. No Authenticode signature, so SmartScreen may warn at first launch. The window exposes the same Small/Medium/Large/Full sweeps as the guided interface and recovers automatically from a cache folder protected against the current user.", }, { file: "encodingdb-client-linux.tar.gz", label: "Linux (x86-64)", sha256: "b1a68a039ce78a6bc9718adbae99409865326ea8a8b3fc29e47aaa090ef0f4d8", - support: "Unsigned archive; unpack it and run the launcher inside. Same guided flow; includes the command-line entry point for scripts.", + support: "Tested on Ubuntu 24.04 x86-64; other Linux distributions are unverified. Unsigned archive; unpack it and run the launcher inside. Same guided flow; includes the command-line entry point for scripts.", }, ]; +export const currentWindowsConsole: ReleaseAsset = { + file: "encodingdb-client-windows-console.exe", + label: "Windows console", + sha256: "05189da160c876f812cd16ae228b76df50bac8925d5763bb3d49ea9f0e42a60e", + support: "Command-line entry point from the current release, built alongside the Windows GUI.", +}; + // Preserve the actual published rc.4 asset identities for rollback and verification. export const supersededAssets: ReleaseAsset[] = [ { diff --git a/release.json b/release.json index 97544988..b671f2c6 100644 --- a/release.json +++ b/release.json @@ -1,9 +1,9 @@ { "schemaVersion": 1, - "projectVersion": "1.3.0-rc.5", - "releaseDate": "2026-09-22", + "projectVersion": "1.3.0-rc.8", + "releaseDate": "2026-09-29", "benchmarkProtocolVersion": "7.1", "plFormulaVersion": "7.0", "suiteVersion": "encodingdb-test-suite-v1", - "clientImplementationVersion": "client/0.3.3" + "clientImplementationVersion": "client/0.3.8" } diff --git a/scripts/assemble_client_release.py b/scripts/assemble_client_release.py new file mode 100644 index 00000000..84eea961 --- /dev/null +++ b/scripts/assemble_client_release.py @@ -0,0 +1,201 @@ +#!/usr/bin/env python3 +"""Verify one candidate's native assets and write a shared release manifest. + +The JSON input declares ``expectedSourceRevision`` and ``expectedProjectVersion`` +and has an ``assets`` array with one entry per role: ``macos-dmg``, +``windows-gui``, ``windows-console`` and ``linux-archive``. Each entry names +``artifact`` and ``releaseManifest`` paths, plus ``packageInfo`` for macOS +and Linux wrappers. Relative paths resolve beside the spec file. Use native +build receipts from one clean committed revision. This gate does not publish +assets or replace the physical G01 acceptance runs. +""" + +import argparse +import json +import re +import sys +from pathlib import Path +from typing import Any + +ROOT_DIR = Path(__file__).resolve().parent.parent +if str(ROOT_DIR) not in sys.path: + sys.path.insert(0, str(ROOT_DIR)) + +from scripts.release_manifest_lib import atomic_write_json, sha256_path + +ROLES = {"macos-dmg", "windows-gui", "windows-console", "linux-archive"} +EXECUTABLE_SUPPORT = { + "macos-dmg": ("Mach-O", "arm64"), + "windows-gui": ("PE", "x86_64"), + "windows-console": ("PE", "x86_64"), + "linux-archive": ("ELF", "x86_64"), +} + + +def _path(base: Path, value: Any) -> Path: + path = Path(str(value or "")) + if not str(value or "").strip(): + raise ValueError("asset path is missing") + return path if path.is_absolute() else base / path + + +def _read_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"expected a JSON object in {path}") + return payload + + +def _check_file(path: Path, identity: dict[str, Any]) -> str: + if path.name != identity.get("fileName"): + raise ValueError(f"file name differs from native receipt: {path.name}") + if path.stat().st_size != identity.get("byteSize"): + raise ValueError(f"file bytes differ from native receipt: {path.name}") + observed_sha = sha256_path(path) + if observed_sha != identity.get("sha256"): + raise ValueError(f"file bytes differ from native receipt: {path.name}") + return observed_sha + + +def _source_revision(receipt: dict[str, Any]) -> str: + source = receipt.get("source") or {} + revision = source.get("revision") + if not isinstance(revision, str) or not re.fullmatch(r"[0-9a-f]{40}", revision) or source.get("trackedChanges") is not False: + raise ValueError("native receipt must identify a clean committed source revision") + return revision + + +def _support(role: str, binary: dict[str, Any], package: dict[str, Any] | None) -> dict[str, str]: + observed = binary.get("executableIdentity") or {} + expected_format, expected_arch = EXECUTABLE_SUPPORT[role] + if observed.get("format") != expected_format or observed.get("architecture") != expected_arch: + raise ValueError(f"{role} executable header differs from its supported architecture") + if role == "macos-dmg": + minimum = (package or {}).get("minimumSystemVersion") + if not isinstance(minimum, str) or not re.fullmatch(r"\d+(?:\.\d+){1,2}", minimum): + raise ValueError("macOS package lacks a verified minimum OS version") + # The current frozen helper bundle was qualified at macOS 27.0. + if tuple(int(part) for part in minimum.split(".")) < (27, 0): + raise ValueError("macOS package minimum is below the qualified runtime floor") + return {"operatingSystem": "macOS", "architecture": expected_arch, + "minimumVersion": minimum} + if role in ("windows-gui", "windows-console"): + return {"operatingSystem": "Windows", "architecture": expected_arch, + "supportedVersion": "11", "otherVersions": "unverified"} + return {"operatingSystem": "Ubuntu Linux", "architecture": expected_arch, + "supportedVersion": "24.04", "otherDistributions": "unverified"} + + +def assemble(spec: dict[str, Any], base: Path) -> dict[str, Any]: + expected_revision = spec.get("expectedSourceRevision") + expected_version = spec.get("expectedProjectVersion") + if not isinstance(expected_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", expected_revision): + raise ValueError("spec must name the reviewed source revision") + if not isinstance(expected_version, str) or not expected_version.strip(): + raise ValueError("spec must name the candidate project version") + entries = spec.get("assets") + if not isinstance(entries, list) or len(entries) != len(ROLES): + raise ValueError("spec must contain one entry for each native release role") + assets = [] + common: dict[str, Any] | None = None + roles_seen: set[str] = set() + runtimes: dict[str, str] = {} + for entry in entries: + if not isinstance(entry, dict): + raise ValueError("each asset entry must be an object") + role = entry.get("role") + if not isinstance(role, str) or role not in ROLES or role in roles_seen: + raise ValueError(f"unexpected or repeated native release role: {role}") + roles_seen.add(role) + artifact = _path(base, entry.get("artifact")) + receipt = _read_json(_path(base, entry.get("releaseManifest"))) + if receipt.get("schemaVersion") != 1: + raise ValueError(f"unsupported native manifest for {role}") + protocol = receipt.get("protocol") or {} + suite = receipt.get("suite") or {} + runtime = receipt.get("runtime") or {} + identity = { + "sourceRevision": _source_revision(receipt), + "projectVersion": receipt.get("projectVersion"), + "clientVersion": protocol.get("clientVersion"), + "protocolVersion": protocol.get("benchmarkProtocolVersion"), + "minimumClientVersion": protocol.get("minimumClientVersion"), + "suiteVersion": suite.get("suiteVersion"), + "suiteManifestVersion": suite.get("manifestVersion"), + "suiteFingerprint": suite.get("suiteFingerprint"), + } + if not all(identity.values()) or suite.get("isFrozen") is not True: + raise ValueError(f"incomplete frozen client identity for {role}") + if identity["sourceRevision"] != expected_revision or identity["projectVersion"] != expected_version: + raise ValueError(f"{role} differs from the reviewed source/version in the spec") + if common is None: + common = identity + elif identity != common: + raise ValueError(f"{role} was built from a different client/protocol/suite identity") + platform = {"macos-dmg": "mac", "linux-archive": "linux"}.get(role, "win") + if receipt.get("platform") != platform or not isinstance(runtime.get("fingerprint"), str): + raise ValueError(f"invalid {role} platform/runtime identity") + if platform in runtimes and runtimes[platform] != runtime["fingerprint"]: + raise ValueError(f"different {platform} runtime identities in one release") + runtimes[platform] = runtime["fingerprint"] + + binary = receipt.get("artifact") or {} + wrapper = None + package = None + if role in {"macos-dmg", "linux-archive"}: + package = _read_json(_path(base, entry.get("packageInfo"))) + if package.get("provisional") is not False or _source_revision(package) != identity["sourceRevision"]: + raise ValueError(f"unverified package source for {role}") + wrapper = package.get("dmg") if role == "macos-dmg" else package.get("archive") + if not isinstance(wrapper, dict): + raise ValueError(f"missing package identity for {role}") + artifact_sha = _check_file(artifact, wrapper) + if role == "macos-dmg": + if package.get("cliEmbeddedSha256") != binary.get("sha256") or package.get("cliBytesChangedByPackaging") is not False: + raise ValueError("macOS DMG embeds a different client executable") + mount = wrapper.get("readOnlyMountVerification") or {} + if mount.get("mounted") is not True or mount.get("innerSha256") != binary.get("sha256"): + raise ValueError("macOS DMG lacks read-only mount verification") + else: + members = wrapper.get("members") or [] + if not any(isinstance(member, dict) and member.get("sha256") == binary.get("sha256") for member in members): + raise ValueError("Linux archive does not contain the audited client executable") + if (package.get("verification") or {}).get("verified") is not True: + raise ValueError("Linux archive lacks launch-contract verification") + else: + artifact_sha = _check_file(artifact, binary) + assets.append({ + "role": role, "fileName": artifact.name, + "sha256": artifact_sha, "byteSize": artifact.stat().st_size, + "embeddedExecutableSha256": binary.get("sha256"), + "runtimeFingerprint": runtime["fingerprint"], + "support": _support(role, binary, package), + }) + if roles_seen != ROLES or common is None: + raise ValueError("native release roles are incomplete") + return { + "schemaVersion": 1, + **common, + "lifecycle": { + "builtFromReviewedSource": True, + "nativeAcceptance": "not_certified", + "published": False, + "independentRedownloadVerified": False, + }, + "assets": sorted(assets, key=lambda item: item["role"]), + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--spec", type=Path, required=True, help="JSON mapping each release role to artifact and native receipts") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + spec = _read_json(args.spec) + atomic_write_json(args.output, assemble(spec, args.spec.parent)) + print(args.output) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/build_linux_client.sh b/scripts/build_linux_client.sh index 3c3d5f9a..7a74d01a 100644 --- a/scripts/build_linux_client.sh +++ b/scripts/build_linux_client.sh @@ -95,6 +95,7 @@ cd "$ROOT_DIR" "${PYI_CMD[@]}" \ --clean \ --onefile \ + --bootloader-ignore-signals \ --name "$APP_NAME" \ --distpath "$PYI_DIST_DIR" \ --workpath "$PYI_WORK_DIR" \ diff --git a/scripts/build_macos_client.sh b/scripts/build_macos_client.sh index 05304379..0413a5b6 100755 --- a/scripts/build_macos_client.sh +++ b/scripts/build_macos_client.sh @@ -103,6 +103,7 @@ fi cd "$ROOT_DIR" "$BUILD_PYTHON" -m PyInstaller.utils.cliutils.makespec \ --onefile \ + --bootloader-ignore-signals \ --name "$APP_NAME" \ --specpath "$PYI_SPEC_DIR" \ --paths "$ROOT_DIR" \ diff --git a/scripts/build_windows_client.ps1 b/scripts/build_windows_client.ps1 index 40028a4b..dfd38653 100644 --- a/scripts/build_windows_client.ps1 +++ b/scripts/build_windows_client.ps1 @@ -187,6 +187,8 @@ function Invoke-PyInstallerBuild { ) if ($Windowed) { $buildArgs += "--windowed" + } else { + $buildArgs += "--bootloader-ignore-signals" } $buildArgs += $Entrypoint diff --git a/scripts/campaign_conservation.py b/scripts/campaign_conservation.py new file mode 100644 index 00000000..b2b08a36 --- /dev/null +++ b/scripts/campaign_conservation.py @@ -0,0 +1,194 @@ +#!/usr/bin/env python3 +"""Independent, read-only conservation check for one V7 campaign journal. + +This intentionally does not import client code. It counts frozen groups, +attempts, finalizable measured records and durable publication identities from +files, then optionally matches an exported read-only server CSV. +""" + +import argparse +import csv +import hashlib +import json +import statistics +from collections import Counter, defaultdict +from pathlib import Path + + +def _json(path): + with path.open(encoding="utf-8") as handle: + return json.load(handle) + + +def _payload_hash(payload): + normalized = dict(payload) + if normalized.get("submissionKind") == "authoritative-artifact-run-v1": + normalized.pop("artifactPath", None) + normalized.pop("artifactManaged", None) + encoded = json.dumps(normalized, sort_keys=True, separators=(",", ":")).encode() + return hashlib.sha256(encoded).hexdigest() + + +def audit(queue: Path, campaign_id: str, server_csv: Path | None = None): + root = queue / "campaigns" / campaign_id + manifest = _json(root / "manifest.json") + plan = manifest["protocolConfig"] + groups_planned = len(manifest["tasks"]) + warmup_required = int(plan["warmup_runs"]) + measured_required = int(plan["minimum_measured_runs"]) + adaptive_max = int(plan["max_adaptive_repeats"]) + stability_limit = float(plan["stability_threshold_ratio"]) + issues = [] + by_group = defaultdict(list) + attempts = {} + for path in sorted(root.glob("attempt-*.json")): + try: + order = int(path.stem.split("-")[1]) + record = _json(path) + schedule = record["schedule"] + if (schedule["campaign_id"] != campaign_id + or int(schedule["execution_order"]) != order + or order in attempts): + raise ValueError("attempt identity mismatch") + attempts[order] = record + by_group[schedule["recipe_id"]].append(record) + except (OSError, ValueError, KeyError, TypeError) as exc: + issues.append(f"{path.name}: {exc}") + if len(by_group) > groups_planned: + issues.append("more recipe groups in attempts than in frozen plan") + + group_states = Counter() + eligible_orders = set() + for recipe, records in by_group.items(): + warmups = [r for r in records if r["schedule"]["phase"] == "warmup"] + measured = [r for r in records if r["schedule"]["phase"] == "measured"] + counted = [float(r["timing"]["elapsed_s"]) for r in measured + if r.get("countedForStability") and isinstance(r.get("timing"), dict)] + stable = False + if len(counted) >= measured_required: + mean = statistics.fmean(counted) + stable = mean > 0 and (max(counted) - min(counted)) / mean <= stability_limit + finished = (len(warmups) >= warmup_required + and len(measured) >= measured_required + and (stable or len(measured) >= measured_required + adaptive_max)) + group_states["finished" if finished else "unfinished"] += 1 + if finished: + for record in measured: + if (record.get("skippedBeforeEncode") + or record.get("overallValidity", {}).get("state") == "invalid" + or not isinstance(record.get("timing"), dict) + or (record.get("metadata", {}).get("info") or {}).get("error")): + continue + eligible_orders.add(int(record["schedule"]["execution_order"])) + group_states["unstarted"] = max(0, groups_planned - len(by_group)) + if sum(group_states.values()) != groups_planned: + issues.append("frozen group accounting does not conserve") + + envelopes = {} + for path in sorted(root.glob("submission-*.json")): + if path.name.endswith(".accepted.json"): + continue + try: + order = int(path.stem.split("-")[1]) + envelopes[order] = _json(path) + except (OSError, ValueError, KeyError, TypeError) as exc: + issues.append(f"{path.name}: {exc}") + markers = {} + for path in sorted(root.glob("submission-*.accepted.json")): + try: + order = int(path.name.split("-")[1].split(".")[0]) + marker = _json(path) + if int(marker["executionOrder"]) != order or not marker.get("benchmarkRunId"): + raise ValueError("accepted marker identity mismatch") + markers[order] = marker + except (OSError, ValueError, KeyError, TypeError) as exc: + issues.append(f"{path.name}: {exc}") + pending_spool = {path.stem for path in queue.glob("*.json") if path.is_file()} + queue_receipts = {path.stem for path in (queue / "receipts").glob("*.json") if path.is_file()} + terminal_spool = {path.stem for path in (queue / "terminal").glob("*.json") if path.is_file()} + publication = Counter() + marker_ids = {} + for order in sorted(eligible_orders): + record = attempts[order] + info = (record.get("metadata", {}).get("info") or {}) + marker = markers.get(order) + envelope = envelopes.get(order) + local_hash = _payload_hash(envelope) if isinstance(envelope, dict) else None + if marker: + if marker.get("artifactSha256") != info.get("artifactSha256"): + issues.append(f"accepted marker {order} artifact hash mismatch") + marker_ids[str(marker["benchmarkRunId"])] = marker.get("artifactSha256") + publication["confirmed"] += 1 + elif local_hash and local_hash in queue_receipts: + publication["receipted_without_journal_marker"] += 1 + elif local_hash and local_hash in terminal_spool: + publication["terminal"] += 1 + elif local_hash and local_hash in pending_spool: + publication["queued"] += 1 + elif envelope: + publication["unstaged_envelope"] += 1 + else: + publication["journal_only"] += 1 + if len(marker_ids) != publication["confirmed"]: + issues.append("duplicate server run ID in accepted markers") + if sum(publication.values()) != len(eligible_orders): + issues.append("eligible publication identities do not conserve") + extra_envelopes = set(envelopes) - eligible_orders + extra_markers = set(markers) - eligible_orders + if extra_envelopes: + issues.append(f"{len(extra_envelopes)} envelopes outside eligible finalized attempts") + if extra_markers: + issues.append(f"{len(extra_markers)} accepted markers outside eligible finalized attempts") + + server = None + if server_csv: + with server_csv.open(newline="", encoding="utf-8") as handle: + rows = list(csv.DictReader(handle)) + server_ids = {row["run_id"]: row for row in rows} + if len(server_ids) != len(rows): + issues.append("duplicate run ID in server export") + for run_id, sha256 in marker_ids.items(): + row = server_ids.get(run_id) + if row is None: + issues.append(f"accepted run {run_id} missing from server export") + elif row.get("artifact_sha256") != sha256: + issues.append(f"accepted run {run_id} server artifact hash mismatch") + server = {"rows": len(rows), "matchedAcceptedMarkers": len(set(server_ids) & set(marker_ids)), + "serverRowsWithoutLocalMarker": len(set(server_ids) - set(marker_ids)), + "disposition": dict(Counter((row["run_status"], row["artifact_state"], row["analysis_status"]) + for row in rows))} + server["disposition"] = {"/".join(key): value for key, value in server["disposition"].items()} + + validity = Counter(r.get("overallValidity", {}).get("state", "unknown") + for r in attempts.values()) + return { + "campaignId": campaign_id, + "frozenGroups": groups_planned, + "groups": dict(group_states), + "attempts": {"total": len(attempts), + "warmup": sum(r["schedule"]["phase"] == "warmup" for r in attempts.values()), + "measured": sum(r["schedule"]["phase"] == "measured" for r in attempts.values()), + "validity": dict(validity), + "skippedBeforeEncode": sum(bool(r.get("skippedBeforeEncode")) + for r in attempts.values())}, + "requiredMeasured": groups_planned * measured_required, + "eligibleFinalized": len(eligible_orders), + "publication": dict(publication), + "server": server, + "issues": issues, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--queue-dir", type=Path, required=True) + parser.add_argument("--campaign-id", required=True) + parser.add_argument("--server-csv", type=Path) + args = parser.parse_args() + report = audit(args.queue_dir, args.campaign_id, args.server_csv) + print(json.dumps(report, indent=2, sort_keys=True)) + return 1 if report["issues"] else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/prepare_client_suite_distribution.py b/scripts/prepare_client_suite_distribution.py index 5ed60e19..ec3d743b 100644 --- a/scripts/prepare_client_suite_distribution.py +++ b/scripts/prepare_client_suite_distribution.py @@ -1,5 +1,7 @@ #!/usr/bin/env python3 import argparse +import hashlib +import json import os import shutil import sys @@ -42,17 +44,60 @@ def prepare_distribution( shutil.rmtree(staged_resource_dir, ignore_errors=True) staged_resource_dir.mkdir(parents=True, exist_ok=True) - for name in ("manifest.json", "finalization-status.json", "suite-pack.json", "suite-lock.json"): + for name in ("manifest.json", "finalization-status.json", "suite-pack.json", "suite-lock.json", "clip-distribution.json"): copy_if_exists(source_suite_dir / name, staged_resource_dir / name) if (source_suite_dir / "notices").exists(): shutil.copytree(source_suite_dir / "notices", staged_resource_dir / "notices") + +def build_clip_bundle(*, source_suite_dir: Path, bundle_out: Path, base_url: str = "", + release_tag: str | None = None, published: bool = False) -> None: + """Stage the separately addressable per-clip bundle (no upload performed). + + The bundle root contains flat, unique ``downloadName`` files suitable for + GitHub release assets, plus clip-distribution.json. Logical per-clip paths + remain in the metadata. Bytes come from the frozen canonical tree. + """ + source_suite_dir = source_suite_dir.resolve() + bundle_out = bundle_out.resolve() + manifest = suite.load_default_suite_manifest() + metadata = suite.build_clip_distribution_metadata( + str(source_suite_dir), manifest=manifest, base_url=base_url, + release_tag=release_tag, published=published, + ) + shutil.rmtree(bundle_out, ignore_errors=True) + bundle_out.mkdir(parents=True, exist_ok=True) + for clip_id, entry in metadata["clips"].items(): + for asset in entry["assets"]: + relative = str(asset["path"]) + if str(asset["role"]) == "clip": + source = source_suite_dir / "canonical" / entry["fileName"] + else: + source = source_suite_dir / "notices" / relative.rsplit("notices/", 1)[-1] + if not source.is_file(): + raise RuntimeError(f"canonical asset missing for staged bundle: {source}") + destination = bundle_out / str(asset["downloadName"]) + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, destination) + digest = hashlib.sha256(destination.read_bytes()).hexdigest() + if digest != asset["sha256"] or destination.stat().st_size != asset["byteSize"]: + raise RuntimeError(f"staged bundle asset does not match frozen identity: {relative}") + with open(bundle_out / "clip-distribution.json", "w", encoding="utf-8") as handle: + json.dump(metadata, handle, indent=2, sort_keys=False) + handle.write("\n") + + def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description="Build the external EncodingDB suite pack and stage client suite resources without embedded canonical media.") parser.add_argument("--source-suite-dir", default=str(ROOT_DIR / "client" / "resources" / "test_suite_v1")) parser.add_argument("--staged-resource-dir", required=True) parser.add_argument("--pack-out", required=True) + parser.add_argument("--clip-bundle-out", default=None, + help="Also stage the per-clip publishable bundle in this directory") + parser.add_argument("--clip-base-url", default="", + help="Public base URL for the per-clip bundle (empty until published)") + parser.add_argument("--clip-release-tag", default=None) return parser.parse_args() @@ -63,6 +108,15 @@ def main() -> int: staged_resource_dir=Path(args.staged_resource_dir), pack_out=Path(args.pack_out), ) + if args.clip_bundle_out: + build_clip_bundle( + source_suite_dir=Path(args.source_suite_dir), + bundle_out=Path(args.clip_bundle_out), + base_url=args.clip_base_url, + release_tag=args.clip_release_tag, + published=bool(args.clip_base_url), + ) + print(f"staged per-clip bundle (not uploaded): {os.path.abspath(args.clip_bundle_out)}") print(f"staged suite resources: {os.path.abspath(args.staged_resource_dir)}") print(f"suite pack: {os.path.abspath(args.pack_out)}") return 0 diff --git a/scripts/provision_pinned_runtime.py b/scripts/provision_pinned_runtime.py new file mode 100644 index 00000000..284063f8 --- /dev/null +++ b/scripts/provision_pinned_runtime.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Recover the reviewed FFmpeg bytes from immutable candidate archives. + +The original BtbN autobuild tag was removed upstream. These existing public +EncodingDB candidate archives embed the *same* FFmpeg/FFprobe bytes. Verify +the archive digest and each extracted binary against the checked-in runtime +lock before CI or a native build may use them. Requires the already-pinned +PyInstaller build dependency; does not register a new runtime identity. +""" + +import argparse +import hashlib +import json +import stat +import tarfile +import tempfile +import urllib.request +import zipfile +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +RELEASE = "https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909" +SOURCES = { + "linux": { + "archive": "encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167.tar.gz", + "archiveSha256": "cb2712b94b705cf447eb4b550e8f3b2be1bfcc6120db1f361a6534234bfb325c", + "archiveBytes": 169129799, + "member": "encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167/encodingdb-client-linux", + "directory": "ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1", + "toc": {"ffmpeg": "bin/linux/ffmpeg", "ffprobe": "bin/linux/ffprobe"}, + }, + "win": { + "archive": "encodingdb-windows-candidate-ci34430919675-2d3ed7d4d167.zip", + "archiveSha256": "9e00b681397044ced552ec9539d10af42c731fe5a8b583eb9efb72de917fbe05", + "archiveBytes": 294384846, + "member": "encodingdb-client-windows-console.exe", + "directory": "ffmpeg-n8.1.2-51-g7ba069f4f1-win64-gpl-8.1", + "toc": {"ffmpeg": r"bin\win\ffmpeg.exe", "ffprobe": r"bin\win\ffprobe.exe"}, + }, +} + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _download(url: str, target: Path) -> None: + with urllib.request.urlopen(url, timeout=30) as response, target.open("wb") as output: + while chunk := response.read(1024 * 1024): + output.write(chunk) + + +def provision(platform: str, output_root: Path, *, archive_path: Path | None = None) -> Path: + spec = SOURCES[platform] + lock = json.loads((ROOT / "client/resources/runtime/ffmpeg-lock.json").read_text())["platforms"][platform] + output_root.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix="reviewed-runtime-") as temp: + temp_root = Path(temp) + candidate = archive_path or temp_root / spec["archive"] + if archive_path is None: + _download(f"{RELEASE}/{spec['archive']}", candidate) + if candidate.stat().st_size != spec["archiveBytes"] or _sha256(candidate) != spec["archiveSha256"]: + raise RuntimeError("candidate archive differs from reviewed SHA-256/size") + + onefile = temp_root / Path(spec["member"]).name + if platform == "linux": + with tarfile.open(candidate, "r:gz") as archive: + member = archive.getmember(spec["member"]) + if not member.isfile(): + raise RuntimeError("candidate executable is not a regular file") + source = archive.extractfile(member) + if source is None: + raise RuntimeError("candidate executable cannot be read") + with source, onefile.open("wb") as output: + while chunk := source.read(1024 * 1024): + output.write(chunk) + else: + with zipfile.ZipFile(candidate) as archive, archive.open(spec["member"]) as source, onefile.open("wb") as output: + while chunk := source.read(1024 * 1024): + output.write(chunk) + + from PyInstaller.archive.readers import CArchiveReader + reader = CArchiveReader(str(onefile)) + bundle = output_root / spec["directory"] / "bin" + bundle.mkdir(parents=True, exist_ok=True) + for name, toc_name in spec["toc"].items(): + data = reader.extract(toc_name) + expected = lock[name] + if len(data) != expected["byteSize"] or hashlib.sha256(data).hexdigest() != expected["sha256"]: + raise RuntimeError(f"embedded {name} differs from reviewed runtime lock") + target = bundle / Path(toc_name.replace("\\", "/")).name + target.write_bytes(data) + if platform == "linux": + target.chmod(target.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + if _sha256(target) != expected["sha256"]: + raise RuntimeError(f"installed {name} changed after verification") + receipt = {"platform": platform, "archive": spec["archive"], + "archiveSha256": spec["archiveSha256"], + "runtime": {name: lock[name]["sha256"] for name in spec["toc"]}} + (bundle.parent / "provision-receipt.json").write_text(json.dumps(receipt, indent=2) + "\n") + return bundle + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--platform", choices=sorted(SOURCES), required=True) + parser.add_argument("--output-root", type=Path, required=True) + parser.add_argument("--archive-path", type=Path, help="Use an already-downloaded candidate archive") + args = parser.parse_args() + bundle = provision(args.platform, args.output_root, archive_path=args.archive_path) + print(f"Reviewed {args.platform} runtime: {bundle}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/release_manifest_lib.py b/scripts/release_manifest_lib.py index 186f9b08..d8d94a57 100644 --- a/scripts/release_manifest_lib.py +++ b/scripts/release_manifest_lib.py @@ -127,6 +127,16 @@ def read_client_minimum_version() -> str: return client_match.group(1) +def read_client_implementation_version() -> str: + source = read_text(ROOT_DIR / "client" / "main.py") + match = re.search(r'^CLIENT_VERSION\s*=\s*"([^"]+)"', source, re.MULTILINE) + release = json.loads(read_text(ROOT_DIR / "release.json")) + version = release.get("clientImplementationVersion") + if not match or not isinstance(version, str) or match.group(1) != version: + raise RuntimeError("client implementation version differs from release.json") + return version + + def load_vmaf_manifest() -> Dict[str, Any]: with (ROOT_DIR / "client" / "resources" / "vmaf" / "manifest.json").open("r", encoding="utf-8") as handle: payload = json.load(handle) @@ -405,6 +415,7 @@ def build_release_manifest( "executableIdentity": executable_identity(artifact_path), }, "protocol": { + "clientVersion": read_client_implementation_version(), "benchmarkProtocolVersion": config.BENCHMARK_PROTOCOL_VERSION, "minimumClientVersion": read_client_minimum_version(), }, diff --git a/scripts/release_preflight.sh b/scripts/release_preflight.sh index c7022445..797b489f 100755 --- a/scripts/release_preflight.sh +++ b/scripts/release_preflight.sh @@ -45,7 +45,7 @@ run_root() { ( cd "$ROOT_DIR" "$@" - ) + ) || die "command failed: $*" } run_shell() { @@ -53,7 +53,7 @@ run_shell() { ( cd "$ROOT_DIR" bash -c "$*" - ) + ) || die "command failed: $*" } json_field() { @@ -89,8 +89,8 @@ check_metadata() { [[ -f "$ROOT_DIR/CHANGELOG.md" ]] || die "CHANGELOG.md missing" [[ -f "$ROOT_DIR/release.json" ]] || die "release.json missing" - python3 - "$ROOT_DIR/release.json" "$ROOT_DIR/client/resources/test_suite_v1/finalization-status.json" "$ROOT_DIR/CHANGELOG.md" <<'PY' -import json, sys + python3 - "$ROOT_DIR/release.json" "$ROOT_DIR/client/resources/test_suite_v1/finalization-status.json" "$ROOT_DIR/CHANGELOG.md" "$ROOT_DIR/client/main.py" <<'PY' || die "metadata identity check failed" +import json, re, sys payload = json.load(open(sys.argv[1], encoding="utf-8")) status = json.load(open(sys.argv[2], encoding="utf-8")) changelog = open(sys.argv[3], encoding="utf-8").read() @@ -98,11 +98,14 @@ expected = { "benchmarkProtocolVersion": "7.1", "plFormulaVersion": "7.0", "suiteVersion": "encodingdb-test-suite-v1", - "clientImplementationVersion": "client/0.3.1", } for key, value in expected.items(): if payload.get(key) != value: raise SystemExit(f"release.json {key} must be {value}") +client_source = open(sys.argv[4], encoding="utf-8").read() +client_match = re.search(r'^CLIENT_VERSION\s*=\s*"([^"]+)"', client_source, re.MULTILINE) +if not client_match or payload.get("clientImplementationVersion") != client_match.group(1): + raise SystemExit("release.json clientImplementationVersion must match client/main.py") if payload.get("projectVersion") is not None and not isinstance(payload["projectVersion"], str): raise SystemExit("release.json projectVersion must be a string or null") if payload.get("releaseDate") is not None and not isinstance(payload["releaseDate"], str): diff --git a/scripts/test-windows-gui-harness-guards.ps1 b/scripts/test-windows-gui-harness-guards.ps1 index 529c64b1..ca7dbb41 100644 --- a/scripts/test-windows-gui-harness-guards.ps1 +++ b/scripts/test-windows-gui-harness-guards.ps1 @@ -9,7 +9,7 @@ $ErrorActionPreference='Stop' $tokens=$null; $parseErrors=$null $ast=[Management.Automation.Language.Parser]::ParseFile($Harness,[ref]$tokens,[ref]$parseErrors) if ($parseErrors.Count) { throw ($parseErrors | Out-String) } -foreach ($name in @('Get-OwnedExitConfirmation','Record-CleanupFailure','Get-OwnedClientTree','Get-ObservedRunControl','Get-ObservedModeControl','Set-OwnedForeground','Select-AdvancedSingleMode','Invoke-RunAction','Wait-Until','Observe-Processes','Get-CompletionMarkers','Wait-Encoder','Wait-NoEncoders','Get-ActiveEncodeEvidence','Get-DurableMeasuredAttempts','Wait-MeasuredThenActiveEncode')) { +foreach ($name in @('Get-OwnedExitConfirmation','Get-OwnedDownloadEstimateConfirmation','Confirm-ObservedDownloadEstimate','Record-CleanupFailure','Get-OwnedClientTree','Get-ObservedRunControl','Get-ObservedModeControl','Set-OwnedForeground','Select-AdvancedSingleMode','Invoke-RunAction','Wait-Until','Observe-Processes','Get-CompletionMarkers','Wait-Encoder','Wait-NoEncoders','Get-ActiveEncodeEvidence','Get-DurableMeasuredAttempts','Wait-MeasuredThenActiveEncode')) { $definitions=@($ast.FindAll({param($node) $node -is [Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq $name},$true)) if ($definitions.Count -ne 1) { throw "Expected one real $name definition." } Invoke-Expression $definitions[0].Extent.Text @@ -28,6 +28,8 @@ public static class EdbWindows { public static long ObservedFocus=0; public static string FocusFailure=null; public static bool SetForegroundWindow(IntPtr h){if(ActivateSucceeds){Foreground=h;return true;}return false;} public static IntPtr GetForegroundWindow(){return Foreground;} + public static long LastMessageTarget=0; public static uint LastMessage=0; + public static IntPtr SendMessageTimeout(IntPtr h,uint m,IntPtr w,IntPtr l,uint flags,uint timeout,out UIntPtr result){LastMessageTarget=h.ToInt64();LastMessage=m;result=new UIntPtr(1);return Data.ContainsKey(h.ToInt64())?new IntPtr(1):IntPtr.Zero;} public static bool GetWindowRect(IntPtr h, out Rect r){ var b=Data[h.ToInt64()].Bounds; r=new Rect{Left=b[0],Top=b[1],Right=b[2],Bottom=b[3]}; return true; } public static IntPtr SetThreadDpiAwarenessContext(IntPtr c){ return new IntPtr(-1); } public static bool ShowWindow(IntPtr h,int mode){return true;} @@ -87,9 +89,9 @@ $script:events=@() function Record-Event([string]$Kind,$Data) { $script:events+=@{kind=$Kind;data=$Data} } function Capture-Ui([string]$Label) { return @() } function Run-Fixture { - # Exact hosted-runner geometry from CI 35651286705 prepare-stop dumps: the real three-button - # run row (Start 99px, Stop 76px, Retry Queued Uploads 128px), the real seven-control mode - # row, plus the decoy log frame whose Text child and ScrollBar child match every purely + # Hosted-runner geometry: the three-button run row (Start, Stop, Retry due uploads) and the + # current three-control guided mode row from CI 35976723286 prepare-stop dumps, plus the + # decoy log frame whose Text child and ScrollBar child match every purely # geometric rule except child count and class. [void][EdbWindows]::Data.Clear() Run-Window 1 0 @(8,8,1016,703) 'TkTopLevel'; $t=[EdbWindows]::Data[1];$t.Text='EncodingDB Windows Client' @@ -103,11 +105,7 @@ function Run-Fixture { Run-Window 40 1 @(40,78,984,99) 'TkChild' Run-Window 41 40 @(40,79,75,98) 'TkChild' Run-Window 42 40 @(83,78,214,99) 'TkChild' - Run-Window 43 40 @(230,78,412,99) 'TkChild' - Run-Window 44 40 @(424,79,463,98) 'TkChild' - Run-Window 45 40 @(469,78,524,98) 'TkChild' - Run-Window 46 40 @(536,79,592,98) 'TkChild' - Run-Window 47 40 @(598,78,653,98) 'TkChild' + Run-Window 43 40 @(230,78,385,99) 'TkChild' } Case 'run-row:real-identity-with-hosted-decoys' Run-Fixture @@ -135,10 +133,10 @@ Run-Fixture;[EdbWindows]::Data[12].Class='TkChild';[EdbWindows]::Data[11].Bounds Run-Window 13 10 @(468,490,700,673) 'TkChild' Assert-Throws {Get-ObservedRunControl 'Start'} '*BLOCKED_GUI_POINT*observed 2 candidate Start/Stop rows*not established*' Case 'run-row:mode-row-uniform-height-never-selected' -# Flattening every mode-row child to the mode row's uniform height must not let the seven-control +# Flattening every mode-row child to the mode row's uniform height must not let the three-control # row impersonate the run row nor displace the real Start/Stop identity. Run-Fixture -foreach ($id in @(41,43,44,45,46,47)) { $row1=[EdbWindows]::Data[$id]; $row1.Bounds=@($row1.Bounds[0],78,$row1.Bounds[2],99) } +foreach ($id in @(41,43)) { $row1=[EdbWindows]::Data[$id]; $row1.Bounds=@($row1.Bounds[0],78,$row1.Bounds[2],99) } $obs=Get-ObservedRunControl 'Start' Assert-True ($obs.child -eq [IntPtr]21 -and $obs.x -eq 89 -and $obs.y -eq 179) 'The uniform-height mode row impersonated the run control.' Case 'run-row:reject-first-not-wider' @@ -156,27 +154,23 @@ Assert-Throws {Get-ObservedRunControl 'Start'} '*BLOCKED_GUI_POINT*not establish Case 'mode-row:real-identity-after-run-row-rejection' $mode=Get-ObservedModeControl Assert-True ($mode.child -eq [IntPtr]42 -and $mode.owner -eq 42 -and $mode.x -eq 148 -and $mode.y -eq 88) 'Real mode observer picked the wrong mode combobox on the hosted geometry.' -Case 'mode-row:reject-eight-control-row' -Run-Fixture;Run-Window 48 40 @(660,78,715,98) 'TkChild' +Case 'mode-row:reject-four-control-row' +Run-Fixture;Run-Window 48 40 @(401,78,456,98) 'TkChild' Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-mixed-class-children' -Run-Fixture;$mflipped=[EdbWindows]::Data[44];$mflipped.Class='ScrollBar' +Run-Fixture;$mflipped=[EdbWindows]::Data[43];$mflipped.Class='ScrollBar' Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-overlapping-children' Run-Fixture;$mover=[EdbWindows]::Data[43];$mover.Bounds=@(200,78,412,99) Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-undersized-combobox' Run-Fixture;$msmall=[EdbWindows]::Data[42];$msmall.Bounds=@(83,78,110,99) -Assert-Throws {Get-ObservedModeControl} '*unexpected size*' -Case 'mode-row:reject-ambiguous-second-seven-control-row' +Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*not established*' +Case 'mode-row:reject-ambiguous-second-three-control-row' Run-Fixture;Run-Window 50 1 @(40,216,984,237) 'TkChild' Run-Window 51 50 @(40,217,75,236) 'TkChild' Run-Window 52 50 @(83,216,214,237) 'TkChild' -Run-Window 53 50 @(230,216,412,237) 'TkChild' -Run-Window 54 50 @(424,217,463,236) 'TkChild' -Run-Window 55 50 @(469,216,524,236) 'TkChild' -Run-Window 56 50 @(536,217,592,236) 'TkChild' -Run-Window 57 50 @(598,216,653,236) 'TkChild' +Run-Window 53 50 @(230,216,385,237) 'TkChild' Assert-Throws {Get-ObservedModeControl} '*observed 2 candidate configuration rows*' function Mode-Fixture { Run-Fixture @@ -277,6 +271,37 @@ Assert-Throws {Get-OwnedExitConfirmation} '*multiple owned native Exit dialogs*' Case 'exit-confirmation:reject-two-yes-buttons' Ready-Fixture;[EdbWindows]::Data.Add(3,[EdbWindows]::Data[2]) Assert-Throws {Get-OwnedExitConfirmation} '*multiple observed enabled native Yes buttons*' +function Download-Estimate-Fixture { + [EdbWindows]::Data.Clear() + Run-Window 1 0 @(8,8,1016,703) 'TkTopLevel';[EdbWindows]::Data[1].Text='EncodingDB Windows Client' + Run-Window 2 0 @(317,310,722,469) '#32770';[EdbWindows]::Data[2].Text='Download and storage estimate' + Run-Window 3 2 @(541,429,616,452) 'Button';[EdbWindows]::Data[3].Text='&Yes';[EdbWindows]::Data[3].Control=6 + Run-Window 4 2 @(624,429,699,452) 'Button';[EdbWindows]::Data[4].Text='&No';[EdbWindows]::Data[4].Control=7 + Run-Window 5 2 @(387,367,686,395) 'Static';[EdbWindows]::Data[5].Text='This run may download 1.4 GB of frozen reference media. Estimated peak extra storage: 2.9 GB. Continue?' + [EdbWindows]::ActivateSucceeds=$true;[EdbWindows]::Foreground=[IntPtr]::Zero + [EdbWindows]::LastMessageTarget=0;[EdbWindows]::LastMessage=0 + $script:harnessDeadline=[DateTime]::UtcNow.AddSeconds(30) + $script:events=@() +} +Case 'download-estimate:exact-native-dialog-and-bounded-yes' +Download-Estimate-Fixture +$estimate=Get-OwnedDownloadEstimateConfirmation +Assert-True ($estimate.dialog -eq [IntPtr]2 -and $estimate.button -eq [IntPtr]3 -and $estimate.owner -eq 42) 'The estimate observer did not identify the owned Yes button.' +Confirm-ObservedDownloadEstimate +Assert-True ([EdbWindows]::LastMessageTarget -eq 3 -and [EdbWindows]::LastMessage -eq 0x00F5) 'The observed estimate Yes button did not receive BM_CLICK.' +Assert-True (@($script:events | Where-Object { $_.kind -eq 'observed-download-estimate-approved' }).Count -eq 1) 'Estimate approval evidence was not recorded.' +Case 'download-estimate:reject-unexpected-title' +Download-Estimate-Fixture;[EdbWindows]::Data[2].Text='Allow Benchmark Publication' +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*unexpected owned dialog*' +Case 'download-estimate:reject-missing-cost-disclosure' +Download-Estimate-Fixture;[EdbWindows]::Data[5].Text='Continue?' +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*lacks one visible Yes/No pair and the expected cost disclosure*' +Case 'download-estimate:reject-wrong-button-identity' +Download-Estimate-Fixture;[EdbWindows]::Data[3].Control=7 +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*lacks one visible Yes/No pair and the expected cost disclosure*' +Case 'download-estimate:reject-foreign-owner' +Download-Estimate-Fixture;[EdbWindows]::Data[5].Owner=99 +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*owner differs from the client*' Case 'cleanup:primary-error-and-owned-identity-preserved' $script:events=@() $receipt=@{status='BLOCKED';error='original failure';primaryError=@{message='original failure';stage='close readiness'};cleanupErrors=@()} diff --git a/scripts/test-windows-native-gui.ps1 b/scripts/test-windows-native-gui.ps1 index f7c5d312..d3a1457c 100644 --- a/scripts/test-windows-native-gui.ps1 +++ b/scripts/test-windows-native-gui.ps1 @@ -396,6 +396,46 @@ function Confirm-ObservedExit { if ($sent -eq [IntPtr]::Zero) { throw 'BLOCKED_GUI_AUTOMATION: native Yes button did not accept the bounded click.' } Record-Event 'observed-native-button-clicked' @{ name='Yes'; processId=$buttonOwner; dialogHandle=$dialog.ToInt64(); buttonHandle=$button.ToInt64(); controlId=6 } } +function Get-OwnedDownloadEstimateConfirmation { + # The guided client asks before extracting the frozen media. Accept only this exact native + # Yes/No dialog, with its visible cost disclosure, from the owned packaged process. + $roots=@(Get-OwnedWindows | Where-Object { [EdbWindows]::Text($_) -eq 'EncodingDB Windows Client' }) + if ($roots.Count -ne 1) { throw 'BLOCKED_GUI_AUTOMATION: expected one owned client window before download consent.' } + $dialogs=@(Get-OwnedWindows | Where-Object { $_ -ne $roots[0] }) + if ($dialogs.Count -eq 0) { return $null } + if ($dialogs.Count -ne 1) { throw 'BLOCKED_GUI_AUTOMATION: multiple owned dialogs appeared before download consent.' } + $dialog=$dialogs[0] + if ([EdbWindows]::Text($dialog) -ne 'Download and storage estimate' -or [EdbWindows]::Class($dialog) -ne '#32770') { + throw 'BLOCKED_GUI_AUTOMATION: unexpected owned dialog before download consent.' + } + $children=@([EdbWindows]::Windows($dialog) | Where-Object { [EdbWindows]::GetParent($_) -eq $dialog }) + $yes=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Button' -and [EdbWindows]::Text($_).Replace('&','') -eq 'Yes' -and [EdbWindows]::GetDlgCtrlID($_) -eq 6 -and [EdbWindows]::IsWindowVisible($_) -and [EdbWindows]::IsWindowEnabled($_) }) + $no=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Button' -and [EdbWindows]::Text($_).Replace('&','') -eq 'No' -and [EdbWindows]::GetDlgCtrlID($_) -eq 7 -and [EdbWindows]::IsWindowVisible($_) -and [EdbWindows]::IsWindowEnabled($_) }) + $disclosure=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Static' -and [EdbWindows]::Text($_) -match '^This run may download .+ of frozen reference media\. Estimated peak extra storage: .+\. Continue\?$' }) + if ($yes.Count -ne 1 -or $no.Count -ne 1 -or $disclosure.Count -ne 1) { + throw 'BLOCKED_GUI_AUTOMATION: download estimate lacks one visible Yes/No pair and the expected cost disclosure.' + } + [uint32]$rootOwner=0; [void][EdbWindows]::GetWindowThreadProcessId($roots[0],[ref]$rootOwner) + foreach ($handle in @($dialog,$yes[0],$no[0],$disclosure[0])) { + [uint32]$owner=0; [void][EdbWindows]::GetWindowThreadProcessId($handle,[ref]$owner) + if ($owner -ne $rootOwner) { throw 'BLOCKED_GUI_AUTOMATION: download estimate control owner differs from the client.' } + } + return @{ dialog=$dialog; button=$yes[0]; owner=$rootOwner } +} +function Confirm-ObservedDownloadEstimate { + $script:operationStage='start:capture-download-estimate' + [void](Capture-Ui 'before-download-estimate-Yes') + $confirmation=Get-OwnedDownloadEstimateConfirmation + if ($null -eq $confirmation) { throw 'BLOCKED_GUI_AUTOMATION: owned download estimate is no longer ready.' } + $dialog=$confirmation.dialog; $button=$confirmation.button + [void][EdbWindows]::SetForegroundWindow($dialog) + Wait-Until { return [EdbWindows]::GetForegroundWindow() -eq $dialog } 5 'BLOCKED_GUI_FOCUS: download estimate did not receive foreground focus.' + [UIntPtr]$result=[UIntPtr]::Zero + $script:operationStage='start:click-download-estimate-yes' + $sent=[EdbWindows]::SendMessageTimeout($button,0x00F5,[IntPtr]::Zero,[IntPtr]::Zero,2,2000,[ref]$result) + if ($sent -eq [IntPtr]::Zero) { throw 'BLOCKED_GUI_AUTOMATION: download estimate Yes button did not accept the bounded click.' } + Record-Event 'observed-download-estimate-approved' @{ processId=$confirmation.owner; dialogHandle=$dialog.ToInt64(); buttonHandle=$button.ToInt64(); controlId=6 } +} function Get-OwnedClientTree { # Enumerate the exact owned client window and every descendant child HWND once, in physical # pixels, so row-structure predicates and click coordinates share one observation space. @@ -426,14 +466,15 @@ function Get-ObservedRunControl([ValidateSet('Start','Stop')][string]$Action) { # Tk widgets expose no accessible name (verified: every descendant is an unnamed UIA Pane and only # the TkTopLevel carries window text). The guided client packs the run controls as three exact # native child HWNDs inside one row container (client/windows_gui.py: 'Start benchmark (Alt+B)', - # 'Stop (Alt+S)', 'Retry Queued Uploads'), so the controls are reobserved from that live Win32 + # 'Stop (Alt+S)', 'Retry due uploads'), so the controls are reobserved from that live Win32 # structure: the unique row at least half the root width whose visible children are exactly # three, share one class and one height tightly equal to the row height, are ordered left to # right flush with the row's left edge with gaps <=32px, and the first button is wider than the # second. Start is the first; Stop is the second. Every other row fails a predicate on the - # hosted runner (CI 35651286705 failure.win32.json: the log frame holds only a Text child and a - # ScrollBar child, mixing classes; configuration rows hold 2, 7 or 8 children or mixed child - # heights). The former two-button contract blocked this phase at run35651286705. + # hosted runner (CI 35976723286 launch.win32.json: the log frame mixes a Text child and a + # ScrollBar child; the current three-control mode row has a narrow first label, and other + # configuration rows fail child count or height). The former two-button contract blocked + # this phase at run35651286705. $tree=Get-OwnedClientTree $rootRect=$tree.rootRect; $byHandle=$tree.byHandle $rootWidth=$rootRect.Right-$rootRect.Left @@ -476,10 +517,11 @@ function Get-ObservedRunControl([ValidateSet('Start','Stop')][string]$Action) { } function Get-ObservedModeControl { # The mode selector has no accessible name either, so it is identified structurally: the unique - # row at least half the root width whose visible same-class children are exactly seven controls + # row at least half the root width whose visible same-class children are exactly three controls # in one non-overlapping left-to-right line flush with the row's left edge (client/windows_gui.py - # row1: Mode label, Mode combobox, No-submit checkbutton, Retries label, Retries spinbox, Batch - # label, Batch spinbox; CI 35651286705 launch.win32.json row at y=78). The combobox is the + # row1: Mode label, Mode combobox, Save locally checkbutton; CI 35976723286 launch.win32.json + # row at y=78). The short label followed by the wider combobox distinguishes this row from + # the three-button Start/Stop row and the saved-work row. The combobox is the # second control from the left. The guarded click that follows must post an aligned owned popup # before any value-changing key is sent, so a structurally stale identity can never commit. $tree=Get-OwnedClientTree @@ -491,11 +533,14 @@ function Get-ObservedModeControl { $pRect=$candidate.rect if (($pRect.Right-$pRect.Left) -lt [Math]::Floor($rootWidth*0.5)) { continue } $kids=@($byHandle.Values | Where-Object { $_.parent -eq $key -and $_.visible }) - if ($kids.Count -ne 7) { continue } + if ($kids.Count -ne 3) { continue } $classes=@($kids | ForEach-Object { $_.class } | Select-Object -Unique) if ($classes.Count -ne 1) { continue } $ordered=@($kids | Sort-Object { $_.rect.Left }) if ($ordered[0].rect.Left -ne $pRect.Left) { continue } + $labelWidth=$ordered[0].rect.Right-$ordered[0].rect.Left + $comboWidth=$ordered[1].rect.Right-$ordered[1].rect.Left + if ($labelWidth -lt 16 -or $labelWidth -gt 64 -or $comboWidth -lt 64 -or $comboWidth -gt 400 -or $labelWidth -ge $comboWidth) { continue } $inside=$true foreach ($kid in $ordered) { $r=$kid.rect @@ -508,7 +553,7 @@ function Get-ObservedModeControl { if (-not $inside) { continue } $rows+=,@{ row=$candidate; ordered=$ordered } } - if ($rows.Count -ne 1) { throw "BLOCKED_GUI_POINT: observed $($rows.Count) candidate configuration rows on this fresh instance; the unique seven-control mode row is not established." } + if ($rows.Count -ne 1) { throw "BLOCKED_GUI_POINT: observed $($rows.Count) candidate configuration rows on this fresh instance; the unique three-control mode row is not established." } $combo=$rows[0].ordered[1] $rect=$combo.rect $width=$rect.Right-$rect.Left; $height=$rect.Bottom-$rect.Top @@ -808,6 +853,22 @@ try { # default; deliberately select Single (advanced) first or Start would run a sweep. Select-AdvancedSingleMode Invoke-RunAction 'Start' + $script:downloadEstimateStatus=$null + Wait-Until { + $estimate=Get-OwnedDownloadEstimateConfirmation + if ($null -ne $estimate) { $script:downloadEstimateStatus='prompt'; return $true } + # ttk's disabled visual state is not Win32 IsWindowEnabled. A grey Stop button + # can report enabled while this modal blocks the Tk callback. Only actual + # preparation or an owned encoder proves that Start passed the decision. + [void](Observe-Processes) + if ($null -ne $phase.preparationProbe) { $script:downloadEstimateStatus='already-running'; return $true } + $encode=Get-ActiveEncodeEvidence + if ($encode.identified.Count -or $encode.unidentified.Count) { $script:downloadEstimateStatus='already-running'; return $true } + if ($script:process.HasExited) { throw 'BLOCKED_GUI_AUTOMATION: client exited before the download decision or source preparation.' } + return $false + } 20 'BLOCKED_GUI_AUTOMATION: neither a download estimate nor an active run followed Start.' + if ($script:downloadEstimateStatus -eq 'prompt') { Confirm-ObservedDownloadEstimate } + else { Record-Event 'download-estimate-not-shown' @{ phase=$name } } if ($name -eq 'prepare-stop') { Wait-Until { [void](Observe-Processes); return $null -ne $phase.preparationProbe } $AcquisitionSeconds 'Source preparation probe was not observed.' if (@(Get-ChildItem -Path $phase.queue -Recurse -Filter 'manifest.json' -ErrorAction SilentlyContinue).Count) { throw 'Campaign already exists; preparation cancellation was not exercised.' } diff --git a/scripts/test_campaign_conservation.py b/scripts/test_campaign_conservation.py new file mode 100644 index 00000000..45265f4e --- /dev/null +++ b/scripts/test_campaign_conservation.py @@ -0,0 +1,157 @@ +"""Seeded state exercises for the independent campaign conservation oracle.""" + +import hashlib +import importlib.util +import json +import random +from pathlib import Path + + +MODULE = Path(__file__).with_name("campaign_conservation.py") +SPEC = importlib.util.spec_from_file_location("campaign_conservation", MODULE) +oracle = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(oracle) + + +def _write(path, value): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value), encoding="utf-8") + + +def _fixture(tmp_path, seed): + rng = random.Random(seed) + campaign_id = f"campaign-{seed:016x}" + queue = tmp_path / "queue" + root = queue / "campaigns" / campaign_id + _write(root / "manifest.json", { + "tasks": [{"recipe": "a"}, {"recipe": "b"}], + "protocolConfig": { + "warmup_runs": 1, "minimum_measured_runs": 2, + "max_adaptive_repeats": 2, "stability_threshold_ratio": 0.03, + }, + }) + order = 0 + expected = [] + for recipe in ("a", "b"): + state = rng.choice(("unstarted", "partial", "stable", "adaptive")) + if state == "unstarted": + continue + phases = ["warmup"] + (["measured"] if state == "partial" else ["measured"] * (4 if state == "adaptive" else 2)) + for index, phase in enumerate(phases): + order += 1 + sha = hashlib.sha256(f"{recipe}:{index}".encode()).hexdigest() + record = { + "schedule": {"campaign_id": campaign_id, "recipe_id": recipe, + "phase": phase, "execution_order": order, + "repetition_index": index if phase == "measured" else 1}, + "timing": {"elapsed_s": float(index + 1) if state == "adaptive" else 1.0}, + "countedForStability": phase == "measured", + "overallValidity": {"state": "valid"}, + "metadata": {"info": {"artifactSha256": sha, "error": None}}, + "skippedBeforeEncode": False, + } + _write(root / f"attempt-{order:06d}.json", record) + if phase != "measured" or state == "partial": + continue + expected.append(order) + verdict = rng.choice(("confirmed", "queued", "terminal", "unstaged", "journal_only")) + if verdict != "journal_only": + payload = {"runCreate": {"campaignId": campaign_id, "payloadHash": sha}, + "artifactSha256": sha} + _write(root / f"submission-{order:06d}.json", payload) + local_hash = oracle._payload_hash(payload) + if verdict == "queued": + _write(queue / f"{local_hash}.json", {"payload": payload}) + if verdict == "terminal": + _write(queue / "terminal" / f"{local_hash}.json", {"payload": payload}) + if verdict == "confirmed": + _write(root / f"submission-{order:06d}.accepted.json", { + "executionOrder": order, "benchmarkRunId": f"run-{seed}-{order}", + "artifactSha256": sha, + }) + return queue, campaign_id, expected + + +def test_seeded_conservation_states(tmp_path): + for seed in range(1, 25): + queue, campaign_id, expected = _fixture(tmp_path / str(seed), seed) + result = oracle.audit(queue, campaign_id) + assert result["issues"] == [], (seed, result) + assert result["frozenGroups"] == sum(result["groups"].values()) + assert result["attempts"]["total"] == sum(result["attempts"]["validity"].values()) + assert result["eligibleFinalized"] == len(expected) + assert result["eligibleFinalized"] == sum(result["publication"].values()) + + +def test_duplicate_acknowledged_server_identity_is_detected(tmp_path): + queue, campaign_id, eligible = _fixture(tmp_path, 49) + root = queue / "campaigns" / campaign_id + assert len(eligible) >= 2 + for order in eligible[:2]: + record = json.loads((root / f"attempt-{order:06d}.json").read_text()) + _write(root / f"submission-{order:06d}.accepted.json", { + "executionOrder": order, "benchmarkRunId": "same-server-run", + "artifactSha256": record["metadata"]["info"]["artifactSha256"], + }) + result = oracle.audit(queue, campaign_id) + assert "duplicate server run ID in accepted markers" in result["issues"] + + +def test_oracle_identifies_staged_copy_by_immutable_payload(): + payload = { + "submissionKind": "authoritative-artifact-run-v1", + "artifactPath": "/journal/original.mp4", + "artifactSha256": "a" * 64, + "runCreate": {"payloadHash": "b" * 64}, + } + staged = dict(payload, artifactPath="/queue/artifacts/copy.mp4", artifactManaged=True) + assert oracle._payload_hash(payload) == oracle._payload_hash(staged) + + +def test_persistence_and_acknowledgment_boundaries_conserve_identity(tmp_path): + queue = tmp_path / "queue" + campaign_id = "campaign-0000000000000001" + root = queue / "campaigns" / campaign_id + _write(root / "manifest.json", { + "tasks": [{"recipe": "r"}], + "protocolConfig": {"warmup_runs": 1, "minimum_measured_runs": 2, + "max_adaptive_repeats": 0, "stability_threshold_ratio": 0.03}, + }) + for order, phase in ((1, "warmup"), (2, "measured"), (3, "measured")): + _write(root / f"attempt-{order:06d}.json", { + "schedule": {"campaign_id": campaign_id, "recipe_id": "r", + "phase": phase, "execution_order": order}, + "timing": {"elapsed_s": 1.0}, + "countedForStability": phase == "measured", + "overallValidity": {"state": "valid"}, + "metadata": {"info": {"artifactSha256": f"{order:064x}"}}, + "skippedBeforeEncode": False, + }) + + def state(): + result = oracle.audit(queue, campaign_id) + assert result["issues"] == [] + assert result["eligibleFinalized"] == sum(result["publication"].values()) == 2 + return result["publication"] + + assert state() == {"journal_only": 2} + payload = {"submissionKind": "authoritative-artifact-run-v1", + "artifactPath": str(root / "measured.mp4"), + "artifactSha256": f"{2:064x}", + "runCreate": {"campaignId": campaign_id, "payloadHash": "b" * 64}} + _write(root / "submission-000002.json", payload) + assert state() == {"unstaged_envelope": 1, "journal_only": 1} + local_hash = oracle._payload_hash(payload) + queued = queue / f"{local_hash}.json" + _write(queued, {"payload": dict(payload, artifactPath="/managed/copy.mp4")}) + assert state() == {"queued": 1, "journal_only": 1} + queued.unlink() + _write(queue / "receipts" / f"{local_hash}.json", { + "localHash": local_hash, "response": {"benchmarkRun": {"id": "run-original"}}, + }) + assert state() == {"receipted_without_journal_marker": 1, "journal_only": 1} + _write(root / "submission-000002.accepted.json", { + "executionOrder": 2, "benchmarkRunId": "run-original", + "artifactSha256": f"{2:064x}", + }) + assert state() == {"confirmed": 1, "journal_only": 1}