From 6a11412697826235dcc3dace9a4793151b1d7700 Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 01:23:39 -0700 Subject: [PATCH 1/9] Reduce first-contribution transfer cost while preserving frozen evidence The quick clip can be acquired and resumed independently with exact hash and license verification. A staged flat asset inventory keeps the full-pack fallback explicit until the per-clip files are published. Constraint: Protocol 7.1 and frozen canonical bytes cannot change Constraint: Per-clip assets remain unpublished pending the release gate Confidence: high Scope-risk: moderate Directive: Publish and rehash every flat asset before setting published=true Tested: 14 acquisition tests; 40 frozen-suite/cancellation checks; suite drift check Not-tested: Public hosted URLs and native packaged first-launch acquisition --- client/ENCODINGDB_TEST_SUITE_V1.md | 4 +- .../test_suite_v1/clip-distribution.json | 215 +++++++ client/suite.py | 576 +++++++++++++++++- client/tests/test_suite_clip_acquisition.py | 397 ++++++++++++ scripts/prepare_client_suite_distribution.py | 56 +- 5 files changed, 1228 insertions(+), 20 deletions(-) create mode 100644 client/resources/test_suite_v1/clip-distribution.json create mode 100644 client/tests/test_suite_clip_acquisition.py diff --git a/client/ENCODINGDB_TEST_SUITE_V1.md b/client/ENCODINGDB_TEST_SUITE_V1.md index 88264131..9533f1be 100644 --- a/client/ENCODINGDB_TEST_SUITE_V1.md +++ b/client/ENCODINGDB_TEST_SUITE_V1.md @@ -18,7 +18,9 @@ python3 scripts/test_suite_drift_check.py The materializer verifies archive, metadata, notices and every reference, then installs matching client/server resources. Run it before native builds or Docker builds. Frozen references have no synthetic fallback; `build_test_suite_v1.py --rewrite-manifest` refuses a frozen suite. -Packaged clients obtain the archive beside the executable or through the manifest download URLs, with verified cache reuse and atomic extraction. Explicit overrides: `ENCODINGDB_SUITE_CACHE_DIR`, `ENCODINGDB_SUITE_PACK_PATH`, `ENCODINGDB_SUITE_PACK_URL`, and `ENCODINGDB_QUICK_CLIP_ID`. +Packaged clients reuse hash-verified cached clips and prefer the separately addressable clips declared in `clip-distribution.json` when those assets are published. The checked-in per-clip inventory is staged but **not published** yet; current public clients still use the external 1.51 GB pack. A full-pack fallback states its transfer size before downloading. `ENCODINGDB_ALLOW_FULL_PACK=0` refuses that fallback, and `ENCODINGDB_SUITE_CLIP_BASE_URL` selects a staged per-clip host for testing. Other overrides: `ENCODINGDB_SUITE_CACHE_DIR`, `ENCODINGDB_SUITE_PACK_PATH`, `ENCODINGDB_SUITE_PACK_URL`, and `ENCODINGDB_QUICK_CLIP_ID`. + +`scripts/prepare_client_suite_distribution.py --clip-bundle-out DIR` stages flat, unique release-asset filenames with exactly the frozen clip bytes and bound license notices. Staging does not upload them or activate public URLs. Publication and actual download verification belong to the release gate in PLA-90. Per-clip attributions, modification notices and license texts travel inside the pack and are hash-bound to its inventory. Third-party CC BY media is not relabeled CC0 or covered by Apache-2.0. diff --git a/client/resources/test_suite_v1/clip-distribution.json b/client/resources/test_suite_v1/clip-distribution.json new file mode 100644 index 00000000..2cffbf70 --- /dev/null +++ b/client/resources/test_suite_v1/clip-distribution.json @@ -0,0 +1,215 @@ +{ + "schemaVersion": 1, + "suiteId": "encodingdb-test-suite", + "suiteVersion": "encodingdb-test-suite-v1", + "manifestVersion": 2, + "source": "staged-unpublished", + "releaseTag": null, + "distribution": { + "baseUrl": "", + "clipUrlOverrideEnv": "ENCODINGDB_SUITE_CLIP_BASE_URL", + "published": false + }, + "manifest": { + "sha256": "e0fa76d96f75f5e88c3c95ff452d155159ba1ac0c282e80e4c60245658309a76", + "byteSize": 15791 + }, + "clips": { + "athletic-action-1080p24-final": { + "fileName": "athletic-action-1080p24-final.mkv", + "license": "CC-BY-3.0", + "assets": [ + { + "role": "clip", + "path": "athletic-action-1080p24-final/athletic-action-1080p24-final.mkv", + "downloadName": "athletic-action-1080p24-final--athletic-action-1080p24-final.mkv", + "sha256": "1e06fe0315d0cb90247a3dae2258327989e657496247ca8974ddd8c2431755de", + "byteSize": 139815565 + }, + { + "role": "notice", + "path": "athletic-action-1080p24-final/notices/athletic-action-1080p24-final.txt", + "downloadName": "athletic-action-1080p24-final--athletic-action-1080p24-final.txt", + "sha256": "b7c8d6954e9a5ead857aa2b31a1a31a9ba24951fae0cbb924827463fb2870338", + "byteSize": 1435 + }, + { + "role": "license", + "path": "athletic-action-1080p24-final/notices/CC-BY-3.0.txt", + "downloadName": "athletic-action-1080p24-final--CC-BY-3.0.txt", + "sha256": "e6bc9e9c474700b708f568bac9e5a8a9bcb2b1dad53442f5ba449fcb848b8e76", + "byteSize": 19467 + } + ] + }, + "natural-detail-1080p24-final": { + "fileName": "natural-detail-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "natural-detail-1080p24-final/natural-detail-1080p24-final.mkv", + "downloadName": "natural-detail-1080p24-final--natural-detail-1080p24-final.mkv", + "sha256": "cfcc51d48372138f6f66b3286fec5a533a808650098e40f119011060d8e51225", + "byteSize": 283872120 + }, + { + "role": "notice", + "path": "natural-detail-1080p24-final/notices/natural-detail-1080p24-final.txt", + "downloadName": "natural-detail-1080p24-final--natural-detail-1080p24-final.txt", + "sha256": "d353a009802973bbfb8910c7440440c3ae59d981b0f9c306515c2d10c671de66", + "byteSize": 1388 + }, + { + "role": "license", + "path": "natural-detail-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "natural-detail-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "film-grain-1080p24-final": { + "fileName": "film-grain-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "film-grain-1080p24-final/film-grain-1080p24-final.mkv", + "downloadName": "film-grain-1080p24-final--film-grain-1080p24-final.mkv", + "sha256": "67d3d2f5a4f8c617f223077e7071aaee625f014950126e0e93a16b27e578d603", + "byteSize": 336898554 + }, + { + "role": "notice", + "path": "film-grain-1080p24-final/notices/film-grain-1080p24-final.txt", + "downloadName": "film-grain-1080p24-final--film-grain-1080p24-final.txt", + "sha256": "046f2cb4547c0223dd072b5109e03394d80418fa915f909fbe4d24a9006ee49c", + "byteSize": 1165 + }, + { + "role": "license", + "path": "film-grain-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "film-grain-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "dark-gradients-1080p24-final": { + "fileName": "dark-gradients-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "dark-gradients-1080p24-final/dark-gradients-1080p24-final.mkv", + "downloadName": "dark-gradients-1080p24-final--dark-gradients-1080p24-final.mkv", + "sha256": "3377e6927fdd256633961520ace19f6b5b1494f689841ed3a29831e48e0d27c3", + "byteSize": 317829621 + }, + { + "role": "notice", + "path": "dark-gradients-1080p24-final/notices/dark-gradients-1080p24-final.txt", + "downloadName": "dark-gradients-1080p24-final--dark-gradients-1080p24-final.txt", + "sha256": "decd3e519d768ec76400ac6de3a7445c68b0df26e9b29adab55fb5847e634c4e", + "byteSize": 1179 + }, + { + "role": "license", + "path": "dark-gradients-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "dark-gradients-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "animation-1080p24-final": { + "fileName": "animation-1080p24-final.mkv", + "license": "CC-BY-4.0", + "assets": [ + { + "role": "clip", + "path": "animation-1080p24-final/animation-1080p24-final.mkv", + "downloadName": "animation-1080p24-final--animation-1080p24-final.mkv", + "sha256": "d70c4d9e85e88c4369b6391a21f0388a4a3a7b72e1b89f545e5897b4d5d820ea", + "byteSize": 173819800 + }, + { + "role": "notice", + "path": "animation-1080p24-final/notices/animation-1080p24-final.txt", + "downloadName": "animation-1080p24-final--animation-1080p24-final.txt", + "sha256": "74d6dce61b31a0039262021ae5bdda18d9e8ab61825000f442afa86d143b9ea9", + "byteSize": 1110 + }, + { + "role": "license", + "path": "animation-1080p24-final/notices/CC-BY-4.0.txt", + "downloadName": "animation-1080p24-final--CC-BY-4.0.txt", + "sha256": "9ba9550ad48438d0836ddab3da480b3b69ffa0aac7b7878b5a0039e7ab429411", + "byteSize": 18657 + } + ] + }, + "screen-text-1080p24-final": { + "fileName": "screen-text-1080p24-final.mkv", + "license": "CC0-1.0", + "assets": [ + { + "role": "clip", + "path": "screen-text-1080p24-final/screen-text-1080p24-final.mkv", + "downloadName": "screen-text-1080p24-final--screen-text-1080p24-final.mkv", + "sha256": "3c024a4aceb4ad09f8fa8cf51b6a4aba9be68460acb3f6a24a223e287bc05c8c", + "byteSize": 35313873 + }, + { + "role": "notice", + "path": "screen-text-1080p24-final/notices/screen-text-1080p24-final.txt", + "downloadName": "screen-text-1080p24-final--screen-text-1080p24-final.txt", + "sha256": "b8e04df7bb96a8f23cde80b4b52a551b05313242623924f5dafba930cc834092", + "byteSize": 1016 + }, + { + "role": "license", + "path": "screen-text-1080p24-final/notices/CC0-1.0.txt", + "downloadName": "screen-text-1080p24-final--CC0-1.0.txt", + "sha256": "a2010f343487d3f7618affe54f789f5487602331c0a8d03f49e9a7c547cf0499", + "byteSize": 7048 + }, + { + "role": "license", + "path": "screen-text-1080p24-final/notices/IBM-Plex-Mono-OFL.txt", + "downloadName": "screen-text-1080p24-final--IBM-Plex-Mono-OFL.txt", + "sha256": "7e6b2818edbd8f6a01ae80641cc8f16a51080d08fb4e532be3a0b6f74adb07da", + "byteSize": 4456 + } + ] + }, + "talking-head-1080p24-final": { + "fileName": "talking-head-1080p24-final.mkv", + "license": "CC-BY-3.0", + "assets": [ + { + "role": "clip", + "path": "talking-head-1080p24-final/talking-head-1080p24-final.mkv", + "downloadName": "talking-head-1080p24-final--talking-head-1080p24-final.mkv", + "sha256": "ac85d1350e5e668d5c0798b25fd7de16f8da05e831f396680a8f0f0ab05bcaf3", + "byteSize": 219125475 + }, + { + "role": "notice", + "path": "talking-head-1080p24-final/notices/talking-head-1080p24-final.txt", + "downloadName": "talking-head-1080p24-final--talking-head-1080p24-final.txt", + "sha256": "63fb7ce61b5ebcdc4aabd02c5075094ef2ebd9ecec6dff6d4a1d12987dc8c143", + "byteSize": 1430 + }, + { + "role": "license", + "path": "talking-head-1080p24-final/notices/CC-BY-3.0.txt", + "downloadName": "talking-head-1080p24-final--CC-BY-3.0.txt", + "sha256": "e6bc9e9c474700b708f568bac9e5a8a9bcb2b1dad53442f5ba449fcb848b8e76", + "byteSize": 19467 + } + ] + } + } +} diff --git a/client/suite.py b/client/suite.py index c4ddfa69..cf413c64 100644 --- a/client/suite.py +++ b/client/suite.py @@ -1,6 +1,8 @@ +import errno import hashlib import json import math +import re import os import shutil import queue @@ -18,7 +20,7 @@ import struct from dataclasses import dataclass from fractions import Fraction -from typing import Any, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple +from typing import Any, Callable, Dict, Iterable, List, Mapping, Optional, Sequence, Tuple from . import config @@ -30,6 +32,11 @@ SUITE_PACK_METADATA_RELATIVE_PATH = os.path.join("resources", "test_suite_v1", "suite-pack.json") SUITE_PACK_SCHEMA_VERSION = 1 DEFAULT_SUITE_PACK_FILE_NAME = f"{SUITE_VERSION}.tar.gz" +CLIP_DISTRIBUTION_RELATIVE_PATH = os.path.join("resources", "test_suite_v1", "clip-distribution.json") +CLIP_DISTRIBUTION_SCHEMA_VERSION = 1 +SUITE_CLIP_BASE_URL_ENV = "ENCODINGDB_SUITE_CLIP_BASE_URL" +SUITE_ALLOW_FULL_PACK_ENV = "ENCODINGDB_ALLOW_FULL_PACK" +SUITE_MIN_FREE_MB_ENV = "ENCODINGDB_SUITE_MIN_FREE_MB" REQUIRED_CONTENT_CLASSES: Tuple[str, ...] = ( "high-motion-sports", @@ -228,6 +235,210 @@ def load_suite_pack_metadata(path: Optional[str] = None) -> Dict[str, Any]: return payload +def get_clip_distribution_path() -> Optional[str]: + for manifest_path in _manifest_resource_candidates(): + candidate = os.path.join(os.path.dirname(manifest_path), os.path.basename(CLIP_DISTRIBUTION_RELATIVE_PATH)) + if candidate and os.path.exists(candidate): + return candidate + return None + + +def _clip_notice_asset_names(manifest: SuiteManifest, suite_root: str) -> Dict[str, List[str]]: + """Map each clip to its attribution notice plus the license texts it names. + + Bindings come from the frozen notice bodies themselves: every clip notice + declares "License: ;" and a font notice beside that clip's license is + included only when the clip notice mentions the font. + """ + notices_root = os.path.join(suite_root, "notices") + available = {name for name in os.listdir(notices_root) if name.endswith(".txt")} if os.path.isdir(notices_root) else set() + bindings: Dict[str, List[str]] = {} + for clip in manifest.clips: + own = f"{clip.clip_id}.txt" + if own not in available: + raise RuntimeError(f"missing attribution notice for {clip.clip_id}") + with open(os.path.join(notices_root, own), "r", encoding="utf-8") as handle: + text = handle.read() + names = [own] + license_match = re.search(r"License:\s*([^;\n]+);", text) + declared = license_match.group(1).strip() if license_match else "" + if declared: + license_file = f"{declared}.txt" + if license_file not in available: + raise RuntimeError(f"missing license text {license_file} for {clip.clip_id}") + names.append(license_file) + if re.search(r"plex|font", text, re.IGNORECASE): + for name in sorted(available): + if name in names or name == own: + continue + with open(os.path.join(notices_root, name), "r", encoding="utf-8") as other_handle: + other = other_handle.read(400) + if re.search(r"font software is licensed", other, re.IGNORECASE): + names.append(name) + bindings[clip.clip_id] = names + return bindings + + +def build_clip_distribution_metadata( + suite_root: str, + *, + manifest: Optional[SuiteManifest] = None, + base_url: str = "", + release_tag: Optional[str] = None, + published: bool = False, +) -> Dict[str, Any]: + """Derive the per-clip distribution manifest from frozen suite resources. + + Every clip/notice hash and size comes from manifest.json and notices/; the + canonical media bytes themselves are not required. ``path`` records the + logical source layout; ``downloadName`` is unique and flat for releases. + """ + suite_root_abs = os.path.abspath(suite_root) + manifest_value = manifest + if manifest_value is None: + with open(os.path.join(suite_root_abs, "manifest.json"), "r", encoding="utf-8") as handle: + manifest_value = manifest_from_payload(json.load(handle)) + manifest_path = os.path.join(suite_root_abs, "manifest.json") + clips: Dict[str, Any] = {} + for clip_id, names in _clip_notice_asset_names(manifest_value, suite_root_abs).items(): + clip = get_clip(manifest_value, clip_id) + assets: List[Dict[str, Any]] = [ + { + "role": "clip", + "path": f"{clip_id}/{clip.file_name}", + "downloadName": f"{clip_id}--{clip.file_name}", + "sha256": clip.sha256, + "byteSize": clip.byte_size, + } + ] + for name in names: + notice_path = os.path.join(suite_root_abs, "notices", name) + assets.append( + { + "role": "notice" if name == f"{clip_id}.txt" else "license", + "path": f"{clip_id}/notices/{name}", + "downloadName": f"{clip_id}--{name}", + "sha256": _sha256_of_file(notice_path), + "byteSize": os.path.getsize(notice_path), + } + ) + clips[clip_id] = { + "fileName": clip.file_name, + "license": str(clip.provenance.get("license") or ""), + "assets": assets, + } + return { + "schemaVersion": CLIP_DISTRIBUTION_SCHEMA_VERSION, + "suiteId": "encodingdb-test-suite", + "suiteVersion": manifest_value.suite_version, + "manifestVersion": manifest_value.manifest_version, + "source": "github-release" if published else "staged-unpublished", + "releaseTag": release_tag, + "distribution": { + "baseUrl": base_url, + "clipUrlOverrideEnv": SUITE_CLIP_BASE_URL_ENV, + "published": bool(published), + }, + "manifest": { + "sha256": _sha256_of_file(manifest_path), + "byteSize": os.path.getsize(manifest_path), + }, + "clips": clips, + } + + +def write_clip_distribution_metadata(suite_root: str, output_path: Optional[str] = None, **kwargs: Any) -> str: + payload = build_clip_distribution_metadata(suite_root, **kwargs) + destination = os.path.abspath(output_path or os.path.join(suite_root, "clip-distribution.json")) + os.makedirs(os.path.dirname(destination), exist_ok=True) + with open(destination, "w", encoding="utf-8") as handle: + json.dump(payload, handle, indent=2, sort_keys=False) + handle.write("\n") + return destination + + +def load_clip_distribution_metadata(path: Optional[str] = None, + manifest: Optional[SuiteManifest] = None) -> Optional[Dict[str, Any]]: + """Load the clip distribution manifest; None when the file is absent. + + Absence only disables the per-clip route (development fixtures). A present + file that contradicts the frozen suite identity fails closed — it would + otherwise redirect acquisition to unreviewed bytes. + """ + resolved = path if path is not None else get_clip_distribution_path() + if resolved is None: + return None + with open(resolved, "r", encoding="utf-8") as handle: + payload = json.load(handle) + if not isinstance(payload, dict): + raise RuntimeError("clip distribution metadata is invalid") + if int(payload.get("schemaVersion") or 0) != CLIP_DISTRIBUTION_SCHEMA_VERSION: + raise RuntimeError("clip distribution metadata has an unsupported schemaVersion") + suite_manifest = manifest if manifest is not None else load_default_suite_manifest() + if str(payload.get("suiteVersion") or "") != suite_manifest.suite_version: + raise RuntimeError("clip distribution metadata suite version mismatch") + manifest_path = os.path.join(os.path.dirname(resolved), "manifest.json") + record = dict(payload.get("manifest") or {}) + if int(record.get("byteSize") or 0) != os.path.getsize(manifest_path) \ + or str(record.get("sha256") or "").lower() != _sha256_of_file(manifest_path): + raise RuntimeError("clip distribution metadata manifest identity mismatch") + clips = dict(payload.get("clips") or {}) + manifest_ids = {clip.clip_id for clip in suite_manifest.clips} + if set(clips) != manifest_ids: + raise RuntimeError("clip distribution metadata clip inventory mismatch") + bindings = _clip_notice_asset_names(suite_manifest, os.path.dirname(resolved)) + for clip in suite_manifest.clips: + entry = dict(clips.get(clip.clip_id) or {}) + if str(entry.get("fileName") or "") != clip.file_name: + raise RuntimeError(f"clip distribution metadata file name mismatch for {clip.clip_id}") + assets = list(entry.get("assets") or []) + roles = [str(asset.get("role") or "") for asset in assets] + if roles.count("clip") != 1: + raise RuntimeError(f"clip distribution metadata needs exactly one clip asset for {clip.clip_id}") + expected_names = bindings[clip.clip_id] + actual_names = [ + str(asset.get("path") or "").rsplit("notices/", 1)[-1] + for asset in assets if str(asset.get("role") or "") in ("notice", "license") + ] + if sorted(actual_names) != sorted(expected_names): + raise RuntimeError(f"clip distribution metadata notice binding mismatch for {clip.clip_id}") + for asset in assets: + relative = str(asset.get("path") or "") + if not relative.startswith(f"{clip.clip_id}/") or ".." in relative.split("/"): + raise RuntimeError(f"clip distribution metadata has an unsafe asset path: {relative}") + expected_download_name = f"{clip.clip_id}--{relative.rsplit('/', 1)[-1]}" + if str(asset.get("downloadName") or "") != expected_download_name: + raise RuntimeError(f"clip distribution metadata download name mismatch: {relative}") + expected_sha = str(asset.get("sha256") or "").lower() + expected_bytes = int(asset.get("byteSize") or 0) + if str(asset.get("role") or "") == "clip": + if expected_bytes != clip.byte_size or expected_sha != clip.sha256.lower(): + raise RuntimeError(f"clip distribution metadata identity mismatch for {clip.clip_id}") + else: + notice_file = relative.rsplit("notices/", 1)[-1] + notice_path = os.path.join(os.path.dirname(resolved), "notices", notice_file) + if not os.path.isfile(notice_path) or os.path.getsize(notice_path) != expected_bytes \ + or _sha256_of_file(notice_path).lower() != expected_sha: + raise RuntimeError(f"clip distribution metadata notice mismatch: {relative}") + return payload + + +def clip_distribution_available(metadata: Optional[Mapping[str, Any]]) -> bool: + if not metadata: + return False + override = os.environ.get(SUITE_CLIP_BASE_URL_ENV, "").strip() + base = str(dict(metadata.get("distribution") or {}).get("baseUrl") or "").strip() + return bool(override or base) + + +def _clip_asset_url(metadata: Mapping[str, Any], asset_path: str) -> str: + override = os.environ.get(SUITE_CLIP_BASE_URL_ENV, "").strip() + base = override or str(dict(metadata.get("distribution") or {}).get("baseUrl") or "").strip() + if not base: + raise RuntimeError("clip distribution has no base URL") + return urljoin(base.rstrip("/") + "/", asset_path.lstrip("/")) + + def _suite_cache_root() -> str: custom = os.environ.get("ENCODINGDB_SUITE_CACHE_DIR", "").strip() if custom: @@ -1186,9 +1397,15 @@ def read_network(): # The reader closes it on return/timeout and cannot mutate retained data. -def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any]) -> None: - distribution = dict(metadata.get("distribution") or {}) - expected_size = int(distribution.get("byteSize") or 0) +def _download_verified_file(url: str, destination: str, *, expected_size: int, + verify: "Callable[[str], ClipVerificationResult]", + exceeds_message: str) -> None: + """Resumable, size-bounded, hash-verified download into `destination`. + + Shared by the frozen suite pack and per-clip assets: one owned reader + thread, `.part` resume via Range, hard stop at the declared size, and a + caller-supplied final verification before the atomic install. + """ temp_path = f"{destination}.part" os.makedirs(os.path.dirname(destination), exist_ok=True) resume_from = 0 @@ -1214,25 +1431,346 @@ def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any] os.remove(temp_path) continue completed = resume_from if mode == "ab" else 0 - with open(temp_path, mode) as handle: - for kind, chunk in events: - if kind != "chunk": - raise RuntimeError("Unexpected suite download event") - check_preparation_cancelled() - if completed + len(chunk) > expected_size: - raise RuntimeError("Suite download exceeds declared pack size") - handle.write(chunk) - completed += len(chunk) - preparation_progress("download", path=destination, completedBytes=completed, totalBytes=expected_size) + try: + with open(temp_path, mode) as handle: + for kind, chunk in events: + if kind != "chunk": + raise RuntimeError("Unexpected suite download event") + check_preparation_cancelled() + if completed + len(chunk) > expected_size: + raise RuntimeError(exceeds_message) + handle.write(chunk) + completed += len(chunk) + preparation_progress("download", path=destination, completedBytes=completed, totalBytes=expected_size) + except OSError as exc: + if getattr(exc, "errno", None) in (errno.ENOSPC, errno.EDQUOT): + raise RuntimeError( + f"the suite cache volume ran out of free space while downloading " + f"{destination}: {exc}; free space on that volume and start the run again" + ) from exc + if getattr(exc, "errno", None) in (errno.EACCES, errno.EPERM, errno.EROFS): + raise RuntimeError( + f"suite content could not be written to {destination}: the folder is " + f"write-protected or unwritable for this account ({exc})" + ) from exc + raise break finally: events.close() - result = _verify_suite_pack_file(temp_path, metadata) + result = verify(temp_path) if not result.ok: + # The bytes are untrusted; never keep a corrupt partial for resume. + try: + os.remove(temp_path) + except FileNotFoundError: + pass raise RuntimeError(result.message) os.replace(temp_path, destination) +def _suite_free_bytes(path: str) -> Optional[int]: + try: + probe_path = path + while not os.path.exists(probe_path): + parent = os.path.dirname(probe_path) + if parent == probe_path: + break + probe_path = parent + return int(shutil.disk_usage(probe_path).free) + except OSError: + return None + + +def _suite_cache_write_error(cache_root: str) -> Optional[str]: + """Return a clear cause when the cache root cannot accept new files.""" + try: + os.makedirs(cache_root, exist_ok=True) + probe_dir = os.path.join(cache_root, "canonical") + os.makedirs(probe_dir, exist_ok=True) + fd, probe_path = tempfile.mkstemp(prefix=".encodingdb-write-probe-", dir=probe_dir) + os.close(fd) + os.remove(probe_path) + except OSError as exc: + return f"the suite cache {cache_root} is not writable by this account ({exc})" + return None + + +def _check_acquisition_storage(cache_root: str, required_bytes: int) -> None: + """Gate a costly fetch on free space + writability before the first byte. + + Requires room for the transfer plus the ENCODINGDB_SUITE_MIN_FREE_MB floor + (default 64 MiB; 0 disables). A declared-but-unwritable cache is reported + as such instead of failing mid-download. + """ + floor_mb = max(0, int(os.environ.get(SUITE_MIN_FREE_MB_ENV, "64") or "0")) + required = int(required_bytes) + floor_mb * 1024 * 1024 + write_error = _suite_cache_write_error(cache_root) + if write_error is not None: + raise RuntimeError(write_error) + free = _suite_free_bytes(cache_root) + if free is not None and free < required: + raise RuntimeError( + f"acquiring suite content needs {required_bytes:,} bytes plus a " + f"{floor_mb * 1024 * 1024:,} byte free-space floor, but the volume holding " + f"{cache_root} has only {free:,} bytes free; free space or set " + f"{SUITE_MIN_FREE_MB_ENV} to 0 to override the floor" + ) + + +def _download_suite_pack(url: str, destination: str, metadata: Mapping[str, Any]) -> None: + distribution = dict(metadata.get("distribution") or {}) + expected_size = int(distribution.get("byteSize") or 0) + _download_verified_file( + url, + destination, + expected_size=expected_size, + verify=lambda path: _verify_suite_pack_file(path, metadata), + exceeds_message="Suite download exceeds declared pack size", + ) + + +def _full_pack_fallback_allowed(allow_full_pack: bool) -> bool: + if not allow_full_pack: + return False + setting = os.environ.get(SUITE_ALLOW_FULL_PACK_ENV, "1").strip().lower() + return setting not in ("0", "false", "no", "off") + + +def _disclose_large_download(clip: SuiteClip, pack_bytes: int, clip_error: Optional[str]) -> None: + """Make an unexpected full-pack transfer visible before it starts.""" + reason = f" ({clip_error})" if clip_error else " (no per-clip assets are published for this suite yet)" + message = ( + f"The small per-clip asset for {clip.clip_id} could not be acquired{reason}. " + f"Falling back to the full frozen suite pack: {pack_bytes:,} bytes " + f"(about {pack_bytes / (2 ** 30):.1f} GiB) will be transferred into the local suite cache. " + f"Set {SUITE_CLIP_BASE_URL_ENV} to a host with published per-clip assets to avoid this." + ) + preparation_progress("large-download", clipId=clip.clip_id, totalBytes=pack_bytes, message=message) + print(f"EncodingDB: {message}", file=sys.stderr) + + +def _clip_asset_target(cache_base: str, clip: SuiteClip, asset: Mapping[str, Any]) -> str: + if str(asset.get("role") or "") == "clip": + return clip_cache_path(clip, cache_base) + name = str(asset.get("path") or "").rsplit("notices/", 1)[-1] + return os.path.join(cache_base, "notices", name) + + +def _verified_asset_result(path: str, asset: Mapping[str, Any], clip: SuiteClip) -> ClipVerificationResult: + if str(asset.get("role") or "") == "clip": + return _verify_suite_clip_bytes(path, clip) + if not os.path.exists(path): + return ClipVerificationResult(False, f"{os.path.basename(path)} not found", {"path": path}) + actual_size = os.path.getsize(path) + if actual_size != int(asset.get("byteSize") or 0): + return ClipVerificationResult(False, f"{os.path.basename(path)} size mismatch", {"path": path}) + if _sha256_of_file(path).lower() != str(asset.get("sha256") or "").lower(): + return ClipVerificationResult(False, f"{os.path.basename(path)} checksum mismatch", {"path": path}) + return ClipVerificationResult(True, "ok", {"path": path}) + + +def _pending_clip_assets(clip: SuiteClip, cache_base: str, + assets: Sequence[Mapping[str, Any]]) -> Tuple[List[Tuple[Mapping[str, Any], str]], int]: + pending: List[Tuple[Mapping[str, Any], str]] = [] + cached_bytes = 0 + for asset in assets: + target = _clip_asset_target(cache_base, clip, asset) + if _verified_asset_result(target, asset, clip).ok: + cached_bytes += int(asset.get("byteSize") or 0) + else: + pending.append((asset, target)) + return pending, cached_bytes + + +def _materialize_clip_from_distribution(clip: SuiteClip, distribution: Mapping[str, Any], + cache_base: str) -> str: + """Fetch only this clip's canonical file plus its license notices. + + Each asset resumes via its own .part, stops at the declared size, and is + hash-verified against the frozen manifest before the atomic install; a + corrupt response never installs. + """ + entry = dict(distribution.get("clips") or {}).get(clip.clip_id) + if not entry: + raise RuntimeError(f"clip distribution metadata has no entry for {clip.clip_id}") + assets = list(dict(entry).get("assets") or []) + pending, _ = _pending_clip_assets(clip, cache_base, assets) + if not pending: + return clip_cache_path(clip, cache_base) + transfer_bytes = sum(int(asset.get("byteSize") or 0) for asset, _ in pending) + headroom_bytes = max(64 * 1024 * 1024, transfer_bytes // 20) + _check_acquisition_storage(cache_base, transfer_bytes + headroom_bytes) + for _, target in pending: + try: + os.makedirs(os.path.dirname(target), exist_ok=True) + except OSError as exc: + raise RuntimeError( + f"the suite cache folder {os.path.dirname(target)} is not writable by this account ({exc})" + ) from exc + label = f"Suite clip download for {clip.clip_id}" + for asset, target in pending: + url = _clip_asset_url(distribution, str(asset.get("downloadName") or "")) + _download_verified_file( + url, + target, + expected_size=int(asset.get("byteSize") or 0), + verify=lambda path, current=asset: _verified_asset_result(path, current, clip), + exceeds_message=f"{label} exceeds declared file size", + ) + result = _verify_suite_clip_bytes(clip_cache_path(clip, cache_base), clip) + if not result.ok: + raise RuntimeError(result.message) + return clip_cache_path(clip, cache_base) + + +def _packaged_canonical_path(clip: SuiteClip) -> Optional[str]: + for manifest_path in _manifest_resource_candidates(): + candidate = os.path.join(os.path.dirname(manifest_path), "canonical", clip.file_name) + if os.path.exists(candidate): + return candidate + return None + + +def _pack_state(pack_metadata: Mapping[str, Any], cache_base: str) -> Dict[str, bool]: + pack_cached = _verify_suite_pack_file(_cache_suite_pack_path(pack_metadata, cache_base), pack_metadata).ok + fingerprint = str(pack_metadata.get("suiteFingerprint") or "").strip() + # Without a fingerprint the extraction directory is unnameable; assume absent. + extracted = bool(fingerprint) and os.path.isdir( + os.path.join(_suite_pack_extract_root(pack_metadata, cache_base), "canonical")) + return {"packCached": pack_cached, "extracted": extracted} + + +def _plan_clip_acquisition(clip: SuiteClip, cache_base: str, *, clip_route_available: bool, + pack_metadata: Mapping[str, Any], pack_state: Mapping[str, bool], + allow_full_pack: bool) -> Dict[str, Any]: + """Decide one clip's acquisition without side effects beyond hash reads.""" + packaged = _packaged_canonical_path(clip) + if packaged is not None and _verify_suite_clip_bytes(packaged, clip).ok: + return {"source": "packaged", "transferBytes": 0, "cachedBytes": clip.byte_size, "peakBytes": 0, + "note": "verified canonical asset is bundled with the client"} + cached_path = clip_cache_path(clip, cache_base) + if _verify_suite_clip_bytes(cached_path, clip).ok: + return {"source": "cache", "transferBytes": 0, "cachedBytes": clip.byte_size, "peakBytes": 0, "note": "hash-verified cache hit"} + pack_bytes = int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0) + if clip_route_available: + distribution = load_clip_distribution_metadata(manifest=None) + entry = dict((distribution or {}).get("clips") or {}).get(clip.clip_id) or {} + assets = list(dict(entry).get("assets") or []) + pending, cached_bytes = _pending_clip_assets(clip, cache_base, assets) + transfer = sum(int(asset.get("byteSize") or 0) for asset, _ in pending) + headroom = max(64 * 1024 * 1024, transfer // 20) + return {"source": "clip", "transferBytes": transfer, "cachedBytes": cached_bytes, + "peakBytes": transfer + headroom, "note": "per-clip assets are published for this suite"} + if not _full_pack_fallback_allowed(allow_full_pack): + return {"source": "unavailable", "transferBytes": 0, "cachedBytes": 0, + "peakBytes": clip.byte_size, + "note": (f"clip is not cached, no per-clip assets are available, and the full pack " + f"({pack_bytes:,} bytes) is disabled via {SUITE_ALLOW_FULL_PACK_ENV}/allow_full_pack")} + transfer = 0 if pack_state.get("packCached") else pack_bytes + extract_bytes = 0 if pack_state.get("extracted") else pack_bytes + return {"source": "pack", "transferBytes": transfer, "cachedBytes": 0, + "peakBytes": transfer + extract_bytes + clip.byte_size, + "note": f"only the full frozen suite pack ({pack_bytes:,} bytes) is published for this suite"} + + +def acquisition_estimate(clip_ids: Optional[Sequence[str]] = None, *, + cache_root: Optional[str] = None, + allow_full_pack: bool = True, + manifest: Optional[SuiteManifest] = None) -> Dict[str, Any]: + """Truthful, side-effect-free transfer/peak-storage estimate for a fetch. + + Stable read-only API for UIs: reports exactly what `ensure_suite_clip`/ + `ensure_suite` would transfer per clip (cache hit, per-clip assets, or the + full pack) before any byte is downloaded. + """ + suite_manifest = manifest or load_default_suite_manifest() + resolved_cache_root = cache_root or _suite_cache_root() + target_ids = list(clip_ids) if clip_ids else [clip.clip_id for clip in suite_manifest.clips] + distribution = load_clip_distribution_metadata(manifest=suite_manifest) + route_available = clip_distribution_available(distribution) + pack_metadata = load_suite_pack_metadata() + pack_state = _pack_state(pack_metadata, resolved_cache_root) + plans: Dict[str, Any] = {} + transfer_total = 0 + cached_total = 0 + peak_total = 0 + warnings: List[str] = [] + worst = "cache" + pack_counted = False + severity = {"cache": 0, "packaged": 0, "pack": 1, "clip": 1, "unavailable": 2} + for clip_id in target_ids: + clip = get_clip(suite_manifest, clip_id) + plan = _plan_clip_acquisition(clip, resolved_cache_root, clip_route_available=route_available, + pack_metadata=pack_metadata, pack_state=pack_state, + allow_full_pack=allow_full_pack) + if plan["source"] == "pack": + if pack_counted: + plan["transferBytes"] = 0 + plan["peakBytes"] = clip.byte_size + plan["note"] = "shared full suite pack counted with the first missing clip" + pack_counted = True + plans[clip_id] = {"clipId": clip_id, "fileName": clip.file_name, **plan} + transfer_total += int(plan["transferBytes"]) + cached_total += int(plan["cachedBytes"]) + peak_total += int(plan["peakBytes"]) + if severity.get(plan["source"], 0) > severity.get(worst, 0): + worst = plan["source"] + if plan["source"] in ("unavailable", "pack"): + warnings.append(f"{clip_id}: {plan['note']}") + free = _suite_free_bytes(resolved_cache_root) + floor_mb = max(0, int(os.environ.get(SUITE_MIN_FREE_MB_ENV, "64") or "0")) + storage_ok = free is None or free >= peak_total + floor_mb * 1024 * 1024 + return { + "schemaVersion": 1, + "suiteVersion": suite_manifest.suite_version, + "cacheRoot": resolved_cache_root, + "strategy": worst, + "clipRouteAvailable": bool(route_available), + "allowFullPack": bool(_full_pack_fallback_allowed(allow_full_pack)), + "fullPackBytes": int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0), + "bytesToTransfer": transfer_total, + "bytesAlreadyVerified": cached_total, + "peakStorageBytes": peak_total, + "freeBytes": free, + "freeFloorBytes": floor_mb * 1024 * 1024, + "storageOk": bool(storage_ok), + "clips": plans, + "warnings": warnings, + } + + +def _acquire_missing_clip(clip: SuiteClip, cache_base: str, *, allow_full_pack: bool) -> str: + distribution = load_clip_distribution_metadata(manifest=None) + clip_error: Optional[str] = None + if clip_distribution_available(distribution): + try: + return _materialize_clip_from_distribution(clip, distribution, cache_base) + except RuntimeError as exc: + clip_error = str(exc) + preparation_progress("recovery", clipId=clip.clip_id, + message=f"per-clip acquisition failed ({clip_error}); considering the full pack") + pack_metadata = load_suite_pack_metadata() + pack_bytes = int(dict(pack_metadata.get("distribution") or {}).get("byteSize") or 0) + if not _full_pack_fallback_allowed(allow_full_pack): + detail = f" Per-clip download failed: {clip_error}." if clip_error else "" + raise RuntimeError( + f"suite clip {clip.clip_id} is not cached and the full frozen suite pack " + f"({pack_bytes:,} bytes) download is disabled. Set {SUITE_CLIP_BASE_URL_ENV} to a host with " + "the published per-clip assets, provide the pack via ENCODINGDB_SUITE_PACK_PATH, or allow " + f"{SUITE_ALLOW_FULL_PACK_ENV}." + detail + ) + state = _pack_state(pack_metadata, cache_base) + needed = clip.byte_size + if not state["packCached"]: + needed += pack_bytes + if not state["extracted"]: + needed += pack_bytes + _check_acquisition_storage(cache_base, needed) + if not state["packCached"] or not state["extracted"]: + _disclose_large_download(clip, pack_bytes, clip_error) + return _materialize_clip_from_suite_pack(clip, pack_metadata, cache_base) + + def _local_suite_pack_candidates(file_name: str) -> List[str]: candidates: List[str] = [] explicit = os.environ.get("ENCODINGDB_SUITE_PACK_PATH", "").strip() @@ -1412,6 +1950,7 @@ def ensure_suite_clip( *, cache_root: Optional[str] = None, regenerate_on_mismatch: bool = True, + allow_full_pack: bool = True, ) -> PreparedSuiteClip: wait_for_owned_acquisition() resolved_cache_root = cache_root or _suite_cache_root() @@ -1430,8 +1969,8 @@ def ensure_suite_clip( path = clip_cache_path(clip, resolved_cache_root) result = verify_suite_clip(path, clip) if not result.ok and regenerate_on_mismatch: - path = _materialize_clip_from_suite_pack(clip, load_suite_pack_metadata(), resolved_cache_root) - # That exact staged stream passed this clip's complete media contract. + path = _acquire_missing_clip(clip, resolved_cache_root, allow_full_pack=allow_full_pack) + # The acquired stream passed this clip's complete media contract. # Recheck the renamed bytes, including SHA, before reusing that result. result = _verify_suite_clip_bytes(path, clip) if not result.ok: @@ -1474,6 +2013,7 @@ def ensure_suite( *, clip_ids: Optional[Sequence[str]] = None, cache_root: Optional[str] = None, + allow_full_pack: bool = True, ) -> List[PreparedSuiteClip]: suite = manifest or load_default_suite_manifest() target_ids = set(clip_ids or [clip.clip_id for clip in suite.clips]) @@ -1481,7 +2021,7 @@ def ensure_suite( for index, clip in enumerate(suite.clips, 1): preparation_progress("clip", clipId=clip.clip_id, completedClips=index - 1, totalClips=len(suite.clips)) if clip.clip_id in target_ids: - prepared.append(ensure_suite_clip(clip, cache_root=cache_root)) + prepared.append(ensure_suite_clip(clip, cache_root=cache_root, allow_full_pack=allow_full_pack)) return prepared diff --git a/client/tests/test_suite_clip_acquisition.py b/client/tests/test_suite_clip_acquisition.py new file mode 100644 index 00000000..f15f033c --- /dev/null +++ b/client/tests/test_suite_clip_acquisition.py @@ -0,0 +1,397 @@ +"""PLA-546 C02: cold Small contributions fetch only their frozen quick clip. + +Synthetic media only (never production suite resources). A loopback HTTP server +stands in for the published per-clip host; request logs prove exactly which +assets each run transferred. +""" +import hashlib +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import json +import os +from pathlib import Path +import posixpath +import shutil +import stat +import threading +from unittest import mock + +import pytest +from client import campaign, suite +from scripts.prepare_client_suite_distribution import build_clip_bundle + +from test_suite_v1 import small_media_fixture + + +def add_notices(root: Path, manifest) -> None: + notices = root / "notices" + notices.mkdir(exist_ok=True) + for clip in manifest.clips: + license_id = str(clip.provenance.get("license") or "CC-BY-4.0") + (notices / f"{license_id}.txt").write_text(f"license text for {license_id}\n") + (notices / f"{clip.clip_id}.txt").write_text( + f"Attribution notice for {clip.clip_id}\nLicense: {license_id}; see {license_id}.txt\n" + ) + + +class ClipServer: + """Serves the staged per-clip bundle; supports Range and tampered assets.""" + + def __init__(self, root: Path, distribution, *, corrupt_paths=(), short_paths=()): + manifest = json.loads((root / "manifest.json").read_text()) + by_id = {clip["id"]: clip for clip in manifest["clips"]} + self.files = {} + for entry in distribution["clips"].values(): + for asset in entry["assets"]: + if str(asset["role"]) == "clip": + clip_id = str(asset["path"]).split("/")[0] + data = (root / "canonical" / by_id[clip_id]["fileName"]).read_bytes() + else: + data = (root / "notices" / posixpath.basename(str(asset["path"]))).read_bytes() + if str(asset["path"]) in corrupt_paths: + data = bytes([data[0] ^ 255]) + data[1:] + self.files["/" + str(asset["downloadName"])] = data + self.short_paths = set(short_paths) + self.requests = [] + self.range_headers = [] + server = self + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_GET(self): + server.requests.append(self.path) + data = server.files.get(self.path) + range_header = self.headers.get("Range") + if range_header: + server.range_headers.append((self.path, range_header)) + if data is None: + self.send_error(404) + return + if self.path in server.short_paths: + # A truncated complete response: declared short, body short. + truncated = data[: max(1, len(data) // 2)] + self.send_response(200) + self.send_header("Content-Length", str(len(truncated))) + self.end_headers() + self.wfile.write(truncated) + return + start = 0 + if range_header and range_header.startswith("bytes=") and range_header.endswith("-"): + start = int(range_header[6:-1]) + if start: + self.send_response(206) + self.send_header("Content-Range", f"bytes {start}-{len(data) - 1}/{len(data)}") + self.send_header("Content-Length", str(len(data) - start)) + self.end_headers() + self.wfile.write(data[start:]) + else: + self.send_response(200) + self.send_header("Content-Length", str(len(data))) + self.end_headers() + self.wfile.write(data) + + self.server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + self.serving = threading.Thread(target=self.server.serve_forever, daemon=True) + self.serving.start() + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self.server.server_port}/" + + def close(self): + self.server.shutdown() + self.server.server_close() + self.serving.join(timeout=2) + campaign.wait_for_owned_acquisition() + + +def staged_fixture(tmp_path, **server_kwargs): + """Staged suite tree: metadata + notices + served media, no local canonical bytes.""" + outer = small_media_fixture() + root, manifest = outer.__enter__() + server = None + try: + add_notices(root, manifest) + distribution_path = suite.write_clip_distribution_metadata(str(root)) + metadata = json.loads(Path(distribution_path).read_text()) + clip = manifest.clips[0] + clip_bytes = (root / "canonical" / clip.file_name).read_bytes() + server = ClipServer(root, metadata, **server_kwargs) + # A distributed client tree carries no canonical bytes; only the pack or + # the published per-clip host can supply the media. + shutil.rmtree(root / "canonical") + cache = str(tmp_path / "cache") + with mock.patch.dict(os.environ, {suite.SUITE_CLIP_BASE_URL_ENV: server.base_url}): + yield server, cache, clip, manifest, clip_bytes + finally: + if server is not None: + server.close() + outer.__exit__(None, None, None) + + +@pytest.fixture +def clip_env(tmp_path): + yield from staged_fixture(tmp_path) + + +def distribution_entry(clip_id): + payload = json.loads(Path(suite.get_clip_distribution_path()).read_text()) + return payload["clips"][clip_id] + + +def test_cold_quick_clip_downloads_only_selected_clip_assets(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + served_ids = {path.lstrip("/").split("--", 1)[0] for path in server.requests} + assert served_ids == {clip.clip_id} + assert not any("pack" in path or path.endswith(".tar.gz") for path in server.requests) + installed = Path(cache) / "canonical" / clip.file_name + assert installed.read_bytes() == clip_bytes + assert Path(prepared.path) == installed + assert prepared.input_hash == clip.sha256 + # License notices install with matching frozen hashes. + for asset in distribution_entry(clip.clip_id)["assets"]: + if str(asset["role"]) == "clip": + continue + installed_notice = Path(suite._clip_asset_target(cache, clip, asset)) + assert hashlib.sha256(installed_notice.read_bytes()).hexdigest() == asset["sha256"] + + +def test_repeat_uses_hash_verified_cache_without_requests(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + suite.ensure_suite_clip(clip, cache_root=cache) + before = len(server.requests) + suite.ensure_suite_clip(clip, cache_root=cache) + assert len(server.requests) == before + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert estimate["strategy"] == "cache" + assert estimate["bytesToTransfer"] == 0 + + +def test_estimate_is_truthful_before_any_download(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + entry = distribution_entry(clip.clip_id) + expected = sum(int(asset["byteSize"]) for asset in entry["assets"]) + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert server.requests == [] # the estimate itself transfers nothing + assert estimate["strategy"] == "clip" + assert estimate["bytesToTransfer"] == expected + assert estimate["peakStorageBytes"] >= expected + assert estimate["clipRouteAvailable"] is True + assert estimate["storageOk"] is True + suite.ensure_suite_clip(clip, cache_root=cache) + assert len(server.requests) == len(entry["assets"]) # one GET per declared asset + + +def test_full_suite_only_downloads_missing_clips(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + suite.ensure_suite_clip(clip, cache_root=cache) + server.requests.clear() + server.range_headers.clear() + prepared = suite.ensure_suite(manifest, cache_root=cache) + assert len(prepared) == len(manifest.clips) + served_ids = {path.lstrip("/").split("--", 1)[0] for path in server.requests} + assert clip.clip_id not in served_ids + assert served_ids == {other.clip_id for other in manifest.clips if other.clip_id != clip.clip_id} + + +def test_truncated_response_never_installs_and_resumes_with_range(clip_env, monkeypatch): + server, cache, clip, manifest, clip_bytes = clip_env + monkeypatch.setenv(suite.SUITE_ALLOW_FULL_PACK_ENV, "0") # surface the clip-route error + entry = distribution_entry(clip.clip_id) + clip_asset = next(asset for asset in entry["assets"] if asset["role"] == "clip") + server.short_paths.add("/" + clip_asset["downloadName"]) + with pytest.raises(RuntimeError, match="size mismatch"): + suite.ensure_suite_clip(clip, cache_root=cache) + target = Path(cache) / "canonical" / clip.file_name + part = Path(str(target) + ".part") + assert not target.exists() + assert not part.exists() # corrupt/incomplete bytes are never retained + # A retained partial from an interrupted run must resume via Range. + target.parent.mkdir(parents=True, exist_ok=True) + part.write_bytes(clip_bytes[: len(clip_bytes) // 2]) + server.short_paths.clear() + server.requests.clear() + server.range_headers.clear() + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + assert Path(prepared.path).read_bytes() == clip_bytes + resumed = [path for path, header in server.range_headers if header.startswith("bytes=")] + assert any(clip.file_name in path for path in resumed) + + +def test_corrupt_clip_response_never_installs(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + server = None + try: + add_notices(root, manifest) + distribution_path = suite.write_clip_distribution_metadata(str(root)) + metadata = json.loads(Path(distribution_path).read_text()) + clip = manifest.clips[0] + corrupt = [asset["path"] for asset in metadata["clips"][clip.clip_id]["assets"] if asset["role"] == "clip"] + server = ClipServer(root, metadata, corrupt_paths=corrupt) + shutil.rmtree(root / "canonical") + cache = str(tmp_path / "cache") + with mock.patch.dict(os.environ, {suite.SUITE_CLIP_BASE_URL_ENV: server.base_url, + suite.SUITE_ALLOW_FULL_PACK_ENV: "0"}): + with pytest.raises(RuntimeError, match="checksum mismatch"): + suite.ensure_suite_clip(clip, cache_root=cache) + target = Path(cache) / "canonical" / clip.file_name + assert not target.exists() + assert not Path(str(target) + ".part").exists() + finally: + if server is not None: + server.close() + outer.__exit__(None, None, None) + + +def test_full_pack_fallback_requires_explicit_size_disclosure(clip_env, capsys): + server, cache, clip, manifest, clip_bytes = clip_env + clip_error = "forced per-clip failure" + with mock.patch.dict(os.environ, {suite.SUITE_ALLOW_FULL_PACK_ENV: "0"}), \ + mock.patch.object(suite, "_materialize_clip_from_distribution", side_effect=RuntimeError(clip_error)): + with pytest.raises(RuntimeError, match="disabled") as blocked: + suite.ensure_suite_clip(clip, cache_root=cache) + assert suite.SUITE_ALLOW_FULL_PACK_ENV in str(blocked.value) + + pack_bytes = int(suite.load_suite_pack_metadata()["distribution"]["byteSize"]) + + def fake_pack(clip_arg, pack_metadata, cache_root=None): + target = Path(suite.clip_cache_path(clip_arg, cache_root)) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(clip_bytes) + return str(target) + + with mock.patch.object(suite, "_materialize_clip_from_distribution", side_effect=RuntimeError(clip_error)), \ + mock.patch.object(suite, "_materialize_clip_from_suite_pack", side_effect=fake_pack) as pack: + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + pack.assert_called_once() + assert Path(prepared.path).read_bytes() == clip_bytes + err = capsys.readouterr().err + assert f"{pack_bytes:,} bytes" in err # size disclosed before the pack transfer + assert clip_error in err + + +def test_full_pack_compatibility_is_preserved_by_default(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + clip = manifest.clips[0] + clip_bytes = (root / "canonical" / clip.file_name).read_bytes() + shutil.rmtree(root / "canonical") # distributed tree: no packaged media + cache = str(tmp_path / "cache") + + def fake_pack(clip_arg, pack_metadata, cache_root=None): + target = Path(suite.clip_cache_path(clip_arg, cache_root)) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(clip_bytes) + return str(target) + + with mock.patch.object(suite, "_materialize_clip_from_suite_pack", side_effect=fake_pack) as pack: + prepared = suite.ensure_suite_clip(clip, cache_root=cache) + pack.assert_called_once() # no clip-distribution.json -> pack route unchanged + assert Path(prepared.path).read_bytes() == clip_bytes + finally: + outer.__exit__(None, None, None) + + +def test_protected_cache_reports_clear_error_without_partial_install(clip_env): + server, cache, clip, manifest, clip_bytes = clip_env + if os.name == "nt": + pytest.skip("POSIX permission semantics") + locked = Path(cache) / "canonical" + locked.mkdir(parents=True) + locked.chmod(stat.S_IRUSR | stat.S_IXUSR) + try: + with pytest.raises(RuntimeError, match="not writable|write-protected"): + suite.ensure_suite_clip(clip, cache_root=cache) + finally: + locked.chmod(stat.S_IRWXU) + assert not list(locked.glob("*.part")) + assert not list(locked.glob(clip.file_name)) + assert server.requests == [] # blocked before any transfer + + +def test_disk_exhaustion_gate_stops_before_any_transfer(clip_env, monkeypatch): + server, cache, clip, manifest, clip_bytes = clip_env + monkeypatch.setenv(suite.SUITE_MIN_FREE_MB_ENV, "4194304") # 4 TiB floor + with pytest.raises(RuntimeError, match="bytes free"): + suite.ensure_suite_clip(clip, cache_root=cache) + assert server.requests == [] + estimate = suite.acquisition_estimate(clip_ids=[clip.clip_id], cache_root=cache, manifest=manifest) + assert estimate["storageOk"] is False + + +def test_tampered_clip_distribution_fails_closed(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + add_notices(root, manifest) + path = suite.write_clip_distribution_metadata(str(root)) + payload = json.loads(Path(path).read_text()) + clip = manifest.clips[0] + payload["clips"][clip.clip_id]["assets"][0]["sha256"] = "0" * 64 + Path(path).write_text(json.dumps(payload)) + with pytest.raises(RuntimeError, match="identity mismatch"): + suite.load_clip_distribution_metadata() + finally: + outer.__exit__(None, None, None) + + +def test_frozen_clip_distribution_matches_repo_suite_locks(): + metadata = suite.load_clip_distribution_metadata() + assert metadata is not None + assert metadata["source"] == "staged-unpublished" + assert metadata["distribution"]["published"] is False + assert metadata["distribution"]["baseUrl"] == "" + frozen = suite.load_default_suite_manifest() + lock = json.loads(Path(suite.get_suite_lock_path()).read_text()) + # The lock binds the manifest's canonical JSON, not its file bytes. + suite_root = Path(suite.get_suite_lock_path()).parent + manifest_payload = json.loads((suite_root / "manifest.json").read_text()) + assert suite._sha256_text(suite._canonical_json(manifest_payload)) == lock["manifestSha256"] + lock_clips = {entry["id"]: entry for entry in lock["clips"]} + for clip in frozen.clips: + entry = metadata["clips"][clip.clip_id] + clip_asset = next(asset for asset in entry["assets"] if asset["role"] == "clip") + assert clip_asset["sha256"] == clip.sha256 == lock_clips[clip.clip_id]["sha256"] + assert clip_asset["byteSize"] == clip.byte_size == lock_clips[clip.clip_id]["byteSize"] + assert clip_asset["downloadName"] == f"{clip.clip_id}--{clip.file_name}" + roles = [asset["role"] for asset in entry["assets"]] + assert roles.count("clip") == 1 + assert "notice" in roles and "license" in roles + + +def test_full_pack_estimate_counts_one_shared_download(tmp_path): + manifest = suite.load_default_suite_manifest() + clips = manifest.clips[:2] + with mock.patch.object(suite, "_packaged_canonical_path", return_value=None), \ + mock.patch.object(suite, "_pack_state", return_value={"packCached": False, "extracted": False}): + estimate = suite.acquisition_estimate( + [clip.clip_id for clip in clips], cache_root=str(tmp_path), manifest=manifest, + ) + pack_bytes = estimate["fullPackBytes"] + assert estimate["strategy"] == "pack" + assert estimate["bytesToTransfer"] == pack_bytes + assert estimate["peakStorageBytes"] == 2 * pack_bytes + sum(clip.byte_size for clip in clips) + assert estimate["clips"][clips[0].clip_id]["transferBytes"] == pack_bytes + assert estimate["clips"][clips[1].clip_id]["transferBytes"] == 0 + + +def test_release_clip_bundle_uses_flat_unique_asset_names(tmp_path): + outer = small_media_fixture() + root, manifest = outer.__enter__() + try: + add_notices(root, manifest) + bundle = tmp_path / "bundle" + with mock.patch.object(suite, "load_default_suite_manifest", return_value=manifest): + build_clip_bundle(source_suite_dir=root, bundle_out=bundle) + metadata = json.loads((bundle / "clip-distribution.json").read_text()) + names = [asset["downloadName"] for entry in metadata["clips"].values() + for asset in entry["assets"]] + assert len(names) == len(set(names)) + assert {path.name for path in bundle.iterdir()} == set(names) | {"clip-distribution.json"} + assert all(path.is_file() for path in bundle.iterdir()) + finally: + outer.__exit__(None, None, None) diff --git a/scripts/prepare_client_suite_distribution.py b/scripts/prepare_client_suite_distribution.py index 5ed60e19..ec3d743b 100644 --- a/scripts/prepare_client_suite_distribution.py +++ b/scripts/prepare_client_suite_distribution.py @@ -1,5 +1,7 @@ #!/usr/bin/env python3 import argparse +import hashlib +import json import os import shutil import sys @@ -42,17 +44,60 @@ def prepare_distribution( shutil.rmtree(staged_resource_dir, ignore_errors=True) staged_resource_dir.mkdir(parents=True, exist_ok=True) - for name in ("manifest.json", "finalization-status.json", "suite-pack.json", "suite-lock.json"): + for name in ("manifest.json", "finalization-status.json", "suite-pack.json", "suite-lock.json", "clip-distribution.json"): copy_if_exists(source_suite_dir / name, staged_resource_dir / name) if (source_suite_dir / "notices").exists(): shutil.copytree(source_suite_dir / "notices", staged_resource_dir / "notices") + +def build_clip_bundle(*, source_suite_dir: Path, bundle_out: Path, base_url: str = "", + release_tag: str | None = None, published: bool = False) -> None: + """Stage the separately addressable per-clip bundle (no upload performed). + + The bundle root contains flat, unique ``downloadName`` files suitable for + GitHub release assets, plus clip-distribution.json. Logical per-clip paths + remain in the metadata. Bytes come from the frozen canonical tree. + """ + source_suite_dir = source_suite_dir.resolve() + bundle_out = bundle_out.resolve() + manifest = suite.load_default_suite_manifest() + metadata = suite.build_clip_distribution_metadata( + str(source_suite_dir), manifest=manifest, base_url=base_url, + release_tag=release_tag, published=published, + ) + shutil.rmtree(bundle_out, ignore_errors=True) + bundle_out.mkdir(parents=True, exist_ok=True) + for clip_id, entry in metadata["clips"].items(): + for asset in entry["assets"]: + relative = str(asset["path"]) + if str(asset["role"]) == "clip": + source = source_suite_dir / "canonical" / entry["fileName"] + else: + source = source_suite_dir / "notices" / relative.rsplit("notices/", 1)[-1] + if not source.is_file(): + raise RuntimeError(f"canonical asset missing for staged bundle: {source}") + destination = bundle_out / str(asset["downloadName"]) + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, destination) + digest = hashlib.sha256(destination.read_bytes()).hexdigest() + if digest != asset["sha256"] or destination.stat().st_size != asset["byteSize"]: + raise RuntimeError(f"staged bundle asset does not match frozen identity: {relative}") + with open(bundle_out / "clip-distribution.json", "w", encoding="utf-8") as handle: + json.dump(metadata, handle, indent=2, sort_keys=False) + handle.write("\n") + + def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description="Build the external EncodingDB suite pack and stage client suite resources without embedded canonical media.") parser.add_argument("--source-suite-dir", default=str(ROOT_DIR / "client" / "resources" / "test_suite_v1")) parser.add_argument("--staged-resource-dir", required=True) parser.add_argument("--pack-out", required=True) + parser.add_argument("--clip-bundle-out", default=None, + help="Also stage the per-clip publishable bundle in this directory") + parser.add_argument("--clip-base-url", default="", + help="Public base URL for the per-clip bundle (empty until published)") + parser.add_argument("--clip-release-tag", default=None) return parser.parse_args() @@ -63,6 +108,15 @@ def main() -> int: staged_resource_dir=Path(args.staged_resource_dir), pack_out=Path(args.pack_out), ) + if args.clip_bundle_out: + build_clip_bundle( + source_suite_dir=Path(args.source_suite_dir), + bundle_out=Path(args.clip_bundle_out), + base_url=args.clip_base_url, + release_tag=args.clip_release_tag, + published=bool(args.clip_base_url), + ) + print(f"staged per-clip bundle (not uploaded): {os.path.abspath(args.clip_bundle_out)}") print(f"staged suite resources: {os.path.abspath(args.staged_resource_dir)}") print(f"suite pack: {os.path.abspath(args.pack_out)}") return 0 From 8557fe713425c54e3d26aa11a4b4fc09d69983cf Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 01:24:31 -0700 Subject: [PATCH 2/9] Make upload cancellation responsive while retaining retry identity Bound create, authorization and upload phases, and let Stop return promptly while the durable spool keeps the same local hash for ambiguous outcomes. Public errors carry safe status and Retry-After information without tokens. Constraint: Requests transport and protocol 7.1 remain in place Rejected: Global socket monkey-patching | process-wide side effects Confidence: medium Scope-risk: moderate Directive: A cancelled daemon socket may linger until its phase timeout; do not claim strict no-orphan acceptance Tested: 22 independent transport-focused tests; 11 artifact cancellation cases Not-tested: Physical Windows Stop/Close during a stalled upload --- client/artifacts.py | 261 ++++++++++--- client/network.py | 247 +++++++++++-- client/tests/test_artifact_cancellation.py | 407 +++++++++++++++++++++ 3 files changed, 842 insertions(+), 73 deletions(-) create mode 100644 client/tests/test_artifact_cancellation.py diff --git a/client/artifacts.py b/client/artifacts.py index 90b9bf43..b30e5197 100644 --- a/client/artifacts.py +++ b/client/artifacts.py @@ -1,9 +1,26 @@ import hashlib import json import os -from typing import Any, Dict, Optional - -from .network import SubmitError, _load_requests, retry_after_seconds +import time +from typing import Any, Callable, Dict, Optional + +from .network import (SubmissionCancelled, SubmitError, _event_cancelled, _load_requests, + _response_error_text, _run_cancellable, retry_after_seconds) + +# C12: bounded, cooperatively cancellable artifact transport. Each phase runs +# its blocking HTTP call in a daemon worker so a cancel_event (threading.Event +# or duck-typed `is_set()`) is observed within ~50 ms even while the call is +# socket-blocked; the worker itself carries a socket inactivity timeout at +# most the phase budget, so a stalled peer cannot pin a worker beyond it. +# Worst-case no-cancel chain: create 30 s + auth 30 s + upload 120 s = 180 s, +# vs the old 60+60+300 s with no cancellation input at all. +CREATE_TIMEOUT_SECONDS = 30.0 +AUTH_TIMEOUT_SECONDS = 30.0 +UPLOAD_TIMEOUT_SECONDS = 120.0 + +# Progress callback signature: (phase, sent_bytes, total_bytes). GUI-safe: +# ints only, no server data. +ProgressCallback = Callable[[str, int, int], None] AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND = "authoritative-artifact-run-v1" AUTHORITATIVE_ANALYZER_VERSION = "authoritative-analysis/v2" @@ -180,17 +197,122 @@ def build_artifact_submission_payload( return submission +class _PhaseBudget: + """Monotonic wall-clock budget for one transport phase (C12).""" + + def __init__(self, seconds: float) -> None: + self.seconds = max(0.05, float(seconds)) + self.deadline = time.monotonic() + self.seconds + + def remaining(self) -> float: + return self.deadline - time.monotonic() + + def timeout(self, phase: str) -> float: + """Socket inactivity timeout clamped to the remaining budget. + + A stalled transaction therefore self-limits to at most the phase + budget even if no cancel ever arrives.""" + remaining = self.remaining() + if remaining <= 0: + raise SubmitError(f"{phase} exceeded {self.seconds:g}s phase budget", retryable=True) + return max(0.1, min(self.seconds, remaining)) + + +def _phase_guard(budget: _PhaseBudget, cancel_event: Optional[Any], phase: str) -> None: + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + if budget.remaining() <= 0: + raise SubmitError(f"{phase} exceeded {budget.seconds:g}s phase budget", retryable=True) + + +def _safe_request_error(exc: Exception) -> SubmitError: + """Sanitized retryable wrapper: exception type only, never str(exc) (may + embed URLs, upload tokens or bodies).""" + return SubmitError(f"artifact transport failed: {type(exc).__name__}", retryable=True) + + +class _UploadBody: + """Sized streaming PUT body with cancel/deadline checks and progress (C12). + + Exposes `__len__` so requests keeps the original Content-Length framing + (no chunked-encoding wire change) while urllib3 still sends it chunk by + chunk; each chunk pull re-checks cancel and the phase budget, so a GUI + Stop ends the upload at the next chunk boundary (or, if the server stops + reading and the socket send blocks, within the armed send timeout — + bounded by the upload phase budget).""" + + def __init__(self, path: str, *, phase: str, budget: _PhaseBudget, + cancel_event: Optional[Any], progress: Optional[ProgressCallback]) -> None: + self._path = path + self._phase = phase + self._budget = budget + self._cancel_event = cancel_event + self._progress = progress + self._total = os.path.getsize(path) + + def __len__(self) -> int: + return self._total + + def __iter__(self): + sent = 0 + chunk_size = 262144 + with open(self._path, "rb") as handle: + while True: + _phase_guard(self._budget, self._cancel_event, self._phase) + try: + chunk = handle.read(chunk_size) + except Exception as exc: + raise _safe_request_error(exc) from exc + if not chunk: + break + sent += len(chunk) + if self._progress is not None: + try: + self._progress(self._phase, sent, self._total) + except Exception: + pass + yield chunk + + +def _read_json_bounded(response: Any, *, phase: str, budget: _PhaseBudget, + cancel_event: Optional[Any]) -> Any: + """Parse a JSON response body with cancel checks and a size cap.""" + text = _response_error_text(_load_requests(), response, cancel_event, budget.deadline, + max_bytes=1 << 22) + try: + return json.loads(text) + except Exception: + return None + + def submit_artifact_submission( base_url: str, submission: Dict[str, Any], *, retries: int = 3, + cancel_event: Optional[Any] = None, + progress: Optional[ProgressCallback] = None, + create_seconds: float = CREATE_TIMEOUT_SECONDS, + auth_seconds: float = AUTH_TIMEOUT_SECONDS, + upload_seconds: float = UPLOAD_TIMEOUT_SECONDS, ) -> Dict[str, Any]: + """Create run → authorize → upload, each phase bounded and cancellable. + + `cancel_event` is a threading.Event (or duck-typed `is_set()`); each + blocking phase runs in a daemon worker so cancel is observed within + ~50 ms even mid-socket-wait, and immediately between phases/chunks. The + worker's own socket inactivity timeout never exceeds its phase budget, so + nothing outlives create 30 s + auth 30 s + upload 120 s even with no + cancel. On cancel/timeout the error is retryable, so the durable spool + keeps the entry and its localHash; an accepted-but-lost response replays + idempotently against the same run. + Progress: `progress(phase, sent_bytes, total_bytes)` during upload only. + """ artifact_path = str(submission.get("artifactPath") or "").strip() if not artifact_path: raise SubmitError("artifact submission missing artifactPath", retryable=False) if not os.path.exists(artifact_path): - raise SubmitError(f"artifact missing at {artifact_path}", retryable=False) + raise SubmitError("artifact missing on disk", retryable=False) run_create = submission.get("runCreate") if not isinstance(run_create, dict): @@ -204,15 +326,24 @@ def submit_artifact_submission( # The durable spool owns retry scheduling, deadlines and backoff. One network # transaction per entry avoids hot retries across multiple clients. for attempt in range(1, 2): + if _event_cancelled(cancel_event): + raise SubmissionCancelled("run create") + create_budget = _PhaseBudget(create_seconds) try: - create_response = requests.post( - create_url, - json=run_create, - timeout=60, - allow_redirects=False, - ) + create_response = _run_cancellable( + lambda: requests.post( + create_url, + json=run_create, + timeout=create_budget.timeout("run create"), + allow_redirects=False, + stream=True, + ), + phase="run create", cancel_event=cancel_event, + deadline=create_budget.deadline, bound_seconds=create_budget.seconds) + except SubmissionCancelled: + raise except Exception as exc: - last_error = SubmitError(str(exc), retryable=True) + last_error = _safe_request_error(exc) continue if create_response.status_code in (429,) or create_response.status_code >= 500: @@ -220,7 +351,8 @@ def submit_artifact_submission( f"run create failed ({create_response.status_code})", retryable=True, status_code=create_response.status_code, - body=create_response.text or "", + body=_response_error_text(requests, create_response, cancel_event, + None), retry_after=retry_after_seconds(create_response.headers), ) continue @@ -229,44 +361,60 @@ def submit_artifact_submission( f"run create rejected ({create_response.status_code})", retryable=False, status_code=create_response.status_code, - body=create_response.text or "", + body=_response_error_text(requests, create_response, cancel_event, + None), retry_after=retry_after_seconds(create_response.headers), ) - try: - create_json = create_response.json() - except Exception as exc: - last_error = SubmitError(f"run create response invalid JSON: {exc}", retryable=True) + create_json = _read_json_bounded(create_response, phase="run create", + budget=create_budget, cancel_event=cancel_event) + if not isinstance(create_json, dict): + last_error = SubmitError("run create response invalid JSON", retryable=True) continue - benchmark_run = create_json.get("benchmarkRun") if isinstance(create_json, dict) else None + benchmark_run = create_json.get("benchmarkRun") run_id = str((benchmark_run or {}).get("id") or "").strip() if not run_id: last_error = SubmitError("run create response missing benchmarkRun.id", retryable=True) continue - artifact = create_json.get("artifact") if isinstance(create_json, dict) else None + artifact = create_json.get("artifact") artifact_state = str((artifact or {}).get("storageState") or "").strip().upper() - analyses = create_json.get("analyses") if isinstance(create_json, dict) else None + analyses = create_json.get("analyses") if artifact_state in ("RETAINED", "VERIFIED") and isinstance(analyses, list) and analyses: return create_json - auth_response = requests.post( - f"{base_url.rstrip('/')}/v7/benchmark-runs/{run_id}/artifacts/ENCODED/upload-authorizations", - json={ - "sha256": run_create["artifact"]["sha256"], - "byteSize": run_create["artifact"]["byteSize"], - "contentType": auth_content_type, - }, - timeout=60, - allow_redirects=False, - ) + if _event_cancelled(cancel_event): + raise SubmissionCancelled("upload authorization") + auth_budget = _PhaseBudget(auth_seconds) + try: + auth_response = _run_cancellable( + lambda: requests.post( + f"{base_url.rstrip('/')}/v7/benchmark-runs/{run_id}/artifacts/ENCODED/upload-authorizations", + json={ + "sha256": run_create["artifact"]["sha256"], + "byteSize": run_create["artifact"]["byteSize"], + "contentType": auth_content_type, + }, + timeout=auth_budget.timeout("upload authorization"), + allow_redirects=False, + stream=True, + ), + phase="upload authorization", cancel_event=cancel_event, + deadline=auth_budget.deadline, bound_seconds=auth_budget.seconds) + except SubmissionCancelled: + raise + except Exception as exc: + last_error = _safe_request_error(exc) + continue + if auth_response.status_code in (429,) or auth_response.status_code >= 500: last_error = SubmitError( f"upload authorization failed ({auth_response.status_code})", retryable=True, status_code=auth_response.status_code, - body=auth_response.text or "", + body=_response_error_text(requests, auth_response, cancel_event, + None), retry_after=retry_after_seconds(auth_response.headers), ) continue @@ -275,10 +423,12 @@ def submit_artifact_submission( f"upload authorization rejected ({auth_response.status_code})", retryable=False, status_code=auth_response.status_code, - body=auth_response.text or "", + body=_response_error_text(requests, auth_response, cancel_event, + None), retry_after=retry_after_seconds(auth_response.headers), ) - auth_json = auth_response.json() + auth_json = _read_json_bounded(auth_response, phase="upload authorization", + budget=auth_budget, cancel_event=cancel_event) if not isinstance(auth_json, dict): last_error = SubmitError("upload authorization response invalid JSON", retryable=True) continue @@ -290,17 +440,34 @@ def submit_artifact_submission( last_error = SubmitError("upload authorization missing token", retryable=True) continue + if _event_cancelled(cancel_event): + raise SubmissionCancelled("artifact upload") + upload_budget = _PhaseBudget(upload_seconds) try: - with open(artifact_path, "rb") as handle: - upload_response = requests.put( + upload_response = _run_cancellable( + lambda: requests.put( f"{base_url.rstrip('/')}/v7/artifact-uploads/{token}", - data=handle, + data=_UploadBody(artifact_path, phase="artifact upload", + budget=upload_budget, cancel_event=cancel_event, + progress=progress), headers={"Content-Type": auth_content_type}, - timeout=300, + timeout=upload_budget.timeout("artifact upload"), allow_redirects=False, - ) + stream=True, + ), + phase="artifact upload", cancel_event=cancel_event, + deadline=upload_budget.deadline, bound_seconds=upload_budget.seconds) + except (SubmissionCancelled, SubmitError): + raise except Exception as exc: - last_error = SubmitError(str(exc), retryable=True) + # A cancel raised inside the streaming body can surface wrapped by + # the HTTP stack; unwrap so the SubmissionCancelled contract holds. + cause = exc.__cause__ or exc.__context__ + while cause is not None: + if isinstance(cause, SubmissionCancelled): + raise cause from exc + cause = getattr(cause, "__cause__", None) or getattr(cause, "__context__", None) + last_error = _safe_request_error(exc) continue if upload_response.status_code in (429,) or upload_response.status_code >= 500: @@ -308,7 +475,8 @@ def submit_artifact_submission( f"artifact upload failed ({upload_response.status_code})", retryable=True, status_code=upload_response.status_code, - body=upload_response.text or "", + body=_response_error_text(requests, upload_response, cancel_event, + None), retry_after=retry_after_seconds(upload_response.headers), ) continue @@ -317,13 +485,16 @@ def submit_artifact_submission( f"artifact upload rejected ({upload_response.status_code})", retryable=False, status_code=upload_response.status_code, - body=upload_response.text or "", + body=_response_error_text(requests, upload_response, cancel_event, + None), retry_after=retry_after_seconds(upload_response.headers), ) - try: - return upload_response.json() - except Exception as exc: - last_error = SubmitError(f"artifact upload response invalid JSON: {exc}", retryable=True) + upload_json = _read_json_bounded(upload_response, phase="artifact upload", + budget=upload_budget, cancel_event=cancel_event) + if not isinstance(upload_json, dict): + last_error = SubmitError("artifact upload response invalid JSON", retryable=True) + continue + return upload_json if last_error is not None: raise last_error diff --git a/client/network.py b/client/network.py index 32819082..c00fcde0 100644 --- a/client/network.py +++ b/client/network.py @@ -2,13 +2,17 @@ import json import re import sys +import threading import time import warnings -from typing import Optional, Dict, Any, List +from typing import Any, Callable, Dict, List, Optional from urllib.parse import urljoin from . import config +# C12: default wall-clock bound on one submit() transaction chain (token fetches, +# POSTs, redirects and waits). Callers may pass a shorter/longer value. +SUBMIT_TRANSACTION_SECONDS = 90.0 def _load_requests(): with warnings.catch_warnings(): @@ -21,6 +25,16 @@ def _load_requests(): class SubmitError(RuntimeError): + """Transport failure with structured fields (status_code, retry_after). + + C04/C12: public text is bounded and redacted — never raw server bodies, + tokens or URLs. Raw response bodies stay private in `_server_body` for + diagnostics only; they must not reach exception text or GUI events. + """ + + _MAX_MESSAGE_CHARS = 300 + _MAX_BODY_CHARS = 4096 + def __init__( self, message: str, @@ -30,15 +44,112 @@ def __init__( body: str = "", retry_after: float = 0.0, ) -> None: - super().__init__(message) + super().__init__(str(message or "")[: self._MAX_MESSAGE_CHARS]) self.retryable = retryable self.status_code = status_code - self.body = body + self._server_body = str(body or "")[: self._MAX_BODY_CHARS] self.retry_after = retry_after -def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: - """Fetch and solve a one-time token for exactly one POST attempt.""" +class SubmissionCancelled(SubmitError): + """Cooperative cancellation (C12). Retryable by contract: the durable spool + keeps the entry and its localHash identity so replay is idempotent.""" + + def __init__(self, phase: str) -> None: + super().__init__(f"submission cancelled during {phase}", retryable=True) + self.phase = phase + + +def _event_cancelled(cancel_event: Optional[Any]) -> bool: + if cancel_event is None: + return False + try: + return bool(cancel_event.is_set()) + except Exception: + return False + + +def _run_cancellable(func: Callable[[], Any], *, phase: str, + cancel_event: Optional[Any], deadline: Optional[float], + bound_seconds: float, + poll_seconds: float = 0.05) -> Any: + """Run a blocking HTTP call in a daemon worker so cooperative cancellation + is observed within ~poll_seconds even while the call is socket-blocked. + + The worker always carries a socket inactivity timeout at most the phase + budget, so an abandoned worker cannot outlive that budget against a + stalled peer — no orphan workers, no monkey-patching. If cancel wins the + race the connection may have been fully sent (ambiguous outcome); the + durable spool keeps the entry and replays it idempotently.""" + done = threading.Event() + outcome: List[Any] = [] + + def _worker() -> None: + try: + outcome.append(("ok", func())) + except BaseException as exc: # noqa: BLE001 — re-raised in caller + outcome.append(("err", exc)) + finally: + done.set() + + threading.Thread(target=_worker, name=f"encodingdb-{phase}", daemon=True).start() + while not done.wait(poll_seconds): + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + remaining = _remaining_seconds(deadline) + if remaining is not None and remaining <= 0: + raise SubmitError(f"{phase} exceeded {bound_seconds:g}s wall-clock bound", + retryable=True) + kind, value = outcome[0] + if kind == "err": + raise value + return value + + +def _remaining_seconds(deadline: Optional[float]) -> Optional[float]: + if deadline is None: + return None + return deadline - time.monotonic() + + +def _check_transaction(cancel_event: Optional[Any], deadline: Optional[float], + phase: str, bound: float) -> None: + """Cooperative cancel + wall-clock bound check between blocking steps.""" + if _event_cancelled(cancel_event): + raise SubmissionCancelled(phase) + remaining = _remaining_seconds(deadline) + if remaining is not None and remaining <= 0: + raise SubmitError(f"{phase} exceeded {bound:g}s wall-clock bound", retryable=True) + + +def _bounded_wait(seconds: float, cancel_event: Optional[Any], deadline: Optional[float], + phase: str, bound: float) -> None: + """Backoff sleep that wakes on cancellation and respects the deadline.""" + remaining = _remaining_seconds(deadline) + if remaining is not None: + seconds = min(seconds, max(0.0, remaining)) + wait = getattr(cancel_event, "wait", None) + if callable(wait): + try: + if wait(seconds): + raise SubmissionCancelled(phase) + except SubmissionCancelled: + raise + except Exception: + if seconds > 0: + time.sleep(seconds) + elif seconds > 0: + time.sleep(seconds) + _check_transaction(cancel_event, deadline, phase, bound) + + +def _get_submit_token_headers(requests: Any, base_url: str, *, + cancel_event: Optional[Any] = None, + deadline: Optional[float] = None) -> Dict[str, str]: + """Fetch and solve a one-time token for exactly one POST attempt. + + C12: cancellation- and deadline-aware — each endpoint GET and the PoW loop + check cancel/deadline so a Stop does not wait out the full 30 s solve.""" headers: Dict[str, str] = {} try: base = base_url.rstrip('/') @@ -49,11 +160,16 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: ] token_resp = None for endpoint in endpoints: + _check_transaction(cancel_event, deadline, "token fetch", SUBMIT_TRANSACTION_SECONDS) try: - response = requests.get(endpoint, timeout=10, verify=config.REQUESTS_VERIFY) + remaining = _remaining_seconds(deadline) + timeout = 10 if remaining is None else max(0.1, min(10.0, remaining)) + response = requests.get(endpoint, timeout=timeout, verify=config.REQUESTS_VERIFY) if response.status_code == 200: token_resp = response break + except SubmissionCancelled: + raise except Exception: continue if token_resp is None: @@ -81,6 +197,14 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: print(f" Solving Proof-of-Work (difficulty={difficulty})...", end='', flush=True) while nonce < max_iters: if nonce % 10000 == 0: + if _event_cancelled(cancel_event): + print(" cancelled") + raise SubmissionCancelled("proof-of-work") + remaining_pow = _remaining_seconds(deadline) + if remaining_pow is not None and remaining_pow <= 0: + print(" deadline reached") + raise SubmitError("proof-of-work exceeded transaction bound", + retryable=True) elapsed_pow = time.time() - pow_start if elapsed_pow > pow_timeout: print(f" timeout after {elapsed_pow:.1f}s") @@ -95,18 +219,66 @@ def _get_submit_token_headers(requests: Any, base_url: str) -> Dict[str, str]: nonce += 1 print(f" exhausted {max_iters} iterations without solution") return {} + except SubmissionCancelled: + raise + except SubmitError: + raise except Exception as exc: try: - print(f"token fetch error: {exc}", file=sys.stderr) + print(f"token fetch error: {type(exc).__name__}", file=sys.stderr) except Exception: pass return {} -def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: int = 3, backoff_seconds: float = 1.0, use_token: Optional[bool] = None) -> None: +def _response_error_text(requests: Any, response: Any, cancel_event: Optional[Any], + deadline: Optional[float], max_bytes: int = 65536) -> str: + """Read an error body for the private field, cancellable and size-capped.""" + chunks: List[bytes] = [] + total = 0 + try: + for chunk in response.iter_content(chunk_size=8192): + _check_transaction(cancel_event, deadline, "response read", SUBMIT_TRANSACTION_SECONDS) + if not chunk: + continue + take = max(0, max_bytes - total) + if take: + chunks.append(bytes(chunk[:take])) + total += len(chunk) + except SubmissionCancelled: + raise + except Exception: + pass + text = b"".join(chunks).decode("utf-8", "replace") + if total > max_bytes: + text += f"...[truncated {total - max_bytes} bytes]" + return text + + +def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: int = 3, + backoff_seconds: float = 1.0, use_token: Optional[bool] = None, + cancel_event: Optional[Any] = None, + transaction_seconds: float = SUBMIT_TRANSACTION_SECONDS) -> None: + """POST a payload with bounded, cancellable retries (C12). + + `cancel_event` (threading.Event-like) and `transaction_seconds` cap the + whole attempt chain on a monotonic wall clock: every step between blocking + calls checks both, and each socket step gets at most the remaining budget + as its inactivity timeout. Cancellation raises SubmissionCancelled + (retryable) so the durable spool keeps identity.""" requests = _load_requests() url = f"{base_url.rstrip('/')}/submit" payload_to_send: Dict[str, Any] = dict(payload) + deadline = time.monotonic() + max(1.0, float(transaction_seconds)) + + def step_timeout() -> float: + remaining = _remaining_seconds(deadline) + if remaining is None: + return 30 + if remaining <= 0: + raise SubmitError(f"submit exceeded {transaction_seconds:g}s wall-clock bound", + retryable=True) + return max(0.1, min(30.0, remaining)) base_headers: Dict[str, str] = {"Content-Type": "application/json"} if use_token is None: @@ -117,10 +289,12 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i attempt = 1 last_hmac_timestamp = 0 while attempt <= retries: + _check_transaction(cancel_event, deadline, "submit", transaction_seconds) body = json.dumps(payload_to_send, separators=(",", ":")) headers = dict(base_headers) if use_token: - headers.update(_get_submit_token_headers(requests, base_url)) + headers.update(_get_submit_token_headers(requests, base_url, + cancel_event=cancel_event, deadline=deadline)) if secret: import hmac ts = max(int(time.time()), last_hmac_timestamp + 1) @@ -129,14 +303,18 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i headers["x-signature"] = sig headers["x-timestamp"] = str(ts) try: - r = requests.post(url, data=body, timeout=30, headers=headers, verify=config.REQUESTS_VERIFY, allow_redirects=False) + r = _run_cancellable( + lambda: requests.post(url, data=body, timeout=step_timeout(), headers=headers, verify=config.REQUESTS_VERIFY, allow_redirects=False, stream=True), + phase="submit", cancel_event=cancel_event, deadline=deadline, + bound_seconds=transaction_seconds) if 300 <= r.status_code < 400: loc = r.headers.get('Location') or r.headers.get('location') if loc: redirect_url = urljoin(url, loc) redirect_headers = dict(base_headers) if use_token: - redirect_headers.update(_get_submit_token_headers(requests, base_url)) + redirect_headers.update(_get_submit_token_headers(requests, base_url, + cancel_event=cancel_event, deadline=deadline)) if secret: import hmac redirect_ts = max(int(time.time()), last_hmac_timestamp + 1) @@ -144,7 +322,11 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i redirect_sig = hmac.new(secret.encode("utf-8"), f"{redirect_ts}.".encode("utf-8") + body.encode("utf-8"), hashlib.sha256).hexdigest() redirect_headers["x-signature"] = redirect_sig redirect_headers["x-timestamp"] = str(redirect_ts) - r = requests.post(redirect_url, data=body, timeout=30, headers=redirect_headers, verify=config.REQUESTS_VERIFY, allow_redirects=False) + _check_transaction(cancel_event, deadline, "submit redirect", transaction_seconds) + r = _run_cancellable( + lambda: requests.post(redirect_url, data=body, timeout=step_timeout(), headers=redirect_headers, verify=config.REQUESTS_VERIFY, allow_redirects=False, stream=True), + phase="submit redirect", cancel_event=cancel_event, deadline=deadline, + bound_seconds=transaction_seconds) if r.status_code == 429: try: ra = r.headers.get('Retry-After') @@ -156,18 +338,20 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i f"submit rate limited ({r.status_code})", retryable=True, status_code=r.status_code, - body=(r.text or ""), + body=_response_error_text(requests, r, cancel_event, deadline), + retry_after=delay, # durable spool keeps the server's own wait (C08) ) - time.sleep(max(0.5, delay)) + _bounded_wait(max(0.5, delay), cancel_event, deadline, "submit retry wait", transaction_seconds) attempt += 1 continue if r.status_code >= 500: + error_body = _response_error_text(requests, r, cancel_event, deadline) if attempt >= retries: raise SubmitError( f"server_error {r.status_code}", retryable=True, - status_code=r.status_code, - body=(r.text or ""), + body=error_body, + retry_after=retry_after_seconds(r.headers), ) raise RuntimeError(f"server_error {r.status_code}") if r.status_code >= 400: @@ -175,10 +359,12 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i f"submit rejected ({r.status_code})", retryable=False, status_code=r.status_code, - body=(r.text or ""), + body=_response_error_text(requests, r, cancel_event, deadline), ) r.raise_for_status() return + except SubmissionCancelled: + raise except Exception as e: retryable = True status_code: Optional[int] = None @@ -186,32 +372,37 @@ def submit(base_url: str, payload: Dict[str, Any], api_key: str = "", retries: i if isinstance(e, SubmitError): retryable = e.retryable status_code = e.status_code - body = e.body + body = e._server_body if attempt == retries: + # C04: never echo server bodies; status/content-type/length only. + status_for_print = status_code + detail = "" try: _req = _load_requests() if isinstance(e, _req.HTTPError) and getattr(e, 'response', None) is not None: resp = e.response - try: - err_text = resp.text - except Exception: - err_text = "" - sent_token = 'x-ingest-token' in headers - sent_nonce = 'x-ingest-nonce' in headers - print(f"submit error body ({resp.status_code}): {err_text}\n(sent_token={sent_token}, sent_nonce={sent_nonce})", file=sys.stderr) + status_for_print = resp.status_code + detail = f" content_type={resp.headers.get('content-type', '')} bytes={len(resp.content or b'')}" except Exception: pass + sent_token = 'x-ingest-token' in headers + sent_nonce = 'x-ingest-nonce' in headers + print( + f"submit failed (status={status_for_print}{detail}; sent_token={sent_token}, sent_nonce={sent_nonce})", + file=sys.stderr, + ) if isinstance(e, SubmitError): raise + # Never put exception str() (may embed URLs/tokens) straight into + # the message: type + safe shape only. raise SubmitError( - str(e), + f"submit failed: {type(e).__name__}", retryable=True, status_code=status_code, - body=body, ) from e if isinstance(e, SubmitError) and not retryable: raise - time.sleep(backoff_seconds * attempt) + _bounded_wait(backoff_seconds * attempt, cancel_event, deadline, "submit retry wait", transaction_seconds) attempt += 1 diff --git a/client/tests/test_artifact_cancellation.py b/client/tests/test_artifact_cancellation.py new file mode 100644 index 00000000..9d8a91d8 --- /dev/null +++ b/client/tests/test_artifact_cancellation.py @@ -0,0 +1,407 @@ +"""C12 revision: phase-aware cooperative cancellation and bounded transport. + +Covers: stalled create/auth/PUT, cancel during each phase, lost response after +server acceptance, idempotent replay with the same identity, redacted public +errors (no raw bodies/tokens), and GUI-safe progress reporting. +""" +import hashlib +import json +import socket +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import ClassVar, List, Optional + +import pytest + +from client.artifacts import SubmitError, SubmissionCancelled, build_payload_hash, submit_artifact_submission +from client.network import submit as legacy_submit + + +def _artifact_bytes() -> bytes: + return b"C" * (1 << 20) # 1 MiB + + +def _submission(tmp_path, run_create=None): + path = tmp_path / "artifact.mp4" + path.write_bytes(_artifact_bytes()) + create = run_create if run_create is not None else { + "campaignId": "campaign-c12", + "repetitionGroupId": "campaign-c12:recipe-1", + "repetitionIndex": 1, + "artifact": {"role": "ENCODED", "sha256": hashlib.sha256(_artifact_bytes()).hexdigest(), + "byteSize": len(_artifact_bytes()), "mediaContainer": "mp4"}, + } + create = dict(create) + create.setdefault("payloadHash", build_payload_hash(create)) + return {"submissionKind": "authoritative-artifact-run-v1", + "artifactPath": str(path), "contentType": "video/mp4", "runCreate": create} + + +class _Server(ThreadingHTTPServer): + daemon_threads = True + allow_reuse_address = True + + +class _StallHandler(BaseHTTPRequestHandler): + """Stalls each phase on a class-level control event.""" + phase: ClassVar[str] = "create" # where to stall: create|auth|put + release: ClassVar[Optional[threading.Event]] = None + first_body_chunk: ClassVar[Optional[threading.Event]] = None + seen: ClassVar[List[str]] = [] + + def _read_body(self) -> bytes: + n = int(self.headers.get("Content-Length", "0")) + return self.rfile.read(n) if n else b"" + + def _stall(self, phase: str) -> None: + if type(self).phase == phase and type(self).release is not None: + type(self).release.wait(10) + + def do_POST(self) -> None: + body = self._read_body() + type(self).seen.append(self.path) + if self.path == "/v7/benchmark-runs": + self._stall("create") + payload = json.loads(body or b"{}") + self.send_response(201) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({ + "benchmarkRun": {"id": "run-c12"}, + "artifact": {"storageState": "PENDING"}, + "analyses": [], + }).encode()) + return + if self.path.endswith("/upload-authorizations"): + self._stall("auth") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": True, "token": "tok-c12"}).encode()) + return + self.send_response(404) + self.end_headers() + + def do_PUT(self) -> None: + # Read the body slowly: one chunk, then wait, so the client generator + # keeps pulling chunks and can observe cancellation mid-upload. + n = int(self.headers.get("Content-Length", "0")) + remaining = n + first = True + while remaining > 0: + chunk = self.rfile.read(min(65536, remaining)) + if not chunk: + break + remaining -= len(chunk) + if first: + first = False + if type(self).first_body_chunk is not None: + type(self).first_body_chunk.set() + self._stall("put") + if remaining > 0: + time.sleep(0.05) # throttle so upload stays in-flight + type(self).seen.append("PUT") + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"benchmarkRun": {"id": "run-c12"}}).encode()) + + def log_message(self, format: str, *args) -> None: # noqa: A003 + return + + +def _start(handler): + server = _Server(("127.0.0.1", 0), handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + return server, f"http://127.0.0.1:{server.server_port}" + + +@pytest.fixture(autouse=True) +def _reset_handler_state(): + _StallHandler.phase = "create" + _StallHandler.release = None + _StallHandler.first_body_chunk = None + _StallHandler.seen = [] + yield + + +def test_cancel_before_call_touches_no_network(tmp_path): + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + cancel.set() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + assert caught.value.retryable + assert _StallHandler.seen == [] + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_stalled_create_is_bounded(tmp_path): + _StallHandler.phase = "create" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + threading.Timer(0.3, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + elapsed = time.monotonic() - started + assert caught.value.phase == "run create" + assert caught.value.retryable + # Bounded shutdown: cancel observed far under the old 300 s PUT window. + assert elapsed < 3.0 + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_stalled_create_hits_phase_budget_without_cancel(tmp_path): + _StallHandler.phase = "create" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), create_seconds=0.5) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 3.0 # hard bound, never the 60 s legacy timeout + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_stalled_auth(tmp_path): + _StallHandler.phase = "auth" + _StallHandler.release = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + threading.Timer(0.3, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), cancel_event=cancel) + assert caught.value.phase == "upload authorization" + assert time.monotonic() - started < 3.0 + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_cancel_during_upload_and_progress_reporting(tmp_path): + _StallHandler.phase = "put" + _StallHandler.release = threading.Event() + _StallHandler.first_body_chunk = threading.Event() + server, base_url = _start(_StallHandler) + try: + cancel = threading.Event() + progress: List[tuple] = [] + + def on_progress(phase, sent, total): + progress.append((phase, sent, total)) + if sent >= 262144: # second chunk already accepted + cancel.set() + + started = time.monotonic() + with pytest.raises(SubmissionCancelled) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), + cancel_event=cancel, progress=on_progress, + upload_seconds=5.0) + elapsed = time.monotonic() - started + assert caught.value.phase == "artifact upload" + assert caught.value.retryable + # The caller returned while the worker was still socket-blocked: the + # transaction was genuinely in-flight (server receives bytes after the + # cancellation returns), and shutdown was fast. + assert _StallHandler.first_body_chunk.wait(5) + assert elapsed < 3.0 + # GUI-safe progress: upload phase, monotonic sent, correct total. + assert progress and all(p[0] == "artifact upload" for p in progress) + assert [p[1] for p in progress] == sorted(p[1] for p in progress) + assert progress[-1][2] == len(_artifact_bytes()) + _StallHandler.release.set() + finally: + server.shutdown() + server.server_close() + + +def test_upload_stall_is_hard_bounded_below_legacy_300s(tmp_path): + """No cancel at all: the upload phase budget alone terminates the stall.""" + class _NeverResponds(_StallHandler): + def do_PUT(self): + n = int(self.headers.get("Content-Length", "0")) + self.rfile.read(min(1024, n)) # read a little, then hang + time.sleep(10) + + server, base_url = _start(_NeverResponds) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path), upload_seconds=1.0) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 8.0 # bounded ~1 s budget, not 300 s + finally: + server.shutdown() + server.server_close() + + +def test_lost_response_after_acceptance_replays_same_identity(tmp_path): + """Server accepts the PUT, then the response is lost. Retry with the same + submission must reuse the existing run and skip re-upload — never create a + second run, never discard the artifact.""" + class _IdempotentFlow(_StallHandler): + runs: ClassVar[dict] = {} + uploaded: ClassVar[bool] = False + lost_once: ClassVar[bool] = False + + def do_POST(self): + body = self._read_body() + if self.path == "/v7/benchmark-runs": + payload = json.loads(body or b"{}") + key = str(payload.get("payloadHash")) + run = _IdempotentFlow.runs.get(key) + if run is None: + run = {"id": f"run-{len(_IdempotentFlow.runs) + 1}"} + _IdempotentFlow.runs[key] = run + state = "RETAINED" if _IdempotentFlow.uploaded else "PENDING" + analyses = [{"id": "analysis-1", "status": "COMPLETE"}] if _IdempotentFlow.uploaded else [] + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({ + "benchmarkRun": {"id": run["id"]}, + "artifact": {"storageState": state}, + "analyses": analyses, + }).encode()) + return + if self.path.endswith("/upload-authorizations"): + if _IdempotentFlow.uploaded: + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": False}).encode()) + return + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(json.dumps({"uploadRequired": True, "token": "t1"}).encode()) + return + self.send_response(404) + self.end_headers() + + def do_PUT(self): + n = int(self.headers.get("Content-Length", "0")) + self.rfile.read(n) + _IdempotentFlow.uploaded = True + if _IdempotentFlow.lost_once: + _IdempotentFlow.lost_once = False + # Accept bytes, then drop the connection without a response. + self.close_connection = True + try: + self.connection.shutdown(socket.SHUT_RDWR) + self.connection.close() + except OSError: + pass + return + super().do_PUT() + + server, base_url = _start(_IdempotentFlow) + try: + submission = _submission(tmp_path) + _IdempotentFlow.lost_once = True + with pytest.raises(SubmitError) as lost: + submit_artifact_submission(base_url, submission) + assert lost.value.retryable # ambiguous: spool retains the entry + + # Replay after restart with the SAME submission identity. + result = submit_artifact_submission(base_url, submission) + assert result["benchmarkRun"]["id"] == "run-1" + assert len(_IdempotentFlow.runs) == 1 # no duplicate run created + finally: + server.shutdown() + server.server_close() + + +def test_public_errors_never_leak_server_bodies_or_tokens(tmp_path): + class _LeakyRejection(_StallHandler): + def do_POST(self): + self._read_body() + self.send_response(400) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(b'{"error":"bad","token":"raw-secret-token","body":"raw-secret-body"}') + + server, base_url = _start(_LeakyRejection) + try: + with pytest.raises(SubmitError) as caught: + submit_artifact_submission(base_url, _submission(tmp_path)) + exc = caught.value + assert exc.retryable is False + assert exc.status_code == 400 + text = str(exc) + assert "raw-secret-token" not in text + assert "raw-secret-body" not in text + assert "raw-secret-token" not in repr(exc) + # Raw body is retained privately for diagnostics only. + assert "raw-secret-body" in exc._server_body + finally: + server.shutdown() + server.server_close() + + +def test_submit_error_message_is_bounded(): + exc = SubmitError("x" * 5000, retryable=True) + assert len(str(exc)) <= SubmitError._MAX_MESSAGE_CHARS + + +def test_legacy_submit_cancel_during_retry_wait(): + """network.submit honours cancel_event while waiting out a 429 backoff.""" + class _RateLimited(_StallHandler): + def do_POST(self): + self._read_body() + self.send_response(429) + self.send_header("Retry-After", "30") + self.end_headers() + + server, base_url = _start(_RateLimited) + try: + cancel = threading.Event() + threading.Timer(0.2, cancel.set).start() + started = time.monotonic() + with pytest.raises(SubmissionCancelled): + legacy_submit(base_url, {"cpuModel": "T"}, retries=3, + backoff_seconds=1, use_token=False, cancel_event=cancel) + assert time.monotonic() - started < 3.0 + finally: + server.shutdown() + server.server_close() + + +def test_legacy_submit_transaction_bound_stops_stalled_attempt(): + """No cancel: the wall-clock transaction bound ends a stalled POST chain.""" + class _Silent(_StallHandler): + def do_POST(self): + time.sleep(10) + + server, base_url = _start(_Silent) + try: + started = time.monotonic() + with pytest.raises(SubmitError) as caught: + legacy_submit(base_url, {"cpuModel": "T"}, retries=3, backoff_seconds=0.1, + use_token=False, transaction_seconds=1.0) + elapsed = time.monotonic() - started + assert caught.value.retryable + assert elapsed < 8.0 # far under 30 s × retries legacy behavior + finally: + server.shutdown() + server.server_close() From f06648e97b6e0ebb2da54b890f42ea0881862137 Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 01:25:13 -0700 Subject: [PATCH 3/9] Let contributors recover saved measurements without repeating encodes Durable journals can rebuild complete unsubmitted groups, while unfinished groups remain unfinished. A host phase lock excludes measurement and upload across queue directories. The terminal menu and Windows GUI expose saved work, safe publication, due retries and truthful pending outcomes. Constraint: Preserve frozen protocol 7.1 recipes, attempts and receipt IDs Constraint: Explicit local-only choice and publication consent remain authoritative Confidence: medium Scope-risk: broad Directive: Legacy journals without clientVersion cannot prove upgrade-stable reconstruction; keep that limitation visible Directive: A storage-limited Publish pass may return deferred with unadmitted envelopes; do not label it published Tested: 547 full client tests with frozen canonical media; 145 focused recovery/GUI/transport checks; 43 publication/durability checks after final truthfulness fix Not-tested: Native packaged Windows/macOS/Linux G01 failure-path acceptance or multi-user host locking --- client/main.py | 818 ++++++++++++++++++++++- client/spool.py | 565 +++++++++++++++- client/tests/test_campaign_durability.py | 131 ++++ client/tests/test_measurement_budget.py | 84 +++ client/tests/test_preparation.py | 2 +- client/tests/test_publication_storage.py | 18 + client/tests/test_spool.py | 233 ++++++- client/tests/test_windows_gui.py | 303 ++++++++- client/windows_gui.py | 553 ++++++++++++--- 9 files changed, 2534 insertions(+), 173 deletions(-) diff --git a/client/main.py b/client/main.py index 3d7bd1bc..5a81e219 100644 --- a/client/main.py +++ b/client/main.py @@ -1,6 +1,7 @@ import argparse from functools import wraps import dataclasses +import errno import json import math import os @@ -74,17 +75,26 @@ EncodeOutcome, EncodeTiming, EnvironmentSnapshot, + EnvironmentThresholds, ProtocolConfig, RecipeSpec, StructuralExpectation, + StructuralTolerance, execute_protocol_campaign, campaign_result_from_records, generate_campaign_id, ) from .spool import ( + campaign_queue_summary, + collector_publication_scope, cleanup_spool, count_pending_entries, + drain_committed_receipts, + host_phase_busy, + host_phase_hold, inspect_spool, + publication_lock_busy, + queue_recovery_summary, replay_spool, spool_payload, SpoolCapacityError, @@ -1232,6 +1242,593 @@ def _retire_uploaded_artifact(journal_root, record: Any, artifact_sha256: str, b pass +def _safe_failure_info(exc: BaseException) -> Dict[str, Any]: + """Structured safe failure fields; raw server bodies and secrets never cross.""" + from .network import SubmitError + info: Dict[str, Any] = {"category": "unexpected", "retryable": True} + if isinstance(exc, SubmitError): + status = exc.status_code or 0 + info["category"] = ("rate_limited" if status == 429 + else "server_error" if status >= 500 + else "rejected" if not exc.retryable else "network") + info["retryable"] = bool(exc.retryable) + if status: + info["statusCode"] = status + elif isinstance(exc, SpoolCapacityError): + info["category"] = "publication_deferred" + elif isinstance(exc, OSError) and getattr(exc, "errno", None) in (errno.ENOSPC, errno.EDQUOT): + info["category"] = "storage" + elif isinstance(exc, (ConnectionError, TimeoutError, OSError)): + info["category"] = "network" + info["reason"] = (str(exc) or exc.__class__.__name__)[:200] + return info + + +def _submit_failure_fields(status: str, message: str) -> Dict[str, Any]: + """Safe cause/category/recovery fields for submit_result events (C04/C05). + + Distinct categories for the outcomes an operator acts on differently: + 429 rate limiting, 5xx upstream failure, permanent protocol rejection, + expired retries, missing bytes, corrupt queue files. Parses the status + code from the transport's own bounded messages (`network`/`artifacts` + embed `(NNN)`); raw server bodies never cross the event boundary.""" + text = (message or "").strip() + match = re.search(r"\((\d{3})\)", text) or re.match(r"(?:server_error|submit failed) (\d{3})", text) + code = int(match.group(1)) if match else 0 + lowered = text.lower() + if code == 429 or "rate limited" in lowered: + return {"errorCategory": "rate_limited", + "safeReason": "Server rate limited the upload (429)" if code else text[:200] or "server rate limited the upload", + "recoveryAction": "No action needed: honoring Retry-After, the durable queue retries when due"} + if 500 <= code <= 599 or lowered.startswith("server_error"): + return {"errorCategory": "server_error", + "safeReason": f"Transient upstream failure ({code})" if code else text[:200] or "transient upstream failure", + "recoveryAction": "No action needed: the payload is retained with the server's Retry-After and retried idempotently"} + if code and 400 <= code < 500 and code != 429: + return {"errorCategory": "protocol_rejected", + "safeReason": f"Server rejected the submission ({code})", + "recoveryAction": "Terminal: verify client/suite versions before republishing; the server will reject identical evidence again"} + if text == "retry_deadline_expired": + return {"errorCategory": "expired", + "safeReason": "Retry deadline expired before the server accepted the upload", + "recoveryAction": "Dead-lettered; resume or publish the campaign later if the evidence still matters"} + if text in ("missing_spooled_artifact", "missing_artifact") or "missing" in lowered: + return {"errorCategory": "unavailable_source", + "safeReason": text[:200] or "Queued artifact bytes are no longer on disk", + "recoveryAction": "Resume the campaign to re-encode only this attempt; accepted groups are unaffected"} + if text.startswith("corrupt"): + return {"errorCategory": "corrupt_queue", + "safeReason": "Queue file could not be parsed and was moved to dead-letter", + "recoveryAction": "Run --queue-cleanup; the campaign journal remains intact for resume"} + if "rejected" in lowered: + return {"errorCategory": "protocol_rejected", + "safeReason": text[:200] or "server rejected the submission", + "recoveryAction": "Terminal: verify client/suite versions; resume re-encodes only if valid evidence is required"} + if status == "queued": + return {"errorCategory": "transient", + "safeReason": text[:200] or "upload deferred; scheduled for retry", + "recoveryAction": "No action needed: the durable queue retries when due"} + return {"errorCategory": "publication_failed", + "safeReason": text[:200] or "upload attempt failed", + "recoveryAction": "Retained for idempotent retry; use --upload-only or Publish saved when due"} + + +def _reconstruct_saved_submissions( + *, + queue_dir: str, + campaign_id: str, + max_storage_mb: int, + cancel_event: Optional[Any] = None, +) -> Dict[str, Any]: + """Materialize submission envelopes for COMPLETE groups lacking them (C09). + + A controlled stop or storage failure can end a campaign AFTER a group's + measured artifacts are durable in the journal but BEFORE the immutable + `submission-*.json` envelopes were written. Publishing must then REBUILD the + exact envelope from retained attempt records - never encode, never download + a source, never add missing repetitions, never invent an accepted receipt. + The journal reopen hash-verifies every retained member, so the rebuilt + payload is byte-faithful to what the live path would have written; frozen + manifest IDs (physicalSourceId) are preserved over current-machine values. + Groups still unfinished (checkpoint could extend them) or individually + invalid records stay exactly as unfinished as they are now. Returns counters + with `failure` set only when publishable evidence cannot be honestly + rebuilt - never a silent drop.""" + outcome = {"reconstructed": 0, "skippedExisting": 0, "skippedAccepted": 0, + "skippedIncomplete": 0, "skippedInvalid": 0, + "cancelled": False, "failure": None} + root = journal_path(queue_dir, campaign_id) + manifest_path = root / "manifest.json" + if not manifest_path.is_file(): + return outcome # legacy journal without plan identity: nothing to reconstruct + try: + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + if not isinstance(manifest, dict): + raise ValueError("journal manifest is not an object") + if str(manifest.get("protocolVersion") or "") != config.BENCHMARK_PROTOCOL_VERSION: + raise ValueError("saved protocol identity differs from this client; use the original compatible client") + saved_client_version = str(manifest.get("clientVersion") or "") + if saved_client_version and saved_client_version != CLIENT_VERSION: + raise ValueError("saved client identity differs from this client; use the original client version") + from .identity import runtime_identity + saved_runtime = manifest.get("runtime") + if saved_runtime is not None and saved_runtime != runtime_identity(): + raise ValueError("saved runtime identity differs from this client; use the original runtime") + pc = dict(manifest.get("protocolConfig") or {}) + protocol_config = ProtocolConfig( + version=str(pc.get("version") or manifest.get("protocolVersion") or ""), + warmup_runs=int(pc.get("warmup_runs", 1)), + minimum_measured_runs=int(pc.get("minimum_measured_runs", 2)), + stability_threshold_ratio=float(pc.get("stability_threshold_ratio", 0.03)), + max_adaptive_repeats=int(pc.get("max_adaptive_repeats", 2)), + environment=EnvironmentThresholds(**dict(pc.get("environment") or {})), + structural_tolerance=StructuralTolerance(**dict(pc.get("structural_tolerance") or {}))) + hardware = HardwareInfo(**{ + key: value for key, value in dict(manifest.get("hardware") or {}).items() + if key in HardwareInfo.__dataclass_fields__}) + # Reopen validates every retained measured member against its recorded + # SHA-256 and refuses bytes that vanished without an accepted receipt. + journal = CampaignJournal(queue_dir, campaign_id, manifest, int(max_storage_mb)) + except (ValueError, TypeError, OSError) as exc: + outcome["failure"] = f"journal evidence cannot be reopened: {exc}"[:200] + return outcome + records = sorted( + (record for record in journal.records.values() + if record.schedule.campaign_id == campaign_id), + key=lambda record: record.schedule.execution_order) + if not records: + return outcome + recipe_ids: List[str] = [] + for record in records: + if record.schedule.recipe_id not in recipe_ids: + recipe_ids.append(record.schedule.recipe_id) + try: + campaign_result = campaign_result_from_records( + campaign_id=campaign_id, config=protocol_config, + seed=int(manifest.get("seed") or 0), + recipes=[RecipeSpec(recipe_id=recipe_id, expectation=StructuralExpectation()) + for recipe_id in recipe_ids], + records=journal.records) + except (TypeError, ValueError, KeyError) as exc: + outcome["failure"] = f"campaign evidence cannot be projected: {exc}"[:200] + return outcome + # Same completeness rule as the live checkpoint path: only recipes the + # scheduler considers final (stable or at the attempt cap) may publish, and + # the group receipt carries exactly the counted timing members. + unfinished = set(getattr(campaign_result, "unfinished_recipes", frozenset())) + groups = _completed_measurement_groups(campaign_result) + ffmpeg_ok, ffmpeg_version = ensure_ffmpeg_and_ffprobe() + if not ffmpeg_ok: + outcome["failure"] = "ffmpeg/ffprobe unavailable; toolchain identity cannot be reconstructed faithfully" + return outcome + for record in records: + if _is_cancelled(cancel_event): + outcome["cancelled"] = True + return outcome + if record.schedule.phase != "measured": + continue + envelope_path = root / f"submission-{record.schedule.execution_order:06d}.json" + if envelope_path.is_file(): + outcome["skippedExisting"] += 1 + continue + if journal.accepted_receipt(record) is not None: + outcome["skippedAccepted"] += 1 + continue # an accepted receipt already authorizes this attempt + if record.schedule.recipe_id in unfinished: + outcome["skippedIncomplete"] += 1 # a later segment could still extend it + continue + info = dict(record.metadata.get("info") or {}) + if (record.skipped_before_encode or record.timing is None + or record.overall_validity.state == "invalid" + or not info or info.get("error") is not None): + outcome["skippedInvalid"] += 1 # the live path never submitted these + continue + suite_clip_data = record.metadata.get("suiteClip") + try: + prepared_clip = (PreparedSuiteClip(**suite_clip_data) + if isinstance(suite_clip_data, dict) else suite_clip_data) + artifact_path = str(info.get("artifactPath") or "") + if not artifact_path or not os.path.isfile(artifact_path): + raise OSError("retained artifact bytes are missing") + artifact_probe = probe_video_stream_metrics(artifact_path) + execution_identity_payload = build_execution_identity_payload( + hardware=hardware, + artifact_info=info, + ffmpeg_version=ffmpeg_version, + client_version=CLIENT_VERSION, + benchmark_protocol_version=config.BENCHMARK_PROTOCOL_VERSION, + ) + run_create = _build_authoritative_run_create_request( + prepared_clip=prepared_clip, + recipe_id=record.schedule.recipe_id, + record=record, + info=info, + metrics=dict(record.metadata.get("metrics") or {}), + hardware=hardware, + source_probe=dict(record.metadata.get("sourceProbe") or {}), + artifact_probe=artifact_probe, + ffmpeg_version=ffmpeg_version, + client_version=CLIENT_VERSION, + execution_identity_payload=execution_identity_payload, + protocol_config=protocol_config, + measurement_group=groups.get(record.schedule.recipe_id), + ) + # Frozen plan identity wins over this machine's current values: the + # campaign was requested and sourced under the manifest IDs. + physical_frozen = str(manifest.get("physicalSourceId") or "") + if physical_frozen and run_create.get("physicalSourceId") != physical_frozen: + run_create["physicalSourceId"] = physical_frozen + run_create["payloadHash"] = build_payload_hash(run_create) + submission = build_artifact_submission_payload( + artifact_path=artifact_path, + media_container=artifact_probe.get("containerFormat"), + run_create=run_create, + ) + except (OSError, ValueError, TypeError, KeyError, RuntimeError, AttributeError) as exc: + outcome["failure"] = ( + f"completed group {record.schedule.recipe_id} cannot be reconstructed: {exc}"[:200]) + return outcome + atomic_json(envelope_path, submission) + outcome["reconstructed"] += 1 + return outcome + + +def publish_saved_campaign( + *, + queue_dir: str, + campaign_id: str, + base_url: str, + api_key: str, + max_storage_mb: Optional[int] = None, + retries: int = 1, + use_token: bool = False, + interactive: bool = False, + cancel_event: Optional[Any] = None, + event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, +) -> Tuple[int, Dict[str, Any]]: + """Publish a saved campaign's completed groups with ZERO encodes (C06/C09/C11). + + Never-encode contract: complete groups whose immutable `submission-*.json` + envelopes already exist are admitted directly; groups whose measured attempts + are durable but whose envelopes never materialized (controlled stop or disk + cap before envelope creation) are REBUILT from retained journal records - + no encoder, suite clip or input source is ever touched, and unfinished + groups stay unfinished. The saved storage budget is restored unless the + caller overrides it; committed receipts drain before any new staging (C10); + accepted groups are skipped by journal receipt and by queue receipt + identity, so replays never re-upload (C07). Publication holds the + host-wide phase lock and defers while any collector measures on this host, + including through a different queue directory (C11). + + Returns (exit_code, info). Exit codes match --upload-only: 0 done, 10 + deferred/pending, 1 terminal failures or corrupt queue evidence. `info` + carries reconciled counters and safe failure fields only.""" + info: Dict[str, Any] = {"campaignId": campaign_id, "status": "working", + "admitted": 0, "skippedAccepted": 0, "terminal": 0, + "deferredReason": None, "failure": None} + if interactive and not _ensure_interactive_publication_consent(queue_dir=queue_dir): + info.update(status="consent_declined", deferredReason="consent_required") + return 10, info + try: + root = journal_path(queue_dir, campaign_id) + except ValueError as exc: + info.update(status="blocked", failure=_safe_failure_info(exc)) + return 1, info + if max_storage_mb is None: + try: + budget = json.loads((root / "budget.json").read_text(encoding="utf-8")) + max_storage_mb = int(budget.get("maxStorageMb") or 0) or 2048 + except (OSError, ValueError, TypeError): + max_storage_mb = 2048 + info["maxStorageMb"] = int(max_storage_mb) + try: + # C11: kernel-backed host phase lock on top of per-queue ownership, so + # two different queue directories on one host can never publish while + # a collector times measurements. Crash releases the flock; nothing is + # ever deleted as "stale". Inspection failure fails closed (defer). + with host_phase_hold("publication"): + return _publish_saved_campaign_gated( + queue_dir=queue_dir, campaign_id=campaign_id, base_url=base_url, + api_key=api_key, max_storage_mb=int(max_storage_mb), retries=retries, + use_token=use_token, cancel_event=cancel_event, event_sink=event_sink) + except SpoolCapacityError as exc: + info.update(status="deferred", deferredReason="measurement_exclusion", + failure=_safe_failure_info(exc)) + _emit_event(event_sink, "publication_deferred", + **{k: v for k, v in info.items() if k != "failure"}, + reason=info["deferredReason"]) + return 10, info + + +def _publish_saved_campaign_gated( + *, + queue_dir: str, + campaign_id: str, + base_url: str, + api_key: str, + max_storage_mb: int, + retries: int = 1, + use_token: bool = False, + cancel_event: Optional[Any] = None, + event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, +) -> Tuple[int, Dict[str, Any]]: + """Publish body that runs only while the host publication phase is held.""" + info: Dict[str, Any] = {"campaignId": campaign_id, "status": "working", + "admitted": 0, "skippedAccepted": 0, "terminal": 0, + "unadmitted": 0, + "deferredReason": None, "failure": None, + "maxStorageMb": int(max_storage_mb)} + root = journal_path(queue_dir, campaign_id) + try: + marker = root / "campaign-complete.json" + if marker.exists(): + result = json.loads(marker.read_text()) + info["campaignFailures"] = bool(result.get("skipped") or result.get("failed")) + # Drain crash-residual receipted entries BEFORE staging (C10): committed + # acceptances own no new bytes and free budget for genuinely new work. + info["drainedResiduals"] = drain_committed_receipts(queue_dir) + # C09: complete groups whose envelopes never materialized (controlled + # stop / disk cap before envelope creation) are rebuilt from retained + # journal records here - never encoded, never re-sourced - so the + # admission loop below publishes the same group identity the live path + # would have submitted. + recon = _reconstruct_saved_submissions( + queue_dir=queue_dir, campaign_id=campaign_id, + max_storage_mb=int(max_storage_mb), cancel_event=cancel_event) + info["reconstructedGroups"] = int(recon.get("reconstructed", 0)) + info["reconstruction"] = {k: v for k, v in recon.items() if k != "reconstructed"} + if recon.get("failure"): + info.update(status="blocked", failure=str(recon["failure"])[:200]) + return 1, info + if recon.get("cancelled"): + info.update(status="cancelled", deferredReason="cancelled") + return 10, info + submission_paths = sorted(root.glob("submission-*.json")) + for index, path in enumerate(submission_paths): + if _is_cancelled(cancel_event): + info.update(status="cancelled", deferredReason="cancelled") + return 10, info + if path.name.endswith(".accepted.json"): + continue + receipt = path.with_name(f"{path.stem}.accepted.json") + if receipt.exists(): + info["skippedAccepted"] += 1 + continue + try: + saved = json.loads(path.read_text()) + except (OSError, ValueError): + info["terminal"] += 1 + continue + if not isinstance(saved, dict): + info["terminal"] += 1 + continue + try: + _spooled_path, entry = spool_payload(queue_dir, saved, max_storage_mb=int(max_storage_mb)) + except SpoolCapacityError as exc: + info.update(status="deferred", deferredReason="storage_or_exclusion", + failure=_safe_failure_info(exc)) + info["unadmitted"] = sum( + not item.name.endswith(".accepted.json") and not item.with_name(f"{item.stem}.accepted.json").exists() + for item in submission_paths[index:] + ) + break + if entry.get("terminal") is True: + info["terminal"] += 1 + else: + info["admitted"] += 1 + except Exception as exc: # unreadable journal root: honest stop, never encode + info.update(status="blocked", failure=_safe_failure_info(exc)) + return 1, info + try: + stats = replay_spool(queue_dir, base_url=base_url, api_key=api_key, + retries=max(1, int(retries)), use_token=use_token, + cancel_event=cancel_event) + except SpoolCapacityError as exc: + info.update(status="deferred", deferredReason="measurement_exclusion", + failure=_safe_failure_info(exc)) + _emit_event(event_sink, "publication_deferred", **{k: v for k, v in info.items() if k != "failure"}, + reason=info["deferredReason"]) + return 10, info + info.update(submitted=stats.submitted, retained=stats.retained, + deadLettered=stats.dead_lettered, corrupt=stats.corrupt, + deferred=stats.deferred, cancelled=stats.cancelled, + pending=count_pending_entries(queue_dir)) + if stats.corrupt or stats.dead_lettered or info["terminal"] or info.get("campaignFailures"): + info["status"] = "corrupt" if stats.corrupt and not (stats.dead_lettered or info["terminal"]) else "terminal_failures" + return 1, info + if info["unadmitted"]: + # Draining the staged prefix cannot make unvisited journal envelopes + # published. Keep the normal Publish saved action available for the + # remaining groups; never return a green result for a partial pass. + info.update(status="deferred", deferredReason=info["deferredReason"] or "storage_or_exclusion") + return 10, info + if info["pending"]: + # Retained, deferred and cancelled work remains pending in the queue; + # pending count is the durable truth for the operator's next step. + info.update(status="pending", deferredReason=info["deferredReason"] or "uploads_pending") + return 10, info + info["status"] = "published" + _emit_event(event_sink, "publication_complete", campaignId=campaign_id, + submitted=info.get("submitted", 0)) + return 0, info + + +def retry_due_uploads( + *, + queue_dir: str, + base_url: str, + api_key: str, + retries: int = 1, + use_token: bool = False, + cancel_event: Optional[Any] = None, +) -> Tuple[int, Dict[str, Any]]: + """Retry DUE queued uploads without encodes and without touching campaigns (C12). + + Bounded (due-first window, time budget) and cancellable; ambiguous network + outcomes stay durable for idempotent retry. Exit codes match publish_saved_campaign.""" + drained = drain_committed_receipts(queue_dir) + try: + stats = replay_spool(queue_dir, base_url=base_url, api_key=api_key, + retries=max(1, int(retries)), use_token=use_token, + cancel_event=cancel_event) + except SpoolCapacityError as exc: + return 10, {"status": "deferred", "deferredReason": "measurement_exclusion", + "drainedResiduals": drained, "failure": _safe_failure_info(exc)} + pending = count_pending_entries(queue_dir) + result = {"status": "pending" if pending else "published", + "drainedResiduals": drained, "submitted": stats.submitted, + "retained": stats.retained, "deadLettered": stats.dead_lettered, + "corrupt": stats.corrupt, "deferred": stats.deferred, + "cancelled": stats.cancelled, "pending": pending} + if stats.corrupt or stats.dead_lettered: + result["status"] = "terminal_failures" + return 1, result + if pending: + return 10, result + return 0, result + + +def campaign_recovery_state(queue_dir: str, campaign_id: str) -> Optional[Dict[str, Any]]: + """Read-only journal projection for one campaign; None when no journal exists. + + Completed groups (saved submissions), accepted uploads, genuinely missing + sources and queue-side publication counters are all reported, so an + operator sees what resume/publish will do before running it. Nothing here + mutates state; rejected/expired evidence stays visible, never revived.""" + try: + root = journal_path(queue_dir, campaign_id) + except ValueError: + return None + if not root.is_dir(): + return None + state: Dict[str, Any] = {"campaignId": campaign_id, "journalBytes": directory_bytes(str(root))} + try: + state["savedBudgetMb"] = int(json.loads((root / "budget.json").read_text()).get("maxStorageMb") or 0) + except (OSError, ValueError, TypeError): + state["savedBudgetMb"] = None + state["complete"] = (root / "campaign-complete.json").is_file() + state["attempts"] = sum(1 for p in root.glob("attempt-*.json") if p.is_file()) + accepted = {p.stem.replace(".accepted", "") for p in root.glob("submission-*.accepted.json")} + submissions = [p for p in sorted(root.glob("submission-*.json")) if not p.name.endswith(".accepted.json")] + state["completedGroups"] = len(submissions) + state["acceptedUploads"] = len(accepted) + unuploaded = [p for p in submissions if p.stem not in accepted] + missing_source = 0 + for path in unuploaded: + try: + saved = json.loads(path.read_text()) + except (OSError, ValueError): + missing_source += 1 + continue + artifact = str((saved or {}).get("artifactPath") or "") if isinstance(saved, dict) else "" + if not artifact or not os.path.isfile(artifact): + missing_source += 1 + state["pendingUploads"] = len(unuploaded) - missing_source + state["unavailableSources"] = missing_source + queue_view = campaign_queue_summary(queue_dir, campaign_id) + state["queuePending"] = queue_view["pendingEntries"] + state["queueDue"] = queue_view["dueEntries"] + state["queueAccepted"] = queue_view["acceptedReceipts"] + state["queueTerminal"] = queue_view["terminalEntries"] + state["nextAttemptAt"] = queue_view["nextAttemptAt"] + actions: List[Dict[str, str]] = [] + if queue_view["terminalEntries"]: + actions.append({"action": "review_terminal", + "why": "a rejected or expired upload is terminal evidence; inspect dead-letter before deciding"}) + if state["pendingUploads"] or queue_view["pendingEntries"]: + actions.append({"action": "publish_saved", + "why": "completed groups can be published with zero encodes via --upload-only/--publish-saved"}) + if state["unavailableSources"]: + actions.append({"action": "resume", + "why": "attempts without accepted receipts and missing artifacts re-encode on resume"}) + if not state["complete"]: + actions.append({"action": "resume", + "why": "campaign never reached its completion marker; resume continues saved plan"}) + if queue_view["nextAttemptAt"]: + actions.append({"action": "wait", + "why": f"earliest Retry-After arrives at {int(queue_view['nextAttemptAt'])}"}) + state["actions"] = actions + return state + + +def recovery_state(queue_dir: str, campaign_id: str = "") -> Dict[str, Any]: + """Documented cross-process recovery projection (C06) for CLI/GUI consumers. + + Read-only: consent, live-collection exclusion, queue counters and per- + campaign journal state, reconciled. Safe strings only - no server bodies, + tokens or credentials.""" + state: Dict[str, Any] = { + "queueDir": str(queue_dir), + "publicationConsent": _has_publication_consent(), + "activeCollection": None, + "publication": queue_recovery_summary(queue_dir), + "publicationLockBusy": publication_lock_busy(queue_dir), + "campaigns": [], + } + try: + state["activeCollection"] = active_collection(queue_dir) + except OSError as exc: + state["activeCollectionError"] = str(exc)[:200] + wanted = [campaign_id] if campaign_id else [ + name for name in sorted(os.listdir(os.path.join(queue_dir, "campaigns"))) + if name.startswith("campaign-") + ] if os.path.isdir(os.path.join(queue_dir, "campaigns")) else [] + for cid in wanted: + view = campaign_recovery_state(queue_dir, cid) + if view is not None: + state["campaigns"].append(view) + return state + + +def _print_recovery_state(queue_dir: str, campaign_id: str) -> None: + state = recovery_state(queue_dir, campaign_id) + pub = state.get("publication") or {} + print_info(f"Queue: {state['queueDir']}") + print_info(f"Publication consent: {'granted' if state['publicationConsent'] else 'NOT granted (publishing stays local)'}") + if state.get("activeCollection"): + print_warning("A measurement is running on this queue; publication work defers until it checkpoints.") + elif state.get("publicationLockBusy"): + print_warning("Another publisher currently owns this queue; retry shortly.") + print_info( + f"Queue: {pub.get('pendingEntries', 0)} pending ({pub.get('dueEntries', 0)} due), " + f"{pub.get('acceptedReceipts', 0)} accepted receipt(s), {pub.get('terminalEntries', 0)} terminal" + ) + if pub.get("nextAttemptAt"): + print_info(f"Earliest scheduled retry: {time.strftime('%Y-%m-%d %H:%M', time.localtime(int(pub['nextAttemptAt'])))}") + for view in state.get("campaigns", []): + print_info( + f"{view['campaignId']}: {view['completedGroups']} completed group(s), " + f"{view['acceptedUploads']} accepted, {view['pendingUploads']} pending upload, " + f"{view['unavailableSources']} unavailable source(s), " + f"{'complete' if view['complete'] else 'incomplete'}" + ) + for act in view.get("actions", []): + print_info(f" -> {act['action']}: {act['why']}") + if not state.get("campaigns"): + print_info("No retained campaigns in this queue.") + + +def _report_recovery_result(info: Dict[str, Any]) -> None: + status = str(info.get("status") or "unknown") + if status == "published": + print_success(f"Published saved campaign {info.get('campaignId')}: {info.get('submitted', 0)} upload(s), " + f"{info.get('skippedAccepted', 0)} already accepted.") + elif status == "consent_declined": + print_warning("Publication needs your consent; nothing was uploaded.") + elif status == "cancelled": + print_warning("Publishing cancelled; accepted uploads remain recorded and retries stay durable.") + elif status == "deferred": + reason = (info.get("failure") or {}).get("reason") or info.get("deferredReason") or "storage or exclusion" + print_warning(f"Publishing deferred: {reason}") + elif status == "terminal_failures": + print_warning(f"{info.get('terminal', 0)} terminal and {info.get('corrupt', 0)} corrupt entr(ies); " + "inspect dead-letter before retrying.") + elif status == "blocked": + print_error((info.get("failure") or {}).get("reason") or "campaign journal unavailable") + else: + print_info(f"Publishing status: {status} ({info.get('pending', 0)} upload(s) still pending)") + + def _submit_payload_with_spool( *, queue_dir: str, @@ -1548,11 +2145,14 @@ def sweep_plan_label(encoder: str) -> str: def active_collection_guard(queue_dir: str, event_sink: Optional[Callable[[Dict[str, Any]], None]] = None, *, scope: str = "batch") -> Optional[int]: - """Refusal code when this queue already hosts a live collection, else None. + """Refusal code when this queue already hosts a live collection or a live + publication pass, else None. A live collector holds the measurement lock for its whole campaign: its checkpoints continue automatically, so a second run must wait for it to - finish or stop/cancel it first - never "resume over" a checkpoint.""" + finish or stop/cancel it first - never "resume over" a checkpoint. The + reverse direction holds too: while a publisher drains this queue, starting + a collector would corrupt timing through the same disk (C11).""" try: active = active_collection(str(queue_dir)) except OSError as exc: @@ -1561,6 +2161,38 @@ def active_collection_guard(queue_dir: str, _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) return 6 if active is None: + try: + publisher_busy = publication_lock_busy(str(queue_dir)) + except OSError as exc: + message = f"Cannot verify publication exclusion: {exc}" + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + if publisher_busy: + # C11 reverse direction: a publisher is draining/staging this queue's + # bytes right now. Its replay is time-bounded; starting a collector + # mid-drain would corrupt measurement timing through the same disk. + message = ("A publication pass currently owns this queue; it ends within its time budget. " + "Wait for it to finish, then start the collection.") + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + try: + # C11 cross-queue: a publisher or collector owns the HOST-wide phase + # through a different queue directory; a non-blocking probe defers + # this start, and an uninspectable lock fails closed (exit 6). + if host_phase_busy(): + message = ("A publication or measurement pass owns this host right now " + "(possibly through another queue directory); it is time-bounded. " + "Wait for it to finish, then start the collection.") + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 + except OSError as exc: + message = f"Cannot verify host phase exclusion: {exc}" + print(message, file=sys.stderr) + _emit_event(event_sink, "run_error", scope=scope, code=6, message=message) + return 6 return None who = f" (campaign {active['campaignId']}, PID {active['pid']})" if active.get("campaignId") else "" message = (f"Another collection is actively running in this queue{who}. Its checkpoints continue " @@ -1572,7 +2204,39 @@ def active_collection_guard(queue_dir: str, +def _collector_publication(function): + """Mark this process as the host's live collector for its whole batch (C11). + + The collector's own checkpoint uploads legitimately run while it holds the + measurement lock and the host measurement phase; the exclusion probes refuse + every OTHER publisher, in-process re-entry (checkpoint upload) passes. The + kernel phase lock is the atomic arbiter across queue directories: a held + lock refuses this collector with exit 6 instead of letting two hosts' worth + of I/O overlap through different queues.""" + @wraps(function) + def wrapped(*args, **kwargs): + queue_dir = str(getattr(kwargs.get("args"), "queue_dir", "") or "") + if not queue_dir: + with collector_publication_scope(): + return function(*args, **kwargs) + try: + hold = host_phase_hold("measurement") + hold.__enter__() + except SpoolCapacityError as exc: + message = str(exc) + print(message, file=sys.stderr) + _emit_event(kwargs.get("event_sink"), "run_error", scope="batch", code=6, message=message) + return 6 + try: + with collector_publication_scope(): + return function(*args, **kwargs) + finally: + hold.__exit__(None, None, None) + return wrapped + + @_preparation_operation +@_collector_publication def run_benchmark_batch( *, hardware: HardwareInfo, @@ -1652,6 +2316,21 @@ def run_benchmark_batch( "tasks": [{"encoder": t["encoder"], "preset": t["preset"], "crf": t.get("crf"), "rateControl": t.get("rateControl"), "clipId": t["suiteClip"].clip_id} for t in tasks], } + existing_manifest = journal_path(args.queue_dir, campaign_id) / "manifest.json" + if existing_manifest.is_file(): + try: + prior_version = str(json.loads(existing_manifest.read_text()).get("clientVersion") or "") + except (OSError, ValueError, TypeError, AttributeError): + prior_version = "" # CampaignJournal reports an unreadable manifest below. + if prior_version and prior_version != client_version: + message = f"Saved campaign requires {prior_version}; this client is {client_version}. Use the original client." + _emit_event(event_sink, "run_error", scope="batch", code=6, message=message) + print(message, file=sys.stderr) + return 6 + if prior_version: + manifest["clientVersion"] = prior_version + else: + manifest["clientVersion"] = client_version if isinstance(plan_metadata, dict): manifest.update(plan_metadata) try: @@ -1696,6 +2375,7 @@ def run_benchmark_batch( submitted_count = 0 skipped_count = 0 failed_count = 0 + locally_complete_count = 0 _emit_event( event_sink, "run_start", @@ -2315,8 +2995,28 @@ def _journal_attempt(record: Any) -> None: else: atomic_json(local_path, authoritative_submission) if args.no_submit: - _emit_event(event_sink, "submit_result", status="locally_complete", campaignId=campaign_id) + # Local-only is a completed unit of work: the global counters and + # task_complete must advance exactly like a submitted path (C04/C05). + # 'submitted' stays the upload count honestly: zero here. + locally_complete_count += 1 completed_count_local += 1 + processed_total += 1 + progress.advance(description=_batch_status("Locally complete", processed_total)) + _emit_event( + event_sink, + "submit_result", + index=next_index, + total=total_tasks, + status="locally_complete", + campaignId=campaign_id, + recipeId=recipe.recipe_id, + repetitionIndex=record.schedule.repetition_index, + executionOrder=record.schedule.execution_order, + ) + _emit_event(event_sink, "task_complete", scope="batch", + processed=processed_total, total=total_tasks) + _emit_counters(event_sink, submitted=submitted_count, skipped=skipped_count, + queued=queued_count, failed=failed_count) continue status, error_text, queued_count = _submit_payload_with_spool( queue_dir=args.queue_dir, @@ -2354,6 +3054,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="queued", error=error_text, + **_submit_failure_fields("queued", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2369,6 +3070,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="failed", error=error_text, + **_submit_failure_fields("failed", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2388,6 +3090,7 @@ def _journal_attempt(record: Any) -> None: total=total_tasks, status="failed", error=error_text, + **_submit_failure_fields("failed", error_text), campaignId=record.schedule.campaign_id, recipeId=recipe.recipe_id, repetitionIndex=record.schedule.repetition_index, @@ -2476,6 +3179,7 @@ def _journal_attempt(record: Any) -> None: "totalBatches": total_batches, "completed": completed_count_local, "submitted": submitted_count, + "locallyComplete": locally_complete_count, "skipped": skipped_count, "queued": queued_count, "failed": failed_count, @@ -2490,6 +3194,7 @@ def _journal_attempt(record: Any) -> None: totalBatches=total_batches, completed=completed_count_local, submitted=submitted_count, + locallyComplete=locally_complete_count, skipped=skipped_count, queued=queued_count, failed=failed_count, @@ -2942,12 +3647,12 @@ def run_legacy_diagnostic( elif status == "retained": print_warning(f"Queued for retry: {effective_preset} ({message})") progress.advance(description=f"{effective_preset} (queued)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="queued", preset=effective_preset, error=message) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="queued", preset=effective_preset, error=message, **_submit_failure_fields("queued", message)) else: failed_count += 1 print(f"Failed to submit {effective_preset}: {message}", file=sys.stderr) progress.advance(description=f"{effective_preset} (failed)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=message) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=message, **_submit_failure_fields("failed", message)) except SpoolCapacityError as exc: print_warning(f"Upload deferred: {exc}") return 10 @@ -2956,7 +3661,7 @@ def run_legacy_diagnostic( print(f"Failed to submit {effective_preset}: {e}", file=sys.stderr) queued_count = count_pending_entries(args.queue_dir) progress.advance(description=f"{effective_preset} (failed)") - _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=str(e)) + _emit_event(event_sink, "submit_result", scope="single", index=task_index, total=len(combos), status="failed", preset=effective_preset, error=str(e), **_submit_failure_fields("failed", str(e))) _emit_counters(event_sink, submitted=submitted_count, skipped=skipped_count, queued=queued_count, failed=failed_count) _emit_event(event_sink, "task_complete", scope="single", processed=task_index, total=len(combos), preset=effective_preset) @@ -3048,6 +3753,34 @@ def _incomplete_campaigns(queue_dir: str) -> List[Tuple[str, float]]: return [] +def _publishable_campaigns(queue_dir: str) -> List[Tuple[str, float, int]]: + """Completed campaigns whose journals still hold un-uploaded submissions. + + These publish with zero encodes (C06/C09): the projection says how many + completed groups await upload, so the menu offers it only where real + pending work exists.""" + try: + root = os.path.join(queue_dir, "campaigns") + if not os.path.isdir(root): + return [] + found: List[Tuple[str, float, int]] = [] + for name in os.listdir(root): + entry = os.path.join(root, name) + if not os.path.isdir(entry): + continue + state = campaign_recovery_state(queue_dir, name) + if state is None or not state.get("complete"): + continue + pending = int(state.get("pendingUploads") or 0) + int(state.get("queuePending") or 0) + if pending <= 0: + continue + found.append((name, os.path.getmtime(entry), pending)) + found.sort(key=lambda item: item[1], reverse=True) + return found[:5] + except Exception: + return [] + + def _sweep_clip_count(plan: sweep_plan.SweepPlan) -> int: return 1 if plan.clip_policy == sweep_plan.CLIP_POLICY_QUICK else len(REQUIRED_CONTENT_CLASSES) @@ -3185,8 +3918,13 @@ def interactive_menu_flow(parser: argparse.ArgumentParser, base_args: argparse.N f"{_guided_mode_label(mode, previews[mode], per_recipe_min, per_recipe_max)}]" ) actions.append(("sweep", mode)) + for campaign_id, _mtime, pending in _publishable_campaigns(base_args.queue_dir): + option_labels.append(f"Publish saved uploads for {campaign_id} ({pending} completed group(s), no re-encode)") + actions.append(("publish", campaign_id)) option_labels.append("Advanced: configure one recipe yourself (encoder, preset, native quality or bitrate)") actions.append(("single", None)) + option_labels.append("Show recovery status (read-only: consent, exclusion, queue and campaign counters)") + actions.append(("recovery", None)) option_labels.append("Exit") actions.append(("exit", None)) default_index = next((index for index, (action, _payload) in enumerate(actions) if action == "sweep"), 0) @@ -3197,6 +3935,18 @@ def interactive_menu_flow(parser: argparse.ArgumentParser, base_args: argparse.N if action == "resume": base_args.resume_campaign = payload return _resume_campaign(base_args, interactive=True) + if action == "publish": + rc, info = publish_saved_campaign( + queue_dir=base_args.queue_dir, campaign_id=str(payload), + base_url=base_args.base_url, api_key=base_args.api_key, + retries=max(1, int(getattr(base_args, "retries", 1) or 1)), + use_token=bool(getattr(base_args, "use_token", False)), + interactive=True) + _report_recovery_result(info) + return rc + if action == "recovery": + _print_recovery_state(base_args.queue_dir, "") + return 0 if action == "sweep": if not confirm_benchmark_readiness(): print("Aborted by user. Please close other programs and try again.") @@ -3243,6 +3993,12 @@ def build_arg_parser() -> argparse.ArgumentParser: p.add_argument("--campaign", choices=("quick", "full"), default="quick", help="One clip (quick) or all seven clips (full), with the selected recipe") p.add_argument("--resume-campaign", default="", help="Resume a retained campaign ID, preserving completed attempts") p.add_argument("--upload-only", action="store_true", help="Retry due queued uploads without encoding") + p.add_argument("--publish-saved", default="", metavar="CAMPAIGN_ID", + help="Publish a saved campaign's completed groups with zero encodes " + "(accepted uploads are skipped; nothing unfinished is materialized)") + p.add_argument("--recovery-status", action="store_true", + help="Print a read-only JSON recovery projection (consent, exclusion, queue and " + "campaign counters with concrete next actions), then exit") p.add_argument("--local-metrics", action="store_true", help="Run optional local quality diagnostics after all measurements") p.add_argument("--max-attempts", type=int, default=100, action=_ExplicitBudgetAction, help="Maximum planned warmup/measured encodes (default 100; guided sweeps size the cap " @@ -3301,6 +4057,15 @@ def main(argv: List[str]) -> int: return 10 if not args.queue_status: return 0 + if args.recovery_status: + print(json.dumps(recovery_state(args.queue_dir, args.publish_saved or args.resume_campaign), + indent=2, sort_keys=True)) + return 0 + if args.publish_saved: + if args.resume_campaign and args.resume_campaign != args.publish_saved: + parser.error("--publish-saved and --resume-campaign name different campaigns") + args.resume_campaign = args.publish_saved + args.upload_only = True if args.queue_status: _print_queue_status(args.queue_dir) return 0 @@ -3322,28 +4087,27 @@ def main(argv: List[str]) -> int: parser.error("--upload-only cannot be combined with --no-submit") try: check_compatibility(args.base_url, CLIENT_VERSION) - campaign_failures = False - publication_deferred = False + storage_mb = (int(args.max_storage_mb) if bool(getattr(args, "max_storage_mb_explicit", False)) + else None) if args.resume_campaign: - root = journal_path(args.queue_dir, args.resume_campaign) - marker = root / "campaign-complete.json" - if marker.exists(): - result = json.loads(marker.read_text()) - campaign_failures = bool(result.get("skipped") or result.get("failed")) - for path in sorted(root.glob("submission-*.json")): - try: - _spooled_path, entry = spool_payload(args.queue_dir, json.loads(path.read_text()), - max_storage_mb=args.max_storage_mb) - except SpoolCapacityError as exc: - print_warning(f"Upload deferred: {exc}") - publication_deferred = True - break - if entry.get("terminal") is True: - campaign_failures = True - print_warning(f"Retained upload is terminal ({entry.get('lastError') or 'terminal_upload'}); see {_spooled_path}.") - stats = replay_spool(args.queue_dir, base_url=args.base_url, api_key=args.api_key, - retries=1, use_token=False) - return 1 if campaign_failures or stats.dead_lettered or stats.corrupt else (10 if publication_deferred or count_pending_entries(args.queue_dir) else 0) + rc, info = publish_saved_campaign( + queue_dir=args.queue_dir, campaign_id=args.resume_campaign, + base_url=args.base_url, api_key=args.api_key, + max_storage_mb=storage_mb, retries=max(1, args.retries)) + else: + rc, info = retry_due_uploads(queue_dir=args.queue_dir, base_url=args.base_url, + api_key=args.api_key, retries=max(1, args.retries)) + for warning in (("Retained uploads are terminal; inspect the dead-letter before retrying." + if info.get("deadLettered") else ""), + (f"Corrupt queue files moved to dead-letter: {info.get('corrupt')}." + if info.get("corrupt") else ""), + (f"Saved campaign has terminal or skipped work: {info.get('terminal', 0)} entry(ies)." + if info.get("terminal") else "")): + if warning: + print_warning(warning) + if info.get("deferredReason") == "storage_or_exclusion": + print_warning(f"Upload deferred: {(info.get('failure') or {}).get('reason') or 'storage budget'}") + return rc except Exception as exc: print(f"Upload deferred: {exc}", file=sys.stderr) return 10 diff --git a/client/spool.py b/client/spool.py index 0e551dfb..92fde29f 100644 --- a/client/spool.py +++ b/client/spool.py @@ -5,7 +5,9 @@ import shutil import time import random +import tempfile from pathlib import Path +import contextvars from contextlib import contextmanager from dataclasses import dataclass from typing import Any, Dict, List, Optional, Tuple @@ -17,6 +19,187 @@ SPOOL_VERSION = 1 MANAGED_ARTIFACT_DIRNAME = "artifacts" SPOOL_METADATA_RESERVE_BYTES = 64 * 1024 +REPLAY_WINDOW = 25 +REPLAY_SECONDS = 60.0 + +# True only inside a collector's own batch: its checkpoint uploads legitimately run +# while it holds this queue's measurement lock. External publishers must defer. +_COLLECTOR_PUBLICATION_SCOPE: contextvars.ContextVar = contextvars.ContextVar( + "encodingdb_collector_publication", default=False) + +# True while this process holds the host phase lock; nested publication passes +# (collector checkpoint uploads, publish wrapper around replay) reuse it. +_HOST_PHASE_HELD: contextvars.ContextVar = contextvars.ContextVar( + "encodingdb_host_phase_held", default=False) + + +@contextmanager +def collector_publication_scope(): + """Declare that the calling process owns the live collection for this queue.""" + token = _COLLECTOR_PUBLICATION_SCOPE.set(True) + try: + yield + finally: + _COLLECTOR_PUBLICATION_SCOPE.reset(token) + + +def _refuse_publication_during_measurement(queue_dir: str) -> None: + """Publication defers to a collector measuring on this queue (C11). + + The probe opens measurement.lock non-blockingly from a fresh descriptor; a + held lock - even one owned by this same process through another descriptor - + means authoritative collection is timing-sensitive right now. Only the + collector's own in-batch uploads (``collector_publication_scope``) skip it.""" + if _COLLECTOR_PUBLICATION_SCOPE.get(): + return + from .campaign import active_collection + try: + active = active_collection(queue_dir) + except OSError as exc: + raise SpoolCapacityError(f"Cannot verify publication exclusion: {exc}") from exc + if active is not None: + raise SpoolCapacityError( + "A collection is measuring in this queue; publication defers to its next checkpoint") + + +def publication_lock_busy(queue_dir: str) -> bool: + """True when another publisher currently owns this queue's publication lock.""" + try: + with _spool_write_lock(queue_dir): + return False + except SpoolCapacityError: + return True + + +def _host_phase_lock_path() -> str: + """Host-scoped lock file shared by every queue of this user on this machine. + + C11: two different queue directories on the same host still share one disk + and network path, so measurement timing and publication cannot overlap even + across queues. The kernel releases flock/msvcrt ownership on process death, + so a crash needs no stale-lock cleanup - and none is ever performed.""" + root = os.environ.get("ENCODINGDB_HOST_PHASE_DIR") or os.path.join( + tempfile.gettempdir(), "encodingdb-host-phase-{}".format(getattr(os, "getuid", lambda: "shared")())) + os.makedirs(root, exist_ok=True) + return os.path.join(root, "phase.lock") + + +def _host_phase_probe_held() -> bool: + """True when some process currently owns the host phase lock (advisory). + + Fail-closed: an uninspectable lock file raises OSError; correctness never + depends on this probe - both phases acquire the lock non-blockingly before + doing timing-sensitive work, and the loser defers.""" + path = _host_phase_lock_path() + try: + with open(path, "a+b") as handle: + if os.fstat(handle.fileno()).st_size == 0: + handle.write(b"0") + handle.flush() + if os.name == "nt": + import msvcrt + handle.seek(0) + try: + msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1) + except OSError: + return True + msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1) + return False + import fcntl + try: + fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError: + return True + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + return False + except OSError: + raise + + +def host_phase_busy() -> bool: + """True when another phase owner currently holds the host phase lock (C11). + + Advisory probe for start guards; the authoritative arbiter remains the + non-blocking acquire inside ``host_phase_hold``. Re-entry by the current + holder reports free - a collector guard runs inside its own measurement + hold and must not refuse itself. Fails closed: an uninspectable lock file + raises SpoolCapacityError.""" + if _HOST_PHASE_HELD.get(): + return False + try: + return _host_phase_probe_held() + except OSError as exc: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + + +@contextmanager +def host_phase_hold(role: str = "publication"): + """Kernel-backed host phase lock for one collector batch or one publication + pass (C11). + + Measurement takes the lock exclusively and publication shared, so a held + collector defers publishers and a held publisher defers the next collector, + in either start order: the non-blocking acquire is the atomic arbiter, so + neither side can slip between the other's probe and acquisition. Two + publishers coexist (uploads are idempotent and time-insensitive); only + measurement timing needs the exclusive phase. (msvcrt has no shared lock, + so on Windows publishers serialize - deferral is always safe.) A busy lock + raises SpoolCapacityError - publishers defer, collectors exit 6 - while any + other inspection failure also fails closed. The kernel releases flock + ownership on process death; no stale-lock deletion exists. Re-entry inside + the same process/thread (collector checkpoint upload, publish wrapper + around replay) reuses the held lock. Only acquisition errors are wrapped: + an OSError raised by the body propagates unchanged.""" + if _HOST_PHASE_HELD.get(): + yield False + return + path = _host_phase_lock_path() + try: + handle = open(path, "a+b") + except OSError as exc: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + try: + if os.fstat(handle.fileno()).st_size == 0: + handle.write(b"0") + handle.flush() + if os.name == "nt": + import msvcrt + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1) + else: + import fcntl + mode = fcntl.LOCK_EX if role == "measurement" else fcntl.LOCK_SH + fcntl.flock(handle.fileno(), mode | fcntl.LOCK_NB) + except OSError as exc: + busy = getattr(exc, "errno", None) in (errno.EAGAIN, errno.EACCES, errno.EWOULDBLOCK, + errno.EDEADLK) or type(exc).__name__ == "BlockingIOError" + try: + handle.close() + except OSError: + pass + if not busy: + raise SpoolCapacityError(f"Cannot inspect host phase lock: {exc}") from exc + if role == "measurement": + raise SpoolCapacityError( + "A publication or measurement pass owns this host right now; " + "wait for it to finish before starting collection") from exc + raise SpoolCapacityError( + "A collector is measuring on this host; publication defers to its next checkpoint") from exc + token = _HOST_PHASE_HELD.set(True) + try: + yield True + finally: + _HOST_PHASE_HELD.reset(token) + try: + if os.name == "nt": + import msvcrt + handle.seek(0) + msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1) + else: + import fcntl + fcntl.flock(handle.fileno(), fcntl.LOCK_UN) + finally: + handle.close() class SpoolCapacityError(OSError): @@ -115,6 +298,8 @@ class ReplayStats: retained: int = 0 dead_lettered: int = 0 corrupt: int = 0 + cancelled: int = 0 + deferred: int = 0 @dataclass @@ -174,6 +359,46 @@ def count_pending_entries(queue_dir: str) -> int: return 0 +def due_first_queue_paths(queue_dir: str, *, limit: int = REPLAY_WINDOW) -> Tuple[List[str], int]: + """Fair bounded due-first selection (C08). + + Returns (paths, deferred). Entries whose Retry-After has arrived - or whose + retry deadline has expired, so the verdict can be finalized - are due; the + window admits only due entries, oldest-scheduled first, so a failing prefix + can never starve healthy entries behind it. Delayed entries are counted as + deferred, never attempted early.""" + now = time.time() + due: List[Tuple[float, int, str, str]] = [] + deferred = 0 + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + return [], 0 + for name in names: + if not name.endswith(".json"): + continue + path = os.path.join(queue_dir, name) + if not os.path.isfile(path): + continue + try: + entry = load_spool_entry(path) + next_at = float(entry.get("nextAttemptAt") or 0) + deadline = float(entry.get("retryDeadlineAt") or 0) + queued = int(entry.get("queuedAt") or 0) + except Exception: + # Corrupt files are due immediately: the cheap terminal verdict + # retires them before healthy traffic waits behind them. + due.append((float("-inf"), 0, name, path)) + continue + if next_at <= now or deadline <= now: + due.append((next_at, queued, name, path)) + else: + deferred += 1 + due.sort(key=lambda item: (item[0], item[1], item[2])) + selected = [item[3] for item in due[:max(0, limit)]] + return selected, deferred + max(0, len(due) - len(selected)) + + def _iter_files(root: str) -> List[str]: files: List[str] = [] if not os.path.isdir(root): @@ -310,6 +535,18 @@ def _managed_artifact_path(queue_dir: str, artifact_sha256: str, source_path: st return os.path.join(_managed_artifact_dir(queue_dir), f"{artifact_sha256}{ext.lower()}") +def spool_payload(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int = 2048) -> Tuple[str, Dict[str, Any]]: + try: + with host_phase_hold("publication"): + _refuse_publication_during_measurement(queue_dir) + with _spool_write_lock(queue_dir): + return _spool_payload_locked(queue_dir, payload, max_storage_mb=max_storage_mb) + except OSError as exc: + if exc.errno in (errno.ENOSPC, errno.EDQUOT): + raise SpoolCapacityError("Publication ran out of disk space; free space and resume the retained campaign") from exc + raise + + def _preserve_artifact_for_spool(queue_dir: str, payload: Dict[str, Any]) -> Dict[str, Any]: if payload.get("submissionKind") != AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: return dict(payload) @@ -376,14 +613,6 @@ def _terminal_spool_entry_locked(queue_dir: str, local_hash: str) -> Optional[Tu return None -def spool_payload(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int = 2048) -> Tuple[str, Dict[str, Any]]: - try: - with _spool_write_lock(queue_dir): - return _spool_payload_locked(queue_dir, payload, max_storage_mb=max_storage_mb) - except OSError as exc: - if exc.errno in (errno.ENOSPC, errno.EDQUOT): - raise SpoolCapacityError("Publication ran out of disk space; free space and resume the retained campaign") from exc - raise def _spool_payload_locked(queue_dir: str, payload: Dict[str, Any], *, max_storage_mb: int) -> Tuple[str, Dict[str, Any]]: @@ -621,6 +850,225 @@ def _submission_success_message(payload: Dict[str, Any], response: Any) -> str: return "" +def _journal_self_publish(queue_dir: str, payload: Dict[str, Any], response: Any) -> None: + """Commit journal acceptance evidence for a completed queue upload (C11). + + A checkpoint upload that finishes outside a live batch (standalone replay, + --upload-only, GUI retry) must still write the campaign journal's own receipt + and retire the owned artifact, so a crash between server acceptance and + journal commit cannot resurrect an uploaded group for re-encode or re-upload. + Without a server run id nothing is treated as accepted; an existing faithful + receipt simply wins (idempotent replay).""" + if payload.get("submissionKind") != AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: + return + run_id = _submission_success_message(payload, response) + if not run_id: + return + local_hash = local_hash_for_payload(payload) + campaigns = Path(queue_dir) / "campaigns" + run_create = payload.get("runCreate") if isinstance(payload.get("runCreate"), dict) else {} + scoped = str(run_create.get("campaignId") or "") + try: + # The payload carries its campaign identity; scope the faithful-identity scan + # to that journal instead of parsing every campaign's submissions. + if scoped and (campaigns / scoped).is_dir(): + candidates = sorted((campaigns / scoped).glob("submission-*.json")) + else: + candidates = sorted(campaigns.glob("*/submission-*.json")) + except OSError: + return + for candidate in candidates: + if candidate.name.endswith(".accepted.json"): + continue + try: + journal_payload = json.loads(candidate.read_text()) + except (OSError, ValueError): + continue + if not isinstance(journal_payload, dict) or local_hash_for_payload(journal_payload) != local_hash: + continue + run_create = journal_payload.get("runCreate") + repetition_group = str((run_create or {}).get("repetitionGroupId") or "") if isinstance(run_create, dict) else "" + artifact_path = str(journal_payload.get("artifactPath") or "").strip() + artifact_sha = str(journal_payload.get("artifactSha256") or "").strip() + try: + order_value = int(candidate.stem.split("-", 1)[1]) + except (ValueError, IndexError): + continue + if (not artifact_path or not artifact_sha or ":" not in repetition_group + or str((run_create or {}).get("campaignId") or "") != candidate.parent.name): + continue + _write_json_atomic(str(candidate.with_name(f"submission-{order_value:06d}.accepted.json")), + {"schemaVersion": 1, + "executionOrder": order_value, + "recipeId": repetition_group.split(":", 1)[1], + "artifactPath": artifact_path, + "artifactSha256": artifact_sha, + "benchmarkRunId": run_id, + "acceptedAt": time.time()}) + owned = Path(artifact_path).resolve() + if candidate.parent.resolve() in owned.parents and owned.is_file(): + try: + owned.unlink() + except OSError: + pass + return + + +def drain_committed_receipts(queue_dir: str) -> int: + """Retire pending entries whose acceptance receipt is already committed (C10). + + A crash between receipt commit and entry unlink leaves both files; the + receipt is terminal evidence, so the entry (and its managed staging copy) + must drain BEFORE new staging consumes the storage budget. Journal + self-publish runs here too, covering a crash inside the other process.""" + drained = 0 + with _spool_write_lock(queue_dir): + try: + names = os.listdir(queue_dir) + except OSError: + return 0 + for name in names: + if not name.endswith(".json") or not os.path.isfile(os.path.join(queue_dir, name)): + continue + receipt_path = os.path.join(queue_dir, "receipts", name) + if not os.path.isfile(receipt_path): + continue + path = os.path.join(queue_dir, name) + try: + entry = load_spool_entry(path) + except Exception: + entry = None + if entry is not None: + try: + receipt = json.loads(Path(receipt_path).read_text()) + except (OSError, ValueError): + receipt = {} + _journal_self_publish(queue_dir, entry.get("payload") or {}, receipt.get("response")) + _cleanup_managed_artifact_if_unreferenced(queue_dir, entry, excluding_entry_path=path) + try: + os.remove(path) + drained += 1 + except OSError: + pass + return drained + + +def campaign_queue_summary(queue_dir: str, campaign_id: str) -> Dict[str, Any]: + """Reconciled publication counters for ONE campaign across queue+receipts+terminal. + + Counts entries by durable identity: pending staging, accepted receipts, + terminal dead letters. Measurement counters live in the journal; this view + shows what publication actually holds for the campaign right now.""" + summary = {"pendingEntries": 0, "pendingBytes": 0, "dueEntries": 0, "acceptedReceipts": 0, + "terminalEntries": 0, "nextAttemptAt": None} + now = time.time() + def belongs(payload: Any) -> bool: + run_create = payload.get("runCreate") if isinstance(payload, dict) else None + return isinstance(run_create, dict) and str(run_create.get("campaignId") or "") == campaign_id + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + return summary + for name in names: + path = os.path.join(queue_dir, name) + if not name.endswith(".json") or not os.path.isfile(path): + continue + try: + entry = load_spool_entry(path) + except Exception: + continue + if not belongs(entry.get("payload")): + continue + summary["pendingEntries"] += 1 + summary["pendingBytes"] += os.path.getsize(path) + if float(entry.get("nextAttemptAt") or 0) <= now: + summary["dueEntries"] += 1 + else: + next_at = float(entry.get("nextAttemptAt") or 0) + if summary["nextAttemptAt"] is None or next_at < summary["nextAttemptAt"]: + summary["nextAttemptAt"] = next_at + receipts = Path(queue_dir) / "receipts" + try: + for receipt_file in sorted(receipts.glob("*.json")): + try: + receipt = json.loads(receipt_file.read_text()) + except (OSError, ValueError): + continue + response = receipt.get("response") if isinstance(receipt, dict) else None + run_id = _submission_success_message({"submissionKind": AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND}, response) + if run_id and _receipt_matches_campaign(queue_dir, receipt_file.stem, campaign_id): + summary["acceptedReceipts"] += 1 + except OSError: + pass + terminal_dir = Path(queue_dir) / "terminal" + try: + for terminal_file in sorted(terminal_dir.glob("*.json")): + try: + terminal = json.loads(terminal_file.read_text()) + except (OSError, ValueError): + continue + if belongs((terminal or {}).get("payload")): + summary["terminalEntries"] += 1 + except OSError: + pass + return summary + + +def _receipt_matches_campaign(queue_dir: str, local_hash: str, campaign_id: str) -> bool: + """A queue receipt names only its hash; the journal holds the campaign link.""" + root = Path(queue_dir) / "campaigns" / campaign_id + if not root.is_dir(): + return False + for submission in root.glob("submission-*.json"): + if submission.name.endswith(".accepted.json"): + continue + try: + payload = json.loads(submission.read_text()) + except (OSError, ValueError): + continue + if isinstance(payload, dict) and local_hash_for_payload(payload) == local_hash: + return True + return False + + +def queue_recovery_summary(queue_dir: str) -> Dict[str, Any]: + """Read-only queue-wide publication view for status/recovery projection (C06).""" + summary = {"pendingEntries": 0, "pendingBytes": 0, "dueEntries": 0, "acceptedReceipts": 0, + "terminalEntries": 0, "nextAttemptAt": None} + now = time.time() + try: + names = sorted(os.listdir(queue_dir)) + except OSError: + names = [] + for name in names: + path = os.path.join(queue_dir, name) + if not name.endswith(".json") or not os.path.isfile(path): + continue + summary["pendingEntries"] += 1 + try: + summary["pendingBytes"] += os.path.getsize(path) + except OSError: + pass + try: + entry = load_spool_entry(path) + next_at = float(entry.get("nextAttemptAt") or 0) + except Exception: + continue + if next_at <= now: + summary["dueEntries"] += 1 + elif summary["nextAttemptAt"] is None or next_at < summary["nextAttemptAt"]: + summary["nextAttemptAt"] = next_at + try: + summary["acceptedReceipts"] = sum(1 for f in (Path(queue_dir) / "receipts").glob("*.json") if f.is_file()) + except OSError: + pass + try: + summary["terminalEntries"] = sum(1 for f in (Path(queue_dir) / "terminal").glob("*.json") if f.is_file()) + except OSError: + pass + return summary + + def _current_spooled_entry_locked(path: str, queue_dir: str): # A different replay may have finished while our network transaction was in # progress. Its receipt/terminal verdict wins; never recreate stale entries. @@ -634,6 +1082,10 @@ def _current_spooled_entry_locked(path: str, queue_dir: str): except ValueError: pending = None if pending is not None: + # Crash recovery (C11): the other process committed the receipt but may + # have died before publishing journal acceptance. Replay the journal + # receipt from the still-pending payload before releasing its bytes. + _journal_self_publish(queue_dir, pending.get("payload") or {}, receipt.get("response")) _cleanup_managed_artifact_if_unreferenced(queue_dir, pending, excluding_entry_path=pending_path) os.remove(pending_path) return None, ("submitted", _submission_success_message( @@ -665,6 +1117,27 @@ def submit_spooled_path( api_key: str, retries: int, use_token: bool, + cancel_event: Optional[Any] = None, +) -> Tuple[str, str]: + # C11: the whole network transaction must live inside the host publication + # phase; otherwise a collector could start mid-upload through the very disk + # and network path the upload is saturating. In-process re-entry (a + # collector's checkpoint upload, or replay_spool's pass-level hold) is free. + with host_phase_hold("publication"): + return _submit_spooled_path_unheld( + path, queue_dir=queue_dir, base_url=base_url, api_key=api_key, + retries=retries, use_token=use_token, cancel_event=cancel_event) + + +def _submit_spooled_path_unheld( + path: str, + *, + queue_dir: str, + base_url: str, + api_key: str, + retries: int, + use_token: bool, + cancel_event: Optional[Any] = None, ) -> Tuple[str, str]: try: with _spool_write_lock(queue_dir): @@ -696,9 +1169,11 @@ def submit_spooled_path( error: Optional[Exception] = None try: if payload.get("submissionKind") == AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND: - response = submit_artifact_submission(base_url, payload, retries=retries) + response = submit_artifact_submission(base_url, payload, retries=retries, + cancel_event=cancel_event) else: - submit(base_url, payload, api_key=api_key, retries=retries, use_token=use_token) + submit(base_url, payload, api_key=api_key, retries=retries, use_token=use_token, + cancel_event=cancel_event) except Exception as exc: error = exc with _spool_write_lock(queue_dir): @@ -714,6 +1189,10 @@ def submit_spooled_path( _write_json_atomic(os.path.join(queue_dir, "receipts", os.path.basename(path)), {"localHash": entry["localHash"], "uploadedAt": time.time(), "response": response, "status": "uploaded_analysis_pending"}) + # Publish journal acceptance while the pending entry still exists: a crash + # here leaves replayable state (receipt + pending), never a journal that + # lost its artifact without a receipt. + _journal_self_publish(queue_dir, payload, response) _cleanup_managed_artifact_if_unreferenced(queue_dir, entry, excluding_entry_path=path) try: os.remove(path) @@ -733,20 +1212,54 @@ def replay_spool( api_key: str, retries: int, use_token: bool, + limit: int = REPLAY_WINDOW, + time_budget: float = REPLAY_SECONDS, + cancel_event: Optional[Any] = None, +) -> ReplayStats: + """Bounded, cancellable, due-first replay (C08/C11/C12). + + - The whole pass holds the host publication phase atomically: the probe and + every upload are one phase, so a collector starting on ANY queue of this + host cannot slip between them (kernel lock; crash releases). + - Only entries whose Retry-After has arrived (or whose deadline lapsed) enter + the window; a failing prefix consumes slots but never blocks later due work. + - A cancel request stops admission between entries; an in-flight network + transaction keeps its durable entry, so the ambiguous outcome is retried + idempotently rather than lost or double-submitted. + - A measuring collector on this queue defers publication (SpoolCapacityError). + """ + with host_phase_hold("publication"): + return _replay_spool_unheld( + queue_dir, base_url=base_url, api_key=api_key, retries=retries, + use_token=use_token, limit=limit, time_budget=time_budget, + cancel_event=cancel_event) + + +def _replay_spool_unheld( + queue_dir: str, + *, + base_url: str, + api_key: str, + retries: int, + use_token: bool, + limit: int = REPLAY_WINDOW, + time_budget: float = REPLAY_SECONDS, + cancel_event: Optional[Any] = None, ) -> ReplayStats: stats = ReplayStats() - try: - files: List[str] = sorted([ - os.path.join(queue_dir, name) - for name in os.listdir(queue_dir) - if name.endswith(".json") and os.path.isfile(os.path.join(queue_dir, name)) - ]) - except Exception: - return stats - + _refuse_publication_during_measurement(queue_dir) + files, deferred = due_first_queue_paths(queue_dir, limit=limit) + stats.deferred += deferred started = time.monotonic() - for path in files[:25]: - if time.monotonic() - started >= 60: + for index, path in enumerate(files): + if _event_cancelled(cancel_event): + # Reconciled counters (C05): this entry counts once, as cancelled; + # only the entries behind it count as deferred. + stats.cancelled += 1 + stats.deferred += len(files) - index - 1 + break + if time.monotonic() - started >= time_budget: + stats.deferred += len(files) - index break status, _message = submit_spooled_path( path, @@ -755,6 +1268,7 @@ def replay_spool( api_key=api_key, retries=retries, use_token=use_token, + cancel_event=cancel_event, ) if status == "submitted": stats.submitted += 1 @@ -765,3 +1279,12 @@ def replay_spool( elif status == "corrupt": stats.corrupt += 1 return stats + + +def _event_cancelled(cancel_event: Optional[Any]) -> bool: + if cancel_event is None: + return False + try: + return bool(cancel_event.is_set()) + except Exception: + return False diff --git a/client/tests/test_campaign_durability.py b/client/tests/test_campaign_durability.py index b8c45609..5ceaf0f5 100644 --- a/client/tests/test_campaign_durability.py +++ b/client/tests/test_campaign_durability.py @@ -167,6 +167,137 @@ def test_completed_local_campaign_publishes_without_source_or_encoder(tmp_path): encode.assert_not_called() +def test_reconstruction_refuses_changed_runtime_identity(tmp_path): + from client.campaign import journal_path + campaign_id = 'campaign-0123456789abcdef' + root = journal_path(str(tmp_path), campaign_id) + root.mkdir(parents=True) + atomic_json(root / 'manifest.json', { + 'protocolVersion': '7.1', 'runtime': {'ffmpeg': {'sha256': 'old-runtime'}}, + }) + with mock.patch('client.identity.runtime_identity', return_value={'ffmpeg': {'sha256': 'new-runtime'}}): + outcome = main._reconstruct_saved_submissions( + queue_dir=str(tmp_path), campaign_id=campaign_id, max_storage_mb=2048, + ) + assert 'saved runtime identity differs' in outcome['failure'] + assert outcome['reconstructed'] == 0 + + +def test_reconstruction_refuses_changed_client_version(tmp_path): + from client.campaign import journal_path + campaign_id = 'campaign-0123456789abcdef' + root = journal_path(str(tmp_path), campaign_id) + root.mkdir(parents=True) + atomic_json(root / 'manifest.json', { + 'protocolVersion': '7.1', 'clientVersion': 'client/0.0.0', + }) + outcome = main._reconstruct_saved_submissions( + queue_dir=str(tmp_path), campaign_id=campaign_id, max_storage_mb=2048, + ) + assert 'saved client identity differs' in outcome['failure'] + assert outcome['reconstructed'] == 0 + + +def test_publish_saved_rebuilds_envelopes_for_complete_group_after_controlled_stop(tmp_path): + # C09: a controlled stop after a group's measured attempts are durable but + # before submission-*.json envelopes exist must NOT publish zero groups. + # Restarting with the original source unavailable, Publish saved rebuilds + # the exact envelope (same group ID) from the retained journal with zero + # encodes, while the still-extendable group stays unfinished. + import threading + from test_main_routing import MainRoutingTests, _DummyDashboard + fixture = MainRoutingTests() + clip_a = fixture._quick_clip() + clip_b = dataclasses.replace(clip_a, clip_id="film-grain-1080p24-final", + workload_id="film-grain-1080p24-final") + args = fixture._batch_args(str(tmp_path), no_submit=True) + args.local_metrics = False + args.campaign_seed = 41 + args.max_duration_minutes = 1.0 + calls = [] + cancel = threading.Event() + def encode(**kwargs): + calls.append(kwargs['artifact_name']) + if len(calls) == 6: # stop during the first group's SECOND measured attempt + cancel.set() # (its outcome is discarded; five attempts stay journaled) + main.check_measurement_budget() + artifact = Path(kwargs['out_dir']) / kwargs['artifact_name'] + artifact.write_bytes(b'encoded') + return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', + 'fileSizeBytes': 7, 'encodeStartMonotonicNs': 1_000_000_000, + 'encodeEndMonotonicNs': 2_000_000_000, 'elapsedMs': 1000, 'error': None} + hardware = main.HardwareInfo('CPU', None, 16, 'OS') + with mock.patch.object(main, 'detect_hardware', return_value=hardware), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, '_build_protocol_config', + return_value=protocol.ProtocolConfig.for_version('7.1', max_adaptive_repeats=0)), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(main, '_probe_artifact_contract', side_effect=lambda path: fixture._artifact_contract()), \ + mock.patch.object(main, '_capture_protocol_environment_snapshot', + return_value=protocol.EnvironmentSnapshot(selected_accelerator='software')), \ + mock.patch.object(main, 'encode_to_artifact', side_effect=encode), \ + mock.patch.object(main, 'BatchRunDashboard', _DummyDashboard): + # Controlled stop: every attempt up to the pause is journalized, but the + # submit loop refuses before materializing any envelope. + assert main.run_benchmark_batch(hardware=hardware, base_url='https://example.invalid', + args=args, cancel_event=cancel, + tasks=[{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_a}, + {'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_b}]) == 130 + root = next((tmp_path / 'campaigns').iterdir()) + measured = {} + for path in sorted(root.glob('attempt-*.json')): + data = json.loads(path.read_text()) + schedule = data.get('schedule', {}) + if schedule.get('phase') == 'measured': + measured.setdefault(schedule['recipe_id'], []).append(schedule['execution_order']) + complete = [rid for rid, orders in measured.items() if len(orders) >= 2] + partial = [rid for rid, orders in measured.items() if len(orders) == 1] + assert complete and partial, f"fixture must yield one complete and one partial group: {measured}" + assert not list(root.glob('submission-*.json')), "stop happened before envelope creation" + sent = [] + def transport(base_url, submission, **kwargs): + sent.append(submission) + return {'benchmarkRun': {'id': f'run-{len(sent)}'}} + source = mock.Mock(side_effect=AssertionError('original source must never be re-fetched')) + encode_after = mock.Mock(side_effect=AssertionError('publish must never encode')) + with mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, '_prepare_named_suite_clip', source), \ + mock.patch.object(main, 'encode_to_artifact', encode_after), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=transport): + assert main.main(['prog', '--publish-saved', root.name, + '--queue-dir', str(tmp_path), '--base-url', 'https://example.invalid']) == 0 + # Rebuilt payloads carry the identical group identity the live path emits. + assert len(sent) == len(measured[complete[0]]), \ + "the complete group publishes every counted attempt, once" + for submission in sent: + run_create = submission['runCreate'] + assert run_create['measurementGroup']['repetitionGroupId'] == f"{root.name}:{complete[0]}" + assert run_create['measurementGroup']['completed'] is True + source.assert_not_called() + encode_after.assert_not_called() + accepted_orders = {path.stem.replace('submission-', '').replace('.accepted', '') + for path in root.glob('submission-*.accepted.json')} + for order in measured[complete[0]]: + assert f"{order:06d}" in accepted_orders # honest server receipts + for order in measured[partial[0]]: + assert not (root / f'submission-{order:06d}.json').exists() + assert not (root / f'submission-{order:06d}.accepted.json').exists() + assert spool.count_pending_entries(str(tmp_path)) == 0 + # Accepted bytes retired; the unfinished group's bytes remain for resume. + assert (root / 'campaign-complete.json').exists() is False + remaining = [path for path in root.glob('*.mp4') if path.read_bytes() == b'encoded'] + assert len(remaining) == 1, measured + # Second Publish saved is a no-op: accepted receipts authorize, nothing re-uploads. + with mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=AssertionError('replay must not re-upload')): + assert main.main(['prog', '--publish-saved', root.name, + '--queue-dir', str(tmp_path), '--base-url', 'https://example.invalid']) == 0 + def test_cancellation_stops_only_owned_process(): import subprocess import sys diff --git a/client/tests/test_measurement_budget.py b/client/tests/test_measurement_budget.py index d3fa49bb..ca555c7e 100644 --- a/client/tests/test_measurement_budget.py +++ b/client/tests/test_measurement_budget.py @@ -222,3 +222,87 @@ def submit(queue_dir, *, base_url, payload, max_storage_mb, api_key, retries, us assert entry['artifactSha256'] != '' and len(entry['artifactSha256']) == 64 assert not Path(entry['artifactPath']).exists() # all accepted bytes retired assert (root / 'campaign-complete.json').exists() + + +def test_batch_counters_reconcile_with_spool_and_journal(tmp_path): + # C04/C05: the end-screen counters (submitted/queued/failed) and the global + # task_complete progress must reconcile against real durable evidence: + # accepted receipts in the journal, one retryable pending queue entry, and + # one permanently rejected dead-letter - with safe, distinct errorCategory + # values and never a raw server body. + from dataclasses import replace + from test_main_routing import MainRoutingTests, _DummyDashboard + from client import spool + from client.network import SubmitError + fixture = MainRoutingTests() + clip_a = fixture._quick_clip() + clip_b = replace(clip_a, clip_id="film-grain-1080p24-final", workload_id="film-grain-1080p24-final") + args = fixture._batch_args(str(tmp_path), no_submit=False) + args.local_metrics = False + args.campaign_seed = 29 + args.max_duration_minutes = 1.0 + def encode(**kwargs): + artifact = Path(kwargs['out_dir']) / kwargs['artifact_name'] + artifact.write_bytes(b'encoded') + return {'artifactPath': str(artifact), 'encoderUsed': 'libx264', 'presetUsed': 'fast', 'fileSizeBytes': 7, + 'encodeStartMonotonicNs': 1_000_000_000, 'encodeEndMonotonicNs': 2_000_000_000, + 'elapsedMs': 1000, 'error': None} + def transport(base_url, submission, **kwargs): + run_create = submission['runCreate'] + group = str(run_create['repetitionGroupId']) + index = int(run_create['repetitionIndex']) + if 'athletic' in group: # first group: both attempts accepted + return {'benchmarkRun': {'id': 'run-counters-ok'}} + if index == 1: # transient upstream outage: durable queue must defer, not fail + raise SubmitError('submit failed (503)', retryable=True) + raise SubmitError('server rejected the evidence (400)', retryable=False) + events = [] + hardware = main.HardwareInfo('CPU', None, 16, 'OS') + with mock.patch.object(main, 'detect_hardware', return_value=hardware), \ + mock.patch.object(main, 'check_compatibility', return_value={}), \ + mock.patch.object(main, 'fetch_baseline_rows', return_value=[]), \ + mock.patch.object(spool, 'submit_artifact_submission', side_effect=transport), \ + mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ + mock.patch.object(main, '_build_protocol_config', + return_value=protocol.ProtocolConfig.for_version('7.1', max_adaptive_repeats=0)), \ + mock.patch.object(main, 'probe_video_stream_metrics', + return_value={'sourceFps': 24, 'sourceDurationSeconds': 5, 'containerFormat': 'mp4'}), \ + mock.patch.object(main, '_probe_artifact_contract', side_effect=lambda path: fixture._artifact_contract()), \ + mock.patch.object(main, '_capture_protocol_environment_snapshot', + return_value=protocol.EnvironmentSnapshot(selected_accelerator='software')), \ + mock.patch.object(main, 'encode_to_artifact', side_effect=encode), \ + mock.patch.object(main, 'BatchRunDashboard', _DummyDashboard): + # One accepted pair, one 503-deferred upload, one terminal 400. + assert main.run_benchmark_batch(hardware=hardware, base_url='https://example.invalid', args=args, + event_sink=events.append, + tasks=[{'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_a}, + {'encoder': 'libx264', 'preset': 'fast', 'crf': 24, 'suiteClip': clip_b}]) == 1 + counters = next(e for e in reversed(events) if e.get('type') == 'counters') + assert (counters['submitted'], counters['skipped'], counters['queued'], counters['failed']) == (2, 0, 1, 1) + complete = next(e for e in events if e.get('type') == 'run_complete') + assert (complete['submitted'], complete['queued'], complete['failed']) == (2, 1, 1) + assert complete['locallyComplete'] == 0 + # Global progress: every measured unit of work advanced the batch, uploads included. + assert [e['processed'] for e in events if e.get('type') == 'task_complete' and e.get('scope') == 'batch'] == [1, 2, 3, 4] + outcomes = {e['status']: e for e in events if e.get('type') == 'submit_result'} + assert outcomes['queued']['errorCategory'] == 'server_error' + assert outcomes['failed']['errorCategory'] == 'protocol_rejected' + assert 'benchmarkRunId' in outcomes['submitted'] + # Durable truth: journal accepted receipts, queue pending entry, dead-letter. + root = next((tmp_path / 'campaigns').iterdir()) + accepted = [json.loads(p.read_text()) for p in sorted(root.glob('submission-*.accepted.json'))] + assert len(accepted) == 2 + assert all(entry['benchmarkRunId'] == 'run-counters-ok' for entry in accepted) + assert all(not Path(entry['artifactPath']).exists() for entry in accepted) + pending = [json.loads(p.read_text()) for p in tmp_path.glob('*.json') + if p.name != 'queue' and json.loads(p.read_text()).get('payload')] + assert len(pending) == 1 + assert 'film-grain' in pending[0]['payload']['runCreate']['repetitionGroupId'] + assert int(pending[0]['payload']['runCreate']['repetitionIndex']) == 1 + dead = [json.loads(p.read_text()) for p in (tmp_path / 'dead-letter').glob('*.json')] + assert len(dead) == 1 + assert int(dead[0]['payload']['runCreate']['repetitionIndex']) == 2 + receipts = [json.loads(p.read_text()) for p in (tmp_path / 'receipts').glob('*.json')] + assert len(receipts) == 2 + assert {r['status'] for r in receipts} == {'uploaded_analysis_pending'} + assert 'run-counters-ok' in json.dumps(receipts) diff --git a/client/tests/test_preparation.py b/client/tests/test_preparation.py index 822fc084..23fbb88c 100644 --- a/client/tests/test_preparation.py +++ b/client/tests/test_preparation.py @@ -86,7 +86,7 @@ def test_menu_exit_does_not_prepare_sources_or_probe_hardware(self) -> None: with mock.patch.object(main, 'ensure_ffmpeg_and_ffprobe', return_value=(True, 'ffmpeg test')), \ mock.patch.object(main, 'detect_hardware', return_value=hardware), \ mock.patch.object(main, 'list_all_available_encoders', return_value=['libx264']), \ - mock.patch.object(main, 'prompt_choice', return_value=5), \ + mock.patch.object(main, 'prompt_choice', return_value=6), \ mock.patch.object(main.sys, 'stdin', None), \ mock.patch.object(main, '_prepare_quick_suite_clip') as prepare, \ mock.patch.object(main, 'is_hardware_encoder_usable') as probe: diff --git a/client/tests/test_publication_storage.py b/client/tests/test_publication_storage.py index 99521e04..f2ea829f 100644 --- a/client/tests/test_publication_storage.py +++ b/client/tests/test_publication_storage.py @@ -12,6 +12,24 @@ MIB = 1024 * 1024 +def test_saved_publish_never_reports_success_with_unadmitted_envelopes(tmp_path): + campaign_id = 'campaign-0123456789abcdef' + root = tmp_path / 'campaigns' / campaign_id + root.mkdir(parents=True) + atomic_json(root / 'submission-000001.json', {'saved': 'immutable'}) + with mock.patch.object(main, 'drain_committed_receipts', return_value=0), \ + mock.patch.object(main, 'spool_payload', side_effect=spool.SpoolCapacityError('volume full')), \ + mock.patch.object(main, 'replay_spool', return_value=spool.ReplayStats()): + rc, info = main.publish_saved_campaign( + queue_dir=str(tmp_path), campaign_id=campaign_id, + base_url='http://127.0.0.1:9', api_key='', max_storage_mb=2048, + ) + assert rc == 10 + assert info['status'] == 'deferred' + assert info['unadmitted'] == 1 + assert info['pending'] == 0 + + def payload_at(path, size=400 * 1024): from test_spool import SpoolTests path.parent.mkdir(parents=True, exist_ok=True) diff --git a/client/tests/test_spool.py b/client/tests/test_spool.py index ecc60258..bd75d084 100644 --- a/client/tests/test_spool.py +++ b/client/tests/test_spool.py @@ -1,6 +1,7 @@ import json import hashlib import os +import sys import tempfile import threading import unittest @@ -9,7 +10,7 @@ from unittest import mock from client.artifacts import AUTHORITATIVE_ARTIFACT_SUBMISSION_KIND -from client.spool import cleanup_spool, count_pending_entries, inspect_spool, load_spool_entry, replay_spool, spool_payload +from client.spool import cleanup_spool, collector_publication_scope, count_pending_entries, due_first_queue_paths, inspect_spool, load_spool_entry, replay_spool, spool_payload class _SpoolHandler(BaseHTTPRequestHandler): @@ -339,6 +340,236 @@ def test_spool_capacity_charges_own_campaign_and_honors_free_floor(self) -> None with self.assertRaisesRegex(OSError, "safety"): spool_payload(queue_dir, payload, max_storage_mb=2048) + def test_publication_defers_to_a_measuring_collector(self) -> None: + # C11, collector-first order: while measurement.lock is held, both new + # staging and replay must defer (recoverable pause), never upload. + import fcntl + with tempfile.TemporaryDirectory() as queue_dir: + source_path = os.path.join(queue_dir, "artifact.mp4") + with open(source_path, "wb") as handle: + handle.write(b"test") + payload = self._authoritative_payload(source_path) + with open(os.path.join(queue_dir, "measurement.lock"), "a+b") as held: + fcntl.flock(held.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + with self.assertRaisesRegex(OSError, "publication defers"): + spool_payload(queue_dir, payload, max_storage_mb=2048) + with self.assertRaisesRegex(OSError, "publication defers"): + replay_spool(queue_dir, base_url="http://127.0.0.1:1", api_key="", retries=1, use_token=False) + with collector_publication_scope(): + # The collector's own checkpoint uploads legitimately proceed. + path, entry = spool_payload(queue_dir, payload, max_storage_mb=2048) + self.assertTrue(os.path.exists(path)) + self.assertFalse(entry.get("terminal")) + + def test_collector_entry_refuses_while_publisher_owns_queue(self) -> None: + # C11 reverse order: a publisher holding publication.lock blocks the next + # collection; releasing it lets the collector proceed. + from client import main as client_main + import fcntl + with tempfile.TemporaryDirectory() as queue_dir: + with open(os.path.join(queue_dir, "publication.lock"), "a+b") as held: + fcntl.flock(held.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB) + events = [] + rc = client_main.active_collection_guard(queue_dir, lambda event: events.append(event)) + self.assertEqual(rc, 6) + self.assertTrue(any("publication pass" in str(event.get("message")) for event in events), + "the refusal must name the publisher, not a phantom collector") + self.assertIsNone(client_main.active_collection_guard(queue_dir)) + + def test_due_first_window_skips_delayed_and_stays_fair(self) -> None: + # C08: delayed (Retry-After in the future) entries never enter the + # window; due entries are admitted oldest-scheduled first and the + # window stays bounded when more than `limit` are due. + import time as _time + with tempfile.TemporaryDirectory() as queue_dir: + for index in range(25): + path1, entry1 = spool_payload(queue_dir, {"cpuModel": f"delayed-{index}", "fps": 1}) + entry1["nextAttemptAt"] = _time.time() + 3600 + entry1["retryDeadlineAt"] = _time.time() + 86400 + with open(path1, "w") as handle: + json.dump(entry1, handle) + due_paths = [] + for index in range(3): + path2, entry2 = spool_payload(queue_dir, {"cpuModel": f"due-{index}", "fps": 1}) + entry2["nextAttemptAt"] = _time.time() - (10 - index) # due, staggered + with open(path2, "w") as handle: + json.dump(entry2, handle) + due_paths.append(path2) + selected, deferred = due_first_queue_paths(queue_dir, limit=25) + self.assertEqual(sorted(selected), sorted(due_paths), + "only due entries may be attempted; delayed must wait") + self.assertEqual(deferred, 25) + # Oldest-scheduled first (fairness) + self.assertEqual([os.path.basename(p) for p in selected], + [os.path.basename(p) for p in sorted(due_paths, key=lambda p: load_spool_entry(p)["nextAttemptAt"])]) + # Bound: with limit=2 the two earliest win; the third is deferred, not dropped silently + capped, deferred2 = due_first_queue_paths(queue_dir, limit=2) + self.assertEqual(len(capped), 2) + self.assertEqual(deferred2, 26) + + def test_replay_cancellation_is_bounded_and_keeps_entries_durable(self) -> None: + # C12: a cancel request stops admission between entries; every entry + # keeps its durable file so the next pass retries idempotently, and a + # pre-set cancel performs zero network attempts. + with tempfile.TemporaryDirectory() as queue_dir: + for index in range(3): + spool_payload(queue_dir, {"cpuModel": f"p{index}", "fps": 1}) + cancelled = type("AlwaysCancelled", (), {"is_set": lambda self: True})() + stats = replay_spool(queue_dir, base_url="http://127.0.0.1:1", api_key="", + retries=1, use_token=False, cancel_event=cancelled) + self.assertEqual(stats.submitted, 0) + self.assertEqual(stats.cancelled, 1) + self.assertEqual(stats.deferred, 2) + self.assertEqual(count_pending_entries(queue_dir), 3, + "cancellation must never lose or duplicate durable work") + + def test_lost_response_stays_durable_and_retries_idempotently(self) -> None: + # C12: a dropped/ambiguous network outcome must keep the exact pending + # entry (same localHash identity) and the next pass completes it once. + _SpoolHandler.mode = "ok" + server, thread, base_url = self._start_server() + try: + with tempfile.TemporaryDirectory() as queue_dir: + path, entry = spool_payload(queue_dir, {"cpuModel": "ambiguous", "fps": 1}) + hash_before = entry["localHash"] + from client.spool import submit as real_submit + calls = {"n": 0} + def flaky(*args, **kwargs): + calls["n"] += 1 + if calls["n"] == 1: + raise ConnectionError("response lost mid-transaction") + return real_submit(*args, **kwargs) + with mock.patch("client.spool.submit", side_effect=flaky): + stats = replay_spool(queue_dir, base_url=base_url, api_key="", retries=1, use_token=False) + self.assertEqual((stats.submitted, stats.retained), (0, 1)) + retained = load_spool_entry(path) + self.assertEqual(retained["localHash"], hash_before, + "ambiguous outcome must not replace retry identity") + retained["nextAttemptAt"] = 0 # due now (test clock shortcut) + with open(path, "w") as handle: + json.dump(retained, handle) + stats2 = replay_spool(queue_dir, base_url=base_url, api_key="", retries=1, use_token=False) + self.assertEqual(stats2.submitted, 1) + self.assertEqual(count_pending_entries(queue_dir), 0) + finally: + server.shutdown() + thread.join(timeout=5) + + +class HostPhaseExclusionTests(unittest.TestCase): + """C11: kernel-backed host phase exclusion across distinct queue dirs.""" + + def _phase_dir_like(self, tmp: str) -> str: + root = os.path.join(tmp, "host-phase") + os.makedirs(root, exist_ok=True) + return root + + + def test_collector_first_defers_publisher_on_other_queue(self) -> None: + # The publisher runs on its own thread: same-thread re-entry is the + # collector's own checkpoint-upload path and legitimately bypasses the + # probe; a *different* publisher (thread/process) must be refused. + from client.spool import SpoolCapacityError, host_phase_hold + errors: list = [] + queue_b: list = [] + def publish(): + try: + spool_payload(queue_b[0], {"cpuModel": "other-queue", "fps": 1}) + except BaseException as exc: # noqa: BLE001 - recorded for assertion + errors.append(exc) + with tempfile.TemporaryDirectory() as tmp, \ + mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": self._phase_dir_like(tmp)}): + queue_b.append(os.path.join(tmp, "qb")) + collector = host_phase_hold("measurement") + collector.__enter__() + try: + thread = threading.Thread(target=publish) + thread.start() + thread.join(5) + self.assertFalse(thread.is_alive()) + self.assertEqual(len(errors), 1) + self.assertIsInstance(errors[0], SpoolCapacityError) + self.assertIn("measuring on this host", str(errors[0])) + finally: + collector.__exit__(None, None, None) + queue_b[0] = os.path.join(tmp, "qb2") + with mock.patch("client.spool.submit_artifact_submission", return_value={}): + spool_payload(queue_b[0], {"cpuModel": "after-release", "fps": 1}) + + def test_publisher_first_defers_collector_and_allows_second_publisher(self) -> None: + from client.spool import SpoolCapacityError, host_phase_hold + errors: list = [] + with tempfile.TemporaryDirectory() as tmp, \ + mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": self._phase_dir_like(tmp)}): + publisher = host_phase_hold("publication") + publisher.__enter__() + def collector_start(): + try: + hold = host_phase_hold("measurement") + hold.__enter__() + except BaseException as exc: # noqa: BLE001 - recorded for assertion + errors.append(exc) + return + errors.append(None) + hold.__exit__(None, None, None) + try: + thread = threading.Thread(target=collector_start) + thread.start() + thread.join(5) + self.assertFalse(thread.is_alive()) + self.assertEqual(len(errors), 1) + self.assertIsInstance(errors[0], SpoolCapacityError) + self.assertIn("owns this host right now", str(errors[0])) + # A second publisher coexists: idempotent uploads are not a + # timing hazard and must not deadlock each other. + with host_phase_hold("publication"): + pass + finally: + publisher.__exit__(None, None, None) + + def test_crashed_owner_releases_host_phase_for_reopen(self) -> None: + import signal + import subprocess + from client.spool import host_phase_busy, host_phase_hold + with tempfile.TemporaryDirectory() as tmp: + phase_dir = self._phase_dir_like(tmp) + env = dict(os.environ, ENCODINGDB_HOST_PHASE_DIR=phase_dir) + script = ( + "import os, fcntl, time\n" + "path = os.path.join(os.environ['ENCODINGDB_HOST_PHASE_DIR'], 'phase.lock')\n" + "handle = open(path, 'a+b')\n" + "fcntl.flock(handle.fileno(), fcntl.LOCK_EX)\n" + "print('HELD', flush=True)\n" + "time.sleep(30)\n") + proc = subprocess.Popen([sys.executable, "-c", script], env=env, + stdout=subprocess.PIPE, text=True) + try: + self.assertEqual(proc.stdout.readline().strip(), "HELD") + with mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": phase_dir}): + self.assertTrue(host_phase_busy()) + os.kill(proc.pid, signal.SIGKILL) + proc.wait(5) + self.assertFalse(host_phase_busy(), + "kernel must release a crashed owner's phase lock") + with host_phase_hold("measurement") as acquired: + self.assertTrue(acquired) + finally: + if proc.poll() is None: + proc.kill() + proc.wait(5) + + def test_uninspectable_host_phase_lock_fails_closed(self) -> None: + from client.spool import SpoolCapacityError, host_phase_busy + with tempfile.TemporaryDirectory() as tmp: + phase_dir = self._phase_dir_like(tmp) + os.chmod(phase_dir, 0o000) + try: + with mock.patch.dict(os.environ, {"ENCODINGDB_HOST_PHASE_DIR": phase_dir}): + with self.assertRaises(SpoolCapacityError): + host_phase_busy() + finally: + os.chmod(phase_dir, 0o755) + if __name__ == "__main__": unittest.main() diff --git a/client/tests/test_windows_gui.py b/client/tests/test_windows_gui.py index 5178a33b..5a33a945 100644 --- a/client/tests/test_windows_gui.py +++ b/client/tests/test_windows_gui.py @@ -37,12 +37,24 @@ def set(self, value): self.value = value +def cached_estimate(_mode): + return {"strategy": "cache", "storageOk": True, "bytesToTransfer": 0, + "peakStorageBytes": 0, "warnings": []} + + +def cached_recovery_state(_queue=None): + return {"publicationConsent": False, "campaigns": [], + "publication": {"pendingEntries": 0, "dueEntries": 0, + "acceptedReceipts": 0, "terminalEntries": 0}} + + class Widget: def __init__(self, *args, **kwargs): self.variable = kwargs.get("textvariable") self.values = kwargs.get("values", []) self.index = -1 self.options = kwargs + self.visible = False def __setitem__(self, key, value): if key == "values": @@ -61,7 +73,10 @@ def configure(self, **kwargs): self.options.update(kwargs) def pack(self, **kwargs): - pass + self.visible = True + + def pack_forget(self): + self.visible = False def grid(self, **kwargs): pass @@ -113,9 +128,12 @@ def test_initialized_controls_and_keyboard_handlers_use_real_running_guards(self with mock.patch.dict(sys.modules, {"tkinter": tk}), mock.patch.object(gui.os, "name", "nt"), \ mock.patch.object(gui, "list_all_available_encoders", return_value=["libx265", "libx264"]), \ mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state), \ mock.patch.object(gui.threading, "Thread") as thread: self.assertEqual(gui.launch_windows_gui(self.args()), 0) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state self.assertEqual((app._selected_encoder(), app._selected_preset(), app.crf_var.get()), ("libx264", "fast", 0)) app._handle_event({"type": "preparation_progress", "stage": "probe", "path": "test.mkv"}) self.assertEqual(app.stage_var.get(), "Preparing: probe") @@ -142,7 +160,7 @@ def test_initialized_controls_and_keyboard_handlers_use_real_running_guards(self self.assertEqual(app.start_btn.options["state"], "disabled") bindings[""](None) self.assertTrue(app.cancel_event.is_set()) - self.assertEqual(app.summary_var.get(), "Stopping owned work; retaining downloads and campaign...") + self.assertEqual(app.summary_var.get(), "Stopping owned work; retaining saved results...") def _build_app(self, mode: str = "Small"): """Instantiate the Tk app against the mocked widget harness.""" @@ -160,9 +178,12 @@ def _build_app(self, mode: str = "Small"): mock.patch.object(gui, "list_all_available_encoders", return_value=["libx264", "h264_videotoolbox"]), \ mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state), \ mock.patch.object(gui.threading, "Thread"): gui.launch_windows_gui(self.args()) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state app.mode_var.set(mode) return app @@ -172,11 +193,13 @@ def test_mode_choices_expose_shared_sweeps_with_single_as_advanced(self): self.assertEqual(app.mode_var.get(), "Small") self.assertIn("native recipes", app.summary_var.get()) self.assertIn("quick clip", app.summary_var.get()) + self.assertFalse(app.advanced_frame.visible) app.mode_var.set("Full") app._update_single_fields_state() self.assertIn("all seven frozen clips", app.summary_var.get()) app.mode_var.set("Single (advanced)") app._update_single_fields_state() + self.assertTrue(app.advanced_frame.visible) self.assertEqual(app.encoder_combo.options["state"], "readonly") def test_sweep_worker_dispatches_to_shared_planner_run(self): @@ -208,8 +231,9 @@ def test_upload_retry_never_encodes(self): from client import main as client_main with mock.patch.object(client_main, "count_pending_entries", side_effect=[2, 0]), \ - mock.patch.object(client_main, "replay_spool", - return_value=mock.Mock(dead_lettered=0, corrupt=0)) as replay_mock, \ + mock.patch.object(client_main, "retry_due_uploads", + return_value=(0, {"status": "published", "pending": 0, + "deadLettered": 0, "corrupt": 0})) as replay_mock, \ mock.patch.object(client_main, "run_sweep_mode") as sweep_mock, \ mock.patch.object(client_main, "run_with_args") as single_mock: app._retry_uploads_worker("queue-dir", "https://example.invalid", "", 2) @@ -266,11 +290,14 @@ def build(self, mode: str = "Small"): setattr(tk.ttk, name, Widget) tk.scrolledtext.ScrolledText = Widget with mock.patch.dict(sys.modules, {"tkinter": tk}), mock.patch.object(gui.os, "name", "nt"), \ - mock.patch.object(gui, "list_all_available_encoders", + mock.patch.object(gui, "list_all_available_encoders", return_value=["libx264", "h264_videotoolbox"]), \ - mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)): + mock.patch.object(gui, "desktop_work_area", return_value=(0, 0, 1024, 720)), \ + mock.patch.object(gui.client_main, "recovery_state", side_effect=cached_recovery_state): gui.launch_windows_gui(self.args()) app = bindings[""].__self__ + app._estimate_acquisition = cached_estimate + app._load_recovery_state = cached_recovery_state app.mode_var.set(mode) return app, root, bindings, tk @@ -280,7 +307,7 @@ def states(self, app): def test_idle_states_then_active_benchmark_locks_retry_and_start(self): app, _root, bindings, _tk = self.build() - self.assertEqual(self.states(app), ("normal", "disabled", "normal")) + self.assertEqual(self.states(app), ("normal", "disabled", "disabled")) with mock.patch.object(gui.threading, "Thread", FakeThread): bindings[""](None) self.assertEqual(app.worker_thread.target.__name__, "_run_worker") @@ -293,27 +320,258 @@ def test_idle_states_then_active_benchmark_locks_retry_and_start(self): app.event_queue.put(("done", 0)) app._poll_events() self.assertFalse(app.running) - self.assertEqual(self.states(app), ("normal", "disabled", "normal"), - "Retry must be available again after the run finishes") + self.assertEqual(self.states(app), ("normal", "disabled", "disabled"), + "Local-only policy keeps upload actions disabled after the run") def test_finished_run_surfaces_pending_uploads(self): app, _root, _bindings, _tk = self.build() app.no_submit_var.set(False) + app._active_no_submit = False + app.event_queue.put(("event", {"type": "submit_result", "status": "submitted"})) with mock.patch.object(gui.client_main, "count_pending_entries", return_value=2): app.event_queue.put(("done", 0)) app._poll_events() self.assertIn("2 upload(s) queued", app.summary_var.get()) - self.assertIn("Uploaded; analysis pending", app.summary_var.get()) + self.assertIn("some uploads are queued", app.summary_var.get()) self.assertEqual(app.upload_btn.options["state"], "normal") + def test_batch_progress_counts_groups_not_maximum_encode_attempts(self): + app, _root, _bindings, _tk = self.build() + app._handle_event({"type": "run_start", "scope": "batch", "totalTasks": 30, + "totalBatches": 1}) + app._handle_event({"type": "batch_start", "batchSize": 3, + "totalBatches": 1, "processedTotal": 0}) + app._handle_event({"type": "task_complete", "scope": "batch", "processed": 1, + "total": 30}) + self.assertEqual(app.overall_total, 3) + self.assertEqual(app.overall_pb.options["maximum"], 3) + self.assertEqual(app.overall_pb.options["value"], 1) + self.assertIn("1/3", app.summary_var.get()) + + def test_invalid_typed_settings_leave_idle_without_worker(self): + for field, value, mode in ( + ("retries_var", "abc", "Small"), ("retries_var", "", "Small"), + ("retries_var", "11", "Small"), ("batch_size_var", " ", "Small"), + ("batch_size_var", "65", "Small"), ("crf_var", "oops", "Single (advanced)"), + ("bitrate_var", "-3", "Single (advanced)"), + ): + with self.subTest(field=field, value=value): + app, _root, _bindings, tk = self.build(mode=mode) + getattr(app, field).set(value) + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.start_btn.options["state"], "normal") + tk.messagebox.showerror.assert_called_once() + + def test_real_tcl_invalid_intvar_becomes_field_error(self): + try: + import tkinter as tkinter_real + interpreter = tkinter_real.Tcl() + except Exception as exc: + self.skipTest(f"Tcl unavailable: {exc}") + value = tkinter_real.IntVar(master=interpreter, value=3) + interpreter.setvar(value._name, "abc") + with self.assertRaises(tkinter_real.TclError): + value.get() + with self.assertRaisesRegex(ValueError, "Retries must be a whole number"): + gui._validated_integer(value, "Retries", 1, 10) + + def test_local_only_run_does_not_require_a_live_server_url(self): + app, _root, _bindings, tk = self.build() + app.base_url_var.set("unconfigured") + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertTrue(app.running) + self.assertTrue(app.worker_thread.args[0].no_submit) + tk.messagebox.showerror.assert_not_called() + + def test_download_cost_is_confirmed_before_worker_starts(self): + app, _root, _bindings, tk = self.build() + app._estimate_acquisition = lambda _mode: { + "strategy": "pack", "storageOk": True, + "bytesToTransfer": 1506890018, "peakStorageBytes": 3100000000, + } + tk.messagebox.askyesno.return_value = False + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertIn("1.4 GB", tk.messagebox.askyesno.call_args.args[1]) + self.assertIn("declined", app.summary_var.get()) + + def test_storage_estimate_refuses_before_worker(self): + app, _root, _bindings, tk = self.build() + app._estimate_acquisition = lambda _mode: { + "strategy": "clip", "storageOk": False, + "bytesToTransfer": 100, "peakStorageBytes": 5000000000, + } + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertIn("disk space", app.summary_var.get()) + tk.messagebox.showerror.assert_called_once() + + def test_acquisition_preview_uses_quick_or_complete_frozen_clip_set(self): + manifest = gui.suite.load_default_suite_manifest() + with mock.patch.object(gui.suite, "acquisition_estimate", return_value={}) as estimate: + gui._acquisition_preview("small") + quick_ids = estimate.call_args.args[0] + gui._acquisition_preview("full") + full_ids = estimate.call_args.args[0] + self.assertEqual(quick_ids, [gui.suite.get_default_quick_clip(manifest).clip_id]) + self.assertEqual(set(full_ids), {clip.clip_id for clip in manifest.clips}) + + def test_consent_save_and_thread_start_failures_restore_idle(self): + app, _root, _bindings, tk = self.build() + app.no_submit_var.set(False) + with mock.patch.object(gui.client_main, "_ensure_interactive_publication_consent", + side_effect=OSError("consent storage unavailable")): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.no_submit_var.get(), False) + tk.messagebox.showerror.assert_called_once() + + app.no_submit_var.set(True) + class FailingThread(FakeThread): + def start(self): + raise RuntimeError("thread unavailable") + with mock.patch.object(gui.threading, "Thread", FailingThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.stage_var.get(), "Idle") + self.assertEqual(app.start_btn.options["state"], "normal") + + app.base_args.max_duration_minutes_explicit = True + app.base_args.max_duration_minutes = "invalid" + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertFalse(app.running) + self.assertIsNone(app.worker_thread) + self.assertEqual(app.start_btn.options["state"], "normal") + + def test_single_worker_uses_snapshot_even_if_widgets_change(self): + app, _root, _bindings, _tk = self.build(mode="Single (advanced)") + app._update_single_fields_state() + app.crf_var.set(19) + with mock.patch.object(gui.threading, "Thread", FakeThread): + app._start_run() + self.assertEqual(app.no_submit_check.options["state"], "disabled") + app.crf_var.set("broken later") + app.no_submit_var.set(False) + with mock.patch.object(gui.client_main, "run_with_args", return_value=0) as run: + app.worker_thread.target(*app.worker_thread.args) + args = run.call_args.args[0] + self.assertEqual(args.crf, 19) + self.assertTrue(args.no_submit) + app._poll_events() + self.assertIn("Saved locally", app.summary_var.get()) + + def test_local_completion_reports_saved_groups_without_upload_claim(self): + app, _root, _bindings, _tk = self.build() + app._active_no_submit = True + app.event_queue.put(("event", {"type": "submit_result", "status": "locally_complete"})) + app.event_queue.put(("event", {"type": "submit_result", "status": "locally_complete"})) + app.event_queue.put(("done", 0)) + app._poll_events() + self.assertIn("2 measurement group(s) locally", app.summary_var.get()) + self.assertNotIn("Uploaded", app.summary_var.get()) + + def test_saved_work_is_visible_and_publish_uses_zero_encode_api(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["publicationConsent"] = True + state["publication"] = {"pendingEntries": 2, "dueEntries": 1, + "acceptedReceipts": 3, "terminalEntries": 1} + state["campaigns"] = [{"campaignId": "campaign-test", "complete": True, + "pendingUploads": 2, "unavailableSources": 0, + "actions": [{"action": "publish_saved"}]}] + app._load_recovery_state = lambda: state + app._refresh_saved_work() + app.no_submit_var.set(False) + app._refresh_controls() + self.assertIn("1 due / 1 delayed", app.saved_summary_var.get()) + self.assertIn("terminal", app.saved_summary_var.get()) + self.assertEqual(app._selected_saved_campaign(), "campaign-test") + self.assertEqual(app.publish_btn.options["state"], "normal") + with mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ + mock.patch.object(gui.threading, "Thread", FakeThread): + app._publish_saved() + self.assertEqual(app.publish_btn.options["state"], "disabled") + with mock.patch.object(gui.client_main, "publish_saved_campaign", + return_value=(0, {"submitted": 2, "pending": 0})) as publish, \ + mock.patch.object(gui.client_main, "run_with_args") as encode: + app.upload_thread.target(*app.upload_thread.args) + publish.assert_called_once() + self.assertEqual(publish.call_args.kwargs["campaign_id"], "campaign-test") + encode.assert_not_called() + self.assertIn("analysis pending", app.event_queue.get_nowait()[1]) + self.assertIn("1 not yet staged", app._publication_result_text( + 10, {"pending": 0, "unadmitted": 1, "deferredReason": "storage_or_exclusion"})) + + def test_resume_selected_campaign_keeps_its_identity(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["campaigns"] = [{"campaignId": "campaign-saved", "complete": False, + "pendingUploads": 0, "unavailableSources": 0, + "actions": [{"action": "resume"}]}] + app._load_recovery_state = lambda: state + app._refresh_saved_work() + with mock.patch.object(app, "_start_run") as start: + app._resume_saved() + start.assert_called_once_with(resume_id="campaign-saved") + args = self.args(resume_campaign="campaign-saved") + with mock.patch.object(gui.client_main, "_resume_campaign", return_value=0) as resume, \ + mock.patch.object(gui.client_main, "run_with_args") as new_run: + app._run_worker(args, "Small", "small") + resume.assert_called_once() + self.assertEqual(resume.call_args.args[0].resume_campaign, "campaign-saved") + new_run.assert_not_called() + + def test_idle_retry_requires_due_work_and_saved_consent(self): + app, _root, _bindings, _tk = self.build() + state = cached_recovery_state() + state["publicationConsent"] = True + state["publication"]["pendingEntries"] = 2 + state["publication"]["dueEntries"] = 1 + app._load_recovery_state = lambda: state + app.no_submit_var.set(False) + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_called_once_with(automatic=True) + state["publicationConsent"] = False + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_not_called() + state["publicationConsent"] = True + app.no_submit_var.set(True) + with mock.patch.object(app, "_retry_uploads") as retry: + app._idle_retry() + retry.assert_not_called() + + def test_stop_cancels_active_upload_replay(self): + app, _root, _bindings, _tk = self.build() + app.upload_thread = FakeThread() + app.upload_thread.start() + app._refresh_controls() + self.assertEqual(app.stop_btn.options["state"], "normal") + app._stop_run() + self.assertTrue(app.upload_cancel_event.is_set()) + def test_active_replay_blocks_benchmark_start_and_restores_on_status(self): app, _root, _bindings, _tk = self.build() + app.no_submit_var.set(False) with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ mock.patch.object(gui.client_main, "count_pending_entries", return_value=3), \ mock.patch.object(gui.client_main, "replay_spool"): app._retry_uploads() self.assertIsNotNone(app.upload_thread) - self.assertEqual(self.states(app), ("disabled", "disabled", "disabled")) + self.assertEqual(self.states(app), ("disabled", "normal", "disabled")) app._start_run() self.assertIsNone(app.worker_thread, "benchmark must not race a live replay") # A second click while replaying must not spawn another uploader. @@ -331,13 +589,16 @@ def test_active_replay_blocks_benchmark_start_and_restores_on_status(self): def test_close_during_replay_waits_then_closes_bounded(self): app, root, _bindings, tk = self.build() + app.no_submit_var.set(False) tk.messagebox.askyesno.return_value = True with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True), \ mock.patch.object(gui.client_main, "count_pending_entries", return_value=1), \ mock.patch.object(gui.client_main, "replay_spool"): app._retry_uploads() uploader = app.upload_thread app._on_close() + self.assertTrue(app.upload_cancel_event.is_set()) root.destroy.assert_not_called() app._close_when_stopped() root.destroy.assert_not_called() @@ -360,6 +621,7 @@ def test_close_during_replay_waits_then_closes_bounded(self): def test_retry_exception_surfaces_and_restores_controls(self): app, _root, _bindings, _tk = self.build() + app.no_submit_var.set(False) with mock.patch.object(gui.client_main, "count_pending_entries", side_effect=OSError("disk gone")): app._retry_uploads_worker("queue-dir", "http://127.0.0.1:9", "", 2) app.upload_thread = FakeThread() @@ -370,7 +632,9 @@ def test_retry_exception_surfaces_and_restores_controls(self): def test_upload_status_does_not_release_a_still_running_worker(self): app, _root, _bindings, _tk = self.build() - with mock.patch.object(gui.threading, "Thread", FakeThread): + app.no_submit_var.set(False) + with mock.patch.object(gui.threading, "Thread", FakeThread), \ + mock.patch.object(app, "_confirm_publication_consent", return_value=True): app._retry_uploads() uploader = app.upload_thread app.event_queue.put(("upload_status", "Upload complete")) @@ -398,6 +662,21 @@ def test_submit_result_events_get_browse_link_not_fake_run_url(self): self.assertNotIn("browse", lines[1], "browse hint appears once per run") self.assertNotIn("browse", lines[2], "queued is not a browsable success claim") + def test_submission_failure_shows_safe_cause_and_run_id(self): + app, _root, _bindings, _tk = self.build() + app._active_no_submit = False + app._handle_event({"type": "submit_result", "status": "failed", + "errorCategory": "protocol_rejected", "safeReason": "Suite identity differs", + "recoveryAction": "Update the client", "reasonCodes": ["SUITE_MISMATCH"], + "benchmarkRunId": "run-123", "error": "token=SECRET"}) + self.assertIn("Suite identity differs", app.summary_var.get()) + self.assertIn("Update the client", app.summary_var.get()) + self.assertIn("run-123", app.summary_var.get()) + self.assertNotIn("SECRET", app.summary_var.get()) + app.event_queue.put(("done", 1)) + app._poll_events() + self.assertIn("Suite identity differs", app.summary_var.get()) + def test_explicit_allowance_reported_when_set(self): app, _root, bindings, _tk = self.build() log_lines = [] diff --git a/client/windows_gui.py b/client/windows_gui.py index 2a2c431a..280528e1 100644 --- a/client/windows_gui.py +++ b/client/windows_gui.py @@ -1,6 +1,7 @@ import argparse import os import queue +import re import threading import time import traceback @@ -8,6 +9,7 @@ from . import main as client_main from . import sweep_plan +from . import suite from .encoders import ( enumerate_supported_presets_for_encoder, get_encoder_friendly_label, @@ -32,6 +34,59 @@ GUI_CLOSE_GRACE_SECONDS = 70.0 +def _validated_integer(value: Any, label: str, minimum: int, maximum: int) -> int: + """Read a Tk variable without leaving the window running on TclError.""" + try: + raw = str(value.get()).strip() + if not re.fullmatch(r"[0-9]+", raw): + raise ValueError + number = int(raw) + except Exception as exc: + raise ValueError(f"{label} must be a whole number from {minimum} to {maximum}.") from exc + if not minimum <= number <= maximum: + raise ValueError(f"{label} must be a whole number from {minimum} to {maximum}.") + return number + + +def _submission_line(event: Dict[str, Any]) -> str: + def display(value: Any, limit: int) -> str: + text = " ".join(str(value or "").split()) + text = re.sub(r"(?i)\b(bearer)\s+\S+", r"\1 [redacted]", text) + text = re.sub(r"(?i)\b(token|api[_-]?key|secret|authorization)\s*[:=]\s*\S+", + r"\1=[redacted]", text) + return text[:limit] + + status = display(event.get("status") or "unknown", 24) + category = display(event.get("errorCategory"), 40) + reason = display(event.get("safeReason"), 240) + action = display(event.get("recoveryAction"), 160) + codes = event.get("reasonCodes") + if isinstance(codes, (list, tuple)): + valid_codes = [str(code) for code in codes if re.fullmatch(r"[A-Za-z0-9_-]{1,48}", str(code))] + else: + valid_codes = [] + run_id = str(event.get("benchmarkRunId") or "") + if not re.fullmatch(r"[A-Za-z0-9_-]{1,80}", run_id): + run_id = "" + details = [item for item in (category, reason, ", ".join(valid_codes[:4]), action) if item] + label = display(event.get("preset") or event.get("codec") or event.get("campaignId"), 80) + line = f"Submission {status}" + (f" ({label})" if label else "") + if details: + line += ": " + "; ".join(details) + if run_id: + line += f" [run {run_id}]" + return line + + +def _acquisition_preview(mode_key: Optional[str]) -> Dict[str, Any]: + manifest = suite.load_default_suite_manifest() + if mode_key in (None, "small"): + clip_ids = [suite.get_default_quick_clip(manifest).clip_id] + else: + clip_ids = [clip.clip_id for clip in manifest.clips] + return suite.acquisition_estimate(clip_ids, manifest=manifest) + + def plan_summary_text(mode: str, encoders: list[str], presets_cfg: dict[str, Any]) -> str: """One-line honest preview of a sweep plan: finite work counts, no wall-clock claims.""" plan = sweep_plan.plan_sweep(mode, encoders, presets_cfg=presets_cfg) @@ -136,15 +191,24 @@ def __init__(self) -> None: self.event_queue: queue.Queue = queue.Queue() self.worker_thread: Optional[threading.Thread] = None self.upload_thread: Optional[threading.Thread] = None + self.upload_cancel_event = threading.Event() self.cancel_event = threading.Event() self.running = False + self._active_no_submit: Optional[bool] = None + self._last_submission_failure = "" + self._run_counts = {"submitted": 0, "locally_complete": 0, "queued": 0, "failed": 0} self._browse_shown = False self._close_deadline = 0.0 + self._estimate_acquisition = _acquisition_preview + self._load_recovery_state = lambda: client_main.recovery_state(str(self.base_args.queue_dir)) + self.saved_campaign_ids: list[str] = [] + self.saved_state: Dict[str, Any] = {} # Causal message from the most recent run_error/unhandled failure in this # run; the done handler must not replace it with a bare exit code. self.last_failure: Optional[str] = None self.mode_var = tk.StringVar(value="Small") + self.advanced_var = tk.BooleanVar(value=False) self.no_submit_var = tk.BooleanVar(value=bool(getattr(base_args, "no_submit", False))) self.base_url_var = tk.StringVar(value=str(getattr(base_args, "base_url", ""))) self.retries_var = tk.IntVar(value=max(1, int(getattr(base_args, "retries", 3)))) @@ -159,7 +223,9 @@ def __init__(self) -> None: self.current_var = tk.StringVar(value="-") self.summary_var = tk.StringVar(value="Ready") self.telemetry_var = tk.StringVar(value="-") - self.counter_var = tk.StringVar(value="ok=0 skip=0 queue=0 fail=0") + self.counter_var = tk.StringVar(value="local=0 uploaded=0 queued=0 failed=0") + self.saved_summary_var = tk.StringVar(value="Checking saved work...") + self.selected_saved_var = tk.StringVar(value="") self.overall_total = 1 self.overall_done = 0 @@ -170,10 +236,13 @@ def __init__(self) -> None: self.preset_values = [] self._build_ui(ttk, tk, scrolledtext) + self._toggle_advanced() self._refresh_encoders() self._update_single_fields_state() + self._refresh_saved_work() self._refresh_controls() self._poll_events() + self.root.after(30_000, self._idle_retry) self.root.protocol("WM_DELETE_WINDOW", self._on_close) self.root.bind("", self._start_shortcut) self.root.bind("", self._stop_shortcut) @@ -198,21 +267,30 @@ def _build_ui(self, ttk: Any, tk: Any, scrolledtext: Any) -> None: self.mode_combo.pack(side="left", padx=(8, 16)) self.mode_combo.bind("<>", lambda _evt: self._update_single_fields_state()) - ttk.Checkbutton(row1, text="No submit (local dry run only)", variable=self.no_submit_var).pack(side="left", padx=(0, 12)) - ttk.Label(row1, text="Retries").pack(side="left") - self.retries_spin = ttk.Spinbox(row1, from_=1, to=10, textvariable=self.retries_var, width=6) + self.no_submit_check = ttk.Checkbutton(row1, text="Save locally; publish later", + variable=self.no_submit_var, command=self._refresh_controls) + self.no_submit_check.pack(side="left", padx=(0, 12)) + + self.advanced_toggle = ttk.Checkbutton(config_frame, text="Advanced settings", variable=self.advanced_var, + command=self._toggle_advanced) + self.advanced_toggle.pack(anchor="w") + self.advanced_frame = ttk.LabelFrame(config_frame, text="Advanced settings", padding=10) + advanced_row = ttk.Frame(self.advanced_frame) + advanced_row.pack(fill="x", pady=(0, 8)) + ttk.Label(advanced_row, text="Retries").pack(side="left") + self.retries_spin = ttk.Spinbox(advanced_row, from_=1, to=10, textvariable=self.retries_var, width=6) self.retries_spin.pack(side="left", padx=(6, 12)) - ttk.Label(row1, text="Batch size").pack(side="left") - self.batch_spin = ttk.Spinbox(row1, from_=0, to=64, textvariable=self.batch_size_var, width=6) + ttk.Label(advanced_row, text="Batch size").pack(side="left") + self.batch_spin = ttk.Spinbox(advanced_row, from_=0, to=64, textvariable=self.batch_size_var, width=6) self.batch_spin.pack(side="left", padx=(6, 0)) - row2 = ttk.Frame(config_frame) + row2 = ttk.Frame(self.advanced_frame) row2.pack(fill="x", pady=(0, 8)) ttk.Label(row2, text="Base URL").pack(side="left") self.base_url_entry = ttk.Entry(row2, textvariable=self.base_url_var) self.base_url_entry.pack(side="left", fill="x", expand=True, padx=(8, 0)) - row3 = ttk.Frame(config_frame) + row3 = ttk.Frame(self.advanced_frame) row3.pack(fill="x") ttk.Label(row3, text="Encoder").pack(side="left") self.encoder_combo = ttk.Combobox(row3, textvariable=self.selected_encoder_var, state="readonly", width=34) @@ -232,14 +310,30 @@ def _build_ui(self, ttk: Any, tk: Any, scrolledtext: Any) -> None: self.bitrate_entry.pack(side="left", padx=(6, 0)) buttons = ttk.Frame(config_frame) + self.buttons_frame = buttons buttons.pack(fill="x", pady=(10, 0)) self.start_btn = ttk.Button(buttons, text="Start benchmark (Alt+B)", underline=6, command=self._start_run) self.start_btn.pack(side="left") self.stop_btn = ttk.Button(buttons, text="Stop (Alt+S)", underline=0, command=self._stop_run, state="disabled") self.stop_btn.pack(side="left", padx=(8, 0)) - self.upload_btn = ttk.Button(buttons, text="Retry Queued Uploads", command=self._retry_uploads) + self.upload_btn = ttk.Button(buttons, text="Retry due uploads", command=self._retry_uploads) self.upload_btn.pack(side="left", padx=(16, 0)) + saved_frame = ttk.LabelFrame(outer, text="Saved work", padding=10) + saved_frame.pack(fill="x", pady=(12, 0)) + ttk.Label(saved_frame, textvariable=self.saved_summary_var).pack(anchor="w") + saved_row = ttk.Frame(saved_frame) + saved_row.pack(fill="x", pady=(6, 0)) + self.saved_combo = ttk.Combobox(saved_row, textvariable=self.selected_saved_var, + values=[], state="readonly", width=64) + self.saved_combo.pack(side="left", fill="x", expand=True) + self.saved_combo.bind("<>", lambda _evt: self._refresh_controls()) + self.resume_btn = ttk.Button(saved_row, text="Resume", command=self._resume_saved) + self.resume_btn.pack(side="left", padx=(8, 0)) + self.publish_btn = ttk.Button(saved_row, text="Publish saved results", command=self._publish_saved) + self.publish_btn.pack(side="left", padx=(8, 0)) + ttk.Label(saved_frame, text="To publish, turn off Save locally and approve uploads.").pack(anchor="w", pady=(6, 0)) + progress_frame = ttk.LabelFrame(outer, text="Live Progress", padding=10) progress_frame.pack(fill="x", pady=(12, 12)) @@ -287,8 +381,68 @@ def _site_root(self) -> str: def _refresh_controls(self) -> None: idle = not self.running and not self._upload_active() self.start_btn.configure(state="normal" if idle else "disabled") - self.stop_btn.configure(state="normal" if self.running else "disabled") - self.upload_btn.configure(state="normal" if idle else "disabled") + self.stop_btn.configure(state="normal" if self.running or self._upload_active() else "disabled") + self.upload_btn.configure(state="normal" if idle and not self.no_submit_var.get() else "disabled") + self.no_submit_check.configure(state="normal" if idle else "disabled") + selected = self._selected_saved_state() + actions = {str(action.get("action") or "") for action in (selected or {}).get("actions", [])} + self.resume_btn.configure(state="normal" if idle and "resume" in actions else "disabled") + self.publish_btn.configure(state="normal" if idle and not self.no_submit_var.get() + and "publish_saved" in actions else "disabled") + + def _selected_saved_campaign(self) -> str: + index = self.saved_combo.current() + if index is None or index < 0 or index >= len(self.saved_campaign_ids): + return "" + return self.saved_campaign_ids[index] + + def _selected_saved_state(self) -> Optional[Dict[str, Any]]: + campaign_id = self._selected_saved_campaign() + return next((item for item in self.saved_state.get("campaigns", []) + if item.get("campaignId") == campaign_id), None) + + def _confirm_publication_consent(self) -> bool: + return client_main._ensure_interactive_publication_consent( + queue_dir=str(self.base_args.queue_dir), + prompt_callback=lambda disclosure: bool(messagebox.askyesno( + "Allow Benchmark Publication", disclosure, icon="warning", + )), + ) + + def _refresh_saved_work(self) -> None: + previous = self._selected_saved_campaign() if self.saved_campaign_ids else "" + try: + state = self._load_recovery_state() + except Exception as exc: + self.saved_state = {} + self.saved_summary_var.set(f"Saved work unavailable: {exc}") + return + self.saved_state = state + publication = state.get("publication") or {} + pending = int(publication.get("pendingEntries") or 0) + due = int(publication.get("dueEntries") or 0) + terminal = int(publication.get("terminalEntries") or 0) + accepted = int(publication.get("acceptedReceipts") or 0) + campaigns = list(state.get("campaigns") or []) + self.saved_summary_var.set( + f"{len(campaigns)} campaign(s) · {due} due / {max(0, pending - due)} delayed uploads · " + f"{accepted} uploaded (analysis pending) · {terminal} terminal" + ) + self.saved_campaign_ids = [str(item.get("campaignId") or "") for item in campaigns] + labels = [ + f"{item.get('campaignId')} — {'measured' if item.get('complete') else 'unfinished'}, " + f"{int(item.get('pendingUploads') or 0)} unpublished, " + f"{int(item.get('queueDue') or 0)} due, " + f"{int(item.get('queueTerminal') or 0)} terminal, " + f"{int(item.get('unavailableSources') or 0)} unavailable" + for item in campaigns + ] + self.saved_combo["values"] = labels + if labels: + self.saved_combo.current(self.saved_campaign_ids.index(previous) if previous in self.saved_campaign_ids else 0) + else: + self.selected_saved_var.set("") + self._refresh_controls() def _set_running(self, running: bool) -> None: self.running = running @@ -302,12 +456,22 @@ def _set_running(self, running: bool) -> None: self.batch_spin.configure(state="normal" if not self.running and not self._upload_active() else "disabled") self.base_url_entry.configure(state="normal" if not self.running and not self._upload_active() else "disabled") self.crf_spin.configure(state=enabled) + self._update_single_fields_state(preview=False) def _selected_mode_key(self) -> Optional[str]: return GUI_MODE_BY_LABEL.get(self.mode_var.get().strip()) + def _toggle_advanced(self) -> None: + if self.advanced_var.get(): + self.advanced_frame.pack(fill="x", pady=(8, 0), before=self.buttons_frame) + else: + self.advanced_frame.pack_forget() + def _update_single_fields_state(self, preview: bool = True) -> None: single = self._selected_mode_key() is None + if single and not self.advanced_var.get(): + self.advanced_var.set(True) + self._toggle_advanced() state = "readonly" if single and not self.running else "disabled" spin_state = "normal" if single and not self.running else "disabled" self.encoder_combo.configure(state=state) @@ -366,7 +530,12 @@ def _stop_shortcut(self, _event: Any = None) -> str: self._stop_run() return "break" - def _start_run(self) -> None: + def _resume_saved(self) -> None: + campaign_id = self._selected_saved_campaign() + if campaign_id: + self._start_run(resume_id=campaign_id) + + def _start_run(self, *, resume_id: str = "") -> None: if self.running or self._upload_active(): return try: # Advisory only; run_benchmark_batch refuses authoritatively before any preparation. @@ -377,73 +546,109 @@ def _start_run(self) -> None: who = f" (campaign {active['campaignId']}, PID {active['pid']})" if active.get("campaignId") else "" self.summary_var.set(f"Another collection is actively running in this queue{who}. " f"Its checkpoints continue automatically - let it finish, or " - f"stop/cancel that run first. Retry Queued Uploads stays available.") + f"stop/cancel that run first. Due uploads resume after measurement.") self._append_log("Start refused: active collection detected") return - mode = self.mode_var.get().strip() - mode_key = GUI_MODE_BY_LABEL.get(mode) - if mode_key is None and ( - not self._selected_encoder() or self._selected_preset() not in self.preset_values - ): - messagebox.showerror("Unsupported configuration", "Select an available encoder and supported preset before starting.") - return - self.cancel_event.clear() - self.last_failure = None - self._set_running(True) - self.summary_var.set("Run started...") - self.stage_var.set("Starting") - self.current_var.set("-") - self.telemetry_var.set("-") - self.counter_var.set("ok=0 skip=0 queue=0 fail=0") - self.overall_total = 1 - self.overall_done = 0 - self.batch_total = 1 - self.batch_done = 0 - self.overall_pb.configure(maximum=1, value=0) - self.batch_pb.configure(maximum=1, value=0) - self._append_log(f"Starting {mode} run") - - run_args = argparse.Namespace(**vars(self.base_args)) - run_args.base_url = self.base_url_var.get().strip() or self.base_args.base_url - run_args.no_submit = bool(self.no_submit_var.get()) - run_args.retries = max(1, int(self.retries_var.get() or 1)) - run_args.batch_size = max(0, int(self.batch_size_var.get() or 0)) - run_args.pause_on_exit = False - run_args.menu = False - self._browse_shown = False - if getattr(run_args, "max_duration_minutes_explicit", False): - self._append_log( - f"Explicit measurement allowance: {float(run_args.max_duration_minutes):g} minutes; " - "the run stops there with the campaign saved for a later continuation." - ) - else: - self._append_log( - f"Checkpoint segments: {getattr(run_args, 'max_duration_minutes', 60):g} minutes each; " - "the run continues automatically until the plan completes. Acquisition and uploads are separate." - ) - bitrate = self.bitrate_var.get().strip() try: - run_args.target_bitrate_kbps = int(bitrate) if bitrate else None - except ValueError: - messagebox.showerror("Bitrate", "Enter a positive integer bitrate in kbps") - self._set_running(False) + mode = self.mode_var.get().strip() + if mode not in GUI_MODE_BY_LABEL: + raise ValueError("Choose a listed contribution mode.") + mode_key = GUI_MODE_BY_LABEL[mode] + run_args = argparse.Namespace(**vars(self.base_args)) + run_args.base_url = self.base_url_var.get().strip() or str(self.base_args.base_url) + run_args.no_submit = bool(self.no_submit_var.get()) + if resume_id: + run_args.resume_campaign = resume_id + run_args.submit = not run_args.no_submit + if not run_args.no_submit and not re.match(r"^https?://[^/\s]+", run_args.base_url): + raise ValueError("Base URL must be an HTTP or HTTPS address.") + run_args.retries = _validated_integer(self.retries_var, "Retries", 1, 10) + run_args.batch_size = _validated_integer(self.batch_size_var, "Batch size", 0, 64) + run_args.pause_on_exit = False + run_args.menu = False + bitrate = str(self.bitrate_var.get()).strip() + if mode_key is None and not resume_id: + encoder, preset = self._selected_encoder(), self._selected_preset() + if not encoder or preset not in self.preset_values: + raise ValueError("Select an available encoder and supported preset before starting.") + quality = _validated_integer(self.crf_var, "Native quality value", 0, 40) + if bitrate and not re.fullmatch(r"[0-9]+", bitrate): + raise ValueError("Bitrate must be a positive whole number in kbps.") + run_args.target_bitrate_kbps = int(bitrate) if bitrate else None + if run_args.target_bitrate_kbps is not None and run_args.target_bitrate_kbps <= 0: + raise ValueError("Bitrate must be a positive whole number in kbps.") + run_args = client_main.build_single_effective_args( + base_args=run_args, encoder=encoder, preset=preset, crf=quality, + ) + if not resume_id: + estimate = self._estimate_acquisition(mode_key) + if not estimate.get("storageOk", True): + raise ValueError( + "Not enough writable disk space for this contribution. " + f"Estimated peak: {client_main._format_byte_count(int(estimate['peakStorageBytes']))}." + ) + if estimate.get("strategy") == "unavailable": + raise ValueError("The selected frozen clips are unavailable. " + "; ".join(estimate.get("warnings") or [])) + transfer = int(estimate.get("bytesToTransfer") or 0) + if transfer: + approved = messagebox.askyesno( + "Download and storage estimate", + f"This run may download {client_main._format_byte_count(transfer)} of frozen reference media. " + f"Estimated peak extra storage: {client_main._format_byte_count(int(estimate.get('peakStorageBytes') or 0))}. " + "Continue?", + ) + if not approved: + self.summary_var.set("Run not started; download estimate declined") + return + if not run_args.no_submit: + consent_ok = self._confirm_publication_consent() + if not consent_ok: + run_args.no_submit = True + self.no_submit_var.set(True) + self._append_log("Publication consent not granted; saving locally.") + except Exception as exc: + messagebox.showerror("Check settings", str(exc)) + self.summary_var.set(f"Check settings: {exc}") return - if not run_args.no_submit: - consent_ok = client_main._ensure_interactive_publication_consent( - queue_dir=str(run_args.queue_dir), - prompt_callback=lambda disclosure: bool(messagebox.askyesno( - "Allow Benchmark Publication", - disclosure, - icon="warning", - )), - ) - if not consent_ok: - run_args.no_submit = True - self.no_submit_var.set(True) - self._append_log("Publication consent not granted; switching to local dry-run mode.") - self.worker_thread = threading.Thread(target=self._run_worker, args=(run_args, mode, mode_key), daemon=False) - self.worker_thread.start() + try: + self._active_no_submit = run_args.no_submit + self._last_submission_failure = "" + self._run_counts = {"submitted": 0, "locally_complete": 0, "queued": 0, "failed": 0} + self.cancel_event.clear() + self.last_failure = None + self._set_running(True) + self.summary_var.set("Run started...") + self.stage_var.set("Starting") + self.current_var.set("-") + self.telemetry_var.set("-") + self.counter_var.set("local=0 uploaded=0 queued=0 failed=0") + self.overall_total = self.batch_total = 1 + self.overall_done = self.batch_done = 0 + self.overall_pb.configure(maximum=1, value=0) + self.batch_pb.configure(maximum=1, value=0) + self._append_log(f"Resuming {resume_id}" if resume_id else f"Starting {mode} run") + self._browse_shown = False + if getattr(run_args, "max_duration_minutes_explicit", False): + self._append_log( + f"Explicit measurement allowance: {float(run_args.max_duration_minutes):g} minutes; " + "the run stops there with the campaign saved for a later continuation." + ) + else: + self._append_log( + f"Checkpoint segments: {getattr(run_args, 'max_duration_minutes', 60):g} minutes each; " + "the run continues automatically until the plan completes. Acquisition and uploads are separate." + ) + self.worker_thread = threading.Thread(target=self._run_worker, args=(run_args, mode, mode_key), daemon=False) + self.worker_thread.start() + except Exception as exc: + self.worker_thread = None + self._active_no_submit = None + self._set_running(False) + self._update_single_fields_state(preview=False) + self.stage_var.set("Idle") + self.summary_var.set(f"Could not start run: {exc}") + messagebox.showerror("Could not start", str(exc)) def _run_worker(self, run_args: argparse.Namespace, mode: str, mode_key: Optional[str]) -> None: def sink(event: Dict[str, Any]) -> None: @@ -451,7 +656,10 @@ def sink(event: Dict[str, Any]) -> None: rc = 1 try: - if mode_key is not None: + if getattr(run_args, "resume_campaign", ""): + rc = client_main._resume_campaign(run_args, event_sink=sink, + cancel_event=self.cancel_event, interactive=False) + elif mode_key is not None: rc = client_main.run_sweep_mode( mode=mode_key, base_args=run_args, @@ -462,11 +670,7 @@ def sink(event: Dict[str, Any]) -> None: presets_cfg=dict(presets_cfg), ) else: - effective_args = client_main.build_single_effective_args( - base_args=run_args, encoder=self._selected_encoder(), - preset=self._selected_preset(), crf=int(self.crf_var.get()), - ) - rc = client_main.run_with_args(effective_args, event_sink=sink, + rc = client_main.run_with_args(run_args, event_sink=sink, cancel_event=self.cancel_event, show_end_screen=False) except Exception as e: self.event_queue.put(("error", f"{e}\n{traceback.format_exc()}")) @@ -474,22 +678,96 @@ def sink(event: Dict[str, Any]) -> None: finally: self.event_queue.put(("done", rc)) - def _retry_uploads(self) -> None: + def _publish_saved(self) -> None: + if self.running or self._upload_active(): + return + if self.no_submit_var.get(): + self.summary_var.set("Turn off Save locally before publishing saved results") + return + campaign_id = self._selected_saved_campaign() + if not campaign_id: + return + try: + retries = _validated_integer(self.retries_var, "Retries", 1, 10) + if not self._confirm_publication_consent(): + self.summary_var.set("Saved work remains local; publication consent was not granted") + return + self.upload_cancel_event.clear() + self.upload_thread = threading.Thread( + target=self._publish_saved_worker, + args=(campaign_id, self.base_url_var.get().strip() or str(self.base_args.base_url), + str(getattr(self.base_args, "api_key", "") or ""), retries), + daemon=False, + ) + self.upload_thread.start() + self.summary_var.set(f"Publishing saved results from {campaign_id}; no encoding") + self._refresh_controls() + except Exception as exc: + self.upload_thread = None + self.summary_var.set(f"Could not publish saved work: {exc}") + messagebox.showerror("Publish saved results", str(exc)) + self._refresh_controls() + + def _publish_saved_worker(self, campaign_id: str, base_url: str, api_key: str, retries: int) -> None: + try: + rc, info = client_main.publish_saved_campaign( + queue_dir=str(self.base_args.queue_dir), campaign_id=campaign_id, + base_url=base_url, api_key=api_key, retries=retries, + interactive=False, cancel_event=self.upload_cancel_event, + event_sink=lambda event: self.event_queue.put(("event", event)), + ) + self.event_queue.put(("upload_status", self._publication_result_text(rc, info))) + except Exception as exc: + self.event_queue.put(("upload_status", f"Saved publication failed: {exc}")) + + @staticmethod + def _publication_result_text(rc: int, info: Dict[str, Any]) -> str: + submitted = int(info.get("submitted") or 0) + pending = int(info.get("pending") or 0) + unadmitted = int(info.get("unadmitted") or 0) + terminal = int(info.get("terminal") or 0) + int(info.get("deadLettered") or 0) + if rc == 0: + return f"Uploaded {submitted} saved result(s); analysis pending" + if rc == 10: + reason = str(info.get("deferredReason") or "uploads_pending") + return (f"Saved work retained: {pending} queued, {unadmitted} not yet staged " + f"({reason}); no encoding") + return f"Saved publication has {terminal} terminal failure(s); review saved work" + + def _retry_uploads(self, *, automatic: bool = False) -> None: if self.running or self._upload_active(): return + if self.no_submit_var.get(): + self.summary_var.set("Turn off Save locally before retrying uploads") + return + try: + retries = _validated_integer(self.retries_var, "Retries", 1, 10) + if not automatic and not self._confirm_publication_consent(): + self.summary_var.set("Queued uploads remain local; publication consent was not granted") + return + except Exception as exc: + messagebox.showerror("Check settings", str(exc)) + self.summary_var.set(f"Check settings: {exc}") + return base_url = self.base_url_var.get().strip() or str(self.base_args.base_url) api_key = str(getattr(self.base_args, "api_key", "") or "") queue_dir = str(self.base_args.queue_dir) - retries = max(1, int(self.retries_var.get() or 1)) self._append_log("Retrying queued uploads (never encodes)...") self.summary_var.set("Retrying queued uploads...") - self.upload_thread = threading.Thread( - target=self._retry_uploads_worker, - args=(queue_dir, base_url, api_key, retries), - daemon=False, - ) - self.upload_thread.start() - self._refresh_controls() + self.upload_cancel_event.clear() + try: + self.upload_thread = threading.Thread( + target=self._retry_uploads_worker, + args=(queue_dir, base_url, api_key, retries), + daemon=False, + ) + self.upload_thread.start() + self._refresh_controls() + except Exception as exc: + self.upload_thread = None + self.summary_var.set(f"Could not start upload retry: {exc}") + messagebox.showerror("Retry uploads", str(exc)) + self._refresh_controls() def _retry_uploads_worker(self, queue_dir: str, base_url: str, api_key: str, retries: int) -> None: try: @@ -497,22 +775,41 @@ def _retry_uploads_worker(self, queue_dir: str, base_url: str, api_key: str, ret if not pending_before: self.event_queue.put(("upload_status", "Upload queue is empty; nothing to retry.")) return - stats = client_main.replay_spool(queue_dir, base_url=base_url, api_key=api_key, - retries=retries, use_token=False) - remaining = client_main.count_pending_entries(queue_dir) + rc, info = client_main.retry_due_uploads( + queue_dir=queue_dir, base_url=base_url, api_key=api_key, + retries=retries, use_token=False, cancel_event=self.upload_cancel_event, + ) + remaining = int(info.get("pending") or client_main.count_pending_entries(queue_dir)) self.event_queue.put(( "upload_status", f"Upload retry: {pending_before} pending before, {remaining} still pending, " - f"dead-lettered={stats.dead_lettered}, corrupt={stats.corrupt}.", + f"dead-lettered={int(info.get('deadLettered') or 0)}, " + f"corrupt={int(info.get('corrupt') or 0)}, status={info.get('status') or rc}.", )) except Exception as e: self.event_queue.put(("upload_status", f"Upload retry failed: {e}")) + def _idle_retry(self) -> None: + try: + if not self.running and not self._upload_active(): + self._refresh_saved_work() + publication = self.saved_state.get("publication") or {} + if (self.saved_state.get("publicationConsent") and not self.no_submit_var.get() + and not self.saved_state.get("activeCollection") + and not self.saved_state.get("publicationLockBusy") + and int(publication.get("dueEntries") or 0)): + self._retry_uploads(automatic=True) + finally: + self.root.after(30_000, self._idle_retry) + def _stop_run(self) -> None: - if not self.running: + if not self.running and not self._upload_active(): return - self.cancel_event.set() - self.summary_var.set("Stopping owned work; retaining downloads and campaign...") + if self.running: + self.cancel_event.set() + if self._upload_active(): + self.upload_cancel_event.set() + self.summary_var.set("Stopping owned work; retaining saved results...") self._append_log("Cancellation requested") def _handle_event(self, event: Dict[str, Any]) -> None: @@ -539,13 +836,15 @@ def _handle_event(self, event: Dict[str, Any]) -> None: return if event_type == "run_start": - total = max(1, int(event.get("totalTasks") or 1)) + # Batch totalTasks is an upper bound on encode attempts, while + # task_complete counts finished measurement groups. + total = max(1, int(event.get("totalGroups") or 1)) if event.get("scope") == "batch" else max(1, int(event.get("totalTasks") or 1)) self.overall_total = total self.overall_done = 0 self.overall_pb.configure(maximum=total, value=0) self.batch_pb.configure(maximum=total, value=0) - self.summary_var.set(f"Running {event.get('scope', 'benchmark')} tasks") - self._append_log(f"Run start: total={total}") + self.summary_var.set(f"Running {event.get('scope', 'benchmark')} measurement groups") + self._append_log(f"Run start: groups={total}") return if event_type == "batch_start": @@ -553,6 +852,10 @@ def _handle_event(self, event: Dict[str, Any]) -> None: self.batch_total = batch_size self.batch_done = 0 self.batch_pb.configure(maximum=batch_size, value=0) + if int(event.get("totalBatches") or 1) == 1: + self.overall_total = batch_size + self.overall_done = min(batch_size, max(0, int(event.get("processedTotal") or 0))) + self.overall_pb.configure(maximum=batch_size, value=self.overall_done) self._append_log( f"Batch {event.get('batchNo')}/{event.get('totalBatches')} start ({batch_size} tasks)" ) @@ -598,7 +901,19 @@ def _handle_event(self, event: Dict[str, Any]) -> None: return if event_type == "submit_result": - line = f"Submit result: {event.get('status')} ({event.get('preset') or event.get('codec')})" + line = _submission_line(event) + status = str(event.get("status") or "") + if status in self._run_counts: + self._run_counts[status] += 1 + if status == "locally_complete": + self.counter_var.set( + f"local={self._run_counts['locally_complete']} " + f"uploaded={self._run_counts['submitted']} " + f"queued={self._run_counts['queued']} failed={self._run_counts['failed']}" + ) + if status in {"failed", "rejected", "queued"}: + self._last_submission_failure = line + self.summary_var.set(line) # The ingest response carries only the BenchmarkRun id, which the site does # not resolve; never fabricate a per-run URL. Point at the corpus browse page. if event.get("status") == "submitted" and not self._browse_shown: @@ -609,10 +924,11 @@ def _handle_event(self, event: Dict[str, Any]) -> None: if event_type == "counters": self.counter_var.set( - f"ok={int(event.get('submitted') or 0)} " - f"skip={int(event.get('skipped') or 0)} " - f"queue={int(event.get('queued') or 0)} " - f"fail={int(event.get('failed') or 0)}" + f"local={self._run_counts['locally_complete']} " + f"uploaded={int(event.get('submitted') or 0)} (analysis pending) " + f"skipped={int(event.get('skipped') or 0)} " + f"queued={int(event.get('queued') or 0)} " + f"failed={int(event.get('failed') or 0)}" ) return @@ -670,16 +986,29 @@ def _poll_events(self) -> None: elif kind == "done": self._set_running(False) rc = int(payload) + active_no_submit = self._active_no_submit + self._active_no_submit = None pending_note = "" - if rc in (0, 10, 11) and not self.no_submit_var.get(): + if rc in (0, 10, 11) and not active_no_submit: try: pending = client_main.count_pending_entries(str(self.base_args.queue_dir)) except Exception: pending = 0 if pending: - pending_note = f" — {pending} upload(s) queued; use Retry Queued Uploads" + pending_note = f" — {pending} upload(s) queued; due work retries while this window is open" if rc == 0: - self.summary_var.set(("Locally complete" if self.no_submit_var.get() else "Uploaded; analysis pending") + pending_note) + if active_no_submit: + saved = self._run_counts["locally_complete"] + count = f"{saved} measurement group(s) " if saved else "" + self.summary_var.set(f"Saved {count}locally; use Publish saved results when ready") + elif self._run_counts["failed"]: + self.summary_var.set(self._last_submission_failure + pending_note) + elif self._run_counts["queued"] or pending_note: + self.summary_var.set("Measurements saved; some uploads are queued" + pending_note) + elif self._run_counts["submitted"]: + self.summary_var.set("Uploaded; analysis pending") + else: + self.summary_var.set("Run finished; review saved work") elif rc == 11: self.summary_var.set("Measurement allowance reached; campaign saved — starting this mode again continues it" + pending_note) elif rc == 10: @@ -687,13 +1016,12 @@ def _poll_events(self) -> None: elif rc == 130: self.summary_var.set("Run cancelled") else: - if self.last_failure: - failure = f"Run failed (exit code {rc}): {self.last_failure}" - else: - failure = f"Run failed (exit code {rc}); see event log for details" + failure = self._last_submission_failure or self.last_failure + failure = f"Run failed (exit code {rc}): {failure}" if failure else f"Run failed (exit code {rc}); see event log for details" self.summary_var.set(failure) self._append_log(failure) self._update_single_fields_state(preview=False) + self._refresh_saved_work() elif kind == "upload_status": self._append_log(payload) if not self.running: @@ -702,6 +1030,7 @@ def _poll_events(self) -> None: pass if self.upload_thread is not None and not self.upload_thread.is_alive(): self.upload_thread = None + self._refresh_saved_work() self._refresh_controls() self.root.after(120, self._poll_events) @@ -711,6 +1040,8 @@ def _on_close(self) -> None: return if self.running: self.cancel_event.set() + if self._upload_active(): + self.upload_cancel_event.set() self.summary_var.set("Stopping owned work before close...") self._close_deadline = time.monotonic() + GUI_CLOSE_GRACE_SECONDS self.root.after(100, self._close_when_stopped) From 74e568014f7c5dba3435cb48c7641151ce52c788 Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 01:25:51 -0700 Subject: [PATCH 4/9] Prevent mixed client assets and unclear platform guidance at release One manifest gate now checks all four native asset hashes, wrapper receipts, source revision, client/protocol/suite identity and runtime fingerprints. The contribution page places platform eligibility before download and points Windows CLI users to the verified current console asset. Constraint: Current rc.5 packages remain published until a new validated candidate exists Confidence: high Scope-risk: narrow Directive: Run the assembler on actual downloaded native assets before promotion; it is not a substitute for G01 Tested: 11 manifest/release tests; 78 frontend tests; ESLint and TypeScript checks Not-tested: Final candidate native builds, public download byte matches and ordinary-user launch flows --- client/tests/test_assemble_client_release.py | 108 +++++++++++++ client/tests/test_release_packaging.py | 1 + docs/client-release-parity.md | 36 +++++ frontend/app/run/page.test.tsx | 7 + frontend/app/run/page.tsx | 15 +- frontend/app/run/releaseAssets.ts | 11 +- scripts/assemble_client_release.py | 162 +++++++++++++++++++ scripts/release_manifest_lib.py | 11 ++ 8 files changed, 345 insertions(+), 6 deletions(-) create mode 100644 client/tests/test_assemble_client_release.py create mode 100644 docs/client-release-parity.md create mode 100644 scripts/assemble_client_release.py diff --git a/client/tests/test_assemble_client_release.py b/client/tests/test_assemble_client_release.py new file mode 100644 index 00000000..d80af6da --- /dev/null +++ b/client/tests/test_assemble_client_release.py @@ -0,0 +1,108 @@ +import hashlib +import json +import tempfile +import unittest +from pathlib import Path + +from scripts.assemble_client_release import assemble + + +def digest(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +class AssembleClientReleaseTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + self.entries = [] + for role, platform, name in ( + ("macos-dmg", "mac", "EncodingDB-macOS-arm64.dmg"), + ("windows-gui", "win", "encodingdb-client-windows.exe"), + ("windows-console", "win", "encodingdb-client-windows-console.exe"), + ("linux-archive", "linux", "encodingdb-client-linux.tar.gz"), + ): + data = role.encode() + artifact = self.root / name + artifact.write_bytes(data) + binary_sha = digest((role + "-binary").encode()) if role in {"macos-dmg", "linux-archive"} else digest(data) + manifest = { + "schemaVersion": 1, "source": {"revision": "a" * 40, "trackedChanges": False}, + "projectVersion": "1.3.0-rc.6", "platform": platform, + "protocol": {"clientVersion": "client/0.3.4", "benchmarkProtocolVersion": "7.1", + "minimumClientVersion": "client/0.3.0"}, + "suite": {"suiteVersion": "encodingdb-test-suite-v1", "manifestVersion": 2, + "suiteFingerprint": "f" * 64, "isFrozen": True}, + "runtime": {"fingerprint": platform + "-runtime"}, + "artifact": {"fileName": name, "sha256": binary_sha, "byteSize": len(data)}, + } + manifest_path = self.root / f"{name}.release-manifest.json" + manifest_path.write_text(json.dumps(manifest)) + entry = {"role": role, "artifact": name, "releaseManifest": manifest_path.name} + if role in {"macos-dmg", "linux-archive"}: + wrapper = {"fileName": name, "sha256": digest(data), "byteSize": len(data)} + if role == "macos-dmg": + wrapper["readOnlyMountVerification"] = {"mounted": True, "innerSha256": binary_sha} + package = {"provisional": False, "source": manifest["source"], "dmg": wrapper, + "cliEmbeddedSha256": binary_sha, "cliBytesChangedByPackaging": False} + else: + wrapper["members"] = [{"sha256": binary_sha}] + package = {"provisional": False, "source": manifest["source"], "archive": wrapper, + "verification": {"verified": True}} + package_path = self.root / f"{name}.package-info.json" + package_path.write_text(json.dumps(package)) + entry["packageInfo"] = package_path.name + self.entries.append(entry) + + def spec(self): + return {"expectedSourceRevision": "a" * 40, "expectedProjectVersion": "1.3.0-rc.6", + "assets": self.entries} + + def test_assembles_four_verified_assets_from_one_clean_identity(self): + release = assemble(self.spec(), self.root) + self.assertEqual(release["sourceRevision"], "a" * 40) + self.assertEqual(release["clientVersion"], "client/0.3.4") + self.assertEqual(len(release["assets"]), 4) + self.assertEqual({asset["role"] for asset in release["assets"]}, + {"macos-dmg", "windows-gui", "windows-console", "linux-archive"}) + + def test_rejects_tampered_asset_and_mixed_revision(self): + (self.root / "encodingdb-client-windows.exe").write_bytes(b"different") + with self.assertRaisesRegex(ValueError, "file bytes differ"): + assemble(self.spec(), self.root) + (self.root / "encodingdb-client-windows.exe").write_bytes(b"windows-gui") + path = self.root / "encodingdb-client-windows.exe.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["source"]["revision"] = "b" * 40 + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "differs from the reviewed source/version"): + assemble(self.spec(), self.root) + + def test_rejects_missing_package_proof_and_dirty_source(self): + path = self.root / "EncodingDB-macOS-arm64.dmg.package-info.json" + package = json.loads(path.read_text()) + package["dmg"]["readOnlyMountVerification"] = None + path.write_text(json.dumps(package)) + with self.assertRaisesRegex(ValueError, "read-only mount verification"): + assemble(self.spec(), self.root) + package["dmg"]["readOnlyMountVerification"] = {"mounted": True, "innerSha256": package["cliEmbeddedSha256"]} + path.write_text(json.dumps(package)) + path = self.root / "encodingdb-client-linux.tar.gz.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["source"]["trackedChanges"] = True + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "clean committed source"): + assemble(self.spec(), self.root) + + def test_rejects_different_runtime_for_windows_pair(self): + path = self.root / "encodingdb-client-windows-console.exe.release-manifest.json" + receipt = json.loads(path.read_text()) + receipt["runtime"]["fingerprint"] = "other-runtime" + path.write_text(json.dumps(receipt)) + with self.assertRaisesRegex(ValueError, "different win runtime identities"): + assemble(self.spec(), self.root) + + +if __name__ == "__main__": + unittest.main() diff --git a/client/tests/test_release_packaging.py b/client/tests/test_release_packaging.py index 75b83c02..3b969da7 100644 --- a/client/tests/test_release_packaging.py +++ b/client/tests/test_release_packaging.py @@ -98,6 +98,7 @@ def test_release_version_is_assigned_and_missing_version_is_rejected(self) -> No def test_read_client_minimum_version_is_coherent(self) -> None: self.assertEqual(release_manifest_lib.read_client_minimum_version(), "client/0.3.0") + self.assertEqual(release_manifest_lib.read_client_implementation_version(), "client/0.3.3") def test_client_patch_version_does_not_change_protocol_minimum(self) -> None: with mock.patch.object(release_manifest_lib, "read_text", side_effect=[ diff --git a/docs/client-release-parity.md b/docs/client-release-parity.md new file mode 100644 index 00000000..9b2f92ff --- /dev/null +++ b/docs/client-release-parity.md @@ -0,0 +1,36 @@ +# Client release parity gate + +For the next candidate, build the macOS app/DMG, Windows GUI and console, and +Linux launcher/archive from the **same clean commit**. Keep the native +`*.release-manifest.json` sidecars and the macOS/Linux `*.package-info.json` +beside the downloaded build artifacts. The native build helpers already emit +these receipts; new sidecars also record `protocol.clientVersion`. + +Create a spec on the machine holding all four artifacts: + +```json +{ + "expectedSourceRevision": "", + "expectedProjectVersion": "", + "assets": [ + {"role": "macos-dmg", "artifact": "EncodingDB-macOS-arm64.dmg", "releaseManifest": "encodingdb-client-macos.release-manifest.json", "packageInfo": "EncodingDB-macOS-arm64.dmg.package-info.json"}, + {"role": "windows-gui", "artifact": "encodingdb-client-windows.exe", "releaseManifest": "encodingdb-client-windows.exe.release-manifest.json"}, + {"role": "windows-console", "artifact": "encodingdb-client-windows-console.exe", "releaseManifest": "encodingdb-client-windows-console.exe.release-manifest.json"}, + {"role": "linux-archive", "artifact": "encodingdb-client-linux.tar.gz", "releaseManifest": "encodingdb-client-linux.release-manifest.json", "packageInfo": "encodingdb-client-linux.tar.gz.package-info.json"} + ] +} +``` + +Paths are relative to the spec. Use the actual sidecar file names emitted by +each build. Then run: + +```bash +python scripts/assemble_client_release.py --spec candidate-spec.json --output client-release-manifest.json +``` + +The command rehashes each artifact, checks wrapper/embedded executable +identity, and rejects mixed source revisions, dirty builds, protocol/client +versions, frozen-suite fingerprints and runtime identities. The output gives +per-asset sizes and hashes for the release page. It does not publish an asset +or replace the platform-specific G01 checks in Linear PLA-90. Promotion also +requires rehashing the public download and matching its bytes to the manifest. diff --git a/frontend/app/run/page.test.tsx b/frontend/app/run/page.test.tsx index c02ff2db..b6d35f1c 100644 --- a/frontend/app/run/page.test.tsx +++ b/frontend/app/run/page.test.tsx @@ -82,6 +82,13 @@ describe("RunPage", () => { expect(screen.getByRole("link", { name: "Download for macOS (Apple Silicon)" })).toHaveAttribute("href", `${plannedBase}/EncodingDB-macOS-arm64.dmg`); expect(screen.getByRole("link", { name: "Download for Windows (GUI)" })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-windows.exe`); expect(screen.getByRole("link", { name: "Download for Linux (x86-64)" })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-linux.tar.gz`); + expect(screen.getByRole("link", { name: `Windows console (${projectTag})` })).toHaveAttribute("href", `${plannedBase}/encodingdb-client-windows-console.exe`); + const macCard = screen.getByRole("link", { name: "Download for macOS (Apple Silicon)" }).closest("article"); + expect(macCard?.textContent?.indexOf("requires macOS 27 or later")).toBeLessThan(macCard?.textContent?.indexOf("Download for macOS") ?? 0); + const windowsCard = screen.getByRole("link", { name: "Download for Windows (GUI)" }).closest("article"); + expect(windowsCard?.textContent?.indexOf("Windows 11 x86-64")).toBeLessThan(windowsCard?.textContent?.indexOf("Download for Windows") ?? 0); + const linuxCard = screen.getByRole("link", { name: "Download for Linux (x86-64)" }).closest("article"); + expect(linuxCard?.textContent?.indexOf("Ubuntu 24.04 x86-64")).toBeLessThan(linuxCard?.textContent?.indexOf("Download for Linux") ?? 0); expect(screen.getByRole("link", { name: new RegExp(`Release notes, checksums, and build evidence`) })).toHaveAttribute("href", `${repoReleases}/tag/${projectTag}`); expect(screen.queryByText(/pending publication/)).toBeNull(); }); diff --git a/frontend/app/run/page.tsx b/frontend/app/run/page.tsx index 62caf52b..401c3dd9 100644 --- a/frontend/app/run/page.tsx +++ b/frontend/app/run/page.tsx @@ -1,5 +1,5 @@ import styles from "./page.module.css"; -import { cliTag, downloadModel, historicalTag, projectTag, repoReleases, supersededAssets, supersededTag } from "./releaseAssets"; +import { cliTag, currentWindowsConsole, downloadModel, historicalTag, projectTag, repoReleases, supersededAssets, supersededTag } from "./releaseAssets"; export const dynamic = "force-dynamic"; @@ -14,6 +14,7 @@ export default function RunPage() { const historical = `${repoReleases}/download/${historicalTag}`; const superseded = `${repoReleases}/download/${supersededTag}`; const cliBase = `${repoReleases}/download/${cliTag}`; + const currentBase = `${repoReleases}/download/${projectTag}`; return

Contribute results

@@ -28,12 +29,12 @@ export default function RunPage() {
{downloads.items.map((asset) =>

{asset.label}

+

{asset.support}

{asset.href ? Download for {asset.label} : Download for {asset.label} (pending publication)}

{asset.file}

    {(platformNotes[asset.file] ?? []).map((line) =>
  1. {line}
  2. )}
-

{asset.support}

{asset.sha256 ? <>SHA-256 {asset.sha256} : "SHA-256 published with the release."}

)}
@@ -75,8 +76,14 @@ export default function RunPage() { )} -

Plain command-line builds ({cliTag})

-

Protocol-compatible executables that predate the packaged apps; the macOS file is a bare extensionless executable.

+

Current command-line access ({projectTag})

+

For Windows scripts, use the console executable from the current release. The current macOS and Linux command-line entry points are inside their packages above.

+

{downloads.published + ? Windows console ({projectTag}) + : Windows console ({projectTag}) available when current downloads are enabled} · {currentWindowsConsole.file} · SHA-256 {currentWindowsConsole.sha256}

+ +

Historical plain executables ({cliTag})

+

These protocol-compatible files predate the packaged apps and current shared-core repairs. The macOS file is a bare extensionless executable. Use the current packages above for contribution.

  • Windows GUI (rc.1) · encodingdb-client-windows.exe
  • Windows console (rc.1) · encodingdb-client-windows-console.exe
  • diff --git a/frontend/app/run/releaseAssets.ts b/frontend/app/run/releaseAssets.ts index d65add79..24096df3 100644 --- a/frontend/app/run/releaseAssets.ts +++ b/frontend/app/run/releaseAssets.ts @@ -36,16 +36,23 @@ export const primaryAssets: ReleaseAsset[] = [ file: "encodingdb-client-windows.exe", label: "Windows (GUI)", sha256: "2ec0a4bcb7d94af340e61a1837bfc9380b51b2dd60bbd1f48eb023dfa24c0379", - support: "No Authenticode signature, so SmartScreen may warn at first launch. The window exposes the same Small/Medium/Large/Full sweeps as the guided interface and recovers automatically from a cache folder protected against the current user.", + support: "Tested on Windows 11 x86-64; other Windows versions are unverified. No Authenticode signature, so SmartScreen may warn at first launch. The window exposes the same Small/Medium/Large/Full sweeps as the guided interface and recovers automatically from a cache folder protected against the current user.", }, { file: "encodingdb-client-linux.tar.gz", label: "Linux (x86-64)", sha256: "b1a68a039ce78a6bc9718adbae99409865326ea8a8b3fc29e47aaa090ef0f4d8", - support: "Unsigned archive; unpack it and run the launcher inside. Same guided flow; includes the command-line entry point for scripts.", + support: "Tested on Ubuntu 24.04 x86-64; other Linux distributions are unverified. Unsigned archive; unpack it and run the launcher inside. Same guided flow; includes the command-line entry point for scripts.", }, ]; +export const currentWindowsConsole: ReleaseAsset = { + file: "encodingdb-client-windows-console.exe", + label: "Windows console", + sha256: "05189da160c876f812cd16ae228b76df50bac8925d5763bb3d49ea9f0e42a60e", + support: "Command-line entry point from the current release, built alongside the Windows GUI.", +}; + // Preserve the actual published rc.4 asset identities for rollback and verification. export const supersededAssets: ReleaseAsset[] = [ { diff --git a/scripts/assemble_client_release.py b/scripts/assemble_client_release.py new file mode 100644 index 00000000..fdea0968 --- /dev/null +++ b/scripts/assemble_client_release.py @@ -0,0 +1,162 @@ +#!/usr/bin/env python3 +"""Verify one candidate's native assets and write a shared release manifest. + +The JSON input declares ``expectedSourceRevision`` and ``expectedProjectVersion`` +and has an ``assets`` array with one entry per role: ``macos-dmg``, +``windows-gui``, ``windows-console`` and ``linux-archive``. Each entry names +``artifact`` and ``releaseManifest`` paths, plus ``packageInfo`` for macOS +and Linux wrappers. Relative paths resolve beside the spec file. Use native +build receipts from one clean committed revision. This gate does not publish +assets or replace the physical G01 acceptance runs. +""" + +import argparse +import json +import re +import sys +from pathlib import Path +from typing import Any + +ROOT_DIR = Path(__file__).resolve().parent.parent +if str(ROOT_DIR) not in sys.path: + sys.path.insert(0, str(ROOT_DIR)) + +from scripts.release_manifest_lib import atomic_write_json, sha256_path + +ROLES = {"macos-dmg", "windows-gui", "windows-console", "linux-archive"} + + +def _path(base: Path, value: Any) -> Path: + path = Path(str(value or "")) + if not str(value or "").strip(): + raise ValueError("asset path is missing") + return path if path.is_absolute() else base / path + + +def _read_json(path: Path) -> dict[str, Any]: + payload = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(payload, dict): + raise ValueError(f"expected a JSON object in {path}") + return payload + + +def _check_file(path: Path, identity: dict[str, Any]) -> str: + if path.name != identity.get("fileName"): + raise ValueError(f"file name differs from native receipt: {path.name}") + if path.stat().st_size != identity.get("byteSize"): + raise ValueError(f"file bytes differ from native receipt: {path.name}") + observed_sha = sha256_path(path) + if observed_sha != identity.get("sha256"): + raise ValueError(f"file bytes differ from native receipt: {path.name}") + return observed_sha + + +def _source_revision(receipt: dict[str, Any]) -> str: + source = receipt.get("source") or {} + revision = source.get("revision") + if not isinstance(revision, str) or not re.fullmatch(r"[0-9a-f]{40}", revision) or source.get("trackedChanges") is not False: + raise ValueError("native receipt must identify a clean committed source revision") + return revision + + +def assemble(spec: dict[str, Any], base: Path) -> dict[str, Any]: + expected_revision = spec.get("expectedSourceRevision") + expected_version = spec.get("expectedProjectVersion") + if not isinstance(expected_revision, str) or not re.fullmatch(r"[0-9a-f]{40}", expected_revision): + raise ValueError("spec must name the reviewed source revision") + if not isinstance(expected_version, str) or not expected_version.strip(): + raise ValueError("spec must name the candidate project version") + entries = spec.get("assets") + if not isinstance(entries, list) or len(entries) != len(ROLES): + raise ValueError("spec must contain one entry for each native release role") + assets = [] + common: dict[str, Any] | None = None + roles_seen: set[str] = set() + runtimes: dict[str, str] = {} + for entry in entries: + if not isinstance(entry, dict): + raise ValueError("each asset entry must be an object") + role = entry.get("role") + if not isinstance(role, str) or role not in ROLES or role in roles_seen: + raise ValueError(f"unexpected or repeated native release role: {role}") + roles_seen.add(role) + artifact = _path(base, entry.get("artifact")) + receipt = _read_json(_path(base, entry.get("releaseManifest"))) + if receipt.get("schemaVersion") != 1: + raise ValueError(f"unsupported native manifest for {role}") + protocol = receipt.get("protocol") or {} + suite = receipt.get("suite") or {} + runtime = receipt.get("runtime") or {} + identity = { + "sourceRevision": _source_revision(receipt), + "projectVersion": receipt.get("projectVersion"), + "clientVersion": protocol.get("clientVersion"), + "protocolVersion": protocol.get("benchmarkProtocolVersion"), + "minimumClientVersion": protocol.get("minimumClientVersion"), + "suiteVersion": suite.get("suiteVersion"), + "suiteManifestVersion": suite.get("manifestVersion"), + "suiteFingerprint": suite.get("suiteFingerprint"), + } + if not all(identity.values()) or suite.get("isFrozen") is not True: + raise ValueError(f"incomplete frozen client identity for {role}") + if identity["sourceRevision"] != expected_revision or identity["projectVersion"] != expected_version: + raise ValueError(f"{role} differs from the reviewed source/version in the spec") + if common is None: + common = identity + elif identity != common: + raise ValueError(f"{role} was built from a different client/protocol/suite identity") + platform = {"macos-dmg": "mac", "linux-archive": "linux"}.get(role, "win") + if receipt.get("platform") != platform or not isinstance(runtime.get("fingerprint"), str): + raise ValueError(f"invalid {role} platform/runtime identity") + if platform in runtimes and runtimes[platform] != runtime["fingerprint"]: + raise ValueError(f"different {platform} runtime identities in one release") + runtimes[platform] = runtime["fingerprint"] + + binary = receipt.get("artifact") or {} + wrapper = None + if role in {"macos-dmg", "linux-archive"}: + package = _read_json(_path(base, entry.get("packageInfo"))) + if package.get("provisional") is not False or _source_revision(package) != identity["sourceRevision"]: + raise ValueError(f"unverified package source for {role}") + wrapper = package.get("dmg") if role == "macos-dmg" else package.get("archive") + if not isinstance(wrapper, dict): + raise ValueError(f"missing package identity for {role}") + artifact_sha = _check_file(artifact, wrapper) + if role == "macos-dmg": + if package.get("cliEmbeddedSha256") != binary.get("sha256") or package.get("cliBytesChangedByPackaging") is not False: + raise ValueError("macOS DMG embeds a different client executable") + mount = wrapper.get("readOnlyMountVerification") or {} + if mount.get("mounted") is not True or mount.get("innerSha256") != binary.get("sha256"): + raise ValueError("macOS DMG lacks read-only mount verification") + else: + members = wrapper.get("members") or [] + if not any(isinstance(member, dict) and member.get("sha256") == binary.get("sha256") for member in members): + raise ValueError("Linux archive does not contain the audited client executable") + if (package.get("verification") or {}).get("verified") is not True: + raise ValueError("Linux archive lacks launch-contract verification") + else: + artifact_sha = _check_file(artifact, binary) + assets.append({ + "role": role, "fileName": artifact.name, + "sha256": artifact_sha, "byteSize": artifact.stat().st_size, + "embeddedExecutableSha256": binary.get("sha256"), + "runtimeFingerprint": runtime["fingerprint"], + }) + if roles_seen != ROLES or common is None: + raise ValueError("native release roles are incomplete") + return {"schemaVersion": 1, **common, "assets": sorted(assets, key=lambda item: item["role"])} + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--spec", type=Path, required=True, help="JSON mapping each release role to artifact and native receipts") + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + spec = _read_json(args.spec) + atomic_write_json(args.output, assemble(spec, args.spec.parent)) + print(args.output) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/release_manifest_lib.py b/scripts/release_manifest_lib.py index 186f9b08..d8d94a57 100644 --- a/scripts/release_manifest_lib.py +++ b/scripts/release_manifest_lib.py @@ -127,6 +127,16 @@ def read_client_minimum_version() -> str: return client_match.group(1) +def read_client_implementation_version() -> str: + source = read_text(ROOT_DIR / "client" / "main.py") + match = re.search(r'^CLIENT_VERSION\s*=\s*"([^"]+)"', source, re.MULTILINE) + release = json.loads(read_text(ROOT_DIR / "release.json")) + version = release.get("clientImplementationVersion") + if not match or not isinstance(version, str) or match.group(1) != version: + raise RuntimeError("client implementation version differs from release.json") + return version + + def load_vmaf_manifest() -> Dict[str, Any]: with (ROOT_DIR / "client" / "resources" / "vmaf" / "manifest.json").open("r", encoding="utf-8") as handle: payload = json.load(handle) @@ -405,6 +415,7 @@ def build_release_manifest( "executableIdentity": executable_identity(artifact_path), }, "protocol": { + "clientVersion": read_client_implementation_version(), "benchmarkProtocolVersion": config.BENCHMARK_PROTOCOL_VERSION, "minimumClientVersion": read_client_minimum_version(), }, From d4418d89a35b56b2c966c209a532c14f51508248 Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 01:41:24 -0700 Subject: [PATCH 5/9] Keep native CI reproducible after the upstream runtime archive disappears The reviewed BtbN release tag now returns 404 before client tests or native builds start. Existing public candidate archives embed the exact locked Linux and Windows FFmpeg/FFprobe bytes, so CI extracts only those binaries after verifying archive and runtime-lock hashes. Runtime identity is unchanged. Constraint: The checked-in FFmpeg runtime lock and protocol identity cannot change Rejected: Runner FFmpeg packages | required xpsnr/libvmaf capabilities vary Confidence: high Scope-risk: moderate Directive: Do not replace the locked binaries without native model execution and a new reviewed lock Tested: Local extraction and SHA-256 match for Linux/Windows; anonymous archive HEAD 200; 9 tests; workflow YAML parse Not-tested: Rerun of all GitHub Actions native jobs and physical packaged-client acceptance --- .github/workflows/build.yml | 27 ++--- .github/workflows/release-preflight.yml | 6 +- client/tests/test_runtime_provisioning.py | 14 +++ docs/CLIENT_BUILD_RUNTIME.md | 13 ++- scripts/provision_pinned_runtime.py | 120 ++++++++++++++++++++++ 5 files changed, 154 insertions(+), 26 deletions(-) create mode 100644 client/tests/test_runtime_provisioning.py create mode 100644 scripts/provision_pinned_runtime.py diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index ed47f6e7..ce937e78 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -138,10 +138,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" @@ -172,10 +170,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" @@ -323,13 +319,8 @@ jobs: run: | $runtimeDir = Join-Path $env:RUNNER_TEMP "ffmpeg-runtime" New-Item -ItemType Directory -Path $runtimeDir -Force | Out-Null - $archive = Join-Path $runtimeDir "ffmpeg-win.zip" - Invoke-WebRequest -Uri "https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-win64-gpl-8.1.zip" -OutFile $archive - $actualHash = (Get-FileHash -Algorithm SHA256 -Path $archive).Hash.ToLowerInvariant() - if ($actualHash -ne "4e40699fa864811312d6c1895ea4873beaf8475e1188563fee57fac860554601") { - throw "Unexpected FFmpeg archive hash: $actualHash" - } - Expand-Archive -LiteralPath $archive -DestinationPath $runtimeDir -Force + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python scripts/provision_pinned_runtime.py --platform win --output-root $runtimeDir $bundleDir = Get-ChildItem -LiteralPath $runtimeDir -Directory | Where-Object { $_.Name -like "ffmpeg-*" } | Select-Object -First 1 if (-not $bundleDir) { throw "Expanded FFmpeg bundle directory not found" } $ffmpeg = Join-Path $bundleDir.FullName "bin\\ffmpeg.exe" @@ -535,10 +526,8 @@ jobs: set -euo pipefail mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" diff --git a/.github/workflows/release-preflight.yml b/.github/workflows/release-preflight.yml index 3a9b16e5..dfb49ae7 100644 --- a/.github/workflows/release-preflight.yml +++ b/.github/workflows/release-preflight.yml @@ -30,10 +30,8 @@ jobs: sudo apt-get install -y --no-install-recommends openssl ripgrep mkdir -p "$RUNNER_TEMP/ffmpeg-runtime" cd "$RUNNER_TEMP/ffmpeg-runtime" - curl -L --fail --silent --show-error -o ffmpeg-linux.tar.xz \ - https://github.com/BtbN/FFmpeg-Builds/releases/download/autobuild-2026-09-09-14-51/ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1.tar.xz - echo "1288c1e46efce263651f5b10755f9ef11419a91558407fe61cc178aa9869fa1b ffmpeg-linux.tar.xz" | shasum -a 256 -c - - tar -xf ffmpeg-linux.tar.xz + python -m pip install --disable-pip-version-check pyinstaller==6.19.0 + python "$GITHUB_WORKSPACE/scripts/provision_pinned_runtime.py" --platform linux --output-root "$PWD" bundle_dir="$(find "$PWD" -mindepth 1 -maxdepth 1 -type d -name 'ffmpeg-*' -print -quit)" test -x "$bundle_dir/bin/ffmpeg" test -x "$bundle_dir/bin/ffprobe" diff --git a/client/tests/test_runtime_provisioning.py b/client/tests/test_runtime_provisioning.py new file mode 100644 index 00000000..9be48b73 --- /dev/null +++ b/client/tests/test_runtime_provisioning.py @@ -0,0 +1,14 @@ +"""The CI runtime recovery path must reject unreviewed archive bytes.""" + +import pytest + +from scripts.provision_pinned_runtime import provision + + +@pytest.mark.parametrize("platform", ["linux", "win"]) +def test_candidate_archive_tamper_fails_before_extract(tmp_path, platform): + archive = tmp_path / "candidate" + archive.write_bytes(b"unreviewed") + with pytest.raises(RuntimeError, match="reviewed SHA-256/size"): + provision(platform, tmp_path / "out", archive_path=archive) + assert not list((tmp_path / "out").rglob("ffmpeg*")) diff --git a/docs/CLIENT_BUILD_RUNTIME.md b/docs/CLIENT_BUILD_RUNTIME.md index de9c96f9..82444ccd 100644 --- a/docs/CLIENT_BUILD_RUNTIME.md +++ b/docs/CLIENT_BUILD_RUNTIME.md @@ -45,9 +45,16 @@ Supported builder overrides: Runtime-lock-sensitive CI does not rely on ambient `apt`, `brew`, or `choco` FFmpeg packages, because those runner packages do not consistently expose the required `xpsnr` filter. -- Linux and Windows use the immutable BtbN `autobuild-2026-09-09-14-51` - FFmpeg `n8.1.2-51-g7ba069f4f1` GPL archives, checked against pinned upstream - SHA256 digests. Platform identities were generated on native runners in +- Linux and Windows use the reviewed BtbN `autobuild-2026-09-09-14-51` + FFmpeg `n8.1.2-51-g7ba069f4f1` GPL runtime bytes. The original upstream + release tag now returns HTTP 404. CI recovers those *same* FFmpeg/FFprobe + bytes from the already-published [Linux candidate archive](https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909/encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167.tar.gz) + (SHA256 `cb2712b94b705cf447eb4b550e8f3b2be1bfcc6120db1f361a6534234bfb325c`) + or [Windows candidate archive](https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909/encodingdb-windows-candidate-ci34430919675-2d3ed7d4d167.zip) + (SHA256 `9e00b681397044ced552ec9539d10af42c731fe5a8b583eb9efb72de917fbe05`). + `scripts/provision_pinned_runtime.py` hashes each archive and extracted binary + against the committed lock before any build or smoke test. Platform identities + were generated on native runners in [CI run 34419226010](https://github.com/oliverdougherC/Encoding_Database/actions/runs/34419226010), then reviewed into the committed lock. A proposed lock is not a packaged build. - macOS uses Evermeet `126386-gc27482a18d7`, containing libvmaf `3.2.0-13`. diff --git a/scripts/provision_pinned_runtime.py b/scripts/provision_pinned_runtime.py new file mode 100644 index 00000000..284063f8 --- /dev/null +++ b/scripts/provision_pinned_runtime.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +"""Recover the reviewed FFmpeg bytes from immutable candidate archives. + +The original BtbN autobuild tag was removed upstream. These existing public +EncodingDB candidate archives embed the *same* FFmpeg/FFprobe bytes. Verify +the archive digest and each extracted binary against the checked-in runtime +lock before CI or a native build may use them. Requires the already-pinned +PyInstaller build dependency; does not register a new runtime identity. +""" + +import argparse +import hashlib +import json +import stat +import tarfile +import tempfile +import urllib.request +import zipfile +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +RELEASE = "https://github.com/oliverdougherC/Encoding_Database/releases/download/encodingdb-beta-review-assets-20260909" +SOURCES = { + "linux": { + "archive": "encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167.tar.gz", + "archiveSha256": "cb2712b94b705cf447eb4b550e8f3b2be1bfcc6120db1f361a6534234bfb325c", + "archiveBytes": 169129799, + "member": "encodingdb-linux-candidate-ci34430919675-2d3ed7d4d167/encodingdb-client-linux", + "directory": "ffmpeg-n8.1.2-51-g7ba069f4f1-linux64-gpl-8.1", + "toc": {"ffmpeg": "bin/linux/ffmpeg", "ffprobe": "bin/linux/ffprobe"}, + }, + "win": { + "archive": "encodingdb-windows-candidate-ci34430919675-2d3ed7d4d167.zip", + "archiveSha256": "9e00b681397044ced552ec9539d10af42c731fe5a8b583eb9efb72de917fbe05", + "archiveBytes": 294384846, + "member": "encodingdb-client-windows-console.exe", + "directory": "ffmpeg-n8.1.2-51-g7ba069f4f1-win64-gpl-8.1", + "toc": {"ffmpeg": r"bin\win\ffmpeg.exe", "ffprobe": r"bin\win\ffprobe.exe"}, + }, +} + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _download(url: str, target: Path) -> None: + with urllib.request.urlopen(url, timeout=30) as response, target.open("wb") as output: + while chunk := response.read(1024 * 1024): + output.write(chunk) + + +def provision(platform: str, output_root: Path, *, archive_path: Path | None = None) -> Path: + spec = SOURCES[platform] + lock = json.loads((ROOT / "client/resources/runtime/ffmpeg-lock.json").read_text())["platforms"][platform] + output_root.mkdir(parents=True, exist_ok=True) + with tempfile.TemporaryDirectory(prefix="reviewed-runtime-") as temp: + temp_root = Path(temp) + candidate = archive_path or temp_root / spec["archive"] + if archive_path is None: + _download(f"{RELEASE}/{spec['archive']}", candidate) + if candidate.stat().st_size != spec["archiveBytes"] or _sha256(candidate) != spec["archiveSha256"]: + raise RuntimeError("candidate archive differs from reviewed SHA-256/size") + + onefile = temp_root / Path(spec["member"]).name + if platform == "linux": + with tarfile.open(candidate, "r:gz") as archive: + member = archive.getmember(spec["member"]) + if not member.isfile(): + raise RuntimeError("candidate executable is not a regular file") + source = archive.extractfile(member) + if source is None: + raise RuntimeError("candidate executable cannot be read") + with source, onefile.open("wb") as output: + while chunk := source.read(1024 * 1024): + output.write(chunk) + else: + with zipfile.ZipFile(candidate) as archive, archive.open(spec["member"]) as source, onefile.open("wb") as output: + while chunk := source.read(1024 * 1024): + output.write(chunk) + + from PyInstaller.archive.readers import CArchiveReader + reader = CArchiveReader(str(onefile)) + bundle = output_root / spec["directory"] / "bin" + bundle.mkdir(parents=True, exist_ok=True) + for name, toc_name in spec["toc"].items(): + data = reader.extract(toc_name) + expected = lock[name] + if len(data) != expected["byteSize"] or hashlib.sha256(data).hexdigest() != expected["sha256"]: + raise RuntimeError(f"embedded {name} differs from reviewed runtime lock") + target = bundle / Path(toc_name.replace("\\", "/")).name + target.write_bytes(data) + if platform == "linux": + target.chmod(target.stat().st_mode | stat.S_IXUSR | stat.S_IXGRP | stat.S_IXOTH) + if _sha256(target) != expected["sha256"]: + raise RuntimeError(f"installed {name} changed after verification") + receipt = {"platform": platform, "archive": spec["archive"], + "archiveSha256": spec["archiveSha256"], + "runtime": {name: lock[name]["sha256"] for name in spec["toc"]}} + (bundle.parent / "provision-receipt.json").write_text(json.dumps(receipt, indent=2) + "\n") + return bundle + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--platform", choices=sorted(SOURCES), required=True) + parser.add_argument("--output-root", type=Path, required=True) + parser.add_argument("--archive-path", type=Path, help="Use an already-downloaded candidate archive") + args = parser.parse_args() + bundle = provision(args.platform, args.output_root, archive_path=args.archive_path) + print(f"Reviewed {args.platform} runtime: {bundle}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From db42acda0a486dae12452032b581b3078e7c598f Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 02:20:09 -0700 Subject: [PATCH 6/9] Restore native GUI acceptance after the guided layout changed The guided Windows client moved retry and batch controls into Advanced settings, leaving the mode selector in a three-child row. The native harness still required seven children and blocked before it could exercise the packaged GUI. Match the observed three-child mode row with a short label followed by a wider combobox, while retaining owned-window, popup-alignment and unique-row guards. Constraint: Tk controls have no accessible names in the hosted Windows UIA tree. Rejected: Hard-coded screen coordinates | cannot establish owned control identity across layouts. Confidence: medium Scope-risk: narrow Directive: Recheck native Win32 control evidence whenever the guided configuration row changes. Tested: Captured CI Windows screenshot and Win32 tree match the new structural predicate; git diff --check. Not-tested: Packaged Windows GUI rerun pending CI. --- scripts/test-windows-native-gui.ps1 | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/scripts/test-windows-native-gui.ps1 b/scripts/test-windows-native-gui.ps1 index f7c5d312..a271e9f5 100644 --- a/scripts/test-windows-native-gui.ps1 +++ b/scripts/test-windows-native-gui.ps1 @@ -426,7 +426,7 @@ function Get-ObservedRunControl([ValidateSet('Start','Stop')][string]$Action) { # Tk widgets expose no accessible name (verified: every descendant is an unnamed UIA Pane and only # the TkTopLevel carries window text). The guided client packs the run controls as three exact # native child HWNDs inside one row container (client/windows_gui.py: 'Start benchmark (Alt+B)', - # 'Stop (Alt+S)', 'Retry Queued Uploads'), so the controls are reobserved from that live Win32 + # 'Stop (Alt+S)', 'Retry due uploads'), so the controls are reobserved from that live Win32 # structure: the unique row at least half the root width whose visible children are exactly # three, share one class and one height tightly equal to the row height, are ordered left to # right flush with the row's left edge with gaps <=32px, and the first button is wider than the @@ -476,10 +476,11 @@ function Get-ObservedRunControl([ValidateSet('Start','Stop')][string]$Action) { } function Get-ObservedModeControl { # The mode selector has no accessible name either, so it is identified structurally: the unique - # row at least half the root width whose visible same-class children are exactly seven controls + # row at least half the root width whose visible same-class children are exactly three controls # in one non-overlapping left-to-right line flush with the row's left edge (client/windows_gui.py - # row1: Mode label, Mode combobox, No-submit checkbutton, Retries label, Retries spinbox, Batch - # label, Batch spinbox; CI 35651286705 launch.win32.json row at y=78). The combobox is the + # row1: Mode label, Mode combobox, Save locally checkbutton; CI 35976723286 launch.win32.json + # row at y=78). The short label followed by the wider combobox distinguishes this row from + # the three-button Start/Stop row and the saved-work row. The combobox is the # second control from the left. The guarded click that follows must post an aligned owned popup # before any value-changing key is sent, so a structurally stale identity can never commit. $tree=Get-OwnedClientTree @@ -491,11 +492,14 @@ function Get-ObservedModeControl { $pRect=$candidate.rect if (($pRect.Right-$pRect.Left) -lt [Math]::Floor($rootWidth*0.5)) { continue } $kids=@($byHandle.Values | Where-Object { $_.parent -eq $key -and $_.visible }) - if ($kids.Count -ne 7) { continue } + if ($kids.Count -ne 3) { continue } $classes=@($kids | ForEach-Object { $_.class } | Select-Object -Unique) if ($classes.Count -ne 1) { continue } $ordered=@($kids | Sort-Object { $_.rect.Left }) if ($ordered[0].rect.Left -ne $pRect.Left) { continue } + $labelWidth=$ordered[0].rect.Right-$ordered[0].rect.Left + $comboWidth=$ordered[1].rect.Right-$ordered[1].rect.Left + if ($labelWidth -lt 16 -or $labelWidth -gt 64 -or $comboWidth -lt 64 -or $comboWidth -gt 400 -or $labelWidth -ge $comboWidth) { continue } $inside=$true foreach ($kid in $ordered) { $r=$kid.rect @@ -508,7 +512,7 @@ function Get-ObservedModeControl { if (-not $inside) { continue } $rows+=,@{ row=$candidate; ordered=$ordered } } - if ($rows.Count -ne 1) { throw "BLOCKED_GUI_POINT: observed $($rows.Count) candidate configuration rows on this fresh instance; the unique seven-control mode row is not established." } + if ($rows.Count -ne 1) { throw "BLOCKED_GUI_POINT: observed $($rows.Count) candidate configuration rows on this fresh instance; the unique three-control mode row is not established." } $combo=$rows[0].ordered[1] $rect=$combo.rect $width=$rect.Right-$rect.Left; $height=$rect.Bottom-$rect.Top From fdb7669ba0360a4a1e62a1e11357cf0dd9092fc1 Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 02:22:45 -0700 Subject: [PATCH 7/9] Keep Windows harness guards aligned with the guided mode row The native selector now recognizes the current three-control guided row, but its pure operator fixtures still supplied the removed retry and batch controls. Refresh the hosted geometry and negative cases so the guard tests exercise the same structural contract before a native build. Constraint: PowerShell is unavailable on the local macOS host; Windows CI executes the guard suite. Confidence: medium Scope-risk: narrow Tested: Reviewed captured Win32 bounds against fixture; git diff --check. Not-tested: Windows harness guard rerun pending CI. --- scripts/test-windows-gui-harness-guards.ps1 | 32 ++++++++------------- 1 file changed, 12 insertions(+), 20 deletions(-) diff --git a/scripts/test-windows-gui-harness-guards.ps1 b/scripts/test-windows-gui-harness-guards.ps1 index 529c64b1..0331808a 100644 --- a/scripts/test-windows-gui-harness-guards.ps1 +++ b/scripts/test-windows-gui-harness-guards.ps1 @@ -87,9 +87,9 @@ $script:events=@() function Record-Event([string]$Kind,$Data) { $script:events+=@{kind=$Kind;data=$Data} } function Capture-Ui([string]$Label) { return @() } function Run-Fixture { - # Exact hosted-runner geometry from CI 35651286705 prepare-stop dumps: the real three-button - # run row (Start 99px, Stop 76px, Retry Queued Uploads 128px), the real seven-control mode - # row, plus the decoy log frame whose Text child and ScrollBar child match every purely + # Hosted-runner geometry: the three-button run row (Start, Stop, Retry due uploads) and the + # current three-control guided mode row from CI 35976723286 prepare-stop dumps, plus the + # decoy log frame whose Text child and ScrollBar child match every purely # geometric rule except child count and class. [void][EdbWindows]::Data.Clear() Run-Window 1 0 @(8,8,1016,703) 'TkTopLevel'; $t=[EdbWindows]::Data[1];$t.Text='EncodingDB Windows Client' @@ -103,11 +103,7 @@ function Run-Fixture { Run-Window 40 1 @(40,78,984,99) 'TkChild' Run-Window 41 40 @(40,79,75,98) 'TkChild' Run-Window 42 40 @(83,78,214,99) 'TkChild' - Run-Window 43 40 @(230,78,412,99) 'TkChild' - Run-Window 44 40 @(424,79,463,98) 'TkChild' - Run-Window 45 40 @(469,78,524,98) 'TkChild' - Run-Window 46 40 @(536,79,592,98) 'TkChild' - Run-Window 47 40 @(598,78,653,98) 'TkChild' + Run-Window 43 40 @(230,78,385,99) 'TkChild' } Case 'run-row:real-identity-with-hosted-decoys' Run-Fixture @@ -135,10 +131,10 @@ Run-Fixture;[EdbWindows]::Data[12].Class='TkChild';[EdbWindows]::Data[11].Bounds Run-Window 13 10 @(468,490,700,673) 'TkChild' Assert-Throws {Get-ObservedRunControl 'Start'} '*BLOCKED_GUI_POINT*observed 2 candidate Start/Stop rows*not established*' Case 'run-row:mode-row-uniform-height-never-selected' -# Flattening every mode-row child to the mode row's uniform height must not let the seven-control +# Flattening every mode-row child to the mode row's uniform height must not let the three-control # row impersonate the run row nor displace the real Start/Stop identity. Run-Fixture -foreach ($id in @(41,43,44,45,46,47)) { $row1=[EdbWindows]::Data[$id]; $row1.Bounds=@($row1.Bounds[0],78,$row1.Bounds[2],99) } +foreach ($id in @(41,43)) { $row1=[EdbWindows]::Data[$id]; $row1.Bounds=@($row1.Bounds[0],78,$row1.Bounds[2],99) } $obs=Get-ObservedRunControl 'Start' Assert-True ($obs.child -eq [IntPtr]21 -and $obs.x -eq 89 -and $obs.y -eq 179) 'The uniform-height mode row impersonated the run control.' Case 'run-row:reject-first-not-wider' @@ -156,27 +152,23 @@ Assert-Throws {Get-ObservedRunControl 'Start'} '*BLOCKED_GUI_POINT*not establish Case 'mode-row:real-identity-after-run-row-rejection' $mode=Get-ObservedModeControl Assert-True ($mode.child -eq [IntPtr]42 -and $mode.owner -eq 42 -and $mode.x -eq 148 -and $mode.y -eq 88) 'Real mode observer picked the wrong mode combobox on the hosted geometry.' -Case 'mode-row:reject-eight-control-row' -Run-Fixture;Run-Window 48 40 @(660,78,715,98) 'TkChild' +Case 'mode-row:reject-four-control-row' +Run-Fixture;Run-Window 48 40 @(401,78,456,98) 'TkChild' Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-mixed-class-children' -Run-Fixture;$mflipped=[EdbWindows]::Data[44];$mflipped.Class='ScrollBar' +Run-Fixture;$mflipped=[EdbWindows]::Data[43];$mflipped.Class='ScrollBar' Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-overlapping-children' Run-Fixture;$mover=[EdbWindows]::Data[43];$mover.Bounds=@(200,78,412,99) Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*configuration rows*not established*' Case 'mode-row:reject-undersized-combobox' Run-Fixture;$msmall=[EdbWindows]::Data[42];$msmall.Bounds=@(83,78,110,99) -Assert-Throws {Get-ObservedModeControl} '*unexpected size*' -Case 'mode-row:reject-ambiguous-second-seven-control-row' +Assert-Throws {Get-ObservedModeControl} '*BLOCKED_GUI_POINT*not established*' +Case 'mode-row:reject-ambiguous-second-three-control-row' Run-Fixture;Run-Window 50 1 @(40,216,984,237) 'TkChild' Run-Window 51 50 @(40,217,75,236) 'TkChild' Run-Window 52 50 @(83,216,214,237) 'TkChild' -Run-Window 53 50 @(230,216,412,237) 'TkChild' -Run-Window 54 50 @(424,217,463,236) 'TkChild' -Run-Window 55 50 @(469,216,524,236) 'TkChild' -Run-Window 56 50 @(536,217,592,236) 'TkChild' -Run-Window 57 50 @(598,216,653,236) 'TkChild' +Run-Window 53 50 @(230,216,385,237) 'TkChild' Assert-Throws {Get-ObservedModeControl} '*observed 2 candidate configuration rows*' function Mode-Fixture { Run-Fixture From 95b1fc4b13960d52bd1d23b66886e12893804f5f Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 02:54:48 -0700 Subject: [PATCH 8/9] Exercise the guided download decision in native GUI acceptance The packaged GUI now shows a transfer and storage estimate before source preparation. The Windows harness clicked Start but never answered that dialog, so it timed out waiting for the preparation probe. Verify the owned native dialog, visible cost disclosure and exact Yes/No controls before a bounded Yes click; preserve a path for runs with no transfer prompt. Add operator fixtures for accepted and rejected dialog identities. Constraint: Native Windows UIA exposes the message box as #32770 with direct Button and Static children. Rejected: Suppress the estimate prompt in test mode | would skip the contributor decision being certified. Confidence: medium Scope-risk: narrow Directive: Keep consent prompt automation tied to captured native title, text and control IDs. Tested: Captured Windows screenshot, UIA and Win32 evidence matched the guarded predicate; git diff --check. Not-tested: Native GUI rerun pending Windows CI. --- scripts/test-windows-gui-harness-guards.ps1 | 35 +++++++++++- scripts/test-windows-native-gui.ps1 | 59 +++++++++++++++++++-- 2 files changed, 90 insertions(+), 4 deletions(-) diff --git a/scripts/test-windows-gui-harness-guards.ps1 b/scripts/test-windows-gui-harness-guards.ps1 index 0331808a..ca7dbb41 100644 --- a/scripts/test-windows-gui-harness-guards.ps1 +++ b/scripts/test-windows-gui-harness-guards.ps1 @@ -9,7 +9,7 @@ $ErrorActionPreference='Stop' $tokens=$null; $parseErrors=$null $ast=[Management.Automation.Language.Parser]::ParseFile($Harness,[ref]$tokens,[ref]$parseErrors) if ($parseErrors.Count) { throw ($parseErrors | Out-String) } -foreach ($name in @('Get-OwnedExitConfirmation','Record-CleanupFailure','Get-OwnedClientTree','Get-ObservedRunControl','Get-ObservedModeControl','Set-OwnedForeground','Select-AdvancedSingleMode','Invoke-RunAction','Wait-Until','Observe-Processes','Get-CompletionMarkers','Wait-Encoder','Wait-NoEncoders','Get-ActiveEncodeEvidence','Get-DurableMeasuredAttempts','Wait-MeasuredThenActiveEncode')) { +foreach ($name in @('Get-OwnedExitConfirmation','Get-OwnedDownloadEstimateConfirmation','Confirm-ObservedDownloadEstimate','Record-CleanupFailure','Get-OwnedClientTree','Get-ObservedRunControl','Get-ObservedModeControl','Set-OwnedForeground','Select-AdvancedSingleMode','Invoke-RunAction','Wait-Until','Observe-Processes','Get-CompletionMarkers','Wait-Encoder','Wait-NoEncoders','Get-ActiveEncodeEvidence','Get-DurableMeasuredAttempts','Wait-MeasuredThenActiveEncode')) { $definitions=@($ast.FindAll({param($node) $node -is [Management.Automation.Language.FunctionDefinitionAst] -and $node.Name -eq $name},$true)) if ($definitions.Count -ne 1) { throw "Expected one real $name definition." } Invoke-Expression $definitions[0].Extent.Text @@ -28,6 +28,8 @@ public static class EdbWindows { public static long ObservedFocus=0; public static string FocusFailure=null; public static bool SetForegroundWindow(IntPtr h){if(ActivateSucceeds){Foreground=h;return true;}return false;} public static IntPtr GetForegroundWindow(){return Foreground;} + public static long LastMessageTarget=0; public static uint LastMessage=0; + public static IntPtr SendMessageTimeout(IntPtr h,uint m,IntPtr w,IntPtr l,uint flags,uint timeout,out UIntPtr result){LastMessageTarget=h.ToInt64();LastMessage=m;result=new UIntPtr(1);return Data.ContainsKey(h.ToInt64())?new IntPtr(1):IntPtr.Zero;} public static bool GetWindowRect(IntPtr h, out Rect r){ var b=Data[h.ToInt64()].Bounds; r=new Rect{Left=b[0],Top=b[1],Right=b[2],Bottom=b[3]}; return true; } public static IntPtr SetThreadDpiAwarenessContext(IntPtr c){ return new IntPtr(-1); } public static bool ShowWindow(IntPtr h,int mode){return true;} @@ -269,6 +271,37 @@ Assert-Throws {Get-OwnedExitConfirmation} '*multiple owned native Exit dialogs*' Case 'exit-confirmation:reject-two-yes-buttons' Ready-Fixture;[EdbWindows]::Data.Add(3,[EdbWindows]::Data[2]) Assert-Throws {Get-OwnedExitConfirmation} '*multiple observed enabled native Yes buttons*' +function Download-Estimate-Fixture { + [EdbWindows]::Data.Clear() + Run-Window 1 0 @(8,8,1016,703) 'TkTopLevel';[EdbWindows]::Data[1].Text='EncodingDB Windows Client' + Run-Window 2 0 @(317,310,722,469) '#32770';[EdbWindows]::Data[2].Text='Download and storage estimate' + Run-Window 3 2 @(541,429,616,452) 'Button';[EdbWindows]::Data[3].Text='&Yes';[EdbWindows]::Data[3].Control=6 + Run-Window 4 2 @(624,429,699,452) 'Button';[EdbWindows]::Data[4].Text='&No';[EdbWindows]::Data[4].Control=7 + Run-Window 5 2 @(387,367,686,395) 'Static';[EdbWindows]::Data[5].Text='This run may download 1.4 GB of frozen reference media. Estimated peak extra storage: 2.9 GB. Continue?' + [EdbWindows]::ActivateSucceeds=$true;[EdbWindows]::Foreground=[IntPtr]::Zero + [EdbWindows]::LastMessageTarget=0;[EdbWindows]::LastMessage=0 + $script:harnessDeadline=[DateTime]::UtcNow.AddSeconds(30) + $script:events=@() +} +Case 'download-estimate:exact-native-dialog-and-bounded-yes' +Download-Estimate-Fixture +$estimate=Get-OwnedDownloadEstimateConfirmation +Assert-True ($estimate.dialog -eq [IntPtr]2 -and $estimate.button -eq [IntPtr]3 -and $estimate.owner -eq 42) 'The estimate observer did not identify the owned Yes button.' +Confirm-ObservedDownloadEstimate +Assert-True ([EdbWindows]::LastMessageTarget -eq 3 -and [EdbWindows]::LastMessage -eq 0x00F5) 'The observed estimate Yes button did not receive BM_CLICK.' +Assert-True (@($script:events | Where-Object { $_.kind -eq 'observed-download-estimate-approved' }).Count -eq 1) 'Estimate approval evidence was not recorded.' +Case 'download-estimate:reject-unexpected-title' +Download-Estimate-Fixture;[EdbWindows]::Data[2].Text='Allow Benchmark Publication' +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*unexpected owned dialog*' +Case 'download-estimate:reject-missing-cost-disclosure' +Download-Estimate-Fixture;[EdbWindows]::Data[5].Text='Continue?' +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*lacks one visible Yes/No pair and the expected cost disclosure*' +Case 'download-estimate:reject-wrong-button-identity' +Download-Estimate-Fixture;[EdbWindows]::Data[3].Control=7 +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*lacks one visible Yes/No pair and the expected cost disclosure*' +Case 'download-estimate:reject-foreign-owner' +Download-Estimate-Fixture;[EdbWindows]::Data[5].Owner=99 +Assert-Throws {Get-OwnedDownloadEstimateConfirmation} '*owner differs from the client*' Case 'cleanup:primary-error-and-owned-identity-preserved' $script:events=@() $receipt=@{status='BLOCKED';error='original failure';primaryError=@{message='original failure';stage='close readiness'};cleanupErrors=@()} diff --git a/scripts/test-windows-native-gui.ps1 b/scripts/test-windows-native-gui.ps1 index a271e9f5..d3a33bff 100644 --- a/scripts/test-windows-native-gui.ps1 +++ b/scripts/test-windows-native-gui.ps1 @@ -396,6 +396,46 @@ function Confirm-ObservedExit { if ($sent -eq [IntPtr]::Zero) { throw 'BLOCKED_GUI_AUTOMATION: native Yes button did not accept the bounded click.' } Record-Event 'observed-native-button-clicked' @{ name='Yes'; processId=$buttonOwner; dialogHandle=$dialog.ToInt64(); buttonHandle=$button.ToInt64(); controlId=6 } } +function Get-OwnedDownloadEstimateConfirmation { + # The guided client asks before extracting the frozen media. Accept only this exact native + # Yes/No dialog, with its visible cost disclosure, from the owned packaged process. + $roots=@(Get-OwnedWindows | Where-Object { [EdbWindows]::Text($_) -eq 'EncodingDB Windows Client' }) + if ($roots.Count -ne 1) { throw 'BLOCKED_GUI_AUTOMATION: expected one owned client window before download consent.' } + $dialogs=@(Get-OwnedWindows | Where-Object { $_ -ne $roots[0] }) + if ($dialogs.Count -eq 0) { return $null } + if ($dialogs.Count -ne 1) { throw 'BLOCKED_GUI_AUTOMATION: multiple owned dialogs appeared before download consent.' } + $dialog=$dialogs[0] + if ([EdbWindows]::Text($dialog) -ne 'Download and storage estimate' -or [EdbWindows]::Class($dialog) -ne '#32770') { + throw 'BLOCKED_GUI_AUTOMATION: unexpected owned dialog before download consent.' + } + $children=@([EdbWindows]::Windows($dialog) | Where-Object { [EdbWindows]::GetParent($_) -eq $dialog }) + $yes=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Button' -and [EdbWindows]::Text($_).Replace('&','') -eq 'Yes' -and [EdbWindows]::GetDlgCtrlID($_) -eq 6 -and [EdbWindows]::IsWindowVisible($_) -and [EdbWindows]::IsWindowEnabled($_) }) + $no=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Button' -and [EdbWindows]::Text($_).Replace('&','') -eq 'No' -and [EdbWindows]::GetDlgCtrlID($_) -eq 7 -and [EdbWindows]::IsWindowVisible($_) -and [EdbWindows]::IsWindowEnabled($_) }) + $disclosure=@($children | Where-Object { [EdbWindows]::Class($_) -eq 'Static' -and [EdbWindows]::Text($_) -match '^This run may download .+ of frozen reference media\. Estimated peak extra storage: .+\. Continue\?$' }) + if ($yes.Count -ne 1 -or $no.Count -ne 1 -or $disclosure.Count -ne 1) { + throw 'BLOCKED_GUI_AUTOMATION: download estimate lacks one visible Yes/No pair and the expected cost disclosure.' + } + [uint32]$rootOwner=0; [void][EdbWindows]::GetWindowThreadProcessId($roots[0],[ref]$rootOwner) + foreach ($handle in @($dialog,$yes[0],$no[0],$disclosure[0])) { + [uint32]$owner=0; [void][EdbWindows]::GetWindowThreadProcessId($handle,[ref]$owner) + if ($owner -ne $rootOwner) { throw 'BLOCKED_GUI_AUTOMATION: download estimate control owner differs from the client.' } + } + return @{ dialog=$dialog; button=$yes[0]; owner=$rootOwner } +} +function Confirm-ObservedDownloadEstimate { + $script:operationStage='start:capture-download-estimate' + [void](Capture-Ui 'before-download-estimate-Yes') + $confirmation=Get-OwnedDownloadEstimateConfirmation + if ($null -eq $confirmation) { throw 'BLOCKED_GUI_AUTOMATION: owned download estimate is no longer ready.' } + $dialog=$confirmation.dialog; $button=$confirmation.button + [void][EdbWindows]::SetForegroundWindow($dialog) + Wait-Until { return [EdbWindows]::GetForegroundWindow() -eq $dialog } 5 'BLOCKED_GUI_FOCUS: download estimate did not receive foreground focus.' + [UIntPtr]$result=[UIntPtr]::Zero + $script:operationStage='start:click-download-estimate-yes' + $sent=[EdbWindows]::SendMessageTimeout($button,0x00F5,[IntPtr]::Zero,[IntPtr]::Zero,2,2000,[ref]$result) + if ($sent -eq [IntPtr]::Zero) { throw 'BLOCKED_GUI_AUTOMATION: download estimate Yes button did not accept the bounded click.' } + Record-Event 'observed-download-estimate-approved' @{ processId=$confirmation.owner; dialogHandle=$dialog.ToInt64(); buttonHandle=$button.ToInt64(); controlId=6 } +} function Get-OwnedClientTree { # Enumerate the exact owned client window and every descendant child HWND once, in physical # pixels, so row-structure predicates and click coordinates share one observation space. @@ -431,9 +471,10 @@ function Get-ObservedRunControl([ValidateSet('Start','Stop')][string]$Action) { # three, share one class and one height tightly equal to the row height, are ordered left to # right flush with the row's left edge with gaps <=32px, and the first button is wider than the # second. Start is the first; Stop is the second. Every other row fails a predicate on the - # hosted runner (CI 35651286705 failure.win32.json: the log frame holds only a Text child and a - # ScrollBar child, mixing classes; configuration rows hold 2, 7 or 8 children or mixed child - # heights). The former two-button contract blocked this phase at run35651286705. + # hosted runner (CI 35976723286 launch.win32.json: the log frame mixes a Text child and a + # ScrollBar child; the current three-control mode row has a narrow first label, and other + # configuration rows fail child count or height). The former two-button contract blocked + # this phase at run35651286705. $tree=Get-OwnedClientTree $rootRect=$tree.rootRect; $byHandle=$tree.byHandle $rootWidth=$rootRect.Right-$rootRect.Left @@ -812,6 +853,18 @@ try { # default; deliberately select Single (advanced) first or Start would run a sweep. Select-AdvancedSingleMode Invoke-RunAction 'Start' + $script:downloadEstimateStatus=$null + Wait-Until { + $estimate=Get-OwnedDownloadEstimateConfirmation + if ($null -ne $estimate) { $script:downloadEstimateStatus='prompt'; return $true } + try { + $stop=Get-ObservedRunControl 'Stop' + if ($stop.childEnabled) { $script:downloadEstimateStatus='already-running'; return $true } + } catch { } + return $false + } 20 'BLOCKED_GUI_AUTOMATION: neither a download estimate nor an active run followed Start.' + if ($script:downloadEstimateStatus -eq 'prompt') { Confirm-ObservedDownloadEstimate } + else { Record-Event 'download-estimate-not-shown' @{ phase=$name } } if ($name -eq 'prepare-stop') { Wait-Until { [void](Observe-Processes); return $null -ne $phase.preparationProbe } $AcquisitionSeconds 'Source preparation probe was not observed.' if (@(Get-ChildItem -Path $phase.queue -Recurse -Filter 'manifest.json' -ErrorAction SilentlyContinue).Count) { throw 'Campaign already exists; preparation cancellation was not exercised.' } From fe94dc167abbf1d222a23bbcda072c67326545ee Mon Sep 17 00:00:00 2001 From: ofhd Date: Thu, 24 Sep 2026 03:21:40 -0700 Subject: [PATCH 9/9] Require real progress before bypassing the guided download prompt The native Windows GUI showed the cost dialog, but Tk exposed Stop as Win32-enabled even while its visual state was disabled. The harness treated that flag as proof Start had advanced and never answered the prompt. Prefer the exact owned dialog; only a source preparation probe or owned encode can establish that no prompt was needed. Constraint: ttk widget state does not reliably map to IsWindowEnabled on hosted Windows. Rejected: Treat the Stop handle as readiness | the captured modal case disproves that signal. Confidence: medium Scope-risk: narrow Tested: Captured failing receipt showed download-estimate-not-shown with the dialog visible; git diff --check. Not-tested: Native rerun pending Windows CI. --- scripts/test-windows-native-gui.ps1 | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/scripts/test-windows-native-gui.ps1 b/scripts/test-windows-native-gui.ps1 index d3a33bff..d3a1457c 100644 --- a/scripts/test-windows-native-gui.ps1 +++ b/scripts/test-windows-native-gui.ps1 @@ -857,10 +857,14 @@ try { Wait-Until { $estimate=Get-OwnedDownloadEstimateConfirmation if ($null -ne $estimate) { $script:downloadEstimateStatus='prompt'; return $true } - try { - $stop=Get-ObservedRunControl 'Stop' - if ($stop.childEnabled) { $script:downloadEstimateStatus='already-running'; return $true } - } catch { } + # ttk's disabled visual state is not Win32 IsWindowEnabled. A grey Stop button + # can report enabled while this modal blocks the Tk callback. Only actual + # preparation or an owned encoder proves that Start passed the decision. + [void](Observe-Processes) + if ($null -ne $phase.preparationProbe) { $script:downloadEstimateStatus='already-running'; return $true } + $encode=Get-ActiveEncodeEvidence + if ($encode.identified.Count -or $encode.unidentified.Count) { $script:downloadEstimateStatus='already-running'; return $true } + if ($script:process.HasExited) { throw 'BLOCKED_GUI_AUTOMATION: client exited before the download decision or source preparation.' } return $false } 20 'BLOCKED_GUI_AUTOMATION: neither a download estimate nor an active run followed Start.' if ($script:downloadEstimateStatus -eq 'prompt') { Confirm-ObservedDownloadEstimate }