diff --git a/crates/cellule-peer-http/docs/routing.md b/crates/cellule-peer-http/docs/routing.md index 99a21b3f..6ef8aea9 100644 --- a/crates/cellule-peer-http/docs/routing.md +++ b/crates/cellule-peer-http/docs/routing.md @@ -254,8 +254,11 @@ same frozen binary. Artifacts retain the coordination trace, revisions, digests, every raw latency sample, per-command publication phases, object-read/hop counts, and exact recovery results. CI uses 1,024 commands per write lane, verifies 6,144 unique durable mutations per run, and checks every -publication phase record. The gate requires median-run p95/p99 within 10% and completed-call -throughput within 10%; paced throughput is excluded because it includes sleep. +publication phase record. The gate compares medians of all four runs: p95 must +be at most 150% of baseline, p99 at most 200%, and completed-call throughput +at least 90%; paced throughput is excluded because it includes sleep. +The manifest and comparison record these limits. Both latency percentiles +still produce review alerts above 110% of baseline, with all raw samples retained. Historical unleased zero-read routing is reported but cannot qualify a fresh authority latency target. @@ -272,9 +275,15 @@ physical read and hop checks, and exact recovery. The final `routing` check requires both mode jobs to pass. `routing.py --mode leased` or `--mode object_only` runs one complete mode; omitting `--mode` runs both. -The larger CI sample sizes and adjacent pairs retain the original 10% limits. -Serial full-profile comparisons still failed calibration after increasing the -sample sizes, including paced local p99 at 1.13–1.15. Adjacent windows control +The shared-runner blocking latency limits were widened after an identical frozen +binary control with the complete CI profiles reported p95 at 1.31 times and p99 +at 1.84 times baseline ([control run](https://github.com/crabbuild/cellule/actions/runs/36908854812)). +These limits catch large latency regressions; a passing job does not establish +latency equivalence within 10% or a production SLO. Throughput, workloads, +read/hop counts, authority checks and recovery requirements remain unchanged. + +Serial full-profile comparisons failed the earlier 10% calibration after +increasing the sample sizes, including paced local p99 at 1.13–1.15. Adjacent windows control time drift without removing lanes or excluding slow calls. An unchanged-source calibration with 128 commands and 12 bursts failed three gates, with paced tail ratios up to 1.22 and serial-write p95 at 1.20. Manual measurements default to diff --git a/crates/cellule-peer-http/qualification/routing.py b/crates/cellule-peer-http/qualification/routing.py index f8a095c0..f95c01f2 100644 --- a/crates/cellule-peer-http/qualification/routing.py +++ b/crates/cellule-peer-http/qualification/routing.py @@ -22,6 +22,10 @@ QUERIES = 4096 COMMANDS = 1024 PACED_BURSTS = 48 +# Shared-runner identical-binary controls reached 1.31x p95 and 1.84x p99. +# Keep a coarse blocking gate and retain 10% latency alerts for review. +GATE_LIMITS = {"p95_max_ratio": 1.50, "p99_max_ratio": 2.00, "throughput_min_ratio": 0.90} +LATENCY_ALERT_RATIO = 1.10 PAIRS = (("baseline", "candidate"), ("candidate", "baseline"), ("candidate", "baseline"), ("baseline", "candidate")) ORDER = tuple(version for pair in PAIRS for version in pair) @@ -221,6 +225,7 @@ def parse_measurement(version, mode, index, evidence): def compare(rows, modes=None): comparisons = [] failures = [] + latency_alerts = [] modes = tuple(SELECTORS) if modes is None else tuple(modes) if not modes or not set(modes) <= set(SELECTORS): raise RuntimeError("Invalid routing modes") @@ -251,13 +256,20 @@ def compare(rows, modes=None): "local_command", "forwarded_command", "local_query_expired_bursts", "forwarded_query_expired_bursts", "forwarded_query_uncached_route", ) - if gated and (ratios["p95_ms"] > 1.10 or ratios["p99_ms"] > 1.10 - or ("expired_bursts" not in lane and ratios["throughput"] < 0.90)): + if gated and (ratios["p95_ms"] > LATENCY_ALERT_RATIO + or ratios["p99_ms"] > LATENCY_ALERT_RATIO): + latency_alerts.append(f"{mode}/{lane}/c{int(concurrency)}") + if gated and (ratios["p95_ms"] > GATE_LIMITS["p95_max_ratio"] + or ratios["p99_ms"] > GATE_LIMITS["p99_max_ratio"] + or ("expired_bursts" not in lane + and ratios["throughput"] < GATE_LIMITS["throughput_min_ratio"])): failures.append(f"{mode}/{lane}/c{int(concurrency)}") comparisons.append(dict(mode=mode, lane=lane, concurrency=concurrency, medians=medians, ratios=ratios, latency_gate=gated)) return {"comparisons": comparisons, "failures": failures, - "threshold": "median of all four runs: p95/p99 <= 110%, throughput >= 90%"} + "gate_limits": GATE_LIMITS, "latency_alert_ratio": LATENCY_ALERT_RATIO, + "latency_alerts": latency_alerts, + "threshold": "median of all four runs: p95 <= 150%, p99 <= 200%, throughput >= 90%"} def main(): @@ -277,6 +289,7 @@ def main(): "candidate": output("git", "rev-parse", "HEAD")} manifest = {"order": ORDER, "pairs": PAIRS, "queries": QUERIES, "commands_per_lane": COMMANDS, "paced_bursts": PACED_BURSTS, "modes": modes, + "gate_limits": GATE_LIMITS, "latency_alert_ratio": LATENCY_ALERT_RATIO, "selectors": SELECTORS, "host": platform.uname()._asdict(), "test_only_transplant": [str(PEER / path) for path in ( "src/performance_tests.rs", "src/lib.rs", "src/tests.rs", "Cargo.toml")] + ["Cargo.lock (peer test dependencies only)"], diff --git a/crates/cellule-peer-http/qualification/test_routing.py b/crates/cellule-peer-http/qualification/test_routing.py index 52e6eac9..4166bc84 100644 --- a/crates/cellule-peer-http/qualification/test_routing.py +++ b/crates/cellule-peer-http/qualification/test_routing.py @@ -18,7 +18,7 @@ def test_identical_complete_comparison_passes(self): self.assertEqual(routing.compare(measurements())["failures"], []) def test_each_latency_and_throughput_regression_fails(self): - for metric, value in (("throughput", 899), ("p95_ms", 2.21), ("p99_ms", 3.31)): + for metric, value in (("throughput", 899), ("p95_ms", 3.01), ("p99_ms", 6.01)): with self.subTest(metric=metric): rows = measurements() for row in rows: @@ -26,6 +26,30 @@ def test_each_latency_and_throughput_regression_fails(self): row[metric] = value self.assertEqual(routing.compare(rows)["failures"], ["leased/local_query/c16"]) + def test_latency_limits_are_inclusive_and_keep_review_alerts(self): + rows = measurements() + for row in rows: + if row["mode"] == "leased" and row["version"] == "candidate": + row["p95_ms"] = 3.0 + row["p99_ms"] = 6.0 + row["throughput"] = 900.0 + result = routing.compare(rows) + self.assertEqual(result["failures"], []) + self.assertEqual(result["latency_alerts"], ["leased/local_query/c16"]) + self.assertEqual(result["gate_limits"], + dict(p95_max_ratio=1.5, p99_max_ratio=2.0, throughput_min_ratio=0.9)) + self.assertEqual(result["latency_alert_ratio"], 1.1) + + def test_old_latency_exceedances_remain_visible(self): + rows = measurements() + for row in rows: + if row["mode"] == "object_only" and row["version"] == "candidate": + row["p95_ms"] = 2.21 + row["p99_ms"] = 3.31 + result = routing.compare(rows) + self.assertEqual(result["failures"], []) + self.assertEqual(result["latency_alerts"], ["object_only/local_query/c16"]) + def test_missing_repeat_is_rejected(self): rows = measurements() rows.pop() @@ -42,7 +66,7 @@ def test_each_partition_retains_repeats_and_regression_gates(self): with self.subTest(mode=mode): rows = [row for row in measurements() if row["mode"] == mode] self.assertEqual(routing.compare(rows, (mode,))["failures"], []) - for metric, value in (("throughput", 899), ("p95_ms", 2.21), ("p99_ms", 3.31)): + for metric, value in (("throughput", 899), ("p95_ms", 3.01), ("p99_ms", 6.01)): regressed = copy.deepcopy(rows) for row in regressed: if row["version"] == "candidate": @@ -64,14 +88,14 @@ def test_cold_forwarded_discovery_regression_fails(self): for row in rows: row["lane"] = "forwarded_query_uncached_route" if row["mode"] == "leased" and row["version"] == "candidate": - row["p99_ms"] = 4.0 + row["p99_ms"] = 6.01 self.assertEqual(routing.compare(rows)["failures"], ["leased/forwarded_query_uncached_route/c16"]) def test_one_outlier_cannot_hide_a_regressed_majority(self): rows = measurements() for row in rows: if row["mode"] == "leased" and row["version"] == "candidate": - row["p99_ms"] = 1.0 if row["run"] == 0 else 4.0 + row["p99_ms"] = 1.0 if row["run"] == 0 else 6.01 self.assertEqual(routing.compare(rows)["failures"], ["leased/local_query/c16"]) def test_historical_zero_read_unleased_path_is_not_latency_target(self):