From c7f71d32e673e80736c0b9a745fc37ffbda16109 Mon Sep 17 00:00:00 2001 From: Lee Yunjin Date: Mon, 5 Oct 2026 20:09:18 +0900 Subject: [PATCH] ci(perf): calibrate HTTPS gates on ratios; seed history; prove #307 detection The gates from #309 failed every run (7/7). The absolute keep-alive and 1 MiB backstops came from a Ryzen 5600X and an EPYC 9V45, and healthy code on the EPYC 7763 runner lands under them (75k vs 110k keep-alive/s, 1.9 vs 2.5 GB/s). History is only recorded after a pass, so it never filled and the same-CPU check never switched on. - Gate on https/http ratios measured in the same run: churn, keep-alive, 1 MiB. They hold within 0.11-0.13 / 0.55-0.66 / 0.34-0.52 across four CPU models while the absolute numbers move ~1.8x. - Absolute backstops drop to ~60% of the slowest healthy runner: they now catch only breakage that is wrong on any hardware. - Seed benchmarks/https_gates.json with the 7 CI runs and one local run, all on healthy trees, so the EPYC 7763 same-CPU check engages on its next run. - https_perf_gates.sh: the plaintext 1 MiB number had no MB/s fallback. Regression check (Ryzen 5600X, #307's TCP_QUICKACK re-arm disabled): tls_rtt_p50_ms 0.083 -> 43.0, churn ratio 0.118 -> 0.052; the gate fails RTT, churn ratio and absolute churn. All 7 recorded CI runs pass. test_https_gates_eval.py replays both, the recorded history, and the same-CPU path. ROADMAP v3.9: (a) status updated; (c) corrected, the #294 premise correction itself was wrong (cwist_app_listen does use the shepherds; measured), #294 closed as completed; (d) done (#310). --- ROADMAP.md | 11 +-- benchmarks/https_gates.json | 124 +++++++++++++++++++++++++++- scripts/ci/https_gates_eval.py | 94 ++++++++++++++------- scripts/ci/https_perf_gates.sh | 14 ++-- scripts/ci/test_https_gates_eval.py | 81 ++++++++++++++++++ 5 files changed, 281 insertions(+), 43 deletions(-) create mode 100644 scripts/ci/test_https_gates_eval.py diff --git a/ROADMAP.md b/ROADMAP.md index 26f7ef32..937b9df0 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -512,10 +512,10 @@ also brings WebRTC DataChannel support into the release scope. | Goal | Status | Notes | |------|--------|-------| -| (a) CI TLS performance gates (HTTPS churn / keep-alive / large-transfer / RTT) | 🔄 In Progress | Tracked in #306; gate PR pending — measured separately so this roadmap and the gates PR stay decoupled | +| (a) CI TLS performance gates (HTTPS churn / keep-alive / large-transfer / RTT) | 🔄 In Progress | Gates landed in #309 but never passed: the absolute backstops came from fast CPUs and healthy code failed on the EPYC 7763 runner, so no history accumulated either. Recalibrated around hardware-normalised https/http ratios; disabling the #307 fix fails RTT (0.08 → 43 ms) and churn ratio (0.118 → 0.052). Done once green on dev. Tracked in #306 | | (b) TLS observability in Prometheus `/metrics` (handshake counts, TLS version/cipher counters, resumption vs full-handshake ratio) | ✅ Done | `cwist_tls_handshakes_total`, `cwist_tls_handshakes_resumed_total`, `cwist_tls_connections_active`, `cwist_tls_handshakes_tls12_total`, `cwist_tls_handshakes_tls13_total`, `cwist_tls_ciphers_{aes128_gcm,aes256_gcm,chacha20,other}_total`; covered by `test_https_metrics` | -| (c) Correct issue #294's premise with measured data | ✅ Done | Corrected in a #294 comment: the `cwist_app_listen` path never uses the sharded handshake shepherds (app.c multiport accept loop calls `https_pool_submit` directly, and the worker handshakes inline via `cwist_https_accept`); shepherds only serve the `async_server.c` path. Shard tuning therefore does not affect the standard app API | -| (d) WebRTC DataChannel support | 🔄 In Progress | SDP offer/answer, ICE (server lite-role acceptable at MVP), DTLS via the vendored BoringSSL, SCTP DataChannels via a new `lib/usrsctp` submodule; browser-interoperable DataChannel echo. Implementation PR follows | +| (c) Settle issue #294 with measured data | ✅ Done | #294 closed as completed. An earlier comment there claimed `cwist_app_listen` never uses the sharded handshake shepherds; that was wrong (it confused `cwist_app_multiport`'s inline path with `cwist_app_listen`). Measured: `cwist_app_listen` adds one thread per `CWIST_HTTPS_HS_SHARDS` shard on the first HTTPS request, in both C1M and classic mode. The default floor of 4 shards / cap of 16 is already on dev, and #294's data shows no gain beyond 4 | +| (d) WebRTC DataChannel support | ✅ Done | #310: SDP offer/answer, ICE-lite, DTLS (vendored BoringSSL), SCTP DataChannels (`lib/usrsctp`) on the cwist reactor. Echo verified against headless Chromium (`make test_webrtc_browser`); Firefox not yet tested | Entry criteria for v3.9: every item must move a measured metric (handshake throughput, resumption ratio, connection-churn latency, or regression @@ -528,8 +528,9 @@ Exit criteria for v3.9: re-introduced #307-class regression (measured by the churn benchmark). - `/metrics` exposes the full TLS counter set above on a live app, and the resumption-vs-full ratio is observable across prefork workers. -- #294 is closed as *not applicable to the standard app API*, with the - shepherd-path shard table recorded for the async-server path only. +- #294 is closed with measured data (done: closed as completed; the + shepherds do serve `cwist_app_listen`, and the shipped default needs no + change). --- diff --git a/benchmarks/https_gates.json b/benchmarks/https_gates.json index f44998bf..e774cdfc 100644 --- a/benchmarks/https_gates.json +++ b/benchmarks/https_gates.json @@ -1,4 +1,49 @@ [ + { + "timestamp": "2026-10-04T12:23:34+0000", + "runner_hw": "4 vCPU | AMD EPYC 9V45 96-Core Processor", + "tls_rtt_p50_ms": 0.07, + "tls_rtt_p90_ms": 0.112, + "https_churn_rps": 2013.37, + "http_churn_rps": 15793.34, + "https_keepalive_rps": 135570.26, + "http_keepalive_rps": 206650.41, + "https_keepalive_ratio": 0.656, + "https_big_gbps": 4.03, + "http_big_gbps": 7.77, + "https_churn_ratio": 0.127, + "https_big_ratio": 0.519 + }, + { + "timestamp": "2026-10-04T13:10:52+0000", + "runner_hw": "4 vCPU | AMD EPYC 9V74 80-Core Processor", + "tls_rtt_p50_ms": 0.115, + "tls_rtt_p90_ms": 0.173, + "https_churn_rps": 1178.17, + "http_churn_rps": 10590.12, + "https_keepalive_rps": 81188.63, + "http_keepalive_rps": 132775.48, + "https_keepalive_ratio": 0.611, + "https_big_gbps": 1.97, + "http_big_gbps": 5.01, + "https_churn_ratio": 0.111, + "https_big_ratio": 0.393 + }, + { + "timestamp": "2026-10-04T13:15:21+0000", + "runner_hw": "4 vCPU | AMD EPYC 7763 64-Core Processor", + "tls_rtt_p50_ms": 0.112, + "tls_rtt_p90_ms": 0.172, + "https_churn_rps": 1155.07, + "http_churn_rps": 9458.71, + "https_keepalive_rps": 75358.57, + "http_keepalive_rps": 126315.92, + "https_keepalive_ratio": 0.597, + "https_big_gbps": 1.9, + "http_big_gbps": 5.53, + "https_churn_ratio": 0.122, + "https_big_ratio": 0.344 + }, { "timestamp": "2026-10-04T21:07:18+0900", "runner_hw": "12 vCPU | AMD Ryzen 5 5600X 6-Core Processor", @@ -10,6 +55,83 @@ "http_keepalive_rps": 274146.39, "https_keepalive_ratio": 0.552, "https_big_gbps": 4.06, - "http_big_gbps": 11.88 + "http_big_gbps": 11.88, + "https_churn_ratio": 0.112, + "https_big_ratio": 0.342 + }, + { + "timestamp": "2026-10-05T09:52:29+0000", + "runner_hw": "4 vCPU | AMD EPYC 7763 64-Core Processor", + "tls_rtt_p50_ms": 0.12, + "tls_rtt_p90_ms": 0.163, + "https_churn_rps": 1196.32, + "http_churn_rps": 9796.62, + "https_keepalive_rps": 79066.14, + "http_keepalive_rps": 135111.21, + "https_keepalive_ratio": 0.585, + "https_big_gbps": 1.97, + "http_big_gbps": 5.68, + "https_churn_ratio": 0.122, + "https_big_ratio": 0.347 + }, + { + "timestamp": "2026-10-05T09:58:14+0000", + "runner_hw": "4 vCPU | AMD EPYC 7763 64-Core Processor", + "tls_rtt_p50_ms": 0.119, + "tls_rtt_p90_ms": 0.162, + "https_churn_rps": 1197.1, + "http_churn_rps": 9577.46, + "https_keepalive_rps": 76266.04, + "http_keepalive_rps": 128778.69, + "https_keepalive_ratio": 0.592, + "https_big_gbps": 1.98, + "http_big_gbps": 5.88, + "https_churn_ratio": 0.125, + "https_big_ratio": 0.337 + }, + { + "timestamp": "2026-10-05T10:00:17+0000", + "runner_hw": "4 vCPU | AMD EPYC 9V74 80-Core Processor", + "tls_rtt_p50_ms": 0.098, + "tls_rtt_p90_ms": 0.134, + "https_churn_rps": 1548.57, + "http_churn_rps": 13566.37, + "https_keepalive_rps": 106169.36, + "http_keepalive_rps": 169138.08, + "https_keepalive_ratio": 0.628, + "https_big_gbps": 3.03, + "http_big_gbps": 6.53, + "https_churn_ratio": 0.114, + "https_big_ratio": 0.464 + }, + { + "timestamp": "2026-10-05T10:05:17+0000", + "runner_hw": "4 vCPU | AMD EPYC 7763 64-Core Processor", + "tls_rtt_p50_ms": 0.126, + "tls_rtt_p90_ms": 0.164, + "https_churn_rps": 1189.89, + "http_churn_rps": 9700.11, + "https_keepalive_rps": 75342.15, + "http_keepalive_rps": 127613.19, + "https_keepalive_ratio": 0.59, + "https_big_gbps": 1.93, + "http_big_gbps": 5.59, + "https_churn_ratio": 0.123, + "https_big_ratio": 0.345 + }, + { + "timestamp": "2026-10-05T20:05:42+0900", + "runner_hw": "12 vCPU | AMD Ryzen 5 5600X 6-Core Processor", + "tls_rtt_p50_ms": 0.083, + "tls_rtt_p90_ms": 0.127, + "https_churn_rps": 1664.86, + "http_churn_rps": 14080.51, + "https_keepalive_rps": 182125.92, + "http_keepalive_rps": 278394.62, + "https_keepalive_ratio": 0.654, + "https_big_gbps": 4.2, + "http_big_gbps": 12.97, + "https_churn_ratio": 0.118, + "https_big_ratio": 0.324 } ] diff --git a/scripts/ci/https_gates_eval.py b/scripts/ci/https_gates_eval.py index 2922b26f..fbf4ba41 100644 --- a/scripts/ci/https_gates_eval.py +++ b/scripts/ci/https_gates_eval.py @@ -7,8 +7,24 @@ wide; the same-CPU check is what catches real regressions early. Throughput metrics are "min" gates (fail when below backstop / median -divided by the allowance); the RTT metric is a "max" gate. The https/http -keep-alive ratio is recorded as a diagnostic signal, not gated. +divided by the allowance); the RTT metric is a "max" gate. + +The absolute throughput numbers swing ~1.8x between the CPUs GitHub hands +out, so their backstops only catch breakage that is wrong on any hardware. +The TLS-specific gates are the https/http *ratios*: both sides are measured +on the same machine in the same run, so the hardware cancels out. Recorded +healthy range (7 CI runs on EPYC 7763 / 9V74 / 9V45, plus Ryzen 5600X): + + ratio healthy backstop pre-#307 tree (Ryzen) + churn_ratio 0.111-0.127 0.08 0.052 + keepalive_ratio 0.552-0.656 0.45 0.705 (unaffected) + big_ratio 0.337-0.519 0.25 0.300 (unaffected) + +Regression check, Ryzen 5600X, with #307's TCP_QUICKACK re-arm disabled: +tls_rtt_p50_ms 0.083 -> 43.0 and churn_ratio 0.118 -> 0.052, so both the +RTT gate and the churn-ratio gate fail it; https_churn_rps (677/s) also +falls under its backstop, but that one alone would not catch it on a fast +CPU. test_https_gates_eval.py replays both rows. Usage: https_gates_eval.py RESULT.json [HISTORY.json] # evaluate, exit 1 on fail @@ -24,30 +40,54 @@ HISTORY = Path(__file__).resolve().parent.parent.parent / "benchmarks" / "https_gates.json" # (metric, direction, absolute backstop, same-CPU allowance) -# Backstops are calibrated so a healthy tree passes on the noisiest shared -# CI runner we have seen, with real margin for run-to-run variance: -# local Ryzen 5600X (12-core): churn ~1650/s, keep-alive ~151-190k/s, -# 1MiB ~4.1GB/s (https) / ~12-14GB/s (http), RTT p50 ~0.09ms. -# shared EPYC 9V45 CI runner (run 37201724744): churn 2013/s https / -# 15793/s http, keep-alive 135570/s https / 206650/s http, -# 1MiB 4.03GB/s https / 7.8GB/s http. -# The pre-#307 tree measured ~600/s churn, ~43ms RTT p50, so these -# backstops still fail it on any hardware. The same-CPU relative check -# (runner_baseline.py convention) is what catches smaller regressions once -# enough history accumulates for a runner CPU. +# +# Absolute backstops sit at ~60% of the slowest healthy runner seen (EPYC +# 7763: churn 1155/9459/s, keep-alive 75k/126k/s, 1MiB 1.9/5.0 GB/s +# https/http). The first calibration took them from a Ryzen and an EPYC 9V45 +# run, and every run on the 7763 then failed keep-alive and 1MiB on healthy +# code (runs 37204551332, 37293341381), which also meant the history below +# never received a row and the same-CPU check never switched on. +# +# Same-CPU allowances: the four EPYC 7763 runs spread 4-7% on throughput +# and under 3% on the ratios. GATES = [ ("tls_rtt_p50_ms", "max", 5.0, 2.0), - ("https_churn_rps", "min", 1000.0, 1.25), - ("https_keepalive_rps", "min", 110000.0, 1.25), - ("https_big_gbps", "min", 2.5, 1.30), - # Plaintext controls: collateral damage to the plain path fails the gate. - ("http_churn_rps", "min", 6000.0, 1.25), - ("http_keepalive_rps", "min", 150000.0, 1.25), - ("http_big_gbps", "min", 5.5, 1.30), + # TLS-path gates, hardware-normalised (see module docstring). + ("https_churn_ratio", "min", 0.08, 1.20), + ("https_keepalive_ratio", "min", 0.45, 1.20), + ("https_big_ratio", "min", 0.25, 1.20), + # Absolute throughput: catastrophic-only backstops. + ("https_churn_rps", "min", 700.0, 1.25), + ("https_keepalive_rps", "min", 45000.0, 1.25), + ("https_big_gbps", "min", 1.2, 1.30), + # Plaintext controls: collateral damage to the plain path. + ("http_churn_rps", "min", 5500.0, 1.25), + ("http_keepalive_rps", "min", 75000.0, 1.25), + ("http_big_gbps", "min", 3.0, 1.30), +] + +# Ratios derived from each row: (name, numerator, denominator). +RATIOS = [ + ("https_churn_ratio", "https_churn_rps", "http_churn_rps"), + ("https_keepalive_ratio", "https_keepalive_rps", "http_keepalive_rps"), + ("https_big_ratio", "https_big_gbps", "http_big_gbps"), ] +def with_ratios(row): + """Copy of `row` with the https/http ratios filled in from its raw + numbers (older history rows only carry some of them).""" + out = dict(row) + for name, num, den in RATIOS: + a, b = out.get(num), out.get(den) + if isinstance(a, (int, float)) and isinstance(b, (int, float)) and b > 0: + out[name] = round(a / b, 3) + return out + + def evaluate(history, row): + row = with_ratios(row) + history = [with_ratios(r) for r in history] failures = [] details = [] for metric, direction, backstop, allowance in GATES: @@ -63,7 +103,7 @@ def evaluate(history, row): continue else: if value < backstop: - failures.append(f"{metric}: {value:.1f} under the {backstop} " + failures.append(f"{metric}: {value:.3f} under the {backstop} " f"absolute backstop") continue @@ -89,17 +129,11 @@ def evaluate(history, row): else: limit = base / allowance if value < limit: - failures.append(f"{metric}: {value:.1f} on {key} under {limit:.1f} " - f"(median {base:.1f} of {len(past)} earlier " + failures.append(f"{metric}: {value:.3f} on {key} under {limit:.3f} " + f"(median {base:.3f} of {len(past)} earlier " f"runs / {allowance})") continue details.append(f"{metric}: {value:.3f} on {key}, within same-CPU gate") - - ratio = row.get("https_keepalive_ratio") - if isinstance(ratio, (int, float)): - details.append(f"https_keepalive_ratio: {ratio:.3f} (signal only, " - f"~0.65 expected; large drops hint at TLS-path damage " - f"even when both absolute gates pass)") return failures, details @@ -120,7 +154,7 @@ def main(): if update: history = [r for r in history if r.get("timestamp") != row.get("timestamp")] - history.append(row) + history.append(with_ratios(row)) history_path.write_text(json.dumps(history[-100:], indent=2) + "\n") print(f"updated {history_path} ({len(history)} rows)") diff --git a/scripts/ci/https_perf_gates.sh b/scripts/ci/https_perf_gates.sh index 1f257147..b747c59b 100755 --- a/scripts/ci/https_perf_gates.sh +++ b/scripts/ci/https_perf_gates.sh @@ -15,13 +15,12 @@ # stall (43 ms before the fix, ~0.2 ms after). # Backstop: p50 <= 5 ms. # 2. https_churn_rps ab -n 6000 -c 32, new connection per request. -# ~600/s before #307, ~1650/s after. Backstop >= 1000. -# 3. https_keepalive_rps wrk -t4 -c100 -d10s. ~135-190k/s observed. Backstop >= 110k. -# 4. https_big_Bps wrk -t4 -c32 -d10s on a 1 MiB body. ~3.9-4.5GB/s. -# Backstop >= 2.5 GB/s. -# 5. http_* controls same workloads plaintext; detect collateral damage -# to the plain path. The https/http keep-alive ratio -# (~0.65) is recorded as a signal, not gated. +# ~680/s before #307, ~1650/s after (Ryzen). +# 3. https_keepalive_rps wrk -t4 -c100 -d10s. +# 4. https_big_gbps wrk -t4 -c32 -d10s on a 1 MiB body. +# 5. http_* controls same workloads plaintext. +# The thresholds, and the https/http ratio gates derived from these numbers, +# are in https_gates_eval.py with the measurements behind them. set -eu OUT=${1:-https-gates-result.json} @@ -174,6 +173,7 @@ HTTP_KEEP=$(wrk -t4 -c100 -d10s http://127.0.0.1:18080/ 2>/dev/null | sed -n 's/ HTTPS_BIG=$(wrk -t4 -c32 -d10s https://127.0.0.1:18443/big 2>/dev/null | sed -n 's/^Transfer\/sec:\s*\([0-9.]*\)GB.*/\1/p') [ -n "$HTTPS_BIG" ] || HTTPS_BIG=$(wrk -t4 -c32 -d10s https://127.0.0.1:18443/big 2>/dev/null | sed -n 's/^Transfer\/sec:\s*\([0-9.]*\)MB.*/0\1/p') HTTP_BIG=$(wrk -t4 -c32 -d10s http://127.0.0.1:18080/big 2>/dev/null | sed -n 's/^Transfer\/sec:\s*\([0-9.]*\)GB.*/\1/p') +[ -n "$HTTP_BIG" ] || HTTP_BIG=$(wrk -t4 -c32 -d10s http://127.0.0.1:18080/big 2>/dev/null | sed -n 's/^Transfer\/sec:\s*\([0-9.]*\)MB.*/0\1/p') RUNNER_HW="$(nproc) vCPU | $(grep -m1 'model name' /proc/cpuinfo | cut -d: -f2- | sed 's/^ *//')" diff --git a/scripts/ci/test_https_gates_eval.py b/scripts/ci/test_https_gates_eval.py new file mode 100644 index 00000000..367380df --- /dev/null +++ b/scripts/ci/test_https_gates_eval.py @@ -0,0 +1,81 @@ +import json +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +from https_gates_eval import HISTORY, evaluate, with_ratios + +RYZEN = '12 vCPU | AMD Ryzen 5 5600X 6-Core Processor' +EPYC = '4 vCPU | AMD EPYC 7763 64-Core Processor' + +# Measured on the same Ryzen 5600X, same tree, minutes apart: the second row +# with #307's TCP_QUICKACK re-arm disabled (the regression the gates exist +# to catch). +HEALTHY = { + 'timestamp': '2026-10-05T20:05:42+0900', 'runner_hw': RYZEN, + 'tls_rtt_p50_ms': 0.083, 'https_churn_rps': 1664.86, 'http_churn_rps': 14080.51, + 'https_keepalive_rps': 182125.92, 'http_keepalive_rps': 278394.62, + 'https_big_gbps': 4.2, 'http_big_gbps': 12.97, +} +PRE_307 = { + 'timestamp': '2026-10-05T20:07:12+0900', 'runner_hw': RYZEN, + 'tls_rtt_p50_ms': 43.04, 'https_churn_rps': 677.46, 'http_churn_rps': 13051.42, + 'https_keepalive_rps': 164250.42, 'http_keepalive_rps': 233103.08, + 'https_big_gbps': 4.03, 'http_big_gbps': 13.43, +} + + +def failed_metrics(failures): + return {line.split(':', 1)[0] for line in failures} + + +def epyc_row(i, churn=1180.0): + return { + 'timestamp': f'2026-10-0{i}T00:00:00+0000', 'runner_hw': EPYC, + 'tls_rtt_p50_ms': 0.12, 'https_churn_rps': churn, 'http_churn_rps': 9600.0, + 'https_keepalive_rps': 76000.0, 'http_keepalive_rps': 129000.0, + 'https_big_gbps': 1.95, 'http_big_gbps': 5.6, + } + + +class GateTests(unittest.TestCase): + def test_healthy_tree_passes(self): + failures, _ = evaluate([], HEALTHY) + self.assertEqual(failures, []) + + def test_pre_307_regression_fails_on_rtt_and_churn_ratio(self): + failures, _ = evaluate([], PRE_307) + self.assertTrue({'tls_rtt_p50_ms', 'https_churn_ratio'} <= failed_metrics(failures), + failures) + + def test_churn_ratio_alone_catches_it_when_absolute_churn_looks_fine(self): + # On a faster CPU the regressed absolute churn can clear its backstop; + # the ratio must still fail. + fast = dict(PRE_307, https_churn_rps=1100.0, http_churn_rps=21000.0, tls_rtt_p50_ms=0.1) + failures, _ = evaluate([], fast) + self.assertEqual(failed_metrics(failures), {'https_churn_ratio'}) + + def test_every_recorded_run_passes_against_the_rest(self): + history = json.loads(HISTORY.read_text()) + self.assertGreaterEqual(len(history), 5) + for i, row in enumerate(history): + failures, _ = evaluate(history[:i] + history[i + 1:], row) + self.assertEqual(failures, [], f"{row['timestamp']} {row['runner_hw']}") + + def test_same_cpu_check_engages_after_five_runs(self): + history = [epyc_row(i) for i in range(1, 6)] + # 25% lower churn: above every absolute backstop and the ratio + # backstop, but under the same-CPU median / 1.25. + failures, details = evaluate(history, epyc_row(7, churn=880.0)) + self.assertIn('https_churn_rps', failed_metrics(failures)) + self.assertTrue(any('within same-CPU gate' in d for d in details)) + + def test_ratios_are_derived_for_old_rows(self): + row = with_ratios({'https_churn_rps': 100.0, 'http_churn_rps': 1000.0}) + self.assertEqual(row['https_churn_ratio'], 0.1) + self.assertNotIn('https_big_ratio', row) + + +if __name__ == '__main__': + unittest.main()