From e04ef529f9153cf8e37b0777e4c36ca7a9628d15 Mon Sep 17 00:00:00 2001 From: Matt Brown Date: Tue, 15 Sep 2026 13:10:10 -0400 Subject: [PATCH] perf(keys): use binary GCD in the batch-GCD shared-prime pass The cross-file batch-GCD check runs a pairwise BigUint::gcd over every RSA modulus in the tree. The O(n^2) loop is capped and expected; the cost was per-gcd. BigUint::gcd was Euclidean (a % b), and operator% goes through the bit-by-bit long-division divmod, which iterates over every bit of the dividend, allocates fresh vectors per bit, and computes a quotient it then discards. A single 2048-bit gcd cost ~37ms, so a rootfs shipping a full CA certificate store (hundreds of RSA CA certs) turned the pass into a multi-minute hang. Replace it with Stein's binary GCD (shifts and subtraction, no division), keeping divmod off the hot path. Same O(bits^2) worst case, but without the per-bit allocation churn: the --keys pass on such a tree drops from 6+ minutes to ~5s, and the remaining time is the now-cheap O(n^2) loop. Correctness is unchanged: the bigint gcd KATs and the shared-prime recovery fixtures still pass, and the detector still recovers real shared factors. Closes #19 --- src/bigint.cpp | 26 +++++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/src/bigint.cpp b/src/bigint.cpp index 0b57fb0..77c5718 100644 --- a/src/bigint.cpp +++ b/src/bigint.cpp @@ -2,6 +2,7 @@ #include "bigint.hpp" #include +#include namespace ft { @@ -187,12 +188,27 @@ BigUint BigUint::operator%(const BigUint& o) const { } BigUint BigUint::gcd(BigUint a, BigUint b) { - while (!b.is_zero()) { - BigUint r = a % b; - a = b; - b = r; + // Binary (Stein's) GCD: only shifts and subtraction, no division. The + // Euclidean form (a % b each step) calls divmod, whose bit-by-bit long + // division makes each gcd cost O(bits) reductions * O(bits^2) per divmod. + // The batch-GCD shared-prime pass runs a gcd per pair of RSA moduli, so on + // a large cert bundle (a full CA store is hundreds of 2048-bit keys) that + // cubic-per-gcd cost turned the O(n^2) pass into a multi-minute hang. + if (a.is_zero()) return b; + if (b.is_zero()) return a; + size_t shift = 0; // common power of two, restored at the end + while (!a.is_odd() && !b.is_odd()) { + a = a.shr_bits(1); + b = b.shr_bits(1); + ++shift; } - return a; + while (!a.is_odd()) a = a.shr_bits(1); + do { + while (!b.is_odd()) b = b.shr_bits(1); + if (cmp(a, b) > 0) std::swap(a, b); // keep a <= b + b = b - a; // non-negative, and now even + } while (!b.is_zero()); + return a.shl_bits(shift); } BigUint BigUint::isqrt() const {