diff --git a/CHANGELOG.md b/CHANGELOG.md index f30123c6..8a108f7e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,6 +17,18 @@ there; `SUPERSLM_TILED_AVX512_MSVC=1` turns it on. A GEMM call on the tiled path per-call packing buffer; an allocation failure there surfaces through the existing `SSLM_ALLOCATION_FAILED` status, with no state committed. No ABI, format or status change. +The load-time integrity hash uses the x86 SHA extensions when the CPU has them (CPUID leaf 7 +EBX bit 29, with SSSE3 and SSE4.1), selected once per process; the portable SHA-256 stays as the +fallback and the only path on other targets. Digests are unchanged. Hashing the 510 MB Qwen2.5-0.5B model on a +Zen 2 desktop (MSVC) takes 0.31 s against 3.25 s before (about 1.5 GB/s against 150 MB/s). Defining +`SUPERSLM_FORCE_PORTABLE_SHA256` when compiling `src/sha256.cpp` pins the portable path; the new +`sha256_portable_forced_tests` target builds that way. +The FP-free scan's check (A) accepts `sha256rnds2`, `sha256msg1` and `sha256msg2`, the three +instructions the hardware path compiles to, through a new three-entry `_X86_INTEGER_HASH_ALLOW` +set in `check_fp_free_scan.py`. Each is 32-bit integer arithmetic on xmm lanes (Intel SDM: +modular adds, rotates, shifts and boolean ops; no rounding, no MXCSR, no floating-point operand). +The SHA-1 instructions stay rejected. + ## [1.9.0] - 2026-09-25 `sslm_seq_save` writes a new save format, `SSB5`: the `SSB4` layout with the four per-site diff --git a/CMakeLists.txt b/CMakeLists.txt index 01a32928..0d757def 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -122,6 +122,21 @@ endif() enable_testing() add_test(NAME superslm_tests COMMAND superslm_tests) + +# The portable arm of src/sha256.cpp's run-time dispatch, pinned by +# SUPERSLM_FORCE_PORTABLE_SHA256 (the SHA-256 analogue of SUPERSLM_FORCE_SCALAR_MATMUL): +# superslm_tests covers whichever path the CPU selects, this binary covers the portable +# path through the same public API. It compiles src/sha256.cpp itself rather than a whole +# forced copy of the core library, since that file is the only one the macro touches. +add_executable(sha256_portable_forced_tests tests/sha256_portable_forced_tests.cpp src/sha256.cpp) +target_include_directories(sha256_portable_forced_tests PRIVATE include src) +target_compile_definitions(sha256_portable_forced_tests PRIVATE SUPERSLM_FORCE_PORTABLE_SHA256) +if(MSVC) + target_compile_options(sha256_portable_forced_tests PRIVATE /W4 /fp:precise) +else() + target_compile_options(sha256_portable_forced_tests PRIVATE -Wall -Wextra -ffp-contract=off) +endif() +add_test(NAME sha256_portable_forced_tests COMMAND sha256_portable_forced_tests) endif() # The §13 item 7 independent converter verifier (S-HARDEN-3, F13): loads a diff --git a/docs/platform-support.md b/docs/platform-support.md index 043aed2c..36568726 100644 --- a/docs/platform-support.md +++ b/docs/platform-support.md @@ -81,7 +81,7 @@ tiers, and for decode, nothing changes. |---|---|---| | GCC / Clang, AVX2 tier | on at M >= 8 | Bit-identity: full suite forced AVX2, tiled golden, cross-tier digest, save-blob equality against 1.9.0 on the in-tree fixture and on synthetic real-width artifacts | | GCC / Clang, AVX-512 tier | on at M >= 8 | Same evidence, forced AVX-512 and auto dispatch | -| MSVC / clang-cl, AVX2 tier | on at M >= 8 | Built by the forced Windows legs; not yet executed on Windows | +| MSVC / clang-cl, AVX2 tier | on at M >= 8 | Full suite, auto and forced AVX2, green on the Windows CI legs and on a Zen 2 desktop (MSVC 19.33, clang-cl 15); digests equal to the GCC/Clang builds | | MSVC / clang-cl, AVX-512 tier | **off** (`SUPERSLM_TILED_AVX512_MSVC=0`) | Held on the shipped per-row kernel until an MSVC AVX-512 build has executed the tiled kernel | **Measured, engine level, one GEMM, on a 4-vCPU cloud Xeon (AVX2 and @@ -106,6 +106,19 @@ are per-layer engine figures on synthetic weights. They are not a statement about any consumer's end-to-end speed, and they are not a measurement on this project's reference hardware. +### Model load integrity hash (unreleased) + +Loading a model hashes the whole file with SHA-256. On x86-64 CPUs with the +SHA extensions (CPUID leaf 7 EBX bit 29, with SSSE3 and SSE4.1; most AMD CPUs +since Zen, Intel since Ice Lake and Goldmont) the hash uses those +instructions; everywhere else it uses the portable implementation. The digest +is the same either way. + +**Measured, Zen 2 desktop (Ryzen 9 3950X), MSVC, 510 MB Qwen2.5-0.5B model, +best of 3:** 0.31 s with the SHA extensions against 3.25 s for 1.9.0's +portable hash. The portable path itself is also about 1.7x faster than in +1.9.0 (1.95 s). + ### Damped-greedy decoding The 1.2 candidate's opt-in decoder was confirmed on Windows x64 through the diff --git a/include/superslm/sha256.h b/include/superslm/sha256.h index ca5ec1cc..85d564d8 100644 --- a/include/superslm/sha256.h +++ b/include/superslm/sha256.h @@ -30,7 +30,6 @@ class Sha256 { // SslmArtifactAccess comment for the full reasoning. friend struct Sha256Access; - void Block(const uint8_t* p); uint32_t h_[8]; uint64_t total_bits_; uint8_t buf_[64]; @@ -45,6 +44,56 @@ SUPERSLM_API void Sha256Hash(const uint8_t* data, size_t len, uint8_t out[32]); // F5). std::string ToHex(const uint8_t digest[32]); +// --- Block compression dispatch (verification seams) -------------------------- +// +// src/sha256.cpp compresses blocks with the x86 SHA extensions (sha256rnds2, +// sha256msg1, sha256msg2) when the CPU reports them at run time, and with the +// portable FIPS 180-4 code otherwise. Both produce the same digest for the same +// bytes; the choice changes only speed. The portable path is the only one on a +// non-x64 target, and SUPERSLM_FORCE_PORTABLE_SHA256 (defined when compiling +// src/sha256.cpp) pins it on x64 too, the way SUPERSLM_FORCE_SCALAR_MATMUL pins +// matmul.cpp's scalar reference. +// +// The declarations below exist for verification, mirroring matmul.h's +// DotRowScalarRef/ResolveDotRowTier pattern: a test can drive each path directly +// and compare them, and can drive the CPUID decision with fabricated register +// values. They are noexcept: none of them allocates. + +// Compile-time capability: the SHA-extension path exists in this build (the same +// target condition as matmul.h's SUPERSLM_MATMUL_HAVE_SIMD_X64). +#if defined(_M_X64) || defined(__x86_64__) +#define SUPERSLM_SHA256_HAVE_SHANI_X64 1 +#else +#define SUPERSLM_SHA256_HAVE_SHANI_X64 0 +#endif + +inline constexpr int kSha256ImplPortable = 0; +inline constexpr int kSha256ImplShaNi = 1; + +// Pure decision over CPUID fields: leaf 0 EAX (highest basic leaf), leaf 1 ECX +// (SSSE3 bit 9, SSE4.1 bit 19), leaf 7 sub-leaf 0 EBX (SHA bit 29; ignored when the +// highest basic leaf is below 7). Returns kSha256ImplShaNi only when all three +// feature bits are set, else kSha256ImplPortable. +int ResolveSha256Impl(int max_basic_leaf, int leaf1_ecx, int leaf7_ebx) noexcept; + +// What this CPU supports: the resolver above applied to the real CPUID fields +// (always kSha256ImplPortable on a non-x64 build). Ignores the force macro. +int DetectSha256ImplForCpu() noexcept; + +// What Sha256 and Sha256Hash actually use in this build on this CPU: the detected +// implementation, or kSha256ImplPortable under SUPERSLM_FORCE_PORTABLE_SHA256. +int ActiveSha256Impl() noexcept; + +// One-shot SHA-256 through the portable compression, whatever the dispatch selects. +void Sha256HashPortableRef(const uint8_t* data, size_t len, uint8_t out[32]) noexcept; + +#if SUPERSLM_SHA256_HAVE_SHANI_X64 +// One-shot SHA-256 through the SHA-extension compression, whatever the dispatch +// selects. Precondition: DetectSha256ImplForCpu() == kSha256ImplShaNi (on any other +// CPU the instructions fault). +void Sha256HashShaNiRef(const uint8_t* data, size_t len, uint8_t out[32]) noexcept; +#endif + } // namespace superslm #endif // SUPERSLM_SHA256_H diff --git a/src/sha256.cpp b/src/sha256.cpp index c91f30a9..8786c18c 100644 --- a/src/sha256.cpp +++ b/src/sha256.cpp @@ -1,12 +1,49 @@ #include "superslm/sha256.h" +#include + #include "bad_alloc_wrap.h" +// Block compression has two implementations with identical output: the portable +// FIPS 180-4 code (the reference, and the only path on a non-x64 target) and the x86 +// SHA extensions (sha256rnds2 / sha256msg1 / sha256msg2, plus SSSE3/SSE4.1 shuffles), +// selected once per process by a CPUID probe. The dispatch mirrors matmul.cpp's +// (T-2149 design §6): a per-function target attribute, never a TU-wide ISA flag; a +// pure resolver over CPUID fields that a test drives with fabricated values; and a +// force macro, SUPERSLM_FORCE_PORTABLE_SHA256, that pins the portable path so a build +// can test it deliberately (see include/superslm/sha256.h). +#if SUPERSLM_SHA256_HAVE_SHANI_X64 +#include // SHA-extension, SSSE3 and SSE4.1 intrinsics +#if defined(_MSC_VER) +#include // __cpuidex +#else +#include // __cpuid_count +#endif +#endif // SUPERSLM_SHA256_HAVE_SHANI_X64 + +// Same reasoning as matmul.cpp's SUPERSLM_AVX2_TARGET (design §6.4): GCC and Clang +// accept these intrinsics only in a function compiled with the ISA enabled, and a +// TU-wide -msha/-msse4.1 would let the compiler use them anywhere in this file. MSVC +// does not gate intrinsics by /arch, so the macro is empty there. clang-cl defines +// _MSC_VER but generates code with LLVM, which gates intrinsics like Clang on Linux, +// and it does not define __GNUC__ -- so it takes the attributed path through +// __clang__. +#if defined(__clang__) || (defined(__GNUC__) && !defined(_MSC_VER)) +#define SUPERSLM_SHANI_TARGET __attribute__((target("sha,sse4.1,ssse3"))) +#else +#define SUPERSLM_SHANI_TARGET +#endif + namespace superslm { namespace { inline uint32_t Ror(uint32_t x, uint32_t n) { return (x >> n) | (x << (32 - n)); } +constexpr uint32_t kInit[8] = { + 0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, + 0x510e527f, 0x9b05688c, 0x1f83d9ab, 0x5be0cd19, +}; + constexpr uint32_t kK[64] = { 0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, @@ -21,16 +58,17 @@ constexpr uint32_t kK[64] = { 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2, }; -} // namespace +// Compresses one 64-byte block at `p` into `h`. One block per call, with the block +// loop in the shared Absorb/Finish below: the loop is then exercised on every host +// whichever implementation runs, so the SHA-extension function has no branch of its +// own that a host without the extensions leaves unmeasured. Measured on the +// development host, the per-block indirect call costs nothing visible next to a +// multi-block loop inside the function (within run-to-run noise). +using CompressFn = void (*)(uint32_t h[8], const uint8_t* p); -void Sha256::Reset() { - h_[0] = 0x6a09e667; h_[1] = 0xbb67ae85; h_[2] = 0x3c6ef372; h_[3] = 0xa54ff53a; - h_[4] = 0x510e527f; h_[5] = 0x9b05688c; h_[6] = 0x1f83d9ab; h_[7] = 0x5be0cd19; - total_bits_ = 0; - buf_len_ = 0; -} +// --- The portable compression (FIPS 180-4 §6.2.2, the reference) ------------------ -void Sha256::Block(const uint8_t* p) { +void CompressPortable(uint32_t h_[8], const uint8_t* p) { uint32_t w[64]; for (int i = 0; i < 16; ++i) { w[i] = (uint32_t(p[i * 4]) << 24) | (uint32_t(p[i * 4 + 1]) << 16) | @@ -56,14 +94,216 @@ void Sha256::Block(const uint8_t* p) { h_[4] += e; h_[5] += f; h_[6] += g; h_[7] += h; } +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + +// --- The SHA-extension compression ------------------------------------------------- +// +// The instructions keep the working variables as two vectors, ABEF and CDGH +// (lane 3 first), so the state is permuted into that layout on entry and back on +// exit. Each round group g (0..15) runs four rounds: sha256rnds2 does two rounds with +// the low two lanes of W+K, and the high two lanes follow after a shuffle. W for +// groups 4..15 is scheduled from the previous four groups: +// W[g] = sha256msg2(sha256msg1(W[g-4], W[g-3]) + alignr(W[g-1], W[g-2], 4), W[g-1]) +// which is FIPS 180-4's W[t] = s1(W[t-2]) + W[t-7] + s0(W[t-15]) + W[t-16], four +// words at a time (msg1 adds s0(W[t-15]), the alignr supplies W[t-7], msg2 adds +// s1(W[t-2]) in order within the group). The rounds are written out, not looped, +// so the four message vectors stay in registers. +#define SUPERSLM_SHANI_ROUNDS4(g, wg) \ + do { \ + const __m128i msg_ = _mm_add_epi32( \ + (wg), _mm_loadu_si128(reinterpret_cast(&kK[4 * (g)]))); \ + cdgh = _mm_sha256rnds2_epu32(cdgh, abef, msg_); \ + abef = _mm_sha256rnds2_epu32(abef, cdgh, _mm_shuffle_epi32(msg_, 0x0E)); \ + } while (0) +#define SUPERSLM_SHANI_SCHEDULE(w0, w1, w2, w3) \ + (w0) = _mm_sha256msg2_epu32( \ + _mm_add_epi32(_mm_sha256msg1_epu32((w0), (w1)), _mm_alignr_epi8((w3), (w2), 4)), \ + (w3)) + +SUPERSLM_SHANI_TARGET void CompressShaNi(uint32_t h[8], const uint8_t* p) { + // Big-endian word loads: reverse the bytes of each 32-bit lane. + const __m128i kBswap32 = _mm_set_epi64x(0x0c0d0e0f08090a0bLL, 0x0405060700010203LL); + + __m128i dcba = _mm_loadu_si128(reinterpret_cast(&h[0])); + __m128i hgfe = _mm_loadu_si128(reinterpret_cast(&h[4])); + const __m128i cdab = _mm_shuffle_epi32(dcba, 0xB1); + const __m128i efgh = _mm_shuffle_epi32(hgfe, 0x1B); + __m128i abef = _mm_alignr_epi8(cdab, efgh, 8); + __m128i cdgh = _mm_blend_epi16(efgh, cdab, 0xF0); + + const __m128i abef_in = abef; + const __m128i cdgh_in = cdgh; + __m128i w0 = _mm_shuffle_epi8(_mm_loadu_si128(reinterpret_cast(p)), kBswap32); + __m128i w1 = _mm_shuffle_epi8(_mm_loadu_si128(reinterpret_cast(p + 16)), kBswap32); + __m128i w2 = _mm_shuffle_epi8(_mm_loadu_si128(reinterpret_cast(p + 32)), kBswap32); + __m128i w3 = _mm_shuffle_epi8(_mm_loadu_si128(reinterpret_cast(p + 48)), kBswap32); + + SUPERSLM_SHANI_ROUNDS4(0, w0); + SUPERSLM_SHANI_ROUNDS4(1, w1); + SUPERSLM_SHANI_ROUNDS4(2, w2); + SUPERSLM_SHANI_ROUNDS4(3, w3); + SUPERSLM_SHANI_SCHEDULE(w0, w1, w2, w3); SUPERSLM_SHANI_ROUNDS4(4, w0); + SUPERSLM_SHANI_SCHEDULE(w1, w2, w3, w0); SUPERSLM_SHANI_ROUNDS4(5, w1); + SUPERSLM_SHANI_SCHEDULE(w2, w3, w0, w1); SUPERSLM_SHANI_ROUNDS4(6, w2); + SUPERSLM_SHANI_SCHEDULE(w3, w0, w1, w2); SUPERSLM_SHANI_ROUNDS4(7, w3); + SUPERSLM_SHANI_SCHEDULE(w0, w1, w2, w3); SUPERSLM_SHANI_ROUNDS4(8, w0); + SUPERSLM_SHANI_SCHEDULE(w1, w2, w3, w0); SUPERSLM_SHANI_ROUNDS4(9, w1); + SUPERSLM_SHANI_SCHEDULE(w2, w3, w0, w1); SUPERSLM_SHANI_ROUNDS4(10, w2); + SUPERSLM_SHANI_SCHEDULE(w3, w0, w1, w2); SUPERSLM_SHANI_ROUNDS4(11, w3); + SUPERSLM_SHANI_SCHEDULE(w0, w1, w2, w3); SUPERSLM_SHANI_ROUNDS4(12, w0); + SUPERSLM_SHANI_SCHEDULE(w1, w2, w3, w0); SUPERSLM_SHANI_ROUNDS4(13, w1); + SUPERSLM_SHANI_SCHEDULE(w2, w3, w0, w1); SUPERSLM_SHANI_ROUNDS4(14, w2); + SUPERSLM_SHANI_SCHEDULE(w3, w0, w1, w2); SUPERSLM_SHANI_ROUNDS4(15, w3); + + abef = _mm_add_epi32(abef, abef_in); + cdgh = _mm_add_epi32(cdgh, cdgh_in); + + const __m128i feba = _mm_shuffle_epi32(abef, 0x1B); + const __m128i dchg = _mm_shuffle_epi32(cdgh, 0xB1); + dcba = _mm_blend_epi16(feba, dchg, 0xF0); + hgfe = _mm_alignr_epi8(dchg, feba, 8); + _mm_storeu_si128(reinterpret_cast<__m128i*>(&h[0]), dcba); + _mm_storeu_si128(reinterpret_cast<__m128i*>(&h[4]), hgfe); +} + +#undef SUPERSLM_SHANI_ROUNDS4 +#undef SUPERSLM_SHANI_SCHEDULE + +// --- Run-time CPUID probe ------------------------------------------------------------ + +#if defined(_MSC_VER) +inline void QueryCpuId(int leaf, int subleaf, int regs[4]) { __cpuidex(regs, leaf, subleaf); } +#else +inline void QueryCpuId(int leaf, int subleaf, int regs[4]) { + unsigned int eax = 0, ebx = 0, ecx = 0, edx = 0; + __cpuid_count(static_cast(leaf), static_cast(subleaf), eax, ebx, + ecx, edx); + regs[0] = static_cast(eax); + regs[1] = static_cast(ebx); + regs[2] = static_cast(ecx); + regs[3] = static_cast(edx); +} +#endif + +#endif // SUPERSLM_SHA256_HAVE_SHANI_X64 + +// Indexed by kSha256ImplPortable / kSha256ImplShaNi. +constexpr CompressFn kCompressByImpl[] = { + &CompressPortable, +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + &CompressShaNi, +#endif +}; + +inline int ResolveImplFromFields(int max_basic_leaf, int leaf1_ecx, int leaf7_ebx) { + const bool ssse3 = (leaf1_ecx & (1 << 9)) != 0; // leaf 1, ECX bit 9 + const bool sse41 = (leaf1_ecx & (1 << 19)) != 0; // leaf 1, ECX bit 19 + // Leaf 7 is architecturally undefined below max basic leaf 7 (some CPUs echo the + // highest leaf's registers), so its bits count only when the CPU reports it. + const bool sha = max_basic_leaf >= 7 && (leaf7_ebx & (1 << 29)) != 0; // leaf 7/0, EBX bit 29 + return (sha && ssse3 && sse41) ? kSha256ImplShaNi : kSha256ImplPortable; +} + +inline int DetectImpl() { +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + int regs0[4] = {0, 0, 0, 0}; + int regs1[4] = {0, 0, 0, 0}; + int regs7[4] = {0, 0, 0, 0}; + QueryCpuId(0, 0, regs0); + QueryCpuId(1, 0, regs1); + // Queried unconditionally: CPUID never faults on an unsupported leaf, and the + // resolver ignores leaf 7's bits when leaf 0 says the CPU does not have it. + QueryCpuId(7, 0, regs7); + return ResolveImplFromFields(regs0[0], regs1[2], regs7[1]); +#else + return kSha256ImplPortable; +#endif +} + +// Resolved once per process (C++11 magic static: thread-safe, and every later read +// sees the first write). +inline int ActiveImpl() { +#if defined(SUPERSLM_FORCE_PORTABLE_SHA256) + return kSha256ImplPortable; +#else + static const int impl = DetectImpl(); + return impl; +#endif +} + +inline CompressFn ActiveCompress() { + static const CompressFn fn = kCompressByImpl[ActiveImpl()]; + return fn; +} + +// Feeds `len` bytes into the running state: tops up and flushes a partial block in +// `buf`, compresses every whole block straight from `data`, and keeps the tail. +void Absorb(CompressFn compress, uint32_t h[8], uint8_t buf[64], size_t& buf_len, + const uint8_t* data, size_t len) { + if (buf_len > 0) { + const size_t take = std::min(64 - buf_len, len); + std::copy_n(data, take, buf + buf_len); + buf_len += take; + data += take; + len -= take; + if (buf_len < 64) return; + compress(h, buf); + buf_len = 0; + } + for (; len >= 64; data += 64, len -= 64) compress(h, data); + std::copy_n(data, len, buf); + buf_len = len; +} + +// Pads (0x80, zeros to 56 mod 64, the 64-bit big-endian message length in bits), +// compresses the last one or two blocks, and writes the big-endian digest. +void Finish(CompressFn compress, uint32_t h[8], const uint8_t buf[64], size_t buf_len, + uint64_t total_bits, uint8_t out[32]) { + // buf_len is always below 64 (Absorb flushes a full block); the mask states that + // to the compiler, which otherwise cannot bound the index into `tail`. + const size_t used = buf_len & 63; + uint8_t tail[128] = {}; + std::copy_n(buf, used, tail); + tail[used] = 0x80; + const size_t nblocks = used < 56 ? 1 : 2; + uint8_t* len_be = tail + nblocks * 64 - 8; + for (int i = 0; i < 8; ++i) len_be[i] = uint8_t(total_bits >> (56 - i * 8)); + compress(h, tail); + if (nblocks == 2) compress(h, tail + 64); + for (int i = 0; i < 8; ++i) { + out[i * 4] = uint8_t(h[i] >> 24); + out[i * 4 + 1] = uint8_t(h[i] >> 16); + out[i * 4 + 2] = uint8_t(h[i] >> 8); + out[i * 4 + 3] = uint8_t(h[i]); + } +} + +void HashOneShotWith(CompressFn compress, const uint8_t* data, size_t len, uint8_t out[32]) { + uint32_t h[8]; + std::copy_n(kInit, 8, h); + uint8_t buf[64]; + size_t buf_len = 0; + Absorb(compress, h, buf, buf_len, data, len); + Finish(compress, h, buf, buf_len, uint64_t(len) * 8, out); +} + +} // namespace + +void Sha256::Reset() { + std::copy_n(kInit, 8, h_); + total_bits_ = 0; + buf_len_ = 0; +} + // S-HARDEN-7 (design Sec3.1): Update's *Impl body needs private access to -// Sha256 (total_bits_, buf_, buf_len_, Block), which a free function cannot +// Sha256 (total_bits_, buf_, buf_len_, h_), which a free function cannot // have. Sha256Access is the sole friend (sha256.h's // `friend struct Sha256Access;`) -- declared and defined only here, never // in the header. See artifact.cpp's identical SslmArtifactAccess comment // for the full reasoning. struct Sha256Access { static void UpdateImpl(Sha256& self, const uint8_t* data, size_t len); + static void FinalImpl(Sha256& self, uint8_t out[32]); }; void Sha256::Update(const uint8_t* data, size_t len) { @@ -73,47 +313,43 @@ void Sha256::Update(const uint8_t* data, size_t len) { void Sha256Access::UpdateImpl(Sha256& self, const uint8_t* data, size_t len) { internal::MaybeThrowInjectedBadAllocFault(); self.total_bits_ += uint64_t(len) * 8; - while (len > 0) { - size_t take = 64 - self.buf_len_; - if (take > len) take = len; - for (size_t i = 0; i < take; ++i) self.buf_[self.buf_len_ + i] = data[i]; - self.buf_len_ += take; - data += take; - len -= take; - if (self.buf_len_ == 64) { - self.Block(self.buf_); - self.buf_len_ = 0; - } - } + Absorb(ActiveCompress(), self.h_, self.buf_, self.buf_len_, data, len); } void Sha256::Final(uint8_t out[32]) { - // Append 0x80, pad with zeros to 56 mod 64, then the 64-bit big-endian length. - // Calls Sha256Access::UpdateImpl directly, not the public wrapped Update - // -- these bytes are fixed, already-trusted padding, never caller-supplied - // artifact bytes, so routing them through the wrap's try/catch on every one of up - // to 63 padding bytes per digest would add overhead for no benefit - // (S-HARDEN-7 design Sec3.1). - uint8_t pad = 0x80; - uint64_t bits = total_bits_; - Sha256Access::UpdateImpl(*this, &pad, 1); - uint8_t zero = 0; - while (buf_len_ != 56) Sha256Access::UpdateImpl(*this, &zero, 1); - uint8_t lenbe[8]; - for (int i = 0; i < 8; ++i) lenbe[i] = uint8_t(bits >> (56 - i * 8)); - // Update() would re-add these 8 bytes to total_bits_; feed them through Block - // directly instead so the length encodes the message, not the padding. - for (int i = 0; i < 8; ++i) buf_[buf_len_ + i] = lenbe[i]; - buf_len_ = 64; - Block(buf_); - buf_len_ = 0; - for (int i = 0; i < 8; ++i) { - out[i * 4] = uint8_t(h_[i] >> 24); - out[i * 4 + 1] = uint8_t(h_[i] >> 16); - out[i * 4 + 2] = uint8_t(h_[i] >> 8); - out[i * 4 + 3] = uint8_t(h_[i]); - } + // Not wrapped: Final takes no caller-supplied bytes and allocates nothing (the + // padding is built in a fixed stack block), so there is nothing for the + // S-HARDEN-7 wrap to narrow. + Sha256Access::FinalImpl(*this, out); +} + +void Sha256Access::FinalImpl(Sha256& self, uint8_t out[32]) { + Finish(ActiveCompress(), self.h_, self.buf_, self.buf_len_, self.total_bits_, out); + self.buf_len_ = 0; +} + +int ResolveSha256Impl(int max_basic_leaf, int leaf1_ecx, int leaf7_ebx) noexcept { + // Test-reachable wrapper around the anonymous-namespace pure resolver (see + // sha256.h), mirroring matmul.cpp's ResolveDotRowTier. + return ResolveImplFromFields(max_basic_leaf, leaf1_ecx, leaf7_ebx); +} + +int DetectSha256ImplForCpu() noexcept { + static const int impl = DetectImpl(); + return impl; +} + +int ActiveSha256Impl() noexcept { return ActiveImpl(); } + +void Sha256HashPortableRef(const uint8_t* data, size_t len, uint8_t out[32]) noexcept { + HashOneShotWith(&CompressPortable, data, len, out); +} + +#if SUPERSLM_SHA256_HAVE_SHANI_X64 +void Sha256HashShaNiRef(const uint8_t* data, size_t len, uint8_t out[32]) noexcept { + HashOneShotWith(&CompressShaNi, data, len, out); } +#endif namespace { diff --git a/tests/ci/check_fp_free_scan.py b/tests/ci/check_fp_free_scan.py index a2171d4e..3927184a 100644 --- a/tests/ci/check_fp_free_scan.py +++ b/tests/ci/check_fp_free_scan.py @@ -985,6 +985,35 @@ def _account_section(section: _CodeSection, isa: str, md): "vorps", "vorpd", "vandps", "vandpd", "vandnps", "vandnpd", "vxorps", "vxorpd", } +# Integer hash-round family: an explicit, reviewed addition under design +# Sec4.1's own vetting law (fold round 8, D-SLM4374: "a mnemonic newly +# observed in a future corpus and not on this list is rejected as unknown +# until an explicit, reviewed addition lands it"), on the same footing as +# `bswap`'s addition to `_X86_GPR_ALLOW` (T-2529). Decided by Dan, +# 2026-09-29 (decision ID pending: ID-PENDING). The measured reject: +# `superslm::(anonymous namespace)::CompressShaNi` in `sha256.cpp.o` / +# `sha256.obj`, the CPUID-dispatched SHA-extension compression function, +# rejected by check (A) for these three mnemonics and nothing else. +# +# Intel SDM Vol. 2B, `SHA256RNDS2`, `SHA256MSG1`, `SHA256MSG2`: each treats +# its xmm operands as four packed 32-bit unsigned integers and computes +# FIPS 180-4's round and message-schedule functions from 32-bit modular +# integer adds, rotates, shifts and boolean operations (Ch, Maj, the Sigma +# and sigma functions). No rounding, no MXCSR read or write, no SIMD +# floating-point exception, and no operand read as a floating-point value. +# The xmm register file is only where the 32-bit lanes live. +# +# Exactly these three. Kept as its own named set, not folded into +# `_X86_VEC_MOVE_ALLOW` (these are arithmetic, not data movement) or +# `_X86_P_VP_STRUCTURAL_ALLOW` (frozen by D-SLM5155/D-SLM5156/D-SLM5229, +# and these are not p/vp-prefixed). The SHA-1 siblings (`sha1rnds4`, +# `sha1nexte`, `sha1msg1`, `sha1msg2`) are equally integer-only but no +# build emits them, so they are not added and still REJECT as unknown; +# a future corpus that emits one brings its own reviewed addition. +_X86_INTEGER_HASH_ALLOW = frozenset({ + "sha256rnds2", "sha256msg1", "sha256msg2", +}) + def _x86_touches_vector_register(op_str: str) -> bool: return bool(_X86_VEC_REG_RE.search(op_str or "")) @@ -1022,6 +1051,10 @@ def _x86_check_a(mnemonic: str, op_str: str) -> Optional[str]: # for a differing-operand vxorps is reconciled to must-accept by the # same fold (D-SLM5002). return "bitwise_fp_family" + if m in _X86_INTEGER_HASH_ALLOW: + # Dan, 2026-09-29 (ID-PENDING): sha256rnds2/sha256msg1/sha256msg2, + # 32-bit integer hash rounds on xmm lanes; see the set's own comment. + return "integer_hash_allow" return None diff --git a/tests/sha256_portable_forced_tests.cpp b/tests/sha256_portable_forced_tests.cpp new file mode 100644 index 00000000..3f726448 --- /dev/null +++ b/tests/sha256_portable_forced_tests.cpp @@ -0,0 +1,105 @@ +// The portable-forced SHA-256 build: src/sha256.cpp compiled with +// SUPERSLM_FORCE_PORTABLE_SHA256 (CMakeLists.txt, target sha256_portable_forced_tests), so the +// streaming object and the one-shot hash take the portable path whatever the CPU reports. +// superslm_tests covers the dispatched path (the x86 SHA extensions where the CPU has them); +// this binary covers the other arm of the same dispatch through the public API, against the +// FIPS 180-4 answers and against the hardware path called directly where the CPU has it. +#include "superslm/sha256.h" + +#include +#include +#include +#include + +#ifndef SUPERSLM_FORCE_PORTABLE_SHA256 +#error "sha256_portable_forced_tests must be built with SUPERSLM_FORCE_PORTABLE_SHA256" +#endif + +using namespace superslm; + +static int GChecks = 0; +static int GFailures = 0; + +#define CHECK(cond) \ + do { \ + ++GChecks; \ + if (!(cond)) { \ + ++GFailures; \ + std::printf("FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + } \ + } while (0) + +static std::vector Bytes(size_t n, uint32_t seed) { + std::vector v(n); + uint32_t x = seed; + for (auto& b : v) { + x = x * 1664525u + 1013904223u; + b = static_cast(x >> 24); + } + return v; +} + +static std::string Hex(const uint8_t* data, size_t len) { + uint8_t d[32]; + Sha256Hash(data, len, d); + return ToHex(d); +} + +int main() { + // The force macro pins the dispatch; the CPU probe itself still reports the hardware. + CHECK(ActiveSha256Impl() == kSha256ImplPortable); + + const std::string abc = "abc"; + CHECK(Hex(reinterpret_cast(""), 0) == + "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"); + CHECK(Hex(reinterpret_cast(abc.data()), abc.size()) == + "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"); + const std::string m896 = + "abcdefghbcdefghicdefghijdefghijkefghijklfghijklmghijklmnhijklmnoijklmnopjklmnopqklmnopqr" + "lmnopqrsmnopqrstnopqrstu"; + CHECK(Hex(reinterpret_cast(m896.data()), m896.size()) == + "cf5b16a778af8380036ce59e7b0492370b249b11e8f07a51afac45037afee9d1"); + const std::string million(1000000, 'a'); + CHECK(Hex(reinterpret_cast(million.data()), million.size()) == + "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"); + const std::vector big = Bytes((size_t(3) << 20) + 17, 0x5A17u); + CHECK(Hex(big.data(), big.size()) == + "29fe1f49486287b54b24dd1461d5de93cf5cc5b8e0842fd8e2dcf64b0eb609dd"); + + // Streaming (portable, forced) at every two-way split for lengths 0..256 and + // byte-at-a-time for 0..1024, against the portable one-shot and, where the CPU has + // the SHA extensions, the hardware one-shot. + const bool hw = DetectSha256ImplForCpu() == kSha256ImplShaNi; + const std::vector bytes = Bytes(1024, 0x51u); + int mismatches = 0; + for (size_t len = 0; len <= 1024; ++len) { + uint8_t ref[32]; + Sha256HashPortableRef(bytes.data(), len, ref); +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + if (hw) { + uint8_t h2[32]; + Sha256HashShaNiRef(bytes.data(), len, h2); + if (std::memcmp(h2, ref, 32) != 0) ++mismatches; + } +#endif + const size_t max_split = len <= 256 ? len : 0; + for (size_t k = 0; k <= max_split; ++k) { + Sha256 h; + h.Update(bytes.data(), k); + h.Update(bytes.data() + k, len - k); + uint8_t d[32]; + h.Final(d); + if (std::memcmp(d, ref, 32) != 0) ++mismatches; + } + Sha256 h; + for (size_t i = 0; i < len; ++i) h.Update(bytes.data() + i, 1); + uint8_t d[32]; + h.Final(d); + if (std::memcmp(d, ref, 32) != 0) ++mismatches; + } + CHECK(mismatches == 0); + std::printf("sha256 portable-forced: hardware cross-check %s\n", + hw ? "executed" : "not executed (SHA extensions unavailable here)"); + std::printf("sha256 portable-forced tests: %d checks, %d failures\n", GChecks, GFailures); + return GFailures == 0 ? 0 : 1; +} diff --git a/tests/t2296-fp-free-open-red-suite/fp_scan_fixtures/p_vp_structural_only_pinned.txt b/tests/t2296-fp-free-open-red-suite/fp_scan_fixtures/p_vp_structural_only_pinned.txt index 2a508a29..d156059d 100644 --- a/tests/t2296-fp-free-open-red-suite/fp_scan_fixtures/p_vp_structural_only_pinned.txt +++ b/tests/t2296-fp-free-open-red-suite/fp_scan_fixtures/p_vp_structural_only_pinned.txt @@ -22,6 +22,13 @@ # leaking again; removed: capstone renamed or retired a mnemonic) -- # with tests/t2296-fp-free-open-red-suite/test_check_fp_free_scan.py's own # _census_check_a_p_vp_structural_reliance(), sorted, one mnemonic per line. +# Regenerated 2026-09-29 (Dan's decision, ID-PENDING): sha256rnds2, +# sha256msg1 and sha256msg2 added, 457 members. They are named by +# _X86_INTEGER_HASH_ALLOW, not by _X86_VEC_MOVE_ALLOW, so the census's own +# definition of "named" (VEC_MOVE only) files them here, as it already +# files the sixteen _X86_BITWISE_FP_FAMILY members. Each was reviewed: a +# 32-bit integer hash round/schedule op, no floating-point semantics. +# # andnpd andnps @@ -159,6 +166,9 @@ pushaw pushf pushfd pushfq +sha256msg1 +sha256msg2 +sha256rnds2 vandnpd vandnps vandpd diff --git a/tests/t2296-fp-free-open-red-suite/test_check_fp_free_scan.py b/tests/t2296-fp-free-open-red-suite/test_check_fp_free_scan.py index 7f8ae227..a632d9e4 100644 --- a/tests/t2296-fp-free-open-red-suite/test_check_fp_free_scan.py +++ b/tests/t2296-fp-free-open-red-suite/test_check_fp_free_scan.py @@ -4030,6 +4030,14 @@ def test_check_a_p_vp_structural_accept_census_and_violation(): test_check_a_p_vp_structural_only_nonempty_violates_fail_closed_claim, is retired (R7, same commit as R3) rather than left describing a decision that has since been made. + + Rebaselined again 2026-09-29 (Dan's decision, ID-PENDING): the + three-mnemonic `_X86_INTEGER_HASH_ALLOW` (sha256rnds2/sha256msg1/ + sha256msg2) moves accept_a 571 -> 574. This census's own "named" is + `_X86_VEC_MOVE_ALLOW` alone, so named_accept stays 117 and the three + land in structural_only, 454 -> 457 -- the same filing the sixteen + `_X86_BITWISE_FP_FAMILY` members already get. Recomputed by running + this census, not derived by hand. """ if not _SCAN_AVAILABLE: _fail_absent("(D-SLM4999 p/vp vitality pin, census)", "") @@ -4039,10 +4047,11 @@ def test_check_a_p_vp_structural_accept_census_and_violation(): "reproduced 1523 -- this suite's own capstone version may have " "changed; got {}".format(len(vocabulary)) ) - assert len(accept_a) == 571, ( + assert len(accept_a) == 574, ( "check (A)'s own ACCEPT count on a vector operand changed from the " - "reproduced 571 (555 pre-D-SLM5037 + 16 vextract/vinsert " - "lane-movement mnemonics D-SLM5037 widened); got {}".format(len(accept_a)) + "reproduced 574 (555 pre-D-SLM5037 + 16 vextract/vinsert " + "lane-movement mnemonics D-SLM5037 widened + 3 _X86_INTEGER_HASH_ALLOW " + "SHA-256 mnemonics, 2026-09-29); got {}".format(len(accept_a)) ) assert len(named_accept) == 117, ( "check (A)'s own explicitly-allow-listed ACCEPT count changed from " @@ -4051,12 +4060,13 @@ def test_check_a_p_vp_structural_accept_census_and_violation(): "addition, unlike D-SLM4987's structural-rule accept, so this count " "moves with it); got {}".format(len(named_accept)) ) - assert len(structural_only) == 454, ( - "check (A)'s own structural-only (no-allow-list) ACCEPT count " - "changed from the reproduced 454 -- D-SLM5037's sixteen-mnemonic " - "widening is an explicit allow-list addition (named_accept, above), " - "not a structural-rule accept, so this count should be unaffected " - "by it; got {}".format(len(structural_only)) + assert len(structural_only) == 457, ( + "check (A)'s own structural-only (not on _X86_VEC_MOVE_ALLOW) ACCEPT " + "count changed from the reproduced 457 (454 + the three " + "_X86_INTEGER_HASH_ALLOW mnemonics, which this census's own " + "definition of named does not cover) -- D-SLM5037's sixteen-mnemonic " + "widening is an explicit _X86_VEC_MOVE_ALLOW addition (named_accept, " + "above), so it never moved this count; got {}".format(len(structural_only)) ) @@ -4725,8 +4735,9 @@ def test_check_a_reason_attribution_census_every_accept_has_a_reason(): """Population fifty-two's must-accept (D-SLM5157): once _x86_check_a adopts the Optional[str] contract (R4), `unattributed` (every acceptance with no reported reason) must be [] -- every member of - accept_a is attributed to exactly one of the three named reasons - (_X86_VEC_MOVE_ALLOW, _X86_BITWISE_FP_FAMILY, _X86_P_VP_STRUCTURAL_ALLOW). + accept_a is attributed to exactly one of the four named reasons + (_X86_VEC_MOVE_ALLOW, _X86_BITWISE_FP_FAMILY, _X86_P_VP_STRUCTURAL_ALLOW, + _X86_INTEGER_HASH_ALLOW). TODAY _x86_check_a still returns bool: every accepted mnemonic's own 'reason' is the literal True, not a string, so unattributed == @@ -4837,3 +4848,53 @@ def test_x86_gpr_allow_population_is_pinned_by_count_not_by_comment(): "bswap (T-2531 C-1's own addition, closing the linux-x64 job's real GCC reject " "on Sha256::Final) is missing from _X86_GPR_ALLOW" ) + + +# --------------------------------------------------------------------------- +# Dan, 2026-09-29 (decision ID pending: ID-PENDING): `_X86_INTEGER_HASH_ALLOW` +# admits exactly sha256rnds2/sha256msg1/sha256msg2 under check (A), on the +# `bswap` precedent (T-2529, D-SLM4374's vetting law). The cells below pin +# that the addition is exactly scoped: the three ACCEPT with their own reason, +# the set holds nothing else, and the neighbouring SHA-1 mnemonics (equally +# integer-only, but never reviewed onto any list) still REJECT as unknown. +# --------------------------------------------------------------------------- + +_SHA256_HASH_MNEMONICS = ("sha256rnds2", "sha256msg1", "sha256msg2") + + +@pytest.mark.parametrize("mnemonic", _SHA256_HASH_MNEMONICS) +def test_check_a_accepts_sha256_integer_hash_mnemonic(mnemonic): + if not _SCAN_AVAILABLE: + _fail_absent("(integer hash allow, must-accept)", "") + for ops in ("xmm1, xmm2", "xmm1, xmmword ptr [rax]", "xmm1, xmm2, xmm0"): + assert scan._x86_check_a(mnemonic, ops) == "integer_hash_allow", ( + "{} {} must ACCEPT under check (A) via _X86_INTEGER_HASH_ALLOW; " + "got {!r}".format(mnemonic, ops, scan._x86_check_a(mnemonic, ops)) + ) + + +def test_x86_integer_hash_allow_is_exactly_the_three_sha256_mnemonics(): + if not _SCAN_AVAILABLE: + _fail_absent("(integer hash allow, exact membership)", "") + assert set(scan._X86_INTEGER_HASH_ALLOW) == set(_SHA256_HASH_MNEMONICS), ( + "_X86_INTEGER_HASH_ALLOW must hold exactly the three reviewed SHA-256 " + "mnemonics; any other member needs its own reviewed addition. Got " + "{}".format(sorted(scan._X86_INTEGER_HASH_ALLOW)) + ) + for other in ("_X86_VEC_MOVE_ALLOW", "_X86_P_VP_STRUCTURAL_ALLOW", + "_X86_BITWISE_FP_FAMILY", "_X86_GPR_ALLOW"): + overlap = set(getattr(scan, other)) & set(_SHA256_HASH_MNEMONICS) + assert overlap == set(), "{} also names {}".format(other, sorted(overlap)) + + +@pytest.mark.parametrize("mnemonic", ("sha1rnds4", "sha1msg1", "sha1msg2", "sha1nexte")) +def test_check_a_still_rejects_neighbouring_sha1_mnemonic(mnemonic): + if not _SCAN_AVAILABLE: + _fail_absent("(integer hash allow, must-reject neighbour)", "") + assert scan._x86_check_a(mnemonic, "xmm1, xmm2") is None, ( + "{} was never reviewed onto any check (A) list and must still REJECT " + "as unknown -- the SHA-256 addition must not widen to it".format(mnemonic) + ) + assert not scan._x86_check_b(mnemonic), ( + "{} must not be on check (B)'s GPR allow-list either".format(mnemonic) + ) diff --git a/tests/test_main.cpp b/tests/test_main.cpp index d5b2f11c..5ce3df02 100644 --- a/tests/test_main.cpp +++ b/tests/test_main.cpp @@ -145,6 +145,189 @@ static void TestSha256KnownVectors() { "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"); } +// --- Hardware SHA-256 (x86 SHA extensions) against the portable reference --- +// +// src/sha256.cpp selects its block compression at run time: the SHA-extension path +// (sha256rnds2/sha256msg1/sha256msg2) when CPUID reports it, the portable FIPS 180-4 +// code otherwise. The digest is a fixed function of the bytes, so every cell below +// demands byte equality between the dispatched path, the portable reference +// (Sha256HashPortableRef) and, on a CPU that has the extensions, the hardware path +// called directly (Sha256HashShaNiRef). On a CPU without them the hardware cells are +// skipped and say so; the portable-forced build (tests/sha256_portable_forced_tests.cpp) +// runs the streaming cells with the portable path as the dispatched one. + +static std::vector Sha256TestBytes(size_t n, uint32_t seed) { + std::vector v(n); + uint32_t x = seed; + for (auto& b : v) { + x = x * 1664525u + 1013904223u; + b = static_cast(x >> 24); + } + return v; +} + +static bool Sha256HostHasShaExtensions() { + return DetectSha256ImplForCpu() == kSha256ImplShaNi; +} + +// Hashes `data` with every implementation this host can run and checks they agree +// with each other and, when `want_hex` is non-null, with the expected digest. +static void CheckSha256AllImplsAgree(const uint8_t* data, size_t len, const char* want_hex, + const char* what) { + uint8_t dispatched[32], portable[32]; + Sha256Hash(data, len, dispatched); + Sha256HashPortableRef(data, len, portable); + CHECK_MSG(std::memcmp(dispatched, portable, 32) == 0, + "%s (len %zu): dispatched %s != portable %s", what, len, ToHex(dispatched).c_str(), + ToHex(portable).c_str()); + if (want_hex != nullptr) { + CHECK_MSG(ToHex(portable) == want_hex, "%s (len %zu): portable %s, want %s", what, len, + ToHex(portable).c_str(), want_hex); + } +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + if (Sha256HostHasShaExtensions()) { + uint8_t hw[32]; + Sha256HashShaNiRef(data, len, hw); + CHECK_MSG(std::memcmp(hw, portable, 32) == 0, "%s (len %zu): hardware %s != portable %s", + what, len, ToHex(hw).c_str(), ToHex(portable).c_str()); + } +#endif +} + +static void TestSha256ImplResolverFromFields() { + // Pure resolver over fabricated CPUID fields: leaf 0 EAX (max basic leaf), leaf 1 + // ECX (SSSE3 bit 9, SSE4.1 bit 19), leaf 7/0 EBX (SHA bit 29). + const int kSsse3 = 1 << 9, kSse41 = 1 << 19, kSha = 1 << 29; + const int all1 = kSsse3 | kSse41; + CHECK(ResolveSha256Impl(7, all1, kSha) == kSha256ImplShaNi); + CHECK(ResolveSha256Impl(0x10, all1, kSha | (1 << 5)) == kSha256ImplShaNi); + // Leaf 7 is architecturally undefined below max basic leaf 7: its bits must be ignored. + CHECK(ResolveSha256Impl(6, all1, kSha) == kSha256ImplPortable); + CHECK(ResolveSha256Impl(7, all1, 0) == kSha256ImplPortable); + CHECK(ResolveSha256Impl(7, all1, ~kSha) == kSha256ImplPortable); + CHECK(ResolveSha256Impl(7, kSse41, kSha) == kSha256ImplPortable); // no SSSE3 + CHECK(ResolveSha256Impl(7, kSsse3, kSha) == kSha256ImplPortable); // no SSE4.1 + CHECK(ResolveSha256Impl(0, 0, 0) == kSha256ImplPortable); +} + +static void TestSha256ActiveImplIsTheDetectedOne() { + // This binary is built without SUPERSLM_FORCE_PORTABLE_SHA256, so the dispatched + // path must be exactly what the CPU probe selects (portable-only on a non-x64 build). + const int detected = DetectSha256ImplForCpu(); + CHECK(detected == kSha256ImplPortable || detected == kSha256ImplShaNi); + CHECK(ActiveSha256Impl() == detected); +#if !SUPERSLM_SHA256_HAVE_SHANI_X64 + CHECK(detected == kSha256ImplPortable); +#endif + std::printf("sha256: dispatched implementation = %s\n", + ActiveSha256Impl() == kSha256ImplShaNi ? "x86 SHA extensions" : "portable"); +} + +static void TestSha256FipsVectorsAllImpls() { + // FIPS 180-4 / NIST known answers, including both multi-block padding cases and the + // one-million-'a' long message. + struct Vec { + std::string msg; + const char* hex; + }; + const Vec vecs[] = { + {"", "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"}, + {"abc", "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"}, + {"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq", + "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"}, + {"abcdefghbcdefghicdefghijdefghijkefghijklfghijklmghijklmnhijklmnoijklmnopjklmnopqklmnopqr" + "lmnopqrsmnopqrstnopqrstu", + "cf5b16a778af8380036ce59e7b0492370b249b11e8f07a51afac45037afee9d1"}, + {std::string(1000000, 'a'), + "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"}, + }; + for (const Vec& v : vecs) { + CheckSha256AllImplsAgree(reinterpret_cast(v.msg.data()), v.msg.size(), + v.hex, "FIPS vector"); + } +#if SUPERSLM_SHA256_HAVE_SHANI_X64 + if (!Sha256HostHasShaExtensions()) { + std::printf("sha256: host has no SHA extensions -- hardware-path cells not executed\n"); + } +#endif +} + +static void TestSha256EveryLength0To1024AllImpls() { + // Every length 0..1024 covers every padding residue (0..63 bytes past a block, both + // the one-block and two-block padding cases) over 17 block counts. Run at an aligned + // and at an odd offset so the hardware path's unaligned loads are exercised too. + const std::vector bytes = Sha256TestBytes(1024 + 3, 0xC0FFEEu); + for (size_t off : {size_t(0), size_t(3)}) { + for (size_t len = 0; len <= 1024; ++len) { + CheckSha256AllImplsAgree(bytes.data() + off, len, nullptr, "every length"); + } + } +} + +static void TestSha256MultiMegabyteAllImpls() { + // Digests pinned from the portable implementation as it stood before the hardware path + // existed, and cross-checked against an independent SHA-256. + const std::vector a = Sha256TestBytes((size_t(3) << 20) + 17, 0x5A17u); + CheckSha256AllImplsAgree(a.data(), a.size(), + "29fe1f49486287b54b24dd1461d5de93cf5cc5b8e0842fd8e2dcf64b0eb609dd", + "3 MiB + 17"); + const std::vector b = Sha256TestBytes(size_t(8) << 20, 0x5A17u); + CheckSha256AllImplsAgree(b.data(), b.size(), + "35ac12ceae8a53ba17582aa94f381a24882eedf6c3c391b4b715cf9433539d87", + "8 MiB"); + CheckSha256AllImplsAgree(b.data() + 1, b.size() - 1, nullptr, "8 MiB - 1 at offset 1"); +} + +static void TestSha256StreamingEverySplitMatchesPortable() { + // The streaming object (dispatched path) fed in pieces must equal the portable + // one-shot digest, whatever the split: every two-way split for lengths 0..256, every + // three-way split for lengths 0..96, and byte-at-a-time for lengths 0..1024. + const std::vector bytes = Sha256TestBytes(1024, 0x51u); + auto portable = [&](size_t len) { + uint8_t d[32]; + Sha256HashPortableRef(bytes.data(), len, d); + return ToHex(d); + }; + int mismatches = 0; + for (size_t len = 0; len <= 256; ++len) { + const std::string want = portable(len); + for (size_t k = 0; k <= len; ++k) { + Sha256 h; + h.Update(bytes.data(), k); + h.Update(bytes.data() + k, len - k); + uint8_t d[32]; + h.Final(d); + if (ToHex(d) != want) ++mismatches; + } + } + CHECK_MSG(mismatches == 0, "two-way splits: %d mismatches", mismatches); + mismatches = 0; + for (size_t len = 0; len <= 96; ++len) { + const std::string want = portable(len); + for (size_t i = 0; i <= len; ++i) { + for (size_t j = i; j <= len; ++j) { + Sha256 h; + h.Update(bytes.data(), i); + h.Update(bytes.data() + i, j - i); + h.Update(bytes.data() + j, len - j); + uint8_t d[32]; + h.Final(d); + if (ToHex(d) != want) ++mismatches; + } + } + } + CHECK_MSG(mismatches == 0, "three-way splits: %d mismatches", mismatches); + mismatches = 0; + for (size_t len = 0; len <= 1024; ++len) { + Sha256 h; + for (size_t i = 0; i < len; ++i) h.Update(bytes.data() + i, 1); + uint8_t d[32]; + h.Final(d); + if (ToHex(d) != portable(len)) ++mismatches; + } + CHECK_MSG(mismatches == 0, "byte-at-a-time: %d mismatches", mismatches); +} + static void TestDtypeSizes() { CHECK(DtypeSize(static_cast(SslmDtype::Raw)) == 1); CHECK(DtypeSize(static_cast(SslmDtype::Int8)) == 1); @@ -29247,6 +29430,12 @@ int main(int argc, char** argv) { #endif // _WIN32 TestSha256KnownVectors(); + TestSha256ImplResolverFromFields(); + TestSha256ActiveImplIsTheDetectedOne(); + TestSha256FipsVectorsAllImpls(); + TestSha256EveryLength0To1024AllImpls(); + TestSha256MultiMegabyteAllImpls(); + TestSha256StreamingEverySplitMatchesPortable(); TestDtypeSizes(); TestKnownSectionTypes();