diff --git a/.gitattributes b/.gitattributes index 8cc6ae28..8f591270 100644 --- a/.gitattributes +++ b/.gitattributes @@ -31,3 +31,8 @@ tests/golden/** -text # T-2814: the recorded dumpbin read that tests/t2807-api-slot/consumed_symbols.txt is generated from, and # whose SHA-256 its header states. Byte-exact on every platform. tests/t2807-api-slot/recorded/*.tsv -text + +# Generated golden-pin headers claim byte-for-byte regeneration by their +# generators, which emit LF; pin them to LF so a Windows checkout matches. +tests/attn_rowsite_golden_pin.h text eol=lf +tests/matmul_tiled_golden_pin.h text eol=lf diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 737ea174..b81e919c 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -361,7 +361,8 @@ jobs: # built by MSVC and by clang-cl, the full suite and the digest each. Before these legs no MSVC build ever executed # an AVX2 or AVX-512 kernel in CI (the auto legs dispatch on whatever the runner has). The AVX-512 # leg builds with SUPERSLM_TILED_AVX512_MSVC=1, so it runs the tiled AVX-512 kernel that the - # default MSVC build still holds off; it probes the CPU first (IsProcessorFeaturePresent 41 = + # default MSVC build still holds off, and (attention and per-row sites plan §3.2) with + # SUPERSLM_SITES_AVX512_MSVC=1, so it runs the attention kernels' AVX-512 bodies too; it probes the CPU first (IsProcessorFeaturePresent 41 = # AVX-512F, which the hosted image's CPUs pair with BW) and reports SKIPPED without it. windows-msvc-avx2-forced: runs-on: windows-latest @@ -389,7 +390,7 @@ jobs: runs-on: windows-latest steps: - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5 - - run: cmake -B build -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1" + - run: cmake -B build -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1 /DSUPERSLM_SITES_AVX512_MSVC=1" - run: cmake --build build --config Release --target superslm_tests_avx512_forced sslm_axis_digest_avx512_forced - name: Probe for AVX-512, run the forced suite and extract the GLOBAL digest if present shell: pwsh @@ -439,7 +440,7 @@ jobs: runs-on: windows-latest steps: - uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5 - - run: cmake -B build -T ClangCL -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1" + - run: cmake -B build -T ClangCL -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1 /DSUPERSLM_SITES_AVX512_MSVC=1" - run: cmake --build build --config Release --target superslm_tests_avx512_forced sslm_axis_digest_avx512_forced - name: Probe for AVX-512, run the forced suite and extract the GLOBAL digest if present shell: pwsh @@ -466,7 +467,10 @@ jobs: # Tiled-matmul plan slice 1, cell 11.3: every tiled and packer symbol in the matmul object is local # (never global, weak, unique or COMDAT), in the auto and both forced builds; the build-configuration # record is the single external exception. The check's own vitality plants are recorded in - # docs/tiled-matmul-slice1-progress.md. + # docs/tiled-matmul-slice1-progress.md. The attention and per-row sites plan's slice S3 adds the + # intmath.cpp objects, whose requant row bodies are that file's first target-attributed functions; + # slice S4 adds the softmax fast path's bodies to the same objects; slice S5 adds the forward_sites.cpp + # objects, for the Q31 score row's bodies. tiled-matmul-linkage: runs-on: ubuntu-latest steps: @@ -477,7 +481,13 @@ jobs: python3 tools/ci/check_tiled_matmul_linkage.py \ build/CMakeFiles/superslm.dir/src/matmul.cpp.o \ build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o \ - build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o + build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o \ + build/CMakeFiles/superslm.dir/src/intmath.cpp.o \ + build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o \ + build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o \ + build/CMakeFiles/superslm.dir/src/forward/forward_sites.cpp.o \ + build/CMakeFiles/superslm_avx2_forced.dir/src/forward/forward_sites.cpp.o \ + build/CMakeFiles/superslm_avx512_forced.dir/src/forward/forward_sites.cpp.o # Ratified into this slot's scope (Sec7.1, D-SLM170): the first executed # x64-vs-ARM comparison this plan's own text (SuperSLM_Plan.md:1436, T-182) @@ -843,6 +853,12 @@ jobs: python3 tools/ci/check_branch_coverage_floors.py \ --record-floors-to build/measured_branch_coverage_floors.json \ build/coverage.json + # Re-pinning a floor copies this leg's recorded value into tools/ci/branch_coverage_floors.json, + # and the artifact store that holds the record is not reachable from every session that does + # the re-pin. Printing the record puts the exact values in the job log as well. + - name: Print the recorded branch-coverage floors + if: always() + run: cat build/measured_branch_coverage_floors.json - name: Check per-file branch-coverage floors run: python3 tools/ci/check_branch_coverage_floors.py build/coverage.json # Tiled-matmul plan slice 1, cell 11.4: nonzero counts on the tiled kernel's named spans (flush @@ -975,7 +991,8 @@ jobs: # C1): the checked chain funnel's structural half -- no translation unit in # the forward composition may name a funnel leaf (MaxAbsReduce, # MaxAbsReduceWide, RowBoundsWide, NormalizeScale, DynamicScaleReciprocal, - # RequantTokenCode, RequantTokenCodeWide, NarrowAccumulatorToI32) directly, + # RequantTokenCode, RequantTokenCodeWide, NarrowAccumulatorToI32, and since the + # attention and per-row sites plan's slice S3 the row leaf RequantRowWide) directly, # outside the funnel's own file and the leaf certification TUs. Mirrors the # two precedents already CI-wired (tests/check_no_pow_operator.py above, # tools/ci/check_bad_alloc_contract.py's own job): a test step, then the diff --git a/CHANGELOG.md b/CHANGELOG.md index 8a108f7e..340dc0c9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,52 @@ set in `check_fp_free_scan.py`. Each is 32-bit integer arithmetic on xmm lanes ( modular adds, rotates, shifts and boolean ops; no rounding, no MXCSR, no floating-point operand). The SHA-1 instructions stay rejected. +On the AVX2 and AVX-512 tiers, attention's probability-times-value step (`GemmProbQ15Accumulate`) +now runs on 16-bit multiply-add when the head dimension is a multiple of 16 and the probability +row fits 16 bits (every p in [0, 32767], sum at most 2^15). Other rows take the 1.9.0 loop. This +covers every softmax row except a one-hot row. Outputs are bit-identical to 1.9.0: the same +tokens, save blobs and digest. Engine level, on one cloud host, at head dimension 64 the step is +about 16-21x faster. At Qwen2.5-0.5B depth (24 layers, 14 heads) that saves about 0.8 / 3.4 / +6.7 ms per prompt token at 128 / 512 / 1,024 tokens, and about 4 ms per decode token at context +300. MSVC and clang-cl builds keep the AVX-512 tier on the 1.9.0 loop until it has executed +there; `SUPERSLM_SITES_AVX512_MSVC=1` turns it on. There is no ABI, format or status change. + +On the AVX2 and AVX-512 tiers, the requantization step every checked projection, norm, activation +and residual ends in (the funnel's per-element conversion to int8 codes) now runs 4 or 8 elements +at a time in 64-bit integer lanes, through the new row function `RequantRowWide`. Every code is +bit-identical to 1.9.0's per-element `RequantTokenCodeWide`: the same tokens, save blobs and +digest. Engine level, on one cloud host, a funnel call at Qwen2.5-0.5B's widths (896 and 4,864) +is about 4.5x faster on AVX2 and 5.5x on AVX-512, which saves about 2.5 ms per token at 24 +layers, prefill and decode alike. The scalar and SSE2 tiers keep the 1.9.0 loop. MSVC and +clang-cl builds keep the AVX-512 tier on it too, under the same `SUPERSLM_SITES_AVX512_MSVC` +switch. There is no ABI, format or status change. + +On the AVX2 and AVX-512 tiers, attention's softmax row (`SoftmaxRowQ15`) now runs 4 or 8 elements +at a time when the row is inside a guard: width at most 2^14, q_ln2 >= 1, q_c >= 0, +q_b^2 + q_c in [1, 2^47], q_ln2 <= 2 q_b + 1 and every score within 2^61. The two divides per +element are integer estimates that one exact integer correction each way makes exact. Rows outside +the guard run the 1.9.0 body unchanged. Every probability and the returned bool are bit-identical +to 1.9.0: the same tokens, save blobs and digest. Engine level, on one cloud host, a row of 512 +keys is about 3.6x faster on AVX2. At Qwen2.5-0.5B depth (24 layers, 14 heads) that saves about +0.1 / 0.46 / 0.91 ms per prompt token at 128 / 512 / 1,024 tokens, and about 0.53 ms per decode +token at context 300. A one-key row is slightly slower. The scalar and SSE2 tiers keep +the 1.9.0 body. MSVC and clang-cl builds keep the AVX-512 tier on it too, under the same +`SUPERSLM_SITES_AVX512_MSVC` switch. There is no ABI, format or status change. + +On the AVX2 and AVX-512 tiers, the Q31 attention score used by QK-norm models (the Qwen3 path) is +now computed for all of a query head's keys in one call (`QkQ31ScoreRow`), instead of one +`QkQ31Score` call per key. Inside a guard (head_dim at most 512, every K-channel ratio in +[0, 2^32)) each channel's q x ratio product is split exactly into three 16-bit pieces, and each +piece's sum over the channels is a 16-bit multiply-add. The rounding is `RoundingDivideByPOT`'s, ties away from zero. Rows outside the +guard run the 1.9.0 per-key loop. Every score is bit-identical to 1.9.0: the same tokens, save +blobs and digest. Engine level, on one cloud host, a score costs about 14 ns per head and key +instead of about 410 on AVX2 (about 330 on AVX-512). At Qwen3-0.6B depth (28 layers, 16 heads) +that saves about 11 / 46 / 94 ms per prompt token at 128 / 512 / 1,024 tokens on AVX2, and about +55 ms per decode token at context 300. Qwen2.5 models do not take this path. A one-key row on +AVX-512 is slightly slower. The scalar and SSE2 tiers keep the 1.9.0 loop. MSVC and clang-cl +builds keep the AVX-512 tier on it too, under the same `SUPERSLM_SITES_AVX512_MSVC` switch. There +is no ABI, format or status change. + ## [1.9.0] - 2026-09-25 `sslm_seq_save` writes a new save format, `SSB5`: the `SSB4` layout with the four per-site diff --git a/CMakeLists.txt b/CMakeLists.txt index 0d757def..27af4bdf 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -91,7 +91,7 @@ else() endif() add_executable(superslm_tests tests/test_main.cpp tests/test_slm18x_saturation_census.cpp - tests/test_slm19x_schema_damped_greedy.cpp tests/test_tiled_gemm.cpp) + tests/test_slm19x_schema_damped_greedy.cpp tests/test_tiled_gemm.cpp tests/test_attn_rowsites.cpp) target_link_libraries(superslm_tests PRIVATE superslm_test_injection) if(MSVC) target_compile_options(superslm_tests PRIVATE /W4 /fp:precise) @@ -358,7 +358,7 @@ foreach(_sslm_tier SSE2 AVX2 AVX512) # target_compile_definitions call. add_executable(superslm_tests_${_sslm_tier_lower}_forced EXCLUDE_FROM_ALL tests/test_main.cpp tests/test_slm18x_saturation_census.cpp tests/test_slm19x_schema_damped_greedy.cpp - tests/test_tiled_gemm.cpp) + tests/test_tiled_gemm.cpp tests/test_attn_rowsites.cpp) target_link_libraries(superslm_tests_${_sslm_tier_lower}_forced PRIVATE superslm_${_sslm_tier_lower}_forced) target_include_directories(superslm_tests_${_sslm_tier_lower}_forced PRIVATE tests) if(MSVC) @@ -417,6 +417,11 @@ foreach(_sslm_bench_tier AVX2 AVX512) string(TOLOWER "${_sslm_bench_tier}" _sslm_bench_tier_lower) add_executable(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced EXCLUDE_FROM_ALL tools/t2147_chunk_batched_pins.cpp) target_link_libraries(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE superslm_${_sslm_bench_tier_lower}_forced) + # The forced library carries SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT PUBLIC, so this tool's own + # compile takes its instrument branch (#include "support/matmul_dispatch_instrument.h") and needs + # tests/ on its include path; without it the target does not compile (found by the attention and + # per-row sites plan's S1 bench). + target_include_directories(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE tests) if(MSVC) target_compile_options(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE /W4 /fp:precise) else() @@ -432,6 +437,29 @@ endforeach() # its D-infinity twin is the tiled kernel's own speedup on that tier. The bench calls only matmul.cpp's # entry points, so the D-infinity library needs no other engine source. All EXCLUDE_FROM_ALL except the # production bench, for the same non-x64 reason as the forced targets above. +# Attention and per-row sites plan, cell 10.1: the engine bench (tools/sslm_sites_bench.cpp; kernel, prefill +# and decode modes). Reports, not gates. Linked here against the production library; the base arm of a +# comparison is the same source linked against the base's library. +add_executable(sslm_sites_bench tools/sslm_sites_bench.cpp) +target_link_libraries(sslm_sites_bench PRIVATE superslm) +if(MSVC) + target_compile_options(sslm_sites_bench PRIVATE /W4 /fp:precise) +else() + target_compile_options(sslm_sites_bench PRIVATE -Wall -Wextra -ffp-contract=off) +endif() + +# Attention and per-row sites plan, §3.3 evidence 3: the golden-pin generator. The committed pin +# (tests/attn_rowsite_golden_pin.h) is generated by this source built against the v1.9.0 TAG's library +# (recipe in the source's header); this target builds it against the current tree, where it must print +# the same hash -- a consistency check on the generator, never the pin's provenance. +add_executable(gen_attn_rowsite_golden tools/gen_attn_rowsite_golden.cpp) +target_link_libraries(gen_attn_rowsite_golden PRIVATE superslm) +if(MSVC) + target_compile_options(gen_attn_rowsite_golden PRIVATE /W4 /fp:precise) +else() + target_compile_options(gen_attn_rowsite_golden PRIVATE -Wall -Wextra -ffp-contract=off) +endif() + add_executable(sslm_gemm_bench tools/sslm_gemm_bench.cpp) target_link_libraries(sslm_gemm_bench PRIVATE superslm) if(MSVC) diff --git a/build.bat b/build.bat index 98bf9d2f..d31c1045 100644 --- a/build.bat +++ b/build.bat @@ -56,7 +56,7 @@ cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude /Itests /DSUPERSLM_ENA src\forward\checked_chain_funnel.cpp src\forward\forward_sites.cpp src\decode_digest.cpp ^ src\sslm_abi.cpp src\damped_greedy_antilm.cpp src\damped_greedy_topk.cpp src\damped_greedy_phaseD.cpp src\damped_greedy_phaseD_loop.cpp ^ src\gpu\superslm_gpu.cpp src\gpu\gpu_1p0.cpp ^ - tests\test_main.cpp tests\test_slm18x_saturation_census.cpp tests\test_slm19x_schema_damped_greedy.cpp tests\test_tiled_gemm.cpp /Fo:out\ /Fe:out\superslm_tests.exe ^ + tests\test_main.cpp tests\test_slm18x_saturation_census.cpp tests\test_slm19x_schema_damped_greedy.cpp tests\test_tiled_gemm.cpp tests\test_attn_rowsites.cpp /Fo:out\ /Fe:out\superslm_tests.exe ^ /link d3d12.lib dxgi.lib dxguid.lib if errorlevel 1 ( goto :hard_fail diff --git a/build_cert.bat b/build_cert.bat index 8fc4a1a1..49e3e825 100644 --- a/build_cert.bat +++ b/build_cert.bat @@ -11,7 +11,7 @@ call %VSDEVCMD% -arch=x64 -no_logo pushd %~dp0 if not exist out mkdir out cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude ^ - src\intmath.cpp tests\cert_intmath.cpp /Fo:out\ /Fe:out\cert_intmath.exe + src\intmath.cpp src\matmul.cpp tests\cert_intmath.cpp /Fo:out\ /Fe:out\cert_intmath.exe if errorlevel 1 (popd & exit /b 1) popd exit /b 0 diff --git a/docs/attention-rowsites-s1-progress.md b/docs/attention-rowsites-s1-progress.md new file mode 100644 index 00000000..a1ae32a0 --- /dev/null +++ b/docs/attention-rowsites-s1-progress.md @@ -0,0 +1,99 @@ +# Attention and per-row sites, slice S1: build progress + +The plan of record is the attention and per-row sites plan, rev 3.1, approved by the owner. This series builds +its slice S1 only (§4.1): 256-entry per-row tables for the SwiGLU sigmoid (`MlpActSite`), the residual landing +(`ResidualReconcileSite`) and the RMSNorm divide (`RmsNormSite`). Same pure function, same arguments; every tier +bit-identical to the v1.9.0 per-element code. + +**Base.** Tiled-matmul slice 1's branch at `fb56397` (the plan names `c3e0004`; `fb56397` is that plus one +documentation commit). Delivered as a patch series (`git am`), not pushed: that branch carries slice 1's draft PR. + +**Host.** A 4-vCPU cloud Xeon with AVX2, AVX-512F/BW/DQ and AVX-512 VNNI; GCC 13.3.0, Clang 18.1.3. The host is +shared with other agents, so every timing is best-of-N and noisy. + +States: **done**, **CI-only** (runs on the hosted legs; not runnable here), **box-only**, **not done** (with why). + +## Resume here + +1. Clone SuperSLM, `git checkout -b attn-s1 origin/claude/project-thread-c8iecr`, `git am` the series. +2. Cell 11.1(d) needs the 0.5B-width 1-layer synthetic artifact (sha256 `f0fd4886…6ed3`; about 9 minutes and + 7.4 GiB with `tools/consumer_reach/synth_artifact.py`). Point `SUPERSLM_ATTN_ROWSITES_ARTIFACT` at it and run each + suite binary from the repository root; without it the cell prints SKIPPED and the rest runs. +3. The golden pin is regenerated with `tools/gen_attn_rowsite_golden.cpp` built against the v1.9.0 tag (recipe in + its header). The CMake target of the same name builds it against the current tree and must print the same hash. +4. Mutants: `docs/attention-rowsites/s1/mutation-scripts/*.py`, each applied from the repository root to a clean + copy of the implementation commit; rebuild `superslm_tests` and `superslm_tests_avx2_forced` and run both. + +## S1 items + +| Item | State | Evidence | +|---|---|---| +| Red-first: cells, row-table counters (declared, not incremented), digest section, golden pin | done (commit 1) | `docs/attention-rowsites/s1/red-suites.txt`: GCC auto and forced SSE2/AVX2/AVX-512, 258 of 1,756 S1 checks fail, every one a counter assertion; every value assertion passes on the base | +| The three tables, `kRowTablesOn` (off only under forced scalar), `kRowTableMinWidth` = 512, counters (§3.6) | done (commit 2) | `src/forward/forward_sites.cpp` | +| GCC 13.3 suites: auto (AVX-512 here), forced SSE2, AVX2, AVX-512 | done, 0 failures | `suites.txt`: 27,288 / 27,230 / 27,246 / 27,246 checks (S1: 1,757 each) at the series head | +| Clang 18.1.3 suites, same four binaries | done, 0 failures | `suites-clang.txt` | +| MSVC / clang-cl forced AVX2, macOS arm64 (tables on, counters asserted) | CI-only | the hosted legs; nothing Windows- or arm64-specific in S1 (no SIMD, no switch) | +| Digests (6.2): auto and scalar/SSE2/AVX2/AVX-512 forced, GCC and Clang | done | all ten equal: GLOBAL `3a829091…`; `c_rowsites` `8836d5eb…`; sections 1–9 byte-identical to slice 1's (`18e45694…` GLOBAL before the new section) | +| Golden pin (6.3) from the v1.9.0 tag | done | `tests/attn_rowsite_golden_pin.h`: `8836d5eb…` over 634,120 values, matched by all eight suite binaries; SiLU-LUT, matmul and tiled goldens unchanged | +| End-to-end blobs (6.4) | done, all EQUAL | `blob-protocol.txt`: in-tree fixture (24 tokens), wide fixture, 0.5B-width 1- and 2-layer at 8/128/512 tokens × chunk budgets 1/8/whole, each + 32 greedy decode steps; blob and decoded tokens equal to the base's in all 33 comparisons (the N = 8 "whole" rows repeat the chunk-budget-8 rows) | +| Mutants (§9 S1 rows and the "all" counter rows) | done, 15 of 15 killed on both binaries | `mutants.txt` (below) | +| Path counters (§3.6, 11.1(a), 11.1(d)) | done | every 4.S1/1.S1/5.S1/7.S1 call asserts its own delta; 11.1(d) on p05_l1: prefill 256 / 128 / 256 taken, decode 64 / 32 / 64, every skipped 0, chain trace records 1,536 and 384 (128 and 32 per site name) | +| 3.S1 concurrency; sanitizers | done | new cell (see deviations). GCC TSan and ASan+UBSan builds of `superslm_tests` with the artifact: 27,288 checks, 0 failures, 0 sanitizer reports each (`suites.txt`); the hosted TSan and ASan legs are the CI copies | +| Branch-coverage floors (11.6) | not owed | S1 edits neither `intmath.cpp` nor `matmul.cpp` | +| Bench (10.1) and the saving | done | `bench.md`: **0.80 ms/token** measured against the plan's 0.87 | + +### Mutants + +Each killed on the auto binary and on forced AVX2 (the tables are tier-independent, so every S1 mutant sits in shared +code). The cell that turns red is the plan's named one in every case: + +| Mutant | Killed by | +|---|---| +| norm / SiLU / landing table read one entry off ("indexed by code + 127") | 4.S1, 1.S1, 5.S1, 6.3 (41 / 77 / 73 failures) | +| SiLU −128 read from an unset table entry | 4.S1's small-gate-scale −128 rows (11) | +| landing −128 read from an unset table entry | 4.S1 residual rows with −128 (32) | +| norm table built for [−127, 127] only (extra) | 4.S1 norm −128 rows (18) | +| landing flag OR-ed over the whole table | 7.S1c (7) | +| norm / SiLU / landing table built once per call site | 1.S1 and 4.S1 (30 / 57 / 41) | +| threshold `n ≥ 0` | counters at widths 1–511 (123) | +| threshold `n ≥ 513` | counters at width 512 (36) | +| threshold `SIZE_MAX` | counters, including 11.1(d) prefill and decode (138) | +| skipped increment deleted ("fallback increment deleted") | counters (123) | +| taken counted before the guard ("fast increment moved before the guard") | counters (123) | + +"AVX-512 dispatch runs the AVX2 body" does not apply to S1 (no SIMD body, no tier split). + +## Deviations from the plan, and why + +- **11.1(c) (the QK-norm attention fixture) is not built in S1.** Its table's first column is "S5 landed"; the + fixture exists to drive the Q31 path, and its row-table rows (every `…_skipped`, widths 64 and 256) add nothing + 4.S1 does not already assert at widths 1 to 511. S5 builds it and adds its forward hash to the pin header. +- **The golden pin is one hash per slice**, not one hash over every section's inputs, so no later slice + regenerates S1's. S1's covers `c_rowsites`' S1 entries (the digest section's value equals it until S3 appends). +- **11.1(d) is an environment-gated cell in the suite binaries**, since the plan does not commit the artifact. + S1 asserts the rows that exist in S1: the `rowtable_*` closed forms and the chain trace records. `requant_row`, + `softmax`, `pv` and `grouped` rows arrive with their slices. +- **3.S1: the plan's "existing concurrent-read stress cell … on the 0.5B-width artifact" does not exist.** The + suite's concurrent-read cells cover `SiluSigmoidQ15` and `GemmInt8AccumulateRow` only, and none takes an artifact. + S1 adds its own cell (8 threads × all three sites at table widths, compared with the reference), which the hosted + TSan leg runs; it was also run here under GCC TSan (below). +- **The blob tool needed two options** the plan's protocol assumes: `--chunk-budget=B` and `--decode=D` in + `--dump-blob` mode (`tools/t2147_chunk_batched_pins.cpp`). The in-tree fixture's context cap refuses 128 + 32 + positions on the base too, so its rows run at 24 tokens. +- **Base defect fixed in passing:** `sslm_bench_prefill_{avx2,avx512}_forced` did not compile on the base (the + forced library exports the instrument macro, and the tool then includes a `tests/` header without `tests/` on its + path). One `target_include_directories` line each. +- **Decode saving not resolved on this host** (bench.md): about 2% of a reduced-layer decode step against ±10% + pair-to-pair noise. The prefill reading agrees with the per-site method. + +## What the plan got wrong or left loose (for the next revision) + +- §0/§4.1's 0.87 ms/token is close: measured 0.80 by the plan's own per-site method, 0.96–0.99 from reduced-layer + prefill. The loop-only speedups the plan quotes (6.0×, 3.2×, 2.57×) are not what a site call sees (1.50×, 1.39×, + 1.25×), because the funnel and the allocation stay; the absolute saving is what matches. +- 3.S1 cites a cell that does not exist (above). +- The engine base is `fb56397`, not `c3e0004` (one documentation commit later). + +## ID-PENDING list + +Nothing minted here. diff --git a/docs/attention-rowsites-s2-progress.md b/docs/attention-rowsites-s2-progress.md new file mode 100644 index 00000000..77eff29a --- /dev/null +++ b/docs/attention-rowsites-s2-progress.md @@ -0,0 +1,76 @@ +# Attention and per-row sites, slice S2: build progress + +The plan of record is the attention and per-row sites plan, rev 3.1, approved by the owner. This series builds +its slice S2 only (§4.2, §5.2): `GemmProbQ15Accumulate` on int16 multiply-add, per head, on the AVX2 and +AVX-512BW tiers, bit-identical to the v1.9.0 loop on every tier. + +**Base.** The S1 series (three patches) on tiled-matmul slice 1's branch at `fb56397`. Delivered as a patch series +(`git am` after S1's), not pushed: that branch carries slice 1's draft PR. + +**Host.** The same 4-vCPU cloud Xeon as S1 (AVX2, AVX-512F/BW/DQ, AVX-512 VNNI; GCC 13.3.0, Clang 18.1.3), shared +with other agents, so every timing is best-of-N and noisy. + +States: **done**, **CI-only**, **box-only**, **not done** (with why). + +## Resume here + +1. Clone SuperSLM, `git checkout -b attn-s2 origin/claude/project-thread-c8iecr`, `git am` the S1 series, then + this one. +2. Cell 11.1(d) needs the 0.5B-width 1-layer synthetic artifact (sha256 `f0fd4886…6ed3`), as in S1: + `SUPERSLM_ATTN_ROWSITES_ARTIFACT=`, suites run from the repository root. + +## S2 items + +| Item | State | Evidence | +|---|---|---| +| Red-first: cells (11.2, 4.S2, 2.S2, 7.S2, 6.1, 6.3, 11.1(d) prob·V rows), per-tier prob·V counters (declared, not incremented), the 11.2 selector as a stub that selects v1.9.0 code everywhere, digest section `c32_attention`, S2 golden hash | done (commit 1) | `docs/attention-rowsites/s2/red-suites.txt`: auto and forced AVX2/AVX-512 269 failures each, forced SSE2 3 (the selector); every value assertion passes on the base. Digests: sections 1–10 byte-identical to S1's on all five legs; `c32_attention` `b0d1a6cd…` on all five | +| 11.1(d) data terms re-derived on the base | done | `pv-data-terms.txt`: prefill 15 rows fail the int16 condition (14 width-1 rows, plus position 2 head 13's {0, 32,768, 0}), decode 0, as the plan measured | +| Golden pin (6.3) from the v1.9.0 tag | done | `golden.txt`: S2 `b0d1a6cd…` over 30,100 values; S1's hash unchanged | +| `GemmProbQ15Accumulate` tiered: the int16 condition and `head_dim % 16` guard, AVX2 and AVX-512BW `vpmaddwd` bodies over register-formed probability pairs, an accumulate-into core shared by the zeroing entry, per-tier counters in each body (fast) and in the dispatcher (fallback) | done (commit 2) | `src/matmul.cpp` | +| The 11.2 selector and its wiring; `SUPERSLM_SITES_AVX512_MSVC`, default 0; both forced AVX-512 Windows legs build with it on | done (commit 2) | `src/matmul.cpp`, `.github/workflows/tests.yml` | +| Linkage (11.3) and isolation checkers name the new attributed functions | done (commit 2) | `check_tiled_matmul_linkage.py` OK on the auto and both forced objects | +| GCC 13.3 suites: auto (AVX-512 here), forced SSE2, AVX2, AVX-512, at the final code plus commit 3's blocking rows | done, 0 failures | `suites.txt`: 27,927 / 27,869 / 27,885 / 27,885 checks (attn-rowsites 2,396 each) | +| Digests (6.2): auto and scalar/SSE2/AVX2/AVX-512 forced | done | all five equal and equal to the red run's: GLOBAL `ec7016a0…`, `c32_attention` `b0d1a6cd…` = the v1.9.0 pin | +| 4.S2 kernel-blocking rows: head_dim {32, 48, 80, 96, 144, 160, 208, 224, 240} x width {1, 2, 3, 64, 65}, every AVX2 and AVX-512 block count and the AVX-512 16-dimension tail behind a full block (not in the golden set) | done (commit 3) | `tests/test_attn_rowsites.cpp`, 90 checks | +| Clang 18.1.3 suites (four binaries) and digests (five legs) | done, 0 failures | `suites-clang.txt`: same counts; digests equal GCC's (GLOBAL `ec7016a0…`) | +| Mutants (§9): every §9 S2 mutant plus four extras; per-tier arithmetic mutants per body (AVX2 body, AVX-512 32-dimension unit, AVX-512 16-dimension tail) | done: 20 of 20 killed on every binary where they execute | `mutants.txt`, `mutation-scripts/`. Run at the final implementation commit before commit 3's 90 rows existed, which only add kills | +| 11.3 linkage vitality (the plan's plant) | done | `mutants.txt` tail: the checker goes red on an external-linkage AVX2 prob·V function; OK at the final code on all three objects | +| ASan+UBSan (auto, forced AVX2, forced AVX-512), TSan (auto) | done, 0 failures, 0 reports | `sanitizers.txt` | +| Save blobs (6.4), as S1 ran them | done, 45 of 45 EQUAL | `blob-protocol.txt`: 30 rows of base vs auto candidate over wide_l8 / p05_l1 / p05_l2 / in-tree at 8, 128 and 512 tokens and three chunk budgets, then 15 rows with the in-tree fixture at ids:8 and the synthetics against the forced-AVX2 candidate; the blob hashes equal S1's | +| Golden generator consistency | done | the generator built against this series prints the v1.9.0 hashes (`golden.txt`) | +| 11.1(d) with the prob·V rows | done | calls = L·H·N per window, prefill `pv_fallback` 15 and decode 0 on the AVX2 and AVX-512 binaries (no counter moves on SSE2), in every suite run above | +| Bench (10.1): the saving against §0 | done | `bench.md`: prefill 0.825 / 3.38 / 6.74 ms/token at T = 128 / 512 / 1,024 (plan 0.95 / 4.38 / 8.47), decode 3.95 at context 300 and 7.97 at 600 (plan 3.83 / 7.72); forced AVX2 within 2% of those savings. Forward-level check agrees at T = 512 and in decode | +| Branch-coverage floors (11.6) | CI-only (local replica recorded) | `coverage.txt`: a local five-binary clang-18 replica passes every floor on this AVX-512 host (matmul.cpp 89.19 → 91.67). Without AVX-512 profiles it would drop to about 70.98 against the 72.22 floor. Allowlist lines added for the AVX-512 bodies and the two implicit switch defaults. The floors are not re-pinned, because re-pinning copies the hosted leg's value and the series is not pushed | +| MSVC / clang-cl: switch compiled in (default 0), selector truth table, both forced AVX-512 Windows legs build with `/DSUPERSLM_SITES_AVX512_MSVC=1` | CI-only | nothing MSVC can be compiled here. The 11.2 truth table covers the (MSVC, switch 0/1) rows on every binary. The `s2_msvc_switch_ignored` mutant dies | +| MSVC AVX-512 execution (the switch's flip to 1 by default) | box-only | §3.2: needs an MSVC AVX-512 build to execute the path | +| CHANGELOG `[Unreleased]` and `docs/platform-support.md` section | done (commit 3) | | + +## Deviations from the plan as written + +1. **The AVX-512 body uses 32-dimension in-lane units, not a 16-dimension one.** A straight port of the AVX2 unit to 512-bit + registers measured slower than AVX2 (0.42 against 0.30 µs per call at width 128). `_mm512_cvtepi8_epi16` of a 256-bit + unpack keeps pairs in 128-bit lanes, so each unit covers 32 dimensions whose two halves are stored 16 apart. A + 16-dimension tail handles head_dim % 32 == 16. It measures 0.266 µs. The arithmetic is unchanged: the same pairs, the + same `vpmaddwd`, and the same lane bound. The half-order store and the tail have their own mutants, and both die. +2. **The golden pin carries one hash per slice.** The S1 hash is unchanged, and S2's is added beside it, generated from the + v1.9.0 tag's library. +3. **The blocking rows are suite rows, not golden-set rows.** Adding them to `attention_cases.h` would change S2's pin. + They test the kernel's own blocking against the test-side v1.9.0 loop, under the path rule. +4. **The build-configuration record does not gain `SUPERSLM_SITES_AVX512_MSVC`.** The record states the tier and the force + macros. The switch acts only on MSVC builds, where the 11.2 truth table and the Windows legs cover it. +5. **Clang's first build segfaulted once.** It was a compiler crash under parallel load. A rebuild at -j3 succeeded, and every + Clang run above is on that rebuild. +6. **Floors are not re-pinned (11.6).** See the table: the plan's step 3 needs the hosted leg's measurement. + +## What the plan got wrong + +- **The prefill saving estimate is 15–25% high.** Measured 0.825 / 3.38 / 6.74 against 0.95 / 4.38 / 8.47. The estimate at + T = 512 and 1,024 exceeds this host's entire measured v1.9.0 prob·V cost (3.56 and 7.18 ms/token). The kernel ratio + (16–21×) beats the spike's 10.7–14.2×, so the error is in the base per-key cost the arithmetic assumed. Decode matches + (103%). +- **§4.2's AVX-512 port is not "the same body in wider registers".** R2's simplest reading is slower than AVX2. The unit + layout has to follow the in-lane widening; see deviation 1. +- **11.6 cannot be satisfied inside a slice delivered as patches.** On a runner without AVX-512, S2 lowers matmul.cpp's + measured branch coverage by about 4.5 points, which is below today's floor. That makes the owner's floor call (G27) likely + for S2, not merely possible. + diff --git a/docs/attention-rowsites-s3-progress.md b/docs/attention-rowsites-s3-progress.md new file mode 100644 index 00000000..cdc8c23d --- /dev/null +++ b/docs/attention-rowsites-s3-progress.md @@ -0,0 +1,80 @@ +# Attention and per-row sites, slice S3: build progress + +The plan of record is the attention and per-row sites plan, rev 3.1, approved by the owner. This series builds its +slice S3 only (§4.3, §5.3): the requant funnel's element loop as one row leaf, `RequantRowWide`, in 4 (AVX2) or 8 +(AVX-512BW) unsigned 64-bit lanes, bit-identical to the per-element `RequantTokenCodeWide` on every tier. + +**Base.** The S1 and S2 series (three patches each) on tiled-matmul slice 1's branch at `fb56397`. Delivered as a +patch series (`git am` after S2's), not pushed. + +**Host.** The same 4-vCPU cloud Xeon as S1 and S2 (AVX2, AVX-512F/BW/DQ, AVX-512 VNNI; GCC 13.3.0, Clang 18.1.3), +shared with other agents, so every timing is best-of-N and noisy. + +States: **done**, **CI-only**, **box-only**, **not done** (with why). + +## Resume here + +1. Clone SuperSLM, `git checkout -b attn-s3 fb56397`, `git am` the S1 series, the S2 series, then this one. +2. Cell 11.1(d) needs the 0.5B-width 1-layer synthetic artifact (sha256 `f0fd4886…6ed3`), as in S1 and S2: + `SUPERSLM_ATTN_ROWSITES_ARTIFACT=`, suites run from the repository root. Run the four suite binaries with + separate `TMPDIR`s if they run at once: an older cell writes a fixed temp-file name. +3. The fp-free scan on a GCC build needs main's 90e48de (a `TiledGemmAvx512` fix) until this series is rebased onto it. +4. Open for the owner: the branch-coverage floors on a runner without AVX-512 (`coverage.txt`, G27). + +## S3 items + +| Item | State | Evidence | +|---|---|---| +| Red-first: cells (4.S3 both fences, 7.S3 corner premises, 6.1, funnel call site, 6.3, 11.1(d) requant rows), per-tier `requant_row` counters (declared, not incremented), `RequantRowWide` declared with a stub that runs the element loop, the S3 rows appended to `c_rowsites`, the S3 golden hash | done (commit 1) | `docs/attention-rowsites/s3/red-suites.txt`: auto and forced AVX2/AVX-512 fail 10 assertions each, every one a path assertion; forced SSE2 0. Every value assertion passes on the base. Digests: every section but `c_rowsites` byte-identical to S2's on all five legs; `c_rowsites` `d0272180…` on all five | +| Golden pin (6.3) from the v1.9.0 tag | done | `golden.txt`: S3 `3e3abed7…` over 3,567,018 values; S1's and S2's unchanged | +| `RequantRowWide` in `src/intmath.cpp`: AVX2 (4 lanes) and AVX-512BW (8 lanes, F and BW only, no mask register) bodies of the §5.3 identity in unsigned 64-bit lanes with a logical H shift, P formed from r's 32-bit halves, clamp and sign restore as the element code; the tail and the scalar/SSE2 tiers run `RequantTokenCodeWide`; per-tier `requant_row` counter inside each body; dispatch through S2's `DispatchSitesKernel` (so `SUPERSLM_SITES_AVX512_MSVC` covers it) | done (commit 2) | `src/intmath.cpp` | +| The funnel's element loop becomes one `RequantRowWide` call | done (commit 2) | `src/forward/checked_chain_funnel.cpp` | +| GCC 13.3 suites at the implementation commit: auto (AVX-512 here), forced SSE2, AVX2, AVX-512 | done, 0 failures | `suites.txt`: 109,744 / 109,686 / 109,702 / 109,702 checks (attn-rowsites 84,213 each); digests equal the red run's on all five legs (GLOBAL `4892b5be…`) | +| 6.2 digests: `c_rowsites` (now S1 and S3 entries, as §3.3 evidence 2 specifies) equal on every leg; slice 1's sections byte-identical to `docs/tiled-matmul-slice1/s1-b/` | done | every section slice 1 recorded is unchanged on all five legs; `c_rowsites` `d0272180…` on all five and on both compilers | +| Clang 18.1 suites, same four binaries and five digest legs | done, 0 failures | `suites-clang.txt`: the same counts; every digest equals GCC's apart from its compiler line | +| 11.1(d) exact counter counts: `requant_row` 1,536 prefill and 384 decode on the artifact, on the running tier's counter only | done | asserted in the suite on every binary; the never-entered mutant reads +0 against both | +| fp-free scan (`scan_build_output.py --target superslm --isa x86-64`), allow-lists unchanged | done (Clang); GCC blocked by the base | `fp-scan.txt`. Clang: PASS. GCC: one REJECT, `TiledGemmAvx512`, which the S2 head's own build also rejects and main fixes in 90e48de; with that fix applied temporarily, PASS. `intmath.cpp.o` is clean on both compilers | +| §9 mutants, each tier's body separately, on auto, forced AVX2 and forced AVX-512 | done, all killed | `mutants.txt`, `mutation-scripts/` (18): the 8 §9 mutants plus the dispatch mutant and 6 extras die on every binary that runs the tier they mutate; the 3 withdrawn-mutant confirmations survive everywhere, as rev 3 predicts | +| Sanitizers: ASan+UBSan (auto, forced AVX2, forced AVX-512), TSan (auto) | done, 0 reports | `sanitizers.txt` | +| Save-blob protocol (6.4) against the S1 head's build | done, 46 of 46 EQUAL | `blob-protocol.txt`: auto and forced-AVX2 candidates | +| 11.6 branch coverage (clang-18 replica) | done locally; floors not re-pinned | `coverage.txt`: full union OK (intmath.cpp 88.83%). Projected for a runner without AVX-512: intmath.cpp 86.70%, **below its 87.93% floor**; allowlist lines added | +| CHANGELOG and `docs/platform-support.md` entries | done (commit 3) | the Unreleased entry and a section beside S2's | +| 10.1 bench against §0's 1.22 ms/token | done | `bench.md`: **2.53 ms/token saved on AVX2, 2.66 on AVX-512** (4.5× and 5.5× on the funnel call) | +| 11.4 forward-leaf check lists `RequantRowWide`; planted call from `forward_sites.cpp` turns it red | done (commit 2) | `check_no_forward_leaf_calls.py` (+ its 83-cell pytest, green); plant in `leaf-plant.txt` | +| 11.3 linkage checker gains the intmath.cpp objects (population `RequantRowAvx`, the record only in matmul objects); CI job passes all six objects; isolation checker's prose names the two bodies | done (commit 2) | OK on all six objects; plant (RequantRowAvx2 given external linkage) red, `linkage-plant.txt` | +| 11.5 the four recipes of G21 gain `src\matmul.cpp`; GCC link check | done (commit 2) | `recipe-link.txt`: without matmul.cpp all eight link units fail on `detail::ActiveGemmTier` / `DispatchSitesKernel`; with it all link. The five t2296 cells include ``, so their engine source sets are linked as closed sets (`-shared -Wl,--no-undefined`) | + +## Deviations from the plan as written + +1. **The AVX2 body does |x|, the clamp and the sign in 32-bit lanes.** The plan's reading (clamp and sign on the 64-bit lane, + as the element code does) compiles on Clang to `vblendvpd` and `vxorpd`, FP-domain encodings the fp-free scan rejects. That + happened for the `blendv` spelling and for the and/andnot spelling alike. The body now uses `vpabsd` (the low dword of |x|, + exact since |x| ≤ 2^31), the unsigned 32-bit minimum with the magnitude's high dword folded into bit 7, and `vpsignd` with + ±1 from x's high dword. The identity and every 64-bit intermediate are unchanged. The AVX-512 body keeps 64-bit lanes + (`vpabsq`, `vpminuq`, `vpsraq`), which Clang does not lower to FP encodings. +2. **The AVX2 byte-pick constant has distinct halves.** Identical halves let Clang load it with `vbroadcasti128`, which is not on + the scan's vetted move list. The upper half's never-read bytes are now 1 instead of −1. +3. **11.5's link check covers the five t2296 cells as closed-set shared links, not as executables.** They include + ``, so they cannot link as programs here. Their engine source sets are linked with `-shared + -Wl,--no-undefined`, which fails without `src\matmul.cpp` and succeeds with it; `build_cert.bat` and both + `build_inspect.bat` lines link as executables (`recipe-link.txt`). +4. **The golden pin carries one hash per slice**, as in S2: S1's and S2's are unchanged, and S3's is added from the v1.9.0 tag's + library. +5. **Floors are not re-pinned (11.6).** Step 3 needs the hosted leg's measurement. +6. **The fp scan on GCC was passed with main's 90e48de applied temporarily.** It is not part of this series. S3 adds nothing + the GCC scan rejects. + +## What the plan got wrong + +- **The saving is about twice the estimate.** 2.53 ms/token on AVX2 against 1.22, and 2.66 on AVX-512 against about 1.6. The + lanes perform as the spike found. The estimate took the spike's base loop cost (19.8 µs at 4,864); here replacing that loop + saves 24.8 µs on AVX2, more than the spike's whole loop. A per-element out-of-line call explains only about 12% of the + base's cost (`bench.md`). +- **The t2296 cells are Windows-only** (deviation 3). "Link all five executables" cannot be run off Windows. +- **§4.3's lane description is not portable to Clang under the fp-free scan** (deviations 1 and 2). The plan anticipated no + compiler-specific lowering, and the scan's allow-list is correctly not widened for it. +- **11.6 cannot pass on a runner without AVX-512.** S3 takes `intmath.cpp` about 1.2 points below its floor there, on top of + S2's `matmul.cpp` shortfall. The owner's floor call (G27) now covers two files. +- **The base's GCC fp scan was already failing** (`TiledGemmAvx512`, fixed on main by 90e48de). The plan assumed a clean base. +- **§9's loop-bound mutant is killed first by an older cell.** In the suite's own order, a stack-row `test_main` cell trips + glibc's stack protector before 4.S3 runs. Run first, 4.S3's sentinel cell kills it as the plan intends (`mutants.txt`). diff --git a/docs/attention-rowsites-s4-progress.md b/docs/attention-rowsites-s4-progress.md new file mode 100644 index 00000000..f3777734 --- /dev/null +++ b/docs/attention-rowsites-s4-progress.md @@ -0,0 +1,93 @@ +# Attention and per-row sites, slice S4: build progress + +The plan of record is the attention and per-row sites plan, rev 3.1, approved by the owner. This series builds its +slice S4 only (§4.4, §5.4): a guarded fast path inside `SoftmaxRowQ15` on the AVX2 and AVX-512BW tiers, with the shipped +body as the fallback, bit-identical (bool and every probability) to the v1.9.0 body on every tier. + +**Base.** The S1, S2 and S3 series (three patches each) on tiled-matmul slice 1's branch at `fb56397`. Delivered as a +patch series (`git am` after S3's), not pushed. + +**Host.** The same 4-vCPU cloud Xeon as S1–S3 (AVX2, AVX-512F/BW/DQ, AVX-512 VNNI; GCC 13.3.0, Clang 18.1.3), shared with +other agents, so every timing is best-of-N and noisy. + +States: **done**, **CI-only**, **box-only**, **not done** (with why). + +## Resume here + +1. Clone SuperSLM, `git checkout -b attn-s4 fb56397`, `git am` the S1, S2 and S3 series, then this one. +2. Cell 11.1(d) needs the 0.5B-width 1-layer synthetic artifact (sha256 `f0fd4886…6ed3`), as in S1–S3: + `SUPERSLM_ATTN_ROWSITES_ARTIFACT=`, suites run from the repository root, one `TMPDIR` per binary when they run + at once. +3. The fp-free scan on a GCC build needs main's 90e48de (a `TiledGemmAvx512` fix) until this series is rebased onto it. + +## S4 items + +| Item | State | Evidence | +|---|---|---| +| Red-first: cells (2.S4, 4.S4 grid, inside corners, correction rows, aliased rows, width 0, 6.1, 6.3, 11.1(d) softmax rows), the test-side guard copy and estimate replica, per-tier softmax counters (declared, not incremented), the S4 rows appended to `c32_attention`, the S4 golden hash | done (commit 1) | `docs/attention-rowsites/s4/red-suites.txt`: auto and forced AVX2/AVX-512 fail 3,125 assertions each, every one a path assertion; forced SSE2 0. Every value assertion passes on the base. Digests: every section but `c32_attention` byte-identical to S3's on all five legs; `c32_attention` `6bb5971d…` on all five | +| Golden pin (6.3) from the v1.9.0 tag | done | `golden.txt`: S4 `2e47ea3c…` over 268,078 values; S1's, S2's and S3's unchanged | +| 11.1(d) data term re-derived on the base | done | `softmax-data-terms.txt`: 0 rows outside the guard in both windows (1,792 prefill, 448 decode rows), as the plan measured | +| Implementation: the row guard, the AVX2 and AVX-512BW bodies (integer estimates, exact corrections), the dispatcher and its fallback counter | done (commit 2) | GCC 13.3 and Clang 18.1: auto (AVX-512 here) and forced SSE2/AVX2/AVX-512 suites 0 failures (116,129 / 116,071 / 116,087 / 116,087 checks); S4 golden `2e47ea3c…` on every binary; digests equal the red run's on all five legs of both compilers | +| 11.3 linkage checker: the S4 bodies join the population; vitality plant | done | `linkage-plant.txt`: a planted external `SoftmaxExpAvx2Planted` turns it red; restored, OK | +| 6.2 digests: every section equal to the red run's on all five legs, both compilers | done | `suites.txt`, `suites-clang.txt`, `sslm_axis_digest*.txt`: GLOBAL `bc7b1cbe…` on all ten; `c32_attention` `6bb5971d…`; every other section equal to S3's record | +| 11.1(d) exact counter counts: `softmax` fast L·H·N (1,792 prefill, 448 decode on the artifact), fallback 0, on the running tier's counters only | done | asserted in the suite on every binary; the always-fallback mutant reads fast +0 against both windows | +| fp-free scan (`scan_build_output.py --target superslm --isa x86-64`), allow-lists unchanged | done (Clang); GCC with 90e48de applied | `fp-scan.txt`. Clang: PASS (490 ACCEPT). GCC: only the base's `TiledGemmAvx512` rejects, as in S3; with 90e48de applied temporarily, PASS (561 ACCEPT). `intmath.cpp.o` is clean on both | +| §9 mutants, each tier's body separately, on auto, forced AVX2 and forced AVX-512 | done, all killed | `mutants.txt`, `mutation-scripts/` (30): the 22 §9 S4 mutants (14 guard, 6 correction, 2 dispatcher), the 3 all-slice rows and 5 extras die on every binary that runs the code they mutate. Three guard-dropped mutants die by SIGFPE in the 4.S4 grid, before the 2.S4 rows run | +| Sanitizers: ASan+UBSan (auto, forced AVX2, forced AVX-512), TSan (auto) | done, 0 reports | `sanitizers.txt` | +| Save-blob protocol (6.4) against the S1 head's build | done, 46 of 46 EQUAL | `blob-protocol.txt`: auto and forced-AVX2 candidates; every hash equals S3's record | +| 11.6 branch coverage (clang-18 replica) | done locally; floors not re-pinned | `coverage.txt`: full union OK (intmath.cpp 91.32%, all 54 new branches covered). Projected for a runner without AVX-512: intmath.cpp 83.47%, **below its 87.93% floor** (S3 already 86.70%); allowlist notes added | +| CHANGELOG and `docs/platform-support.md` entries | done (commit 3) | the Unreleased entry and a section beside S3's | +| 10.1 bench against §0's 0.15 / 0.54 / 1.15 prefill and 0.53 decode ms/token | done | `bench.md`: AVX2 **0.10 / 0.46 / 0.91 prefill, 0.53 decode** (AVX-512 0.10 / 0.48 / 0.91, 0.54); 3.4–3.6× on rows of 128–1,024 keys; a one-key row 0.03 µs slower | + +## Deviations from the plan as written + +1. **The estimates are integer, not IEEE double.** §4.4 and §5.4 estimate z = floor(a / q_ln2) and p = floor(e·2¹⁵ / denom) + in IEEE double. The library is floating-point-free: `tests/ci/scan_build_output.py` gates `libsuperslm.a` on "no + floating-point arithmetic", and its allow-list is not to be widened. §5.4's own floating-point bullet names what carries + over: the estimates only need to be within one of the floor, because exactness comes from the exact integer corrections. + The integer estimates keep the plan's structure (estimate, then one exact correction each way): + - z: `(a·inv_z) >> kz` with `inv_z = floor(2^kz / q_ln2)`, `kz = 30 + bit_width(q_ln2)`. A floored reciprocal never + overestimates, so, as §5.4 step 3 argues for the double, **only the upward z correction can fire**; the downward one + stays as defensive code with no mutant owed. + - p: `(e·R) >> 47` with `R = round(2^62 / denom)`. |error| ≤ e / 2^48 ≤ 1/2, so **both p corrections are live**, as + §5.4 step 4 has them. + - The plan's p-downward steering constants (q_b = 11,863,283, M within 2⁻²⁴ of 2⁴⁷) were tuned to the double. With the + integer estimate, M·R / 2⁴⁷ sits just below the integer R when M ≈ 2⁴⁷, so the estimate for e = M never overshoots + there. The steered generator keeps the plan's shape (denominator first, then a row summing to it; the plan's two + constant shapes) with M near ¾ and 7⁄10 of 2⁴⁷ (`tests/support/attention_cases.h`, `SmSteeredPDownRows`). +2. **The golden pin carries one hash per slice**, as in S2 and S3. +3. **The off-ratio witness fails three conjuncts, not two.** §8 2.S4 says `kSoftmaxRowOffRatioWitness` fails q_c ≥ 0 and + M ≥ 1. Its q_ln2 is 3,000,000,001 against q_b = 10, so it fails q_ln2 ≤ 2·q_b + 1 too. It stays in the set as a + realistic hostile row; the cell asserts "more than one". + +4. **The fp scan on GCC was passed with main's 90e48de applied temporarily.** It is not part of this series. S4 adds nothing + the GCC scan rejects. +5. **Floors are not re-pinned (11.6).** Step 3 needs the hosted leg's measurement. +6. **The 11.3 vitality plant is an added external function, not a moved body.** S3's plant (a body given external linkage) + does not work for S4: every S4 body takes `SoftmaxFastRow`, an anonymous-namespace type, so the moved function stays + local and the checker rightly stays green. A planted external target-attributed function turns it red + (`linkage-plant.txt`). + +## What the plan got wrong + +- **Its estimates are floating point, which the library forbids** (deviation 1). §4.4 and §5.4 estimate both divides in IEEE + double; `scan_build_output.py` gates `libsuperslm.a` on no floating-point arithmetic, with an allow-list not to be widened. The + argument carries over to integer estimates unchanged, because it only ever needed an estimate within one. +- **The p-downward steering constants are tied to the double** (deviation 1). At q_b = 11,863,283 the integer estimate never + overshoots, so the plan's generator finds no row. With M near ¾ and 7⁄10 of 2⁴⁷ it finds them at once. The integer estimate + also fires the p-downward correction on ordinary grid rows (44 of the grid's rows die under its skip mutant), where the plan + measured about 2⁻³⁸ per row for the double. +- **The off-ratio witness fails three conjuncts, not two** (deviation 3). +- **§0's prefill figures assume a larger per-element saving than its decode figure.** Prefill 0.15 / 0.54 / 1.15 implies 6.9 / + 6.3 / 6.7 ns saved per element, decode 0.53 implies 5.2. Measured, the AVX2 body saves about 5.2 ns per element on long rows, + so decode lands on the estimate and prefill at 65% / 86% / 79% of it. A fixed per-row cost (the guard pass and two + divides) makes short rows cost more, which takes a further share at T = 128 (`bench.md`). +- **11.6 cannot pass on a runner without AVX-512**, now by 4.5 points on `intmath.cpp` (S3 1.2). The owner's floor call (G27) + covers it. +- **§9's killing cell for three guard-dropped mutants is reached late.** q_ln2 ≥ 1, q_c ≥ 0 and M ≥ 1 dropped each trap + (integer divide by zero) in the 4.S4 grid, whose realistic triples include constants outside the guard, before the 2.S4 rows + §9 names run. The mutants are killed; the named signal (bool and counter) is not the one observed. + +## ID-PENDING list + +Nothing minted here. diff --git a/docs/attention-rowsites-s5-progress.md b/docs/attention-rowsites-s5-progress.md new file mode 100644 index 00000000..9ccb7e00 --- /dev/null +++ b/docs/attention-rowsites-s5-progress.md @@ -0,0 +1,95 @@ +# Attention and per-row sites, slice S5: build progress + +The plan of record is the attention and per-row sites plan, rev 3.1, approved by the owner. This series builds its +slice S5 only (§4.5, §5.5): `QkQ31ScoreRow`, every key's Q31 score for one query head (the Qwen3 QK-norm path), in +three 16-bit pieces on the AVX2 and AVX-512BW tiers, bit-identical to the per-key `QkQ31Score` on every tier, and +called by both layer loops in place of their per-key loops. + +**Base.** The S1, S2, S3 and S4 series (three patches each) on tiled-matmul slice 1's branch at `fb56397`. Delivered as +a patch series (`git am` after S4's), not pushed. + +**Host.** The same 4-vCPU cloud Xeon as S1–S4 (AVX2, AVX-512F/BW/DQ, AVX-512 VNNI; GCC 13.3.0, Clang 18.1.3), shared +with other agents, so every timing is best-of-N and noisy. + +States: **done**, **CI-only**, **box-only**, **not done** (with why). + +## Resume here + +1. Clone SuperSLM, `git checkout -b attn-s5 fb56397`, `git am` the S1, S2, S3 and S4 series, then this one. +2. Cell 11.1(d) needs the 0.5B-width 1-layer synthetic artifact (sha256 `f0fd4886…6ed3`), as in S1–S4: + `SUPERSLM_ATTN_ROWSITES_ARTIFACT=`, suites run from the repository root, one `TMPDIR` per binary when they run + at once. Cell 11.1(c) needs nothing: its fixture is in-tree (`tests/support/qk_attention_fixture.h`). +3. The fp-free scan on a GCC build needs main's 90e48de (a `TiledGemmAvx512` fix) until this series is rebased onto it. + +## S5 items + +| Item | State | Evidence | +|---|---|---| +| Red-first: cells (4.S5 grid, 7.S5b margin corners, 7.S5c ties, 7.S5d inside corners, width 0, 2.S5 hostile rows, 6.1, 6.3, 11.1(c) on the widened QK-norm fixture, 11.1(d)'s q31_row rows), the test-side guard copy, per-tier q31_row counters (declared, not incremented), `QkQ31ScoreRow` declared with a stub that runs the per-key loop, the S5 rows appended to `c32_attention`, the S5 golden hash and the fixture hash | done (commit 1) | `docs/attention-rowsites/s5/red-suites.txt`: auto and forced AVX2/AVX-512 fail 358 assertions each, every one a q31_row path assertion; forced SSE2 0. Every value assertion passes on the base. Digests: every section but `c32_attention` byte-identical to S4's on all five legs; `c32_attention` `ddbdb76e…` on all five | +| The widened QK-norm fixture (11.1(c)): its premise on the base | done (commit 1) | `fixture-premise.txt`: every step Ok in all three runs, which hash alike; 96 softmax rows, all inside §5.4's guard; 4 prob-V rows fail the int16 condition (position 0's width-1 rows) | +| Golden pin (6.3) from the v1.9.0 tag | done (commit 1) | `golden.txt`: S5 Q31-row `daea9a39…` over 33,618 values; fixture `336b8d41…` over 14,384; S1–S4 unchanged | +| `QkQ31ScoreRow` on the AVX2 and AVX-512BW tiers, both layer loops calling it | done (commit 2) | GCC 13.3 and Clang 18.1, auto (AVX-512 here) and forced SSE2/AVX2/AVX-512: 0 failures on all eight binaries; S5 and fixture hashes equal the pins; digests equal the red run's on all five legs, both compilers (GLOBAL `f740f833…`) | +| 11.3 linkage checker: the S5 bodies and the forward_sites.cpp objects join; vitality plant | done (commit 2) | `linkage-plant.txt`: a planted external `Q31RoundAvx2Planted` turns it red on all three forward_sites objects; restored, OK. The CI job passes the three forward_sites.cpp objects | +| Isolation checker prose names the S5 bodies | done (commit 2) | `check_matmul_avx_isolation.py` exits 0; its population test passes (31) | +| 4.S5 channel-tail rows (head_dim not a multiple of 4 or 16, inside the guard), from the coverage replica | done (commit 3) | `coverage.txt`: lines 654 and 699 were untaken; 90 rows added outside the golden set; `x_pads_nonzero` dies on them and nowhere else | +| Suites and digests, GCC and Clang, every tier | done (commit 3) | `suites.txt`, `suites-clang.txt`: 0 failures on all eight binaries; S5 `daea9a39…` and fixture `336b8d41…` everywhere; digests GLOBAL `f740f833…` on all ten legs, equal to red | +| fp-free scan, both compilers | done (commit 3) | `fp-scan.txt`: Clang PASS on auto and both forced AVX libraries; GCC fails only on the base's `TiledGemmAvx512` in matmul.cpp.o and PASSes on all three with main's 90e48de applied temporarily; forward_sites.cpp.o clean everywhere; allow-lists unchanged | +| §9 mutants | done (commit 3) | `mutants.txt`: the 11 killable §9 rows (12 scripts, ties per tier) and 11 extras killed on every binary that runs their code; "a₂ by logical shift" equivalent (see below) | +| Sanitizers | done (commit 3) | `sanitizers.txt`: ASan+UBSan on auto, forced AVX2 and forced AVX-512 (both suites) and TSan on auto: 0 reports | +| Save-blob protocol (6.4) and the QK-norm equality | done (commit 3) | `blob-protocol.txt`: 66 of 66 rows EQUAL (auto and forced AVX2), row for row S4's hashes; the QK-norm fixture pinned to v1.9.0 in both loops on every binary; the Qwen3-width forward probe's outputs equal across base and S5, auto and AVX2, at T = 128, 512 and 1,024 | +| Coverage replica (11.6), projected without AVX-512 | done (commit 3), indicative | `coverage.txt`: intmath.cpp and matmul.cpp unchanged from S4 (S5 edits neither; forward_sites.cpp is outside the leg's glob and has no floor); every S5 side covered with five profiles; without AVX-512 only QkQ31RowAvx512 and the dispatcher's AVX-512 arms are lost. Floors not re-pinned | +| Bench against §0 (10.1) | done (commit 3) | `bench.md`: AVX2 11.5 → 0.47 ms/token at T = 128 and 97.6 → 3.23 at T = 1,024, against §0's 10.9 → 0.5 and 87 → 3.9; AVX-512 9.5 → 0.47 and 76.1 → 3.1; a one-layer forward at Qwen3-0.6B width agrees (0.42 / 3.66 ms per layer and token on AVX2) | +| CHANGELOG and platform-support entries | done (commit 3) | `CHANGELOG.md` [Unreleased]; `docs/platform-support.md` "Q31 attention score rows" | +| Box runs B0–B2 (the real Qwen3 artifact, S5-cut timing) | box-only | §7; the only real-artifact run of the Q31 kernel | + +## Deviations from the plan as written + +1. **The golden pin carries one hash per slice**, as in S2–S4, and S5 has two: the Q31 set's and the fixture's (§3.3 + names both). The Q31 set is driven through the v1.9.0 per-key `QkQ31Score` in the generator (v1.9.0 has no row + entry) and through `QkQ31ScoreRow` in the suite and the digest. +2. **The fixture's constants are tuned, not canonical.** `QkNormWiringFixture` uses the canonical site constant + everywhere; at hidden 256 with random weights that drives three sites out of domain (the post-norm funnel preflight, + the SiLU gate scale, the kernel's softmax constants) and clamps every V code. `fixture-premise.txt` records each + sweep. The RoPE table is built from Pythagorean triples, so no platform's libm enters the fixture. +3. **4.S5's widths add 15, 16 and 17** to the plan's {1, 7, 8, 9, 1,024}: the AVX-512 body packs keys in blocks of 16, + and those three are its full block and both partials. + +4. **The linkage checker reports a stale name on a Clang build, before and after this slice.** On Clang 18 objects + `TiledWidenActivations` resolves to no symbol and no inlined signature (the S4 tree gives the same FAIL); the CI + job builds with GCC 13, where the check is OK. Every S5 name resolves on both compilers. Not changed here. + +5. **The saving is measured on synthetic Qwen3-width data, not a real artifact.** A QK-norm artifact is still refused at map + time on this host (R4), so the kernel bench (the plan's own §0 method, at head_dim 128 and 28 × 16 heads) and a + one-layer forward probe at Qwen3-0.6B width stand in. The probe is the in-tree fixture re-parameterised; at that width it + needs hidden gain 4,096 and q/k-norm gains in [2,048, 4,096] to stay Ok through T = 1,024 (`probe_q31_forward.cpp`). +6. **A coverage-found cell is added at the evidence commit, after green.** It is not red-first: the kernel already handled + the channel pads, and no cell reached them. The pad mutant (`x_pads_nonzero`) shows that the cell is load-bearing. It is + outside the golden set, so the pin is unchanged. + +## What the plan got wrong + +1. **§9's "a₂ by logical shift" mutant is equivalent, for any kernel that stores a₂ as an int16 lane.** The low 16 bits of + w >> 30 are bits 30–45 of w, whichever shift forms it. The two shifts differ only in bits 34–63 of the int64 result, and + the int16 store (vpmaddwd's operand) drops them. Inside the guard |w| < 2³⁹, so the int16 value is exactly a₂ both ways. + Negative q cannot kill it, and no input can. Executed: it survives on all three binaries (`mutants.txt`). The row should + be withdrawn as equivalent, as rev 3 withdrew two S3 rows. +2. **4.S5's head_dim set has no in-guard value that is not a multiple of 4.** {4, 8, 60, 64, 128, 132, 256, 512} leaves the + limb and key-pack channel pads unexercised, so a pad bug survives the whole plan-specified suite. The coverage replica + found it; the channel-tail rows close it (`coverage.txt`, `mutants.txt`). +3. **§0's S5 figures hold for AVX2 but not for AVX-512.** §0 is AVX2 (per-key 376–385 ns). Here the shipped per-key AVX-512 + tier costs about 330 ns, while the row kernel is barely faster than on AVX2 (12.8 against 13.5 ns). So AVX-512 saves + about 20% less: 9.0 / 73.4 against AVX2's 11.0 / 94.3 ms per token at T = 128 / 1,024. On AVX2 the estimate holds and is + slightly exceeded. +4. **A one-key row is slower on AVX-512, which the plan does not anticipate.** The body packs a whole 16-key block and builds + 128 channels of limbs whatever the width, so width 1 costs 388 ns against the per-key 326 ns. This happens once per head, + for the first prompt token. On AVX2 (8-key blocks) the one-key row is still faster. +5. **§3.4 / 11.6 name only intmath.cpp and matmul.cpp, and src/forward/ is outside the coverage leg's glob.** S5's kernel + lives in forward_sites.cpp, so no hosted measurement of it exists or can exist until the leg's export includes + src/forward/. G27 notes the missing floor, but not that the file is never measured. The replica measures it separately. +6. **§6's whole-prefill figure for S5 (0.6B at T = 1,024, ≈ 140 → ≈ 57 ms per token) cannot be checked on the cloud host.** + It needs a 28-layer forward on real weights (B1/B2). The one-layer probe's saving, scaled by 28, is 102 ms per token on + AVX2 and 84.5 on AVX-512, against the ≈ 83 implied by 140 → 57. + +## ID-PENDING list + +Nothing minted here. diff --git a/docs/attention-rowsites/s1/bench-forward.txt b/docs/attention-rowsites/s1/bench-forward.txt new file mode 100644 index 00000000..70e89f68 --- /dev/null +++ b/docs/attention-rowsites/s1/bench-forward.txt @@ -0,0 +1,28 @@ +base prefill best-of-3 ms/token: 0.5615 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.6647 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5381 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.5435 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.5741 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.5522 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5589 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.4806 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.5365 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.6716 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5160 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.4429 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.6093 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.6836 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5323 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.7362 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.5814 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.5779 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5086 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.5160 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.5608 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.5845 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5197 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.4395 (T=300 D=32 layers=1) +base prefill best-of-3 ms/token: 0.6273 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.6140 (T=300 D=32 layers=1) +cand prefill best-of-3 ms/token: 0.5572 (T=128 D=0 layers=1) | decode best-of-3 ms/token: 1.4569 (T=300 D=32 layers=1) +base l1 decode best-of-3 ms/token: 1.6443 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.0974 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1529 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.5754 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.2615 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0367 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.5646 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.5364 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1079 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.4506 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.3856 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0276 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.4872 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.3041 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1695 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.5985 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 4.2175 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0471 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.6571 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.0424 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1623 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.5665 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.3777 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0365 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.7313 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.2138 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1050 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.6175 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.3354 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.1025 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.5132 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.3672 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0828 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.5165 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.2826 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0310 (T=128 D=0 layers=2) +base l1 decode best-of-3 ms/token: 1.6290 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.4758 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0917 (T=128 D=0 layers=2) +cand l1 decode best-of-3 ms/token: 1.4524 (T=300 D=64 layers=1) | l2 decode best-of-3 ms/token: 3.2948 (T=300 D=64 layers=2) | l2p prefill best-of-3 ms/token: 1.0561 (T=128 D=0 layers=2) diff --git a/docs/attention-rowsites/s1/bench-kernel.txt b/docs/attention-rowsites/s1/bench-kernel.txt new file mode 100644 index 00000000..f31f7d7d --- /dev/null +++ b/docs/attention-rowsites/s1/bench-kernel.txt @@ -0,0 +1,10 @@ +base kernel best-of-15 us/call: rmsnorm_896 9.360 mlp_act_4864 61.054 residual_896 19.271 (refused 0) +cand kernel best-of-15 us/call: rmsnorm_896 6.719 mlp_act_4864 38.730 residual_896 13.921 (refused 0) +base kernel best-of-15 us/call: rmsnorm_896 8.662 mlp_act_4864 57.994 residual_896 19.050 (refused 0) +cand kernel best-of-15 us/call: rmsnorm_896 6.953 mlp_act_4864 39.221 residual_896 13.665 (refused 0) +base kernel best-of-15 us/call: rmsnorm_896 8.587 mlp_act_4864 58.417 residual_896 19.024 (refused 0) +cand kernel best-of-15 us/call: rmsnorm_896 6.837 mlp_act_4864 38.712 residual_896 13.698 (refused 0) +base kernel best-of-15 us/call: rmsnorm_896 10.528 mlp_act_4864 57.522 residual_896 18.985 (refused 0) +cand kernel best-of-15 us/call: rmsnorm_896 6.950 mlp_act_4864 39.192 residual_896 13.859 (refused 0) +base kernel best-of-15 us/call: rmsnorm_896 8.700 mlp_act_4864 58.099 residual_896 19.368 (refused 0) +cand kernel best-of-15 us/call: rmsnorm_896 9.047 mlp_act_4864 38.802 residual_896 13.739 (refused 0) diff --git a/docs/attention-rowsites/s1/bench.md b/docs/attention-rowsites/s1/bench.md new file mode 100644 index 00000000..eb3a7f8e --- /dev/null +++ b/docs/attention-rowsites/s1/bench.md @@ -0,0 +1,41 @@ +# S1 bench: what the per-row tables save + +Reports, not gates (plan §8 10.1). Host: the shared 4-vCPU cloud Xeon (AVX-512 here; the tables are not SIMD, so the tier does +not matter to them), GCC 13.3 -O3. `tools/sslm_sites_bench.cpp` built twice from the same source: against the base library +(`fb56397`) and against the S1 implementation commit's. Every reading is a best-of-R inside one process; the two builds are +run alternately (interleaved pairs) and the medians are over pairs. Raw output: `bench-kernel.txt`, `bench-forward.txt`. + +## Method 1 (the plan's): per-site savings times per-token site counts + +Kernel mode calls each site through its public entry point at the Qwen2.5-0.5B widths with realistic constants (16 distinct +rows, best of 15 batches, 5 interleaved pairs). The site call includes the funnel and the site's own allocation, which S1 leaves +alone, so the difference is the loop the table replaces. + +| Site (width) | Base µs/call | S1 µs/call | Saved µs/call | Calls per token (24 layers) | Saved ms/token | +|---|---|---|---|---|---| +| `MlpActSite` (4,864) | 58.10 | 38.80 | 19.30 | 24 | 0.463 | +| `ResidualReconcileSite` (896) | 19.05 | 13.74 | 5.31 | 48 | 0.255 | +| `RmsNormSite` (896) | 8.70 | 6.95 | 1.75 | 48 | 0.084 | +| **Total** | | | | | **0.80** | + +The plan estimates **0.87 ms/token** (§0, §4.1: the spike's loop-only figures 25.2 → 4.2 µs, 8.5 → 2.6 µs, 3.0 → 1.2 µs over the +same counts). Measured: **0.80 ms/token**, 92% of the estimate. The saved microseconds per call (19.3, 5.3, 1.75) are close to +the spike's (21.0, 5.9, 1.8); the site-level ratios are 1.50×, 1.39× and 1.25×, lower than the spike's loop-only 6.0×, 3.2× and +2.6× because the funnel and allocation stay. + +## Method 2: whole forward on reduced-layer artifacts, scaled to 24 layers + +| Reading | Base | S1 | Paired saving, median (Q1, Q3) | Per layer | × 24 | +|---|---|---|---|---|---| +| Prefill, p05_l1, T = 128, ms per prompt token (7 pairs) | 0.574 | 0.532 | 0.041 (0.021, 0.073) | 0.041 | 0.99 | +| Prefill, p05_l2, T = 128 (7 pairs) | 1.108 | 1.037 | 0.080 (0.036, 0.122) | 0.040 | 0.96 | +| Decode, p05_l1, context 300, 32 steps (7 pairs) | 1.614 | 1.481 | 0.121 (0.062, 0.157) | — | not resolved | +| Decode, p05_l1, context 300, 64 steps (7 pairs) | 1.629 | 1.567 | 0.091 (−0.003, 0.114) | — | not resolved | +| Decode, p05_l2, context 300, 64 steps (7 pairs) | 3.304 | 3.335 | −0.122 (−0.335, 0.151) | — | not resolved | + +Prefill agrees with method 1 within its noise (0.96–0.99 against 0.80; the forward's per-element loops run on colder caches than +the kernel bench's 16 hot rows, which plausibly widens the saving). **Decode is not resolved on this host**: S1 saves about +0.033 ms per layer per decode token (method 1), about 2% of a 1-layer decode step, and the pair-to-pair spread here is ±10%; the +2-layer decode median even goes the wrong way. The decode saving is the same site work per token as prefill's (the plan's G9: the +same functions), so method 1's 0.80 ms/token stands for decode too; a quiet host (or the box's B2) is where a decode reading can +resolve it. diff --git a/docs/attention-rowsites/s1/blob-protocol.txt b/docs/attention-rowsites/s1/blob-protocol.txt new file mode 100644 index 00000000..8e2142f7 --- /dev/null +++ b/docs/attention-rowsites/s1/blob-protocol.txt @@ -0,0 +1,45 @@ +# Attention and per-row sites S1, end-to-end blobs (plan §3.3, cell 6.4). tools/t2147_chunk_batched_pins.cpp built against the base +# library (fb56397) and the candidate library (the S1 implementation commit), GCC 13.3 -O3, auto dispatch (AVX-512 here). Token-id +# prompts ids:N (ids:N:128 on the in-tree fixture), --chunk-budget=B then --decode=32 greedy steps, then sslm_seq_save. +# Artifacts: in-tree fixture (8 layers), wide fixture wide_l8, the 0.5B-width 1- and 2-layer synthetics (p05_l1 sha256 f0fd4886...6ed3). +# At N = 8 the "whole" budget is 8, so those rows repeat. The in-tree fixture refuses 128 + 32 positions (its context cap), on base +# and candidate alike; its rows are run at N = 24 instead. +intree ids:8 chunk_budget=1 +32 decode: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=1 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +intree ids:128 cb=1: refused on the base (context cap), not compared +wide_l8 ids:128 chunk_budget=1 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +intree ids:128 cb=8: refused on the base (context cap), not compared +wide_l8 ids:128 chunk_budget=8 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +intree ids:128 cb=128: refused on the base (context cap), not compared +wide_l8 ids:128 chunk_budget=128 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +intree ids:512 cb=1: refused on the base (context cap), not compared +wide_l8 ids:512 chunk_budget=1 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:512 cb=8: refused on the base (context cap), not compared +wide_l8 ids:512 chunk_budget=8 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:512 cb=512: refused on the base (context cap), not compared +wide_l8 ids:512 chunk_budget=512 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:24 chunk_budget=1 +32 decode (the fixture's context cap refuses 128+32): blob base 4dd83f12817436f7 cand 4dd83f12817436f7; tokens base 5707bd0c cand 5707bd0c +intree ids:24 chunk_budget=8 +32 decode (the fixture's context cap refuses 128+32): blob base 4dd83f12817436f7 cand 4dd83f12817436f7; tokens base 5707bd0c cand 5707bd0c +intree ids:24 chunk_budget=24 +32 decode (the fixture's context cap refuses 128+32): blob base 4dd83f12817436f7 cand 4dd83f12817436f7; tokens base 5707bd0c cand 5707bd0c diff --git a/docs/attention-rowsites/s1/golden.txt b/docs/attention-rowsites/s1/golden.txt new file mode 100644 index 00000000..873d52c3 --- /dev/null +++ b/docs/attention-rowsites/s1/golden.txt @@ -0,0 +1,6 @@ +# S1 golden pin provenance (plan §3.3 evidence 3, cell 6.3) +# tools/gen_attn_rowsite_golden.cpp compiled with GCC 13.3 -O2 against the v1.9.0 tag (d870d27) include/ and its Release libsuperslm.a: +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +# The same source built against this series (CMake target gen_attn_rowsite_golden, GCC and the auto library): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +# Re-running the v1.9.0 build with the header path as argument reproduced tests/attn_rowsite_golden_pin.h byte for byte. diff --git a/docs/attention-rowsites/s1/mutants.txt b/docs/attention-rowsites/s1/mutants.txt new file mode 100644 index 00000000..2b34ee56 --- /dev/null +++ b/docs/attention-rowsites/s1/mutants.txt @@ -0,0 +1,136 @@ +# S1 mutation evidence (GCC 13.3 Release, this host: AVX-512BW; auto binary dispatches AVX-512). Each mutant is applied to a synced copy +# of the implementation commit, superslm_tests and superslm_tests_avx2_forced rebuilt, and each run from the repository root with +# SUPERSLM_ATTN_ROWSITES_ARTIFACT set (p05_l1, sha256 f0fd4886...6ed3). 'none' is the unmutated control. Scripts: mutation-scripts/. +# Up to six distinct failing lines per run are shown. +== none on superslm_tests: exit 0; attn-rowsites cells (plan slice S1): 1756 checks, 0 failures|superslm tests: 27287 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slice S1): 1756 checks, 0 failures|superslm tests: 27245 checks, 0 failures| +== s1_idx_norm on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 41 failures|superslm tests: 27287 checks, 46 failures| +FAIL tests/test_main.cpp:9545: wide[idx] == e.floor_wide — RmsNormSite's own internal wide_row[0], read from the emitted trace record's x_int, == -10829700, want -9024800 (the independently-derived +FAIL tests/test_main.cpp:9634: out_codes[idx] == expected_codes[idx] — RmsNormSite out_codes[0] == -40, want -33 (the floor-division oracle's own code, via the already-shipped funnel) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=512: scale (2142937600, 24), reference (2126196480, 24) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 0 n=512: 254 output bytes differ from the reference, first at 0 (-11 vs 11) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 940478d4a8b6fc4f0667f0fde591 +== s1_idx_norm on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 41 failures|superslm tests: 27245 checks, 46 failures| +FAIL tests/test_main.cpp:9545: wide[idx] == e.floor_wide — RmsNormSite's own internal wide_row[0], read from the emitted trace record's x_int, == -10829700, want -9024800 (the independently-derived +FAIL tests/test_main.cpp:9634: out_codes[idx] == expected_codes[idx] — RmsNormSite out_codes[0] == -40, want -33 (the floor-division oracle's own code, via the already-shipped funnel) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=512: scale (2142937600, 24), reference (2126196480, 24) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 0 n=512: 254 output bytes differ from the reference, first at 0 (-11 vs 11) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 940478d4a8b6fc4f0667f0fde591 +== s1_idx_silu on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 77 failures|superslm tests: 27287 checks, 77 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 silu shape 0 gate e=-35 n=512: scale (1262868421, 36), reference (1263655205, 36) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-35 n=512: 35 output bytes differ from the reference, first at 10 (85 vs 2) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: d8a4553ad14eb8a3cd5117742c52 +== s1_idx_silu on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 77 failures|superslm tests: 27245 checks, 77 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 silu shape 0 gate e=-35 n=512: scale (1262868421, 36), reference (1263655205, 36) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-35 n=512: 35 output bytes differ from the reference, first at 10 (85 vs 2) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: d8a4553ad14eb8a3cd5117742c52 +== s1_idx_land on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 73 failures|superslm tests: 27287 checks, 73 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 495 output bytes differ from the reference, first at 1 (-119 vs -118) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 1 n=512: scale (1235004944, 0), reference (1236339428, 0) +FAIL tests/test_attn_rowsites.cpp:2preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single ten +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 48ecc3c5eb793445a72444efcbaf +== s1_idx_land on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 73 failures|superslm tests: 27245 checks, 73 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 495 output bytes differ from the reference, first at 1 (-119 vs -118) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 1 n=512: scale (1235004944, 0), reference (1236339428, 0) +FAIL tests/test_attn_rowsites.cpp:277:preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 48ecc3c5eb793445a72444efcbaf +== s1_m128_silu on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 11 failures|superslm tests: 27287 checks, 11 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-35 n=512: 4 output bytes differ from the reference, first at 0 (0 vs -1) +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/do +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: eed62f54893779598da60f1539a2 +== s1_m128_silu on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 11 failures|superslm tests: 27245 checks, 11 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-35 n=512: 4 output bytes differ from the reference, first at 0 (0 vs -1) +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: eed62f54893779598da60f1539a2 +== s1_m128_land on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 32 failures|superslm tests: 27287 checks, 32 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 0 n=512: scale (1237916547, 0), reference (1250048226, 0) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 280 output bytes differ from the reference, first at 0 (-2 vs -127) +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 ro +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: b55ee5fc6789d7708f67eabbd285 +== s1_m128_land on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 32 failures|superslm tests: 27245 checks, 32 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 0 n=512: scale (1237916547, 0), reference (1250048226, 0) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 280 output bytes differ from the reference, first at 0 (-2 vs -127) +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: b55ee5fc6789d7708f67eabbd285 +== s1_m128_norm on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 18 failures|superslm tests: 27287 checks, 18 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 0 n=512: 7 output bytes differ from the reference, first at 0 (0 vs 11) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=4864: scale (1074009024, 25), reference (1082477088, 25) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 574a5b33bdc82022d2877c2da735 +== s1_m128_norm on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 18 failures|superslm tests: 27245 checks, 18 failures| +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 0 n=512: 7 output bytes differ from the reference, first at 0 (0 vs 11) +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=4864: scale (1074009024, 25), reference (1082477088, 25) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 574a5b33bdc82022d2877c2da735 +== s1_or_flag on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 7 failures|superslm tests: 27287 checks, 7 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 7.S1c flag read per element present n=512: scale (1300000000, 60), reference (1300000000, 7) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 7.S1c flag read per element present n=512: 508 output bytes differ from the reference, first at 0 (0 vs -127) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 3afc79cb2160ffc5c2532192999f +== s1_or_flag on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 7 failures|superslm tests: 27245 checks, 7 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 7.S1c flag read per element present n=512: scale (1300000000, 60), reference (1300000000, 7) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 7.S1c flag read per element present n=512: 508 output bytes differ from the reference, first at 0 (0 vs -127) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 3afc79cb2160ffc5c2532192999f +== s1_static_norm on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 30 failures|superslm tests: 27287 checks, 30 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=512: scale (1329520660, 29), reference (2126196480, 24) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 2 n=512: 1 output bytes differ from the reference, first at 494 (-64 vs -63) +FAIL tests/test_attn_ropreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWe +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 9c0606f9b6c090bac94192596155 +== s1_static_norm on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 30 failures|superslm tests: 27245 checks, 30 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 norm shape 0 n=512: scale (1329520660, 29), reference (2126196480, 24) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 norm shape 2 n=512: 1 output bytes differ from the reference, first at 494 (-64 vs -63) +FAIL tests/test_attn_rowsipreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (Laye +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 9c0606f9b6c090bac94192596155 +== s1_static_silu on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 57 failures|superslm tests: 27287 checks, 57 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 silu shape 0 gate e=-34 n=512: scale (1193109974, 37), reference (1217342350, 37) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-34 n=512: 408 output bytes differ from the reference, first at 1 (4 vs 1) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1Golpreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/ +== s1_static_silu on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 57 failures|superslm tests: 27245 checks, 57 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 silu shape 0 gate e=-34 n=512: scale (1193109974, 37), reference (1217342350, 37) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 silu shape 0 gate e=-34 n=512: 408 output bytes differ from the reference, first at 1 (4 vs 1) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1Goldenpreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v +== s1_static_land on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 41 failures|superslm tests: 27287 checks, 41 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 0 n=512: scale (1366137696, 2), reference (1615625000, -2) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 511 output bytes differ from the reference, first at 0 (7 vs 88) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 526df9315044ffe7f74109ed3398 +== s1_static_land on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 41 failures|superslm tests: 27245 checks, 41 failures| +FAIL tests/test_attn_rowsites.cpp:267: scale.m == want.scale.m && scale.e == want.scale.e -- 4.S1 residual shape 0 n=512: scale (1366137696, 2), reference (1615625000, -2) +FAIL tests/test_attn_rowsites.cpp:277: count == 0 -- 4.S1 residual shape 0 n=512: 511 output bytes differ from the reference, first at 0 (7 vs 88) +FAIL tests/test_attn_rowsites.cpp:513: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: 526df9315044ffe7f74109ed3398 +== s1_thr0 on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27287 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +1 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+160/+0) +== s1_thr0 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27245 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +1 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+160/+0) +== s1_thr513 on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 36 failures|superslm tests: 27287 checks, 36 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=512: rowtable_norm taken +0 skipped +1, want +1/+0 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 +== s1_thr513 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 36 failures|superslm tests: 27245 checks, 36 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=512: rowtable_norm taken +0 skipped +1, want +1/+0 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4 +== s1_thrmax on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 138 failures|superslm tests: 27287 checks, 138 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=512: rowtable_norm taken +0 skipped +1, want +1/+0 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (Layer +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+0/+160) +FAIL tests/test_attn_rowsites.cpp:660: d.taken[s] == want_taken[s] && d.skipped[s] == 0 -- 11.1(d) prefill: rowtable_norm taken +0 skipped +256, want +256/+0 +== s1_thrmax on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 138 failures|superslm tests: 27245 checks, 138 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=512: rowtable_norm taken +0 skipped +1, want +1/+0 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (La +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+0/+160) +FAIL tests/test_attn_rowsites.cpp:660: d.taken[s] == want_taken[s] && d.skipped[s] == 0 -- 11.1(d) prefill: rowtable_norm taken +0 skipped +256, want +256/+0 +== all_skip_deleted on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27287 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +0 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+80/+0) +== all_skip_deleted on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27245 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +0 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+80/+0) +== all_taken_before_guard on superslm_tests: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27287 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +1 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+160/+0) +== all_taken_before_guard on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slice S1): 1756 checks, 123 failures|superslm tests: 27245 checks, 123 failures| +FAIL tests/test_attn_rowsites.cpp:125: d.taken[s] == want_taken && d.skipped[s] == want_skipped -- 4.S1 norm shape 0 n=1: rowtable_norm taken +1 skipped +0, want +0/+1 +FAIL (line interleaved with the marshaller's own stdout) preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate +FAIL tests/test_attn_rowsites.cpp:521: d.taken[s] > 0 && d.skipped[s] > 0 -- 6.3: the golden set takes and skips the norm table (+160/+0) diff --git a/docs/attention-rowsites/s1/mutation-scripts/all_skip_deleted.py b/docs/attention-rowsites/s1/mutation-scripts/all_skip_deleted.py new file mode 100644 index 00000000..7cde8cb7 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/all_skip_deleted.py @@ -0,0 +1,3 @@ +# all: "fallback increment deleted" (row tables: the skipped counters never move). +p='src/forward/forward_sites.cpp'; s=open(p).read(); old='inline void CountRowTableDecision(RowTableSite site, bool taken) {'; assert s.count(old)==1 +s=s.replace(old,old+'\n\tif (!taken) return; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/all_taken_before_guard.py b/docs/attention-rowsites/s1/mutation-scripts/all_taken_before_guard.py new file mode 100644 index 00000000..72d4d512 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/all_taken_before_guard.py @@ -0,0 +1,3 @@ +# all: "fast increment moved before the guard" (row tables: every call counts as taken). +p='src/forward/forward_sites.cpp'; s=open(p).read(); old='inline void CountRowTableDecision(RowTableSite site, bool taken) {'; assert s.count(old)==1 +s=s.replace(old,old+'\n\ttaken = true; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_idx_land.py b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_land.py new file mode 100644 index 00000000..1da9793a --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_land.py @@ -0,0 +1,6 @@ +# S1 "table indexed by code + 127" (landing table, whose own offset is 127): value and flag read at code + 126. +p='src/forward/forward_sites.cpp'; s=open(p).read() +for old in ['landed = landed_table[static_cast(other_code[i]) + 127];','magnitude_exceeded = exceeded_table[static_cast(other_code[i]) + 127];']: + assert s.count(old)==1 + s=s.replace(old, old.replace('[static_cast(other_code[i]) + 127]','[(static_cast(other_code[i]) + 126 + 255) % 255]')+' // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_idx_norm.py b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_norm.py new file mode 100644 index 00000000..ea4a8136 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_norm.py @@ -0,0 +1,4 @@ +# S1 "table indexed by code + 127" (norm table): every element reads its neighbour's entry. +p='src/forward/forward_sites.cpp'; s=open(p).read() +old='wide[i] = divided[static_cast(h[i]) + 128]'; assert s.count(old)==1 +s=s.replace(old,'wide[i] = divided[(static_cast(h[i]) + 127) & 255] /* MUTANT */'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_idx_silu.py b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_silu.py new file mode 100644 index 00000000..f4155fc3 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_idx_silu.py @@ -0,0 +1,4 @@ +# S1 "table indexed by code + 127" (SiLU table, whose own offset is 127): read at code + 126. +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=': sigmoid[static_cast(code) + 127];'; assert s.count(old)==1 +s=s.replace(old,': sigmoid[(static_cast(code) + 126 + 255) % 255]; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_m128_land.py b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_land.py new file mode 100644 index 00000000..b101dce8 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_land.py @@ -0,0 +1,13 @@ +# S1 "-128 read from the table (entry left unset)": landing tables of 256 entries, entry for -128 never built (value 0, flag false). +p='src/forward/forward_sites.cpp'; s=open(p).read() +rep=[(''' int64_t landed_table[255]; + bool exceeded_table[255];''',''' int64_t landed_table[256] = {}; // MUTANT + bool exceeded_table[256] = {};'''), +(''' landed_table[c + 127] = LandingRescale(''',''' landed_table[c + 128] = LandingRescale('''), +(''' exceeded_table[c + 127] = exceeded;''',''' exceeded_table[c + 128] = exceeded;'''), +(''' if (use_table && other_code[i] != INT8_MIN) {''',''' if (use_table) { // MUTANT'''), +('''landed = landed_table[static_cast(other_code[i]) + 127];''','''landed = landed_table[static_cast(other_code[i]) + 128];'''), +('''magnitude_exceeded = exceeded_table[static_cast(other_code[i]) + 127];''','''magnitude_exceeded = exceeded_table[static_cast(other_code[i]) + 128];''')] +for a,b in rep: + assert s.count(a)==1,a; s=s.replace(a,b) +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_m128_norm.py b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_norm.py new file mode 100644 index 00000000..f5257e57 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_norm.py @@ -0,0 +1,7 @@ +# S1 norm table built for [-127, 127] only (as the SiLU and landing tables are), -128 entry left 0. +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' int64_t divided[256]; + for (int c = -128; c <= 127; ++c)''' +assert s.count(old)==1 +s=s.replace(old,''' int64_t divided[256] = {}; // MUTANT + for (int c = -127; c <= 127; ++c)'''); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_m128_silu.py b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_silu.py new file mode 100644 index 00000000..c412d3f4 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_m128_silu.py @@ -0,0 +1,14 @@ +# S1 "-128 read from the table (entry left unset)": SiLU table of 256 entries, entry for -128 never built (0). +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' int32_t sigmoid[255]; + for (int c = -127; c <= 127; ++c) + sigmoid[c + 127] = SiluSigmoidQ15(sigmoid_lut_table, static_cast(c), gate_scale.m, gate_e);''' +assert s.count(old)==1 +s=s.replace(old,''' int32_t sigmoid[256] = {}; // MUTANT + for (int c = -127; c <= 127; ++c) + sigmoid[c + 128] = SiluSigmoidQ15(sigmoid_lut_table, static_cast(c), gate_scale.m, gate_e);''') +old=''' const int32_t sig = code == INT8_MIN ? SiluSigmoidQ15(sigmoid_lut_table, code, gate_scale.m, gate_e) + : sigmoid[static_cast(code) + 127];''' +assert s.count(old)==1 +s=s.replace(old,''' const int32_t sig = sigmoid[static_cast(code) + 128]; // MUTANT''') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_or_flag.py b/docs/attention-rowsites/s1/mutation-scripts/s1_or_flag.py new file mode 100644 index 00000000..27150320 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_or_flag.py @@ -0,0 +1,14 @@ +# S1 "landing flag OR-ed over the whole table": the candidate is refused if any code's entry is flagged. +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' exceeded_table[c + 127] = exceeded; + } + }''' +assert s.count(old)==1 +s=s.replace(old,''' exceeded_table[c + 127] = exceeded; + } + for (int c = 0; c < 255; ++c) // MUTANT + if (exceeded_table[c]) { + candidate.status = SslmForwardStatus::ResidualReconciliationMagnitudeOutOfDomain; + return candidate; + } + }'''); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_static_land.py b/docs/attention-rowsites/s1/mutation-scripts/s1_static_land.py new file mode 100644 index 00000000..cb8f3185 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_static_land.py @@ -0,0 +1,10 @@ +# S1 "table built once per call site and reused across rows" (landing; one table for both candidates too). +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' int64_t landed_table[255]; + bool exceeded_table[255]; + if (use_table) {''' +assert s.count(old)==1 +s=s.replace(old,''' static int64_t landed_table[255]; // MUTANT + static bool exceeded_table[255]; + static bool built = false; + if (use_table && !built) { built = true;'''); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_static_norm.py b/docs/attention-rowsites/s1/mutation-scripts/s1_static_norm.py new file mode 100644 index 00000000..651bba0c --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_static_norm.py @@ -0,0 +1,11 @@ +# S1 "table built once per call site and reused across rows" (norm). +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' int64_t divided[256]; + for (int c = -128; c <= 127; ++c) + divided[c + 128] = FloorDivI64(static_cast(c) << (2 * kNormFracBits), root);''' +assert s.count(old)==1 +s=s.replace(old,''' static int64_t divided[256]; // MUTANT + static bool built = false; + if (!built) { built = true; + for (int c = -128; c <= 127; ++c) + divided[c + 128] = FloorDivI64(static_cast(c) << (2 * kNormFracBits), root); }'''); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_static_silu.py b/docs/attention-rowsites/s1/mutation-scripts/s1_static_silu.py new file mode 100644 index 00000000..935257c7 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_static_silu.py @@ -0,0 +1,11 @@ +# S1 "table built once per call site and reused across rows" (SiLU). +p='src/forward/forward_sites.cpp'; s=open(p).read() +old=''' int32_t sigmoid[255]; + for (int c = -127; c <= 127; ++c) + sigmoid[c + 127] = SiluSigmoidQ15(sigmoid_lut_table, static_cast(c), gate_scale.m, gate_e);''' +assert s.count(old)==1 +s=s.replace(old,''' static int32_t sigmoid[255]; // MUTANT + static bool built = false; + if (!built) { built = true; + for (int c = -127; c <= 127; ++c) + sigmoid[c + 127] = SiluSigmoidQ15(sigmoid_lut_table, static_cast(c), gate_scale.m, gate_e); }'''); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_thr0.py b/docs/attention-rowsites/s1/mutation-scripts/s1_thr0.py new file mode 100644 index 00000000..4df5e873 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_thr0.py @@ -0,0 +1,3 @@ +# S1 threshold n >= 0. +p='src/forward/forward_sites.cpp'; s=open(p).read(); old='constexpr size_t kRowTableMinWidth = 512;'; assert s.count(old)==1 +s=s.replace(old,'constexpr size_t kRowTableMinWidth = 0; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_thr513.py b/docs/attention-rowsites/s1/mutation-scripts/s1_thr513.py new file mode 100644 index 00000000..14bbffe3 --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_thr513.py @@ -0,0 +1,3 @@ +# S1 threshold n >= 513. +p='src/forward/forward_sites.cpp'; s=open(p).read(); old='constexpr size_t kRowTableMinWidth = 512;'; assert s.count(old)==1 +s=s.replace(old,'constexpr size_t kRowTableMinWidth = 513; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/mutation-scripts/s1_thrmax.py b/docs/attention-rowsites/s1/mutation-scripts/s1_thrmax.py new file mode 100644 index 00000000..41b6e82a --- /dev/null +++ b/docs/attention-rowsites/s1/mutation-scripts/s1_thrmax.py @@ -0,0 +1,3 @@ +# S1 threshold SIZE_MAX (tables never taken). +p='src/forward/forward_sites.cpp'; s=open(p).read(); old='constexpr size_t kRowTableMinWidth = 512;'; assert s.count(old)==1 +s=s.replace(old,'constexpr size_t kRowTableMinWidth = SIZE_MAX; // MUTANT'); open(p,'w').write(s) diff --git a/docs/attention-rowsites/s1/red-sslm_axis_digest.txt b/docs/attention-rowsites/s1/red-sslm_axis_digest.txt new file mode 100644 index 00000000..8d1bb5a1 --- /dev/null +++ b/docs/attention-rowsites/s1/red-sslm_axis_digest.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/red-sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s1/red-sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..d82ae0a0 --- /dev/null +++ b/docs/attention-rowsites/s1/red-sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/red-sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s1/red-sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..564291b6 --- /dev/null +++ b/docs/attention-rowsites/s1/red-sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/red-sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s1/red-sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..a391bb53 --- /dev/null +++ b/docs/attention-rowsites/s1/red-sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/red-sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s1/red-sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..3ae104e7 --- /dev/null +++ b/docs/attention-rowsites/s1/red-sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/red-suites.txt b/docs/attention-rowsites/s1/red-suites.txt new file mode 100644 index 00000000..ef44d8f1 --- /dev/null +++ b/docs/attention-rowsites/s1/red-suites.txt @@ -0,0 +1,44 @@ +== superslm_tests (auto), GCC 13.3 Release, red commit, SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1756 checks, 258 failures +superslm tests: 27287 checks, 258 failures +failures by kind: + 249 test_attn_rowsites.cpp:125 + 3 test_attn_rowsites.cpp:521 + 6 test_attn_rowsites.cpp:660 +failures outside test_attn_rowsites.cpp: 0 + +== superslm_tests (sse2), GCC 13.3 Release, red commit, SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1756 checks, 258 failures +superslm tests: 27229 checks, 258 failures +failures by kind: + 249 test_attn_rowsites.cpp:125 + 3 test_attn_rowsites.cpp:521 + 6 test_attn_rowsites.cpp:660 +failures outside test_attn_rowsites.cpp: 0 + +== superslm_tests (avx2), GCC 13.3 Release, red commit, SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1756 checks, 258 failures +superslm tests: 27245 checks, 258 failures +failures by kind: + 249 test_attn_rowsites.cpp:125 + 3 test_attn_rowsites.cpp:521 + 6 test_attn_rowsites.cpp:660 +failures outside test_attn_rowsites.cpp: 0 + +== superslm_tests (avx512), GCC 13.3 Release, red commit, SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1756 checks, 258 failures +superslm tests: 27245 checks, 258 failures +failures by kind: + 249 test_attn_rowsites.cpp:125 + 3 test_attn_rowsites.cpp:521 + 6 test_attn_rowsites.cpp:660 +failures outside test_attn_rowsites.cpp: 0 + diff --git a/docs/attention-rowsites/s1/sslm_axis_digest.txt b/docs/attention-rowsites/s1/sslm_axis_digest.txt new file mode 100644 index 00000000..8d1bb5a1 --- /dev/null +++ b/docs/attention-rowsites/s1/sslm_axis_digest.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s1/sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..d82ae0a0 --- /dev/null +++ b/docs/attention-rowsites/s1/sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s1/sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..564291b6 --- /dev/null +++ b/docs/attention-rowsites/s1/sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s1/sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..a391bb53 --- /dev/null +++ b/docs/attention-rowsites/s1/sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s1/sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..3ae104e7 --- /dev/null +++ b/docs/attention-rowsites/s1/sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,18 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s1/suites-clang.txt b/docs/attention-rowsites/s1/suites-clang.txt new file mode 100644 index 00000000..78da3c02 --- /dev/null +++ b/docs/attention-rowsites/s1/suites-clang.txt @@ -0,0 +1,42 @@ +== superslm_tests, Clang 18.1.3 Release, series head, with the p05_l1 artifact +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27288 checks, 0 failures + +== superslm_tests_sse2_forced, Clang 18.1.3 Release, series head, with the p05_l1 artifact +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27230 checks, 0 failures + +== superslm_tests_avx2_forced, Clang 18.1.3 Release, series head, with the p05_l1 artifact +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27246 checks, 0 failures + +== superslm_tests_avx512_forced, Clang 18.1.3 Release, series head, with the p05_l1 artifact +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27246 checks, 0 failures + +== digests, Clang 18.1.3 (auto, scalar/SSE2/AVX2/AVX-512 forced) +sslm_axis_digest.txt: c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +sslm_axis_digest_avx2_forced.txt: c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +sslm_axis_digest_avx512_forced.txt: c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +sslm_axis_digest_scalar_forced.txt: c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 +sslm_axis_digest_sse2_forced.txt: c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 GLOBAL 3a82909147af69b598fe2a41617a8b2a1d67cc612a206e9011d5062e22e7fc57 diff --git a/docs/attention-rowsites/s1/suites.txt b/docs/attention-rowsites/s1/suites.txt new file mode 100644 index 00000000..fd28ae68 --- /dev/null +++ b/docs/attention-rowsites/s1/suites.txt @@ -0,0 +1,47 @@ +== superslm_tests, GCC 13.3.0 Release, series head (implementation + 3.S1 cell), SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27288 checks, 0 failures + +== superslm_tests_sse2_forced, GCC 13.3.0 Release, series head (implementation + 3.S1 cell), SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the SSE2 tier; tiled path not taken on this tier (SKIPPED reach); tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27230 checks, 0 failures + +== superslm_tests_avx2_forced, GCC 13.3.0 Release, series head (implementation + 3.S1 cell), SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX2 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27246 checks, 0 failures + +== superslm_tests_avx512_forced, GCC 13.3.0 Release, series head (implementation + 3.S1 cell), SUPERSLM_ATTN_ROWSITES_ARTIFACT=p05_l1 (sha256 f0fd4886...6ed3) +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27246 checks, 0 failures + +== superslm_tests, GCC 13.3.0 RelWithDebInfo -fsanitize=thread, series head, with the p05_l1 artifact: exit 0, 0 ThreadSanitizer reports +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27288 checks, 0 failures + +== superslm_tests, GCC 13.3.0 RelWithDebInfo -fsanitize=address,undefined -fno-sanitize-recover=undefined, series head, with the p05_l1 artifact: exit 0, 0 sanitizer reports +attn-rowsites cells (plan slice S1): 1757 checks, 0 failures +superslm tests: 27288 checks, 0 failures diff --git a/docs/attention-rowsites/s2/bench-forward.txt b/docs/attention-rowsites/s2/bench-forward.txt new file mode 100644 index 00000000..9fee14b6 --- /dev/null +++ b/docs/attention-rowsites/s2/bench-forward.txt @@ -0,0 +1,106 @@ +# sslm_sites_bench prefill/decode --repeat=3, 7 interleaved rounds (round, build, artifact, mode, T or context, D :: output). Builds as bench-pv.txt. cand_avx2 forces AVX2 for every GEMM too, so it is not comparable with base (auto, AVX-512 GEMMs) at forward level; bench.md uses base vs cand only. +1 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5371 (T=128 D=0 layers=1) +1 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6651 (T=512 D=0 layers=1) +1 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3889 (T=512 D=0 layers=2) +1 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4764 (T=300 D=32 layers=1) +1 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.0433 (T=300 D=32 layers=2) +1 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.4730 (T=128 D=0 layers=1) +1 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5343 (T=512 D=0 layers=1) +1 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.0605 (T=512 D=0 layers=2) +1 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.3097 (T=300 D=32 layers=1) +1 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.5709 (T=300 D=32 layers=2) +1 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.6497 (T=128 D=0 layers=1) +1 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6244 (T=512 D=0 layers=1) +1 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.2819 (T=512 D=0 layers=2) +1 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.3926 (T=300 D=32 layers=1) +1 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.8761 (T=300 D=32 layers=2) +2 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5355 (T=128 D=0 layers=1) +2 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.7646 (T=512 D=0 layers=1) +2 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.4187 (T=512 D=0 layers=2) +2 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4606 (T=300 D=32 layers=1) +2 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.9285 (T=300 D=32 layers=2) +2 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.4970 (T=128 D=0 layers=1) +2 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5711 (T=512 D=0 layers=1) +2 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.1129 (T=512 D=0 layers=2) +2 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.2733 (T=300 D=32 layers=1) +2 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.6012 (T=300 D=32 layers=2) +2 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.6026 (T=128 D=0 layers=1) +2 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6973 (T=512 D=0 layers=1) +2 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3057 (T=512 D=0 layers=2) +2 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5283 (T=300 D=32 layers=1) +2 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.2212 (T=300 D=32 layers=2) +3 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5143 (T=128 D=0 layers=1) +3 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6556 (T=512 D=0 layers=1) +3 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3743 (T=512 D=0 layers=2) +3 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5166 (T=300 D=32 layers=1) +3 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.7323 (T=300 D=32 layers=2) +3 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5090 (T=128 D=0 layers=1) +3 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5611 (T=512 D=0 layers=1) +3 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.0399 (T=512 D=0 layers=2) +3 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5624 (T=300 D=32 layers=1) +3 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.8000 (T=300 D=32 layers=2) +3 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5977 (T=128 D=0 layers=1) +3 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6526 (T=512 D=0 layers=1) +3 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3915 (T=512 D=0 layers=2) +3 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4359 (T=300 D=32 layers=1) +3 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.0520 (T=300 D=32 layers=2) +4 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5766 (T=128 D=0 layers=1) +4 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6841 (T=512 D=0 layers=1) +4 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3898 (T=512 D=0 layers=2) +4 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.7055 (T=300 D=32 layers=1) +4 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.4378 (T=300 D=32 layers=2) +4 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5294 (T=128 D=0 layers=1) +4 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5760 (T=512 D=0 layers=1) +4 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.2004 (T=512 D=0 layers=2) +4 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.2738 (T=300 D=32 layers=1) +4 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.2253 (T=300 D=32 layers=2) +4 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5469 (T=128 D=0 layers=1) +4 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6364 (T=512 D=0 layers=1) +4 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.1086 (T=512 D=0 layers=2) +4 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5416 (T=300 D=32 layers=1) +4 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.1159 (T=300 D=32 layers=2) +5 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5350 (T=128 D=0 layers=1) +5 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.7139 (T=512 D=0 layers=1) +5 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.5390 (T=512 D=0 layers=2) +5 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4840 (T=300 D=32 layers=1) +5 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.5904 (T=300 D=32 layers=2) +5 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.4906 (T=128 D=0 layers=1) +5 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5725 (T=512 D=0 layers=1) +5 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.0433 (T=512 D=0 layers=2) +5 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.3688 (T=300 D=32 layers=1) +5 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.5450 (T=300 D=32 layers=2) +5 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5880 (T=128 D=0 layers=1) +5 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6895 (T=512 D=0 layers=1) +5 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.2605 (T=512 D=0 layers=2) +5 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4634 (T=300 D=32 layers=1) +5 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.1861 (T=300 D=32 layers=2) +6 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5494 (T=128 D=0 layers=1) +6 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.7636 (T=512 D=0 layers=1) +6 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.4423 (T=512 D=0 layers=2) +6 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5210 (T=300 D=32 layers=1) +6 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.5234 (T=300 D=32 layers=2) +6 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5211 (T=128 D=0 layers=1) +6 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5492 (T=512 D=0 layers=1) +6 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.0792 (T=512 D=0 layers=2) +6 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.3557 (T=300 D=32 layers=1) +6 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.1434 (T=300 D=32 layers=2) +6 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5951 (T=128 D=0 layers=1) +6 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.6534 (T=512 D=0 layers=1) +6 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.2824 (T=512 D=0 layers=2) +6 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5877 (T=300 D=32 layers=1) +6 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.0265 (T=300 D=32 layers=2) +7 base p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.5438 (T=128 D=0 layers=1) +7 base p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.7098 (T=512 D=0 layers=1) +7 base p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.3963 (T=512 D=0 layers=2) +7 base p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.5307 (T=300 D=32 layers=1) +7 base p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.2776 (T=300 D=32 layers=2) +7 cand p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.4786 (T=128 D=0 layers=1) +7 cand p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.5457 (T=512 D=0 layers=1) +7 cand p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.0424 (T=512 D=0 layers=2) +7 cand p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.4149 (T=300 D=32 layers=1) +7 cand p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 2.6538 (T=300 D=32 layers=2) +7 cand_avx2 p05_l1.sslm prefill 128 --layers=1 :: prefill best-of-3 ms/token: 0.6186 (T=128 D=0 layers=1) +7 cand_avx2 p05_l1.sslm prefill 512 --layers=1 :: prefill best-of-3 ms/token: 0.7224 (T=512 D=0 layers=1) +7 cand_avx2 p05_l2.sslm prefill 512 --layers=2 :: prefill best-of-3 ms/token: 1.4734 (T=512 D=0 layers=2) +7 cand_avx2 p05_l1.sslm decode 300 32 :: decode best-of-3 ms/token: 1.6836 (T=300 D=32 layers=1) +7 cand_avx2 p05_l2.sslm decode 300 32 :: decode best-of-3 ms/token: 3.2261 (T=300 D=32 layers=2) diff --git a/docs/attention-rowsites/s2/bench-pv.txt b/docs/attention-rowsites/s2/bench-pv.txt new file mode 100644 index 00000000..d7a839f0 --- /dev/null +++ b/docs/attention-rowsites/s2/bench-pv.txt @@ -0,0 +1,190 @@ +# sslm_sites_bench pv --repeat=50, 9 interleaved triples: bench_base (the S1 head's library, auto = AVX-512 here, v1.9.0 prob-V loop), bench_cand (S2, auto = AVX-512 body), bench_cand_avx2 (S2 library built with SUPERSLM_FORCE_AVX2_MATMUL, AVX2 body). GCC 13.3 -O3 -ffp-contract=off. +== pair 1 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2409 w=301 12.5153 w=512 21.0625 w=601 24.8863 w=1024 40.9210 +pv prefill T=128: 0.8774 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5606 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.2079 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1799 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.4376 ms/token at 24 layers x 14 heads +== pair 1 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0558 w=128 0.2673 w=301 0.6013 w=512 1.0105 w=601 1.2822 w=1024 2.9256 +pv prefill T=128: 0.0759 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.2881 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4798 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2114 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4225 ms/token at 24 layers x 14 heads +== pair 1 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0632 w=128 0.4110 w=301 0.9919 w=512 1.7756 w=601 1.3229 w=1024 2.3906 +pv prefill T=128: 0.0540 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1954 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4353 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2264 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4443 ms/token at 24 layers x 14 heads +== pair 2 base +pv best-of-50 us/call at head_dim 64: w=1 0.0553 w=128 5.2871 w=301 14.0768 w=512 21.1137 w=601 24.7519 w=1024 42.2993 +pv prefill T=128: 0.8771 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5549 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.2171 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1436 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3299 ms/token at 24 layers x 14 heads +== pair 2 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0558 w=128 0.2655 w=301 0.5993 w=512 1.0079 w=601 1.1836 w=1024 2.3166 +pv prefill T=128: 0.0523 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1763 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4354 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2015 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3977 ms/token at 24 layers x 14 heads +== pair 2 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0569 w=128 0.2996 w=301 0.6731 w=512 1.1284 w=601 1.3222 w=1024 2.4264 +pv prefill T=128: 0.0540 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1966 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4461 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2264 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4444 ms/token at 24 layers x 14 heads +== pair 3 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2430 w=301 12.3318 w=512 21.3991 w=601 24.7269 w=1024 42.2382 +pv prefill T=128: 0.8775 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.3532 ms/token at 24 layers x 14 heads +pv prefill T=1024: 6.9828 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1480 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3541 ms/token at 24 layers x 14 heads +== pair 3 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0563 w=128 0.2660 w=301 0.5999 w=512 1.0079 w=601 1.1802 w=1024 2.3240 +pv prefill T=128: 0.0530 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1753 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4158 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.1953 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3970 ms/token at 24 layers x 14 heads +== pair 3 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0571 w=128 0.2994 w=301 0.6734 w=512 1.1285 w=601 1.3220 w=1024 2.4266 +pv prefill T=128: 0.0654 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1958 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4443 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2263 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4441 ms/token at 24 layers x 14 heads +== pair 4 base +pv best-of-50 us/call at head_dim 64: w=1 0.0545 w=128 5.2467 w=301 12.4527 w=512 21.0875 w=601 24.7565 w=1024 42.5533 +pv prefill T=128: 0.8773 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5712 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.1697 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1434 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.4036 ms/token at 24 layers x 14 heads +== pair 4 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0557 w=128 0.2666 w=301 0.5992 w=512 1.0078 w=601 1.1785 w=1024 2.2558 +pv prefill T=128: 0.0523 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1750 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4274 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2016 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3968 ms/token at 24 layers x 14 heads +== pair 4 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0569 w=128 0.2995 w=301 0.6736 w=512 1.1285 w=601 1.4497 w=1024 2.4200 +pv prefill T=128: 0.0540 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1952 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4218 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2263 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4330 ms/token at 24 layers x 14 heads +== pair 5 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2435 w=301 12.4070 w=512 20.9945 w=601 24.7924 w=1024 42.2008 +pv prefill T=128: 0.8775 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5579 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.1960 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1463 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3144 ms/token at 24 layers x 14 heads +== pair 5 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0558 w=128 0.2665 w=301 0.6022 w=512 1.0084 w=601 1.1794 w=1024 2.2419 +pv prefill T=128: 0.0525 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1749 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4189 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2018 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3962 ms/token at 24 layers x 14 heads +== pair 5 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0552 w=128 0.2995 w=301 0.6734 w=512 1.1279 w=601 1.3218 w=1024 2.3916 +pv prefill T=128: 0.0539 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1944 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4457 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2266 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4444 ms/token at 24 layers x 14 heads +== pair 6 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2316 w=301 12.4048 w=512 21.1082 w=601 24.7878 w=1024 42.2654 +pv prefill T=128: 0.8785 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5903 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.1748 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.2242 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3959 ms/token at 24 layers x 14 heads +== pair 6 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0559 w=128 0.2663 w=301 0.6006 w=512 1.0115 w=601 1.2007 w=1024 2.2641 +pv prefill T=128: 0.0526 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1749 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4335 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2021 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3963 ms/token at 24 layers x 14 heads +== pair 6 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0569 w=128 0.2995 w=301 0.6735 w=512 1.1283 w=601 1.3226 w=1024 2.4533 +pv prefill T=128: 0.0540 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1964 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4341 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2266 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4448 ms/token at 24 layers x 14 heads +== pair 7 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2390 w=301 12.3333 w=512 21.1725 w=601 24.8000 w=1024 42.9264 +pv prefill T=128: 0.8780 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5487 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.2190 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1666 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3817 ms/token at 24 layers x 14 heads +== pair 7 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0574 w=128 0.3358 w=301 0.6905 w=512 1.0126 w=601 1.1819 w=1024 2.2926 +pv prefill T=128: 0.0478 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1748 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4410 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2014 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3960 ms/token at 24 layers x 14 heads +== pair 7 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0569 w=128 0.2995 w=301 0.6735 w=512 1.1279 w=601 1.3220 w=1024 2.3907 +pv prefill T=128: 0.0539 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1953 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4514 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2263 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4440 ms/token at 24 layers x 14 heads +== pair 8 base +pv best-of-50 us/call at head_dim 64: w=1 0.0558 w=128 5.2312 w=301 12.3709 w=512 21.1039 w=601 24.7549 w=1024 42.4241 +pv prefill T=128: 0.8775 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5581 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.0197 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1821 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3210 ms/token at 24 layers x 14 heads +== pair 8 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0563 w=128 0.2661 w=301 0.6001 w=512 1.0082 w=601 1.1803 w=1024 2.2572 +pv prefill T=128: 0.0525 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1766 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4422 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2247 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4790 ms/token at 24 layers x 14 heads +== pair 8 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0569 w=128 0.2990 w=301 0.6736 w=512 1.1568 w=601 1.7637 w=1024 2.3931 +pv prefill T=128: 0.0539 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1962 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4444 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2263 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.4442 ms/token at 24 layers x 14 heads +== pair 9 base +pv best-of-50 us/call at head_dim 64: w=1 0.0556 w=128 5.2318 w=301 11.9839 w=512 19.9958 w=601 24.8073 w=1024 40.8769 +pv prefill T=128: 0.8776 ms/token at 24 layers x 14 heads +pv prefill T=512: 3.5785 ms/token at 24 layers x 14 heads +pv prefill T=1024: 7.1829 ms/token at 24 layers x 14 heads +pv decode ctx=300: 4.1559 ms/token at 24 layers x 14 heads +pv decode ctx=600: 8.3686 ms/token at 24 layers x 14 heads +== pair 9 cand +pv best-of-50 us/call at head_dim 64: w=1 0.0558 w=128 0.2659 w=301 0.5994 w=512 1.0111 w=601 1.1838 w=1024 2.3212 +pv prefill T=128: 0.0490 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1779 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4215 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2014 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.3961 ms/token at 24 layers x 14 heads +== pair 9 cand_avx2 +pv best-of-50 us/call at head_dim 64: w=1 0.0544 w=128 0.3422 w=301 0.7722 w=512 1.1315 w=601 1.3229 w=1024 2.3911 +pv prefill T=128: 0.0540 ms/token at 24 layers x 14 heads +pv prefill T=512: 0.1955 ms/token at 24 layers x 14 heads +pv prefill T=1024: 0.4830 ms/token at 24 layers x 14 heads +pv decode ctx=300: 0.2735 ms/token at 24 layers x 14 heads +pv decode ctx=600: 0.6173 ms/token at 24 layers x 14 heads diff --git a/docs/attention-rowsites/s2/bench.md b/docs/attention-rowsites/s2/bench.md new file mode 100644 index 00000000..b995bf27 --- /dev/null +++ b/docs/attention-rowsites/s2/bench.md @@ -0,0 +1,76 @@ +# S2 bench: what prob·V on int16 multiply-add saves + +These are reports, not gates (plan §8 10.1). The host is the shared 4-vCPU cloud Xeon (AVX2, AVX-512BW), with GCC 13.3 -O3. +`tools/sslm_sites_bench.cpp` was built three times from the same source: + +- against the base library (the S1 head, which runs the v1.9.0 loop); +- against S2's library (auto dispatch, which runs the AVX-512 body here); +- against S2's library built with `SUPERSLM_FORCE_AVX2_MATMUL` (the AVX2 body). + +Each reading is a best-of-R inside one process. The builds run alternately, and the medians are taken over rounds. The raw +output is in `bench-pv.txt` and `bench-forward.txt`. + +## Method 1 (the plan's): the kernel, scaled to per-token call counts + +`pv` mode calls `GemmProbQ15Accumulate` at head_dim 64 (the 0.5B's) over realistic rows: every p is formed as the softmax +forms it, with weights 2^(0..12) plus jitter. It uses one KV head's value rows. The mode reports three things: + +- The per-call cost at single widths. +- **Prefill** of T tokens: the calls at widths 1..T, summed, times 24 layers × 14 heads, divided by T. +- **Decode** at context C: one call at width C + 1, times 24 × 14. + +The runs used best of 50 and 9 interleaved rounds. The saving is the median of the paired differences; the range is the minimum +and maximum over the rounds. + +| Reading | Base (v1.9.0 loop) | S2 AVX-512 | Saved (range) | × | S2 AVX2 | Saved (range) | × | Plan §0 estimate | +|---|---|---|---|---|---|---|---|---| +| µs per call, width 128 | 5.241 | 0.266 | 4.97 | 19.7 | 0.300 | 4.94 | 17.5 | | +| µs per call, width 301 | 12.405 | 0.600 | 11.80 | 20.7 | 0.674 | 11.70 | 18.4 | | +| µs per call, width 601 | 24.788 | 1.182 | 23.59 | 21.0 | 1.323 | 23.47 | 18.7 | | +| µs per call, width 1,024 | 42.265 | 2.293 | 39.98 | 18.4 | 2.393 | 39.81 | 17.7 | | +| **Prefill T = 128**, ms/token | 0.878 | 0.053 | **0.825** (0.80–0.83) | 16.7 | 0.054 | 0.824 (0.81–0.82) | 16.2 | **0.95** | +| **Prefill T = 512** | 3.558 | 0.175 | **3.38** (3.18–3.42) | 20.3 | 0.196 | 3.36 (3.16–3.39) | 18.2 | **4.38** | +| **Prefill T = 1,024** | 7.183 | 0.434 | **6.74** (6.57–6.78) | 16.6 | 0.444 | 6.75 (6.54–6.77) | 16.2 | **8.47** | +| **Decode, context 300**, ms/token | 4.156 | 0.202 | **3.95** (3.94–4.02) | 20.6 | 0.226 | 3.92 (3.88–4.00) | 18.4 | **3.83** | +| **Decode, context 600** | 8.369 | 0.397 | **7.97** (7.84–8.02) | 21.1 | 0.444 | 7.91 (7.75–7.99) | 18.8 | **7.72** | + +At width 1 there is no saving (0.056 µs on every build). A realistic width-1 row is one-hot (p = 2^15), so it fails the +int16 condition and takes the shipped loop by design. + +**Against the estimate.** + +- **Decode matches it:** 3.95 and 7.97 ms/token against 3.83 and 7.72, which is 103%. +- **Prefill comes in at 77–87% of it:** 0.825 / 3.38 / 6.74 against 0.95 / 4.38 / 8.47. + +The kernel ratio (16–21× AVX-512, 16–19× AVX2) is above the spike's 10.7–14.2×, so the shortfall is not in the kernel. It is in +the estimate's base. The plan's T = 512 and T = 1,024 savings (4.38, 8.47) exceed this host's *entire* measured v1.9.0 prob·V +cost at those lengths (3.56, 7.18 ms/token). The estimate therefore assumed a per-key base cost about 20–25% higher than the +0.041 µs per key per head measured here: 5.24 µs at width 128. + +The AVX2 body is within 2–13% of the AVX-512 body. The AVX-512 body's 32-dimension in-lane units are what put it ahead; see the +progress file's deviations. + +## Method 2: the whole forward on the reduced-layer artifacts, scaled to 24 layers + +This is `prefill` and `decode` mode on the 0.5B-width synthetics, with p05_l1 at sha256 f0fd4886…6ed3. It used best of 3 and +7 interleaved rounds. It compares base (auto) with S2 (auto). The forced-AVX2 build also forces AVX2 GEMMs, so it is not +comparable with the auto base at this level. Its rows are in the raw file but are not used here. + +| Reading | Base | S2 | Paired saving, median (Q1, Q3) | Per layer | × 24 | Method 1 | +|---|---|---|---|---|---|---| +| Prefill, p05_l1, T = 128, ms per prompt token | 0.537 | 0.497 | 0.044 (0.028, 0.064) | 0.044 | 1.07 | 0.825 | +| Prefill, p05_l1, T = 512 | 0.710 | 0.561 | 0.141 (0.108, 0.194) | 0.141 | 3.39 | 3.38 | +| Prefill, p05_l2, T = 512 | 1.396 | 1.061 | 0.334 (0.306, 0.363) | 0.167 | 4.01 | 3.38 | +| Decode, p05_l1, context 300, 32 steps, ms per step | 1.517 | 1.356 | 0.165 (0.115, 0.187) | 0.165 | 3.97 | 3.95 | +| Decode, p05_l2, context 300, 32 steps | 3.438 | 2.654 | 0.472 (0.327, 0.932) | 0.236 | 5.7 | 3.95 | + +The 1-layer readings agree with method 1: + +- prefill at T = 512: 3.39 against 3.38; +- decode at context 300: 3.97 against 3.95; +- prefill at T = 128: 1.07 against 0.83, within its quartiles' noise. + +The 2-layer readings run higher and wider, which is the shared host's noise. Method 1's figures are the ones reported. + +These are per-layer engine figures on synthetic weights. They say nothing about any consumer's end-to-end speed, and they were +not measured on the project's reference hardware (the box's B2, §8 10.2, is box-only). diff --git a/docs/attention-rowsites/s2/blob-protocol.txt b/docs/attention-rowsites/s2/blob-protocol.txt new file mode 100644 index 00000000..54447e07 --- /dev/null +++ b/docs/attention-rowsites/s2/blob-protocol.txt @@ -0,0 +1,50 @@ +# S2 save-blob protocol (plan 6.4, as S1 ran it): tools/t2147_chunk_batched_pins.cpp built against the base library (the S1 head) +# and against the S2 candidate's (auto dispatch, AVX-512 here), same flags (GCC 13.3 -O3 -ffp-contract=off). Each row prefills the +# prompt at the chunk budget, decodes 32 greedy tokens, dumps the SSB5 blob; EQUAL = blobs byte-equal (cmp) and decoded tokens equal. +# Artifacts: the S1 synthetic set (wide_l8, p05_l1 sha256 f0fd4886...6ed3, p05_l2) and the in-tree 8-layer fixture. +wide_l8 ids:8 chunk_budget=1 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=1 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=1 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:24:128 chunk_budget=1 +32 decode: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=8 +32 decode: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=24 +32 decode: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) + +# Second pass: the in-tree fixture at ids:8 against the auto candidate, and the synthetic set against the candidate's forced-AVX2 library. +intree ids:8:128 chunk_budget=1 +32 decode, candidate pins_cand: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL +intree ids:8:128 chunk_budget=8 +32 decode, candidate pins_cand: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate pins_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate pins_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate pins_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate pins_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate pins_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate pins_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate pins_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL diff --git a/docs/attention-rowsites/s2/coverage.txt b/docs/attention-rowsites/s2/coverage.txt new file mode 100644 index 00000000..6ec7f8f6 --- /dev/null +++ b/docs/attention-rowsites/s2/coverage.txt @@ -0,0 +1,34 @@ +# S2 branch coverage (plan §3.4, cell 11.6). INDICATIVE ONLY: a floor is pinned only from the hosted branch-coverage leg's own +# recorded measurement (tools/ci/branch_coverage_floors.json "_measured_cell"), and this series is delivered as patches, not +# pushed, so that leg has not run on it. These are local replicas of the leg's commands (clang-18, llvm-cov/llvm-profdata 18, +# RelWithDebInfo, -fprofile-instr-generate -fcoverage-mapping; the five binaries, merged, exported over src/*.cpp +# include/superslm/*.h), on this host (AVX-512BW), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. All five binaries exit 0 on +# both trees. +# +# Base = the S1 head; candidate = S2's implementation commit plus the evidence commit's test rows. Branches covered / total. +# +# (1) All five profiles (the leg's union), this host (the auto and forced AVX-512 binaries run AVX-512): +# base src/intmath.cpp 153/174 87.93103448275862 src/matmul.cpp 198/222 89.1891891891892 +# candidate src/intmath.cpp 153/174 87.93103448275862 src/matmul.cpp 286/312 91.66666666666666 +# check_branch_coverage_floors.py: OK on both (19 files at or above their pinned floor). +# +# (2) Approximating a runner WITHOUT AVX-512: only the sse2-forced, avx2-forced and scalar-forced-digest profiles merged (on +# such a runner the auto binary dispatches AVX2 and the forced AVX-512 binary faults at its first AVX-512 instruction; +# llvm-profdata warns of 16 functions with mismatched data across the objects, so treat this as a rough figure): +# base src/intmath.cpp 87.93103448275862 src/matmul.cpp 148/196 75.51020408163265 +# candidate src/intmath.cpp 87.93103448275862 src/matmul.cpp 203/286 70.97902097902097 +# On such a runner src/matmul.cpp would measure about 4.5 points lower with S2 than without, which is below today's floor +# (72.22222222222221). Every branch it loses is in the AVX-512 prob-V body; see the uncovered list below. +# +# Uncovered S2 branches (lines of src/matmul.cpp at the implementation commit): +# With all five profiles: 913 and 1059 only. Each is the implicit default of a switch over an enum whose every value +# has a case (DispatchSitesKernel's three SitesKernel values; SelectSitesKernel's four GemmTier values). No input +# reaches it. +# Without AVX-512 profiles, additionally: 821-901 (ProbVBlockAvx512, ProbVTail16Avx512, ProbVAccumulateIntoAvx512) and +# 923-924 (the dispatcher's kAvx512 arm). They cannot execute on a runner without AVX-512, like DotRowAvx512 already. +# No reachable S2 branch is uncovered, so no cell is added (plan §3.4 step 1). +# +# Plan §3.4 steps 2 and 3: allowlist lines for the above are added in tools/ci/branch_coverage_allowlist.txt, in the file's +# existing reviewer-note form. The floors in tools/ci/branch_coverage_floors.json are NOT re-pinned: step 3 copies the leg's +# recorded value, which does not exist until the series runs in CI. If the hosted runner lacks AVX-512 and matmul.cpp +# measures below 72.22222222222221 there, lowering the floor is the owner's one-line call at S2's code review (G27). diff --git a/docs/attention-rowsites/s2/golden.txt b/docs/attention-rowsites/s2/golden.txt new file mode 100644 index 00000000..8e8465f1 --- /dev/null +++ b/docs/attention-rowsites/s2/golden.txt @@ -0,0 +1,10 @@ +# S2 golden pin provenance (plan §3.3 evidence 3, cell 6.3) +# tools/gen_attn_rowsite_golden.cpp (now one hash per slice) compiled with GCC 13.3 -O2 against the v1.9.0 tag (d870d27) include/ +# and its Release libsuperslm.a (the recipe in the generator's header): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +S2 prob-V golden: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed over 30100 values +# The same source built against this series (CMake target gen_attn_rowsite_golden, GCC, the auto library): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +S2 prob-V golden: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed over 30100 values +# Re-running the v1.9.0 build with the header path as argument wrote tests/attn_rowsite_golden_pin.h; S1's hash and value count are +# unchanged, and S2's constants are added beside them. diff --git a/docs/attention-rowsites/s2/mutants.txt b/docs/attention-rowsites/s2/mutants.txt new file mode 100644 index 00000000..3cec10cf --- /dev/null +++ b/docs/attention-rowsites/s2/mutants.txt @@ -0,0 +1,126 @@ +# S2 mutation evidence (plan §9; GCC 13.3 Release; this host has AVX-512BW, so the auto binary dispatches AVX-512), run at the final +# implementation commit (the 32-dimension AVX-512 unit). Each mutant is applied to a synced copy of that commit by its script +# (mutation-scripts/.py, run from the repository root). superslm_tests (auto), superslm_tests_avx2_forced and +# superslm_tests_avx512_forced are then rebuilt and run from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT set (p05_l1, +# sha256 f0fd4886...6ed3). 'none' is the unmutated control. At most one failing line per run is shown (paths shortened). The run +# predates the 4.S2 kernel-blocking rows (90 checks) added in the evidence commit, which is why it counts 2,306 attn-rowsites +# checks rather than 2,396; those rows only add kills. +# +# Arithmetic mutants are applied to each tier's body separately (§9 preamble). Each AVX2-body mutant must die on the forced AVX2 +# binary, and it survives on the AVX-512 binaries, which never run that body. Each AVX-512-body mutant (the 32-dimension unit, +# and separately the 16-dimension tail unit) must die on the forced AVX-512 binary and on auto, and it survives on forced AVX2. +# Guard, dispatch and counter mutants sit in code both tiers share, so they die on all three binaries. +# 'all_avx512_runs_avx2_body' can die only where the AVX-512 kernel is selected (auto here, and forced AVX-512). +# +# Summary (K = killed, s = survives by construction, see above; the number is the failure count): +# mutant auto(AVX-512) forced AVX2 forced AVX-512 killing cell (plan §9) +# s2_hd16_dropped K abort K 259 K abort output (head_dim 4/8/12/60/100/132 rows; test_main head_dim 2 cells). AVX-512 writes past +# a head_dim-12 buffer; glibc aborts ("malloc(): unaligned tcache chunk detected") +# s2_hd16_to_32 K 22 K 22 K 22 counter: fast expected at head_dim 16 (4.S2) +# s2_pnonneg_dropped K 5 K 5 K 5 output and counter (2.S2 p = -32,769; alternating row) +# s2_pmax_32768 K 32 K 32 K 32 output and counter (2.S2 / 4.S2 width-1 one-hot) +# s2_pmax_32766 K 5 K 5 K 5 counter: fast expected (p = 32,767 rows, 4.S2) +# s2_sum_dropped K 5 K 5 K 5 output and counter (2.S2 W = 1,024, p = 32,767, v = 127) +# s2_sum_lt K 6 K 6 K 6 counter: fast expected (Sum p = 2^15 exactly, 4.S2) +# s2_sum_2p16 (output-equivalent) K 2 K 2 K 2 counter: fallback expected (Sum p = 2^15 + 1, 4.S2) +# s2_pair_swapped_avx2 s K 102 s output (grid, 6.1) and 6.3 +# s2_pair_swapped_avx512 K 81 s K 81 output (grid, 6.1) and 6.3 +# s2_pair_swapped_avx512_tail (extra) K 22 s K 22 output (head_dim 16 rows, the only 16-dimension tail in that run) +# s2_odd_dropped_avx2 s K 47 s output (odd widths, 4.S2) +# s2_odd_dropped_avx512 K 37 s K 37 output (odd widths, 4.S2) +# s2_odd_dropped_avx512_tail (extra) K 11 s K 11 output (odd widths at head_dim 16, the tail unit) +# s2_half_misplaced_avx512 (extra) K 82 s K 82 output (every head_dim >= 32: the in-lane halves' store order) +# s2_always_fallback K 109 K 109 K 109 counter (4.S2 and 11.1(d): prefill +1,792 fallback, decode +448) +# all_fallback_increment_deleted K 158 K 158 K 158 counter (11.1(a): every fallback row) +# all_fast_before_guard K 158 K 158 K 158 counter (11.1(a): every fallback row) +# all_avx512_runs_avx2_body K 109 s K 109 counter (11.1(b): the AVX2 counter moves on an AVX-512 binary) +# s2_msvc_switch_ignored (extra) K 1 K 1 K 1 11.2 truth table (the MSVC, switch 0, AVX-512 row) +# 20 mutants: every one killed on every binary where it can execute; 0 survivors. +# +# Cell 11.3 vitality (linkage checker), the plan's plant: an AVX2-attributed ProbV function given external linkage in +# src/matmul.cpp, auto library object. The checker goes red: + ProbQ15AccumulateInto: inlined into superslm::GemmProbQ15Accumulate in build/CMakeFiles/superslm.dir/src/matmul.cpp.o +check_tiled_matmul_linkage: FAIL + build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 'superslm::ProbVAccumulateIntoAvx2Planted(long const*, signed char const*, unsigned long, unsigned long, long*)' is not local (nm type 'T') + +# Raw per-run lines follow. + +== none on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27837 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== none on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 109 failures|superslm tests: 27837 checks, 109 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, realistic row (head_dim 16, width 2, guard copy: fast): pv_fast_avx2 +1 pv_fal +== all_avx512_runs_avx2_body on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 109 failures|superslm tests: 27795 checks, 109 failures| +== all_fallback_increment_deleted on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27837 checks, 158 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, realistic row (head_dim 4, width 1, guard copy: fallback): pv_fast_avx2 +0 pv_ +== all_fallback_increment_deleted on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27795 checks, 158 failures| +== all_fallback_increment_deleted on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27795 checks, 158 failures| +== all_fast_before_guard on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27837 checks, 158 failures| +== all_fast_before_guard on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27795 checks, 158 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, realistic row (head_dim 4, width 1, guard copy: fallback): pv_fast_avx2 +1 pv_ +== all_fast_before_guard on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 158 failures|superslm tests: 27795 checks, 158 failures| +== s2_always_fallback on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 109 failures|superslm tests: 27837 checks, 109 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, realistic row (head_dim 16, width 2, guard copy: fast): pv_fast_avx2 +0 pv_fal +== s2_always_fallback on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 109 failures|superslm tests: 27795 checks, 109 failures| +== s2_always_fallback on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 109 failures|superslm tests: 27795 checks, 109 failures| +== s2_half_misplaced_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 82 failures|superslm tests: 27837 checks, 82 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 64, width 2): 64 of 64 outputs differ from the v1.9.0 loop, first at 0 (-2695168 vs 401408) +== s2_half_misplaced_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_half_misplaced_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 82 failures|superslm tests: 27795 checks, 82 failures| +== s2_hd16_dropped on superslm_tests: exit 134; +== s2_hd16_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 244 failures|superslm tests: 27795 checks, 259 failures| +FAIL tests/test_main.cpp:11110: out_ctx[0] == 1700 — out_ctx[0] == 0, want 1700 (100*3 + 200*7) +== s2_hd16_dropped on superslm_tests_avx512_forced: exit 134; +== s2_hd16_to_32 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 22 failures|superslm tests: 27837 checks, 22 failures| +== s2_hd16_to_32 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 22 failures|superslm tests: 27795 checks, 22 failures| +== s2_hd16_to_32 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 22 failures|superslm tests: 27795 checks, 22 failures| +== s2_msvc_switch_ignored on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 1 failures|superslm tests: 27837 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:770: bad == 0 -- cell 11.2: 1 of 16 selector rows wrong +== s2_msvc_switch_ignored on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 1 failures|superslm tests: 27795 checks, 1 failures| +== s2_msvc_switch_ignored on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 1 failures|superslm tests: 27795 checks, 1 failures| +== s2_odd_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27837 checks, 0 failures| +== s2_odd_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 47 failures|superslm tests: 27795 checks, 47 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 16, width 3): 16 of 16 outputs differ from the v1.9.0 loop, first at 0 (-3367744 vs -2921746) +== s2_odd_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_odd_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 37 failures|superslm tests: 27837 checks, 37 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 64, width 3): 64 of 64 outputs differ from the v1.9.0 loop, first at 0 (317624 vs 2998616) +== s2_odd_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_odd_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 37 failures|superslm tests: 27795 checks, 37 failures| +== s2_odd_dropped_avx512_tail on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 11 failures|superslm tests: 27837 checks, 11 failures| +== s2_odd_dropped_avx512_tail on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_odd_dropped_avx512_tail on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 11 failures|superslm tests: 27795 checks, 11 failures| +== s2_pair_swapped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27837 checks, 0 failures| +== s2_pair_swapped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 102 failures|superslm tests: 27795 checks, 102 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 16, width 2): 16 of 16 outputs differ from the v1.9.0 loop, first at 0 (-1443862 vs 428085) +== s2_pair_swapped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_pair_swapped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 81 failures|superslm tests: 27837 checks, 81 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 64, width 2): 63 of 64 outputs differ from the v1.9.0 loop, first at 0 (-237568 vs 401408) +== s2_pair_swapped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_pair_swapped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 81 failures|superslm tests: 27795 checks, 81 failures| +== s2_pair_swapped_avx512_tail on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 22 failures|superslm tests: 27837 checks, 22 failures| +== s2_pair_swapped_avx512_tail on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2306 checks, 0 failures|superslm tests: 27795 checks, 0 failures| +== s2_pair_swapped_avx512_tail on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 22 failures|superslm tests: 27795 checks, 22 failures| +== s2_pmax_32766 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27837 checks, 5 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, peaked row (head_dim 256, width 3, guard copy: fast): pv_fast_avx2 +0 pv_fallb +== s2_pmax_32766 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_pmax_32766 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_pmax_32768 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 32 failures|superslm tests: 27837 checks, 32 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 4.S2 grid, realistic row (head_dim 16, width 1): 16 of 16 outputs differ from the v1.9.0 loop, first at 0 (-1867776 vs 1867776) +== s2_pmax_32768 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 32 failures|superslm tests: 27795 checks, 32 failures| +== s2_pmax_32768 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 32 failures|superslm tests: 27795 checks, 32 failures| +== s2_pnonneg_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27837 checks, 5 failures| +FAIL tests/test_attn_rowsites.cpp:742: bad == 0 -- 2.S2 p = -32,769 at one key (head_dim 64, width 64): 64 of 64 outputs differ from the v1.9.0 loop, first at 0 (3683222 vs -3656810) +== s2_pnonneg_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_pnonneg_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_sum_2p16 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 2 failures|superslm tests: 27837 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 corner p = 32,767 and 2 (Sum p = 2^15 + 1) (head_dim 16, width 2, guard copy: fallba +== s2_sum_2p16 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 2 failures|superslm tests: 27795 checks, 2 failures| +== s2_sum_2p16 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 2 failures|superslm tests: 27795 checks, 2 failures| +== s2_sum_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27837 checks, 5 failures| +== s2_sum_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_sum_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 5 failures|superslm tests: 27795 checks, 5 failures| +== s2_sum_lt on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 6 failures|superslm tests: 27837 checks, 6 failures| +FAIL tests/test_attn_rowsites.cpp:709: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 4.S2 grid, realistic row (head_dim 64, width 2, guard copy: fast): pv_fast_avx2 +0 pv_fal +== s2_sum_lt on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 6 failures|superslm tests: 27795 checks, 6 failures| +== s2_sum_lt on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2): 2306 checks, 6 failures|superslm tests: 27795 checks, 6 failures| diff --git a/docs/attention-rowsites/s2/mutation-scripts/all_avx512_runs_avx2_body.py b/docs/attention-rowsites/s2/mutation-scripts/all_avx512_runs_avx2_body.py new file mode 100644 index 00000000..f8ffbf00 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/all_avx512_runs_avx2_body.py @@ -0,0 +1,9 @@ +# all: the AVX-512 dispatch runs the AVX2 body. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('\t\t\t\tProbVAccumulateIntoAvx512(probs, values, width, head_dim, out_ctx);\n', '\t\t\t\tProbVAccumulateIntoAvx2(probs, values, width, head_dim, out_ctx); // MUTANT\n') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/all_fallback_increment_deleted.py b/docs/attention-rowsites/s2/mutation-scripts/all_fallback_increment_deleted.py new file mode 100644 index 00000000..9120a8c0 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/all_fallback_increment_deleted.py @@ -0,0 +1,9 @@ +# all: the fallback increments deleted (both tiers). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('\t\t\tsuperslm_test::g_pv_fallback_avx2.fetch_add(1, std::memory_order_relaxed);\n', '\t\t\t(void)0; // MUTANT\n'); sub('\t\t\tsuperslm_test::g_pv_fallback_avx512.fetch_add(1, std::memory_order_relaxed);\n', '\t\t\t(void)0; // MUTANT\n') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/all_fast_before_guard.py b/docs/attention-rowsites/s2/mutation-scripts/all_fast_before_guard.py new file mode 100644 index 00000000..453039ce --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/all_fast_before_guard.py @@ -0,0 +1,9 @@ +# all: the fast increments moved before the guard (counted at dispatch, removed from the bodies). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('\tsuperslm_test::g_pv_fast_avx2.fetch_add(1, std::memory_order_relaxed);\n', ''); sub('\tsuperslm_test::g_pv_fast_avx512.fetch_add(1, std::memory_order_relaxed);\n', ''); sub('\t\tcase detail::SitesKernel::kAvx2:\n\t\t\tif (ProbVFastPathAdmits', '\t\tcase detail::SitesKernel::kAvx2:\n\t\t\tsuperslm_test::g_pv_fast_avx2.fetch_add(1); // MUTANT\n\t\t\tif (ProbVFastPathAdmits'); sub('\t\tcase detail::SitesKernel::kAvx512:\n\t\t\tif (ProbVFastPathAdmits', '\t\tcase detail::SitesKernel::kAvx512:\n\t\t\tsuperslm_test::g_pv_fast_avx512.fetch_add(1); // MUTANT\n\t\t\tif (ProbVFastPathAdmits') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_always_fallback.py b/docs/attention-rowsites/s2/mutation-scripts/s2_always_fallback.py new file mode 100644 index 00000000..7a34b398 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_always_fallback.py @@ -0,0 +1,9 @@ +# S2: always fall back (the fast path never admits a row). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('inline bool ProbVFastPathAdmits(const int64_t* probs, size_t width, size_t head_dim) {\n', 'inline bool ProbVFastPathAdmits(const int64_t* probs, size_t width, size_t head_dim) {\n\treturn false; // MUTANT\n') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_half_misplaced_avx512.py b/docs/attention-rowsites/s2/mutation-scripts/s2_half_misplaced_avx512.py new file mode 100644 index 00000000..8d30f93a --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_half_misplaced_avx512.py @@ -0,0 +1,9 @@ +# S2 AVX-512 32-dimension unit only (extra): the in-lane halves stored at each other's dimensions (o + 16 and o swapped). +p='src/matmul.cpp'; s=open(p).read() +old = ''' _mm512_storeu_si512(o, _mm512_add_epi64(_mm512_loadu_si512(o), lo)); + _mm512_storeu_si512(o + 16, _mm512_add_epi64(_mm512_loadu_si512(o + 16), hi));''' +new = ''' _mm512_storeu_si512(o + 16, _mm512_add_epi64(_mm512_loadu_si512(o + 16), lo)); // MUTANT + _mm512_storeu_si512(o, _mm512_add_epi64(_mm512_loadu_si512(o), hi));''' +assert s.count(old) == 1 +s = s.replace(old, new) +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_dropped.py b/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_dropped.py new file mode 100644 index 00000000..d4e4f37b --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_dropped.py @@ -0,0 +1,9 @@ +# S2 guard: head_dim % 16 dropped. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('return head_dim % 16 == 0 && ProbVInt16Condition(probs, width);', 'return ProbVInt16Condition(probs, width); // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_to_32.py b/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_to_32.py new file mode 100644 index 00000000..b7b0087c --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_hd16_to_32.py @@ -0,0 +1,9 @@ +# S2 guard: head_dim % 16 -> % 32. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('return head_dim % 16 == 0 && ProbVInt16Condition', 'return head_dim % 32 == 0 && ProbVInt16Condition /* MUTANT */') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_msvc_switch_ignored.py b/docs/attention-rowsites/s2/mutation-scripts/s2_msvc_switch_ignored.py new file mode 100644 index 00000000..ea6474b2 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_msvc_switch_ignored.py @@ -0,0 +1,9 @@ +# 11.2 (extra): the selector ignores the MSVC switch (AVX-512 kernels on in every MSVC build). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('return (is_msvc_build && msvc_avx512_switch == 0) ? SitesKernel::kShipped : SitesKernel::kAvx512;', 'return (void)is_msvc_build, (void)msvc_avx512_switch, SitesKernel::kAvx512; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx2.py b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx2.py new file mode 100644 index 00000000..beeda58f --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx2.py @@ -0,0 +1,9 @@ +# S2 AVX2 body only: odd last key dropped. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +i = s.index('SUPERSLM_AVX2_TARGET inline void ProbVBlockAvx2'); j = s.index('if (k < width) { // the unpaired last key', i); s = s[:j] + 'if (false && k < width) { // MUTANT' + s[j+len('if (k < width) { // the unpaired last key'):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512.py b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512.py new file mode 100644 index 00000000..ec9a3bd0 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512.py @@ -0,0 +1,9 @@ +# S2 AVX-512 body only: odd last key dropped. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +i = s.index('SUPERSLM_AVX512_TARGET inline void ProbVBlockAvx512'); j = s.index('if (k < width) { // the unpaired last key', i); s = s[:j] + 'if (false && k < width) { // MUTANT' + s[j+len('if (k < width) { // the unpaired last key'):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512_tail.py b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512_tail.py new file mode 100644 index 00000000..78f99270 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_odd_dropped_avx512_tail.py @@ -0,0 +1,7 @@ +# S2 AVX-512 16-dimension tail unit only: odd last key dropped. +p='src/matmul.cpp'; s=open(p).read() +i = s.index('SUPERSLM_AVX512_TARGET inline void ProbVTail16Avx512') +old = 'if (k < width) { // the unpaired last key' +j = s.index(old, i) +s = s[:j] + 'if (false && k < width) { // MUTANT' + s[j+len(old):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx2.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx2.py new file mode 100644 index 00000000..79d9b4f1 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx2.py @@ -0,0 +1,9 @@ +# S2 AVX2 body only: pair order swapped (p_{k+1} with v_k). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +i = s.index('SUPERSLM_AVX2_TARGET inline void ProbVBlockAvx2'); j = s.index('ProbVPair(probs[k], probs[k + 1])', i); s = s[:j] + 'ProbVPair(probs[k + 1], probs[k]) /* MUTANT */' + s[j+len('ProbVPair(probs[k], probs[k + 1])'):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512.py new file mode 100644 index 00000000..d7faa005 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512.py @@ -0,0 +1,9 @@ +# S2 AVX-512 body only: pair order swapped (p_{k+1} with v_k). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +i = s.index('SUPERSLM_AVX512_TARGET inline void ProbVBlockAvx512'); j = s.index('ProbVPair(probs[k], probs[k + 1])', i); s = s[:j] + 'ProbVPair(probs[k + 1], probs[k]) /* MUTANT */' + s[j+len('ProbVPair(probs[k], probs[k + 1])'):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512_tail.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512_tail.py new file mode 100644 index 00000000..9ac66ce1 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pair_swapped_avx512_tail.py @@ -0,0 +1,6 @@ +# S2 AVX-512 16-dimension tail unit only: pair order swapped (p_{k+1} with v_k). +p='src/matmul.cpp'; s=open(p).read() +i = s.index('SUPERSLM_AVX512_TARGET inline void ProbVTail16Avx512') +j = s.index('ProbVPair(probs[k], probs[k + 1])', i) +s = s[:j] + 'ProbVPair(probs[k + 1], probs[k]) /* MUTANT */' + s[j+len('ProbVPair(probs[k], probs[k + 1])'):] +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32766.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32766.py new file mode 100644 index 00000000..44c9308d --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32766.py @@ -0,0 +1,9 @@ +# S2 guard: p <= 32,767 -> <= 32,766. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('constexpr int64_t kProbVMaxP = 32767;', 'constexpr int64_t kProbVMaxP = 32766; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32768.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32768.py new file mode 100644 index 00000000..2ea52d4f --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pmax_32768.py @@ -0,0 +1,9 @@ +# S2 guard: p <= 32,767 -> <= 32,768. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('constexpr int64_t kProbVMaxP = 32767;', 'constexpr int64_t kProbVMaxP = 32768; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_pnonneg_dropped.py b/docs/attention-rowsites/s2/mutation-scripts/s2_pnonneg_dropped.py new file mode 100644 index 00000000..893443c0 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_pnonneg_dropped.py @@ -0,0 +1,9 @@ +# S2 guard: p >= 0 dropped. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('if (p < 0 || p > kProbVMaxP) return false;', 'if (p > kProbVMaxP) return false; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_sum_2p16.py b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_2p16.py new file mode 100644 index 00000000..511f7201 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_2p16.py @@ -0,0 +1,9 @@ +# S2 guard: Sum p <= 2^15 -> <= 2^16 (output-equivalent, section 5.2). +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('constexpr int64_t kProbVMaxSum = INT64_C(1) << 15;', 'constexpr int64_t kProbVMaxSum = INT64_C(1) << 16; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_sum_dropped.py b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_dropped.py new file mode 100644 index 00000000..3d58b733 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_dropped.py @@ -0,0 +1,9 @@ +# S2 guard: Sum p <= 2^15 dropped. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('return sum <= kProbVMaxSum;', 'return (void)sum, true; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/mutation-scripts/s2_sum_lt.py b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_lt.py new file mode 100644 index 00000000..a515da39 --- /dev/null +++ b/docs/attention-rowsites/s2/mutation-scripts/s2_sum_lt.py @@ -0,0 +1,9 @@ +# S2 guard: Sum p <= 2^15 -> < 2^15. +import re +p='src/matmul.cpp'; s=open(p).read() +def sub(old, new, count=1): + global s + assert s.count(old) == count, (old, s.count(old)) + s = s.replace(old, new) +sub('return sum <= kProbVMaxSum;', 'return sum < kProbVMaxSum; // MUTANT') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s2/pv-data-terms.txt b/docs/attention-rowsites/s2/pv-data-terms.txt new file mode 100644 index 00000000..75500b63 --- /dev/null +++ b/docs/attention-rowsites/s2/pv-data-terms.txt @@ -0,0 +1,11 @@ +# S2 cell 11.1(d) data terms, re-derived on the S1 series head (the base for S2) with the plan's probe: +# probes/rev3/probe_hooks.diff applied to a scratch copy (hooks at function entries only), probes/rev3/probe_111d.cpp, +# GCC 13.3 -O2, artifact p05_l1 (sha256 f0fd4886...6ed3), mode 'direct' (the 11.1(d) drive: 128-token chunk, then 32 decode steps). +# pv_int16_fail is the S2 fallback data term; pv_hd16_fail is 0 (head_dim 64). Same figures as the plan's c3e0004 probe. +L=1 H=14 KV=2 head_dim=64 hidden=896 inter=4864 cap=2048 +pv fail: width=3 sum=32768 max=32768 at=1 nonzero=1 (call 42) +context_length after decode = 160 +[direct prefill] softmax=1792 guard_fail=0 pv=1792 pv_int16_fail=15 pv_hd16_fail=0 requant=1536 chain_records=1536 kv_records=0 norm_taken=256 norm_skip=0 silu_taken=128 silu_skip=0 land_taken=256 land_skip=0 +[direct prefill] records by site: attn_ctx=128 attn_norm=128 attn_residual=128 down_proj.requant=128 embed=128 gate_proj.requant=128 mlp_act=128 mlp_norm=128 mlp_residual=128 o_proj.requant=128 q_proj.requant=128 up_proj.requant=128 +[direct decode] softmax=448 guard_fail=0 pv=448 pv_int16_fail=0 pv_hd16_fail=0 requant=384 chain_records=384 kv_records=0 norm_taken=64 norm_skip=0 silu_taken=32 silu_skip=0 land_taken=64 land_skip=0 +[direct decode] records by site: attn_ctx=32 attn_norm=32 attn_residual=32 down_proj.requant=32 embed=32 gate_proj.requant=32 mlp_act=32 mlp_norm=32 mlp_residual=32 o_proj.requant=32 q_proj.requant=32 up_proj.requant=32 diff --git a/docs/attention-rowsites/s2/red-sslm_axis_digest.txt b/docs/attention-rowsites/s2/red-sslm_axis_digest.txt new file mode 100644 index 00000000..f5f22044 --- /dev/null +++ b/docs/attention-rowsites/s2/red-sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/red-sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s2/red-sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..5b7372bc --- /dev/null +++ b/docs/attention-rowsites/s2/red-sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/red-sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s2/red-sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..16135e6e --- /dev/null +++ b/docs/attention-rowsites/s2/red-sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/red-sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s2/red-sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..11e56cde --- /dev/null +++ b/docs/attention-rowsites/s2/red-sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/red-sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s2/red-sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..0a9ea5d4 --- /dev/null +++ b/docs/attention-rowsites/s2/red-sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/red-suites.txt b/docs/attention-rowsites/s2/red-suites.txt new file mode 100644 index 00000000..bb7c4b0e --- /dev/null +++ b/docs/attention-rowsites/s2/red-suites.txt @@ -0,0 +1,58 @@ +# S2 red cells (commit 1 of the S2 series), GCC 13.3.0 Release on this host (AVX-512BW; the auto binary dispatches AVX-512), +# SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1 (sha256 f0fd4886...6ed3). Every failure is a path-counter assertion (the prob-V counters are +# declared and never incremented) or the 11.2 selector (the red stub selects v1.9.0 code on every tier). Every value assertion +# passes on the base: the S2 grid, corners and hostile rows equal the v1.9.0 loop, and the S2 golden equals the v1.9.0 tag's pin. +== superslm_tests +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2306 checks, 269 failures +superslm tests: 27837 checks, 269 failures + 266 test_attn_rowsites.cpp:709 + 1 test_attn_rowsites.cpp:770 + 2 test_attn_rowsites.cpp:774 +cell 11.2: 7 of 16 selector rows wrong +cell 11.2 wiring: DispatchSitesKernel(tier 2) = v1.9.0 code, want AVX2 (switch 0, msvc 0) +cell 11.2 wiring: DispatchSitesKernel(tier 3) = v1.9.0 code, want AVX-512 (switch 0, msvc 0) +4.S2 grid, realistic row (head_dim 4, width 1, guard copy: fallback): pv_fast_avx2 +0 pv_fallback_avx2 +0 pv_fast_avx512 +0 pv_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== superslm_tests_sse2_forced +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2306 checks, 3 failures +superslm tests: 27779 checks, 3 failures + 1 test_attn_rowsites.cpp:770 + 2 test_attn_rowsites.cpp:774 +cell 11.2: 7 of 16 selector rows wrong +cell 11.2 wiring: DispatchSitesKernel(tier 2) = v1.9.0 code, want AVX2 (switch 0, msvc 0) +cell 11.2 wiring: DispatchSitesKernel(tier 3) = v1.9.0 code, want AVX-512 (switch 0, msvc 0) +== superslm_tests_avx2_forced +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2306 checks, 269 failures +superslm tests: 27795 checks, 269 failures + 265 test_attn_rowsites.cpp:709 + 1 test_attn_rowsites.cpp:770 + 2 test_attn_rowsites.cpp:774 +cell 11.2: 7 of 16 selector rows wrong +cell 11.2 wiring: DispatchSitesKernel(tier 2) = v1.9.0 code, want AVX2 (switch 0, msvc 0) +cell 11.2 wiring: DispatchSitesKernel(tier 3) = v1.9.0 code, want AVX-512 (switch 0, msvc 0) +4.S2 grid, realistic row (head_dim 4, width 1, guard copy: fallback): pv_fast_avx2 +0 pv_fallback_avx2 +0 pv_fast_avx512 +0 pv_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== superslm_tests_avx512_forced +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2306 checks, 269 failures +superslm tests: 27795 checks, 269 failures + 266 test_attn_rowsites.cpp:709 + 1 test_attn_rowsites.cpp:770 + 2 test_attn_rowsites.cpp:774 +cell 11.2: 7 of 16 selector rows wrong +cell 11.2 wiring: DispatchSitesKernel(tier 2) = v1.9.0 code, want AVX2 (switch 0, msvc 0) +cell 11.2 wiring: DispatchSitesKernel(tier 3) = v1.9.0 code, want AVX-512 (switch 0, msvc 0) +4.S2 grid, realistic row (head_dim 4, width 1, guard copy: fallback): pv_fast_avx2 +0 pv_fallback_avx2 +0 pv_fast_avx512 +0 pv_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) diff --git a/docs/attention-rowsites/s2/sanitizers.txt b/docs/attention-rowsites/s2/sanitizers.txt new file mode 100644 index 00000000..0d73cff7 --- /dev/null +++ b/docs/attention-rowsites/s2/sanitizers.txt @@ -0,0 +1,6 @@ +# S2 sanitizer runs at the final code (GCC 13.3; ASan+UBSan and TSan builds as the CI legs configure them), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. +# 'reports' counts AddressSanitizer errors, UBSan runtime errors and ThreadSanitizer warnings in the run's output. +== build-asan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures|superslm tests: 27927 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures|superslm tests: 27885 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures|superslm tests: 27885 checks, 0 failures| reports: 0 +== build-tsan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures|superslm tests: 27927 checks, 0 failures| reports: 0 diff --git a/docs/attention-rowsites/s2/sslm_axis_digest.txt b/docs/attention-rowsites/s2/sslm_axis_digest.txt new file mode 100644 index 00000000..f5f22044 --- /dev/null +++ b/docs/attention-rowsites/s2/sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s2/sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..5b7372bc --- /dev/null +++ b/docs/attention-rowsites/s2/sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s2/sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..16135e6e --- /dev/null +++ b/docs/attention-rowsites/s2/sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s2/sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..11e56cde --- /dev/null +++ b/docs/attention-rowsites/s2/sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s2/sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..0a9ea5d4 --- /dev/null +++ b/docs/attention-rowsites/s2/sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 values=634120 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s2/suites-clang.txt b/docs/attention-rowsites/s2/suites-clang.txt new file mode 100644 index 00000000..785b0678 --- /dev/null +++ b/docs/attention-rowsites/s2/suites-clang.txt @@ -0,0 +1,59 @@ +# S2 final implementation commit plus the evidence commit's 4.S2 kernel-blocking rows, Clang 18.1.3 Release, the four suite binaries and five digest legs, SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. +== superslm_tests (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27927 checks, 0 failures +== superslm_tests_sse2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the SSE2 tier; tiled path not taken on this tier (SKIPPED reach); tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 195 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27869 checks, 0 failures +== superslm_tests_avx2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX2 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27885 checks, 0 failures +== superslm_tests_avx512_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27885 checks, 0 failures +== sslm_axis_digest: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 diff --git a/docs/attention-rowsites/s2/suites.txt b/docs/attention-rowsites/s2/suites.txt new file mode 100644 index 00000000..355c657d --- /dev/null +++ b/docs/attention-rowsites/s2/suites.txt @@ -0,0 +1,59 @@ +# S2 final implementation commit plus the evidence commit's 4.S2 kernel-blocking rows, GCC 13.3.0 Release, this host (AVX-512BW; auto dispatches AVX-512), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. Digests: the five legs. +== superslm_tests (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27927 checks, 0 failures +== superslm_tests_sse2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the SSE2 tier; tiled path not taken on this tier (SKIPPED reach); tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 195 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27869 checks, 0 failures +== superslm_tests_avx2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX2 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27885 checks, 0 failures +== superslm_tests_avx512_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM: this binary dispatches on the AVX-512 tier; tiled path expected at M >= 8; tiled-entry counter read +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites 11.1(d): prefill and decode windows driven on ../slice1/art/p05_l1.sslm +attn-rowsites cells (plan slices S1, S2): 2396 checks, 0 failures +superslm tests: 27885 checks, 0 failures +== sslm_axis_digest: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL ec7016a0162c8f6061dd94deee7d839056069efe8be30ff9250a82bfa296c371 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 diff --git a/docs/attention-rowsites/s3/bench-forward.txt b/docs/attention-rowsites/s3/bench-forward.txt new file mode 100644 index 00000000..74d77f8d --- /dev/null +++ b/docs/attention-rowsites/s3/bench-forward.txt @@ -0,0 +1,58 @@ +# sslm_sites_bench prefill/decode --repeat=3, 7 interleaved rounds (round, build, artifact, mode, args :: output). base vs cand, both auto. +1 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5282 (T=128 D=0 layers=1) +1 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5871 (T=512 D=0 layers=1) +1 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.4281 (T=300 D=32 layers=1) +1 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.4078 (T=300 D=32 layers=2) +1 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.3903 (T=128 D=0 layers=1) +1 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4322 (T=512 D=0 layers=1) +1 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.2704 (T=300 D=32 layers=1) +1 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.7333 (T=300 D=32 layers=2) +2 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5342 (T=128 D=0 layers=1) +2 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5799 (T=512 D=0 layers=1) +2 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.5545 (T=300 D=32 layers=1) +2 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.5073 (T=300 D=32 layers=2) +2 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.4188 (T=128 D=0 layers=1) +2 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4744 (T=512 D=0 layers=1) +2 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.3355 (T=300 D=32 layers=1) +2 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 2.6916 (T=300 D=32 layers=2) +3 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5071 (T=128 D=0 layers=1) +3 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5848 (T=512 D=0 layers=1) +3 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.4599 (T=300 D=32 layers=1) +3 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 6.0219 (T=300 D=32 layers=2) +3 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.3921 (T=128 D=0 layers=1) +3 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4614 (T=512 D=0 layers=1) +3 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.4262 (T=300 D=32 layers=1) +3 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.5047 (T=300 D=32 layers=2) +4 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5329 (T=128 D=0 layers=1) +4 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5829 (T=512 D=0 layers=1) +4 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.6014 (T=300 D=32 layers=1) +4 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.9142 (T=300 D=32 layers=2) +4 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.3939 (T=128 D=0 layers=1) +4 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4293 (T=512 D=0 layers=1) +4 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.1968 (T=300 D=32 layers=1) +4 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.1610 (T=300 D=32 layers=2) +5 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5012 (T=128 D=0 layers=1) +5 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5553 (T=512 D=0 layers=1) +5 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.4433 (T=300 D=32 layers=1) +5 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.7694 (T=300 D=32 layers=2) +5 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.4228 (T=128 D=0 layers=1) +5 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4541 (T=512 D=0 layers=1) +5 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.1610 (T=300 D=32 layers=1) +5 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.2446 (T=300 D=32 layers=2) +6 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.4992 (T=128 D=0 layers=1) +6 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.5829 (T=512 D=0 layers=1) +6 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.4514 (T=300 D=32 layers=1) +6 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.4390 (T=300 D=32 layers=2) +6 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.3782 (T=128 D=0 layers=1) +6 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4744 (T=512 D=0 layers=1) +6 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.3401 (T=300 D=32 layers=1) +6 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.5370 (T=300 D=32 layers=2) +7 base p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.5179 (T=128 D=0 layers=1) +7 base p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.6220 (T=512 D=0 layers=1) +7 base p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.5636 (T=300 D=32 layers=1) +7 base p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.5613 (T=300 D=32 layers=2) +7 cand p05_l1 prefill 128 :: prefill best-of-3 ms/token: 0.3973 (T=128 D=0 layers=1) +7 cand p05_l1 prefill 512 :: prefill best-of-3 ms/token: 0.4243 (T=512 D=0 layers=1) +7 cand p05_l1 decode 300 32 :: decode best-of-3 ms/token: 1.1466 (T=300 D=32 layers=1) +7 cand p05_l2 decode 300 32 :: decode best-of-3 ms/token: 5.3566 (T=300 D=32 layers=2) +BENCH-DONE diff --git a/docs/attention-rowsites/s3/bench-requant.txt b/docs/attention-rowsites/s3/bench-requant.txt new file mode 100644 index 00000000..5bd460b6 --- /dev/null +++ b/docs/attention-rowsites/s3/bench-requant.txt @@ -0,0 +1,44 @@ +# sslm_sites_bench requant --repeat=50, 9 interleaved rounds (round, build :: output). base = S2 head (v1.9.0 element loop); cand = S3 auto (AVX-512 row leaf); cand_avx2 = S3 linked against libsuperslm_avx2_forced.a (AVX2 row leaf). +1 base :: requant best-of-50 us/call: funnel_896 5.2189 funnel_4864 31.8384 (refused 0) requant per token: 3.2996 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +1 cand :: requant best-of-50 us/call: funnel_896 1.2470 funnel_4864 7.4275 (refused 0) requant per token: 0.7755 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +1 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.7628 funnel_4864 9.1991 (refused 0) requant per token: 1.0026 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 base :: requant best-of-50 us/call: funnel_896 5.0381 funnel_4864 31.5712 (refused 0) requant per token: 3.2455 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 cand :: requant best-of-50 us/call: funnel_896 0.9865 funnel_4864 5.4637 (refused 0) requant per token: 0.5838 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2334 funnel_4864 6.6997 (refused 0) requant per token: 0.7204 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 base :: requant best-of-50 us/call: funnel_896 5.2794 funnel_4864 29.2328 (refused 0) requant per token: 3.1237 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 cand :: requant best-of-50 us/call: funnel_896 1.0085 funnel_4864 5.5941 (refused 0) requant per token: 0.5974 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.1974 funnel_4864 6.6965 (refused 0) requant per token: 0.7133 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 base :: requant best-of-50 us/call: funnel_896 5.2666 funnel_4864 31.4490 (refused 0) requant per token: 3.2808 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 cand :: requant best-of-50 us/call: funnel_896 1.0083 funnel_4864 5.6079 (refused 0) requant per token: 0.5984 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.4003 funnel_4864 6.7288 (refused 0) requant per token: 0.7547 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 base :: requant best-of-50 us/call: funnel_896 5.0214 funnel_4864 31.4203 (refused 0) requant per token: 3.2314 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 cand :: requant best-of-50 us/call: funnel_896 1.1778 funnel_4864 5.6073 (refused 0) requant per token: 0.6310 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2339 funnel_4864 6.6659 (refused 0) requant per token: 0.7181 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +6 base :: requant best-of-50 us/call: funnel_896 5.2828 funnel_4864 31.4590 (refused 0) requant per token: 3.2846 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +6 cand :: requant best-of-50 us/call: funnel_896 1.0228 funnel_4864 5.7040 (refused 0) requant per token: 0.6081 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +6 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2340 funnel_4864 6.6849 (refused 0) requant per token: 0.7195 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +7 base :: requant best-of-50 us/call: funnel_896 5.1334 funnel_4864 31.6616 (refused 0) requant per token: 3.2704 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +7 cand :: requant best-of-50 us/call: funnel_896 1.0092 funnel_4864 5.5586 (refused 0) requant per token: 0.5950 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +7 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2326 funnel_4864 6.7201 (refused 0) requant per token: 0.7217 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +8 base :: requant best-of-50 us/call: funnel_896 5.2924 funnel_4864 31.5560 (refused 0) requant per token: 3.2935 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +8 cand :: requant best-of-50 us/call: funnel_896 1.0098 funnel_4864 5.6796 (refused 0) requant per token: 0.6038 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +8 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2342 funnel_4864 6.6952 (refused 0) requant per token: 0.7202 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +9 base :: requant best-of-50 us/call: funnel_896 5.1563 funnel_4864 30.7984 (refused 0) requant per token: 3.2126 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +9 cand :: requant best-of-50 us/call: funnel_896 1.0092 funnel_4864 5.6112 (refused 0) requant per token: 0.5988 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +9 cand_avx2 :: requant best-of-50 us/call: funnel_896 1.2344 funnel_4864 6.7948 (refused 0) requant per token: 0.7275 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +# Supplementary: the candidate linked against libsuperslm_sse2_forced.a (RequantRowWide's element loop, in intmath.cpp) beside base and cand, --repeat=50, 5 interleaved rounds. +1 base :: requant best-of-50 us/call: funnel_896 5.1583 funnel_4864 31.4662 (refused 0) requant per token: 3.2611 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +1 cand_sse2 :: requant best-of-50 us/call: funnel_896 4.4915 funnel_4864 23.8544 (refused 0) requant per token: 2.5844 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +1 cand :: requant best-of-50 us/call: funnel_896 1.0254 funnel_4864 4.6008 (refused 0) requant per token: 0.5292 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 base :: requant best-of-50 us/call: funnel_896 5.1029 funnel_4864 31.3059 (refused 0) requant per token: 3.2389 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 cand_sse2 :: requant best-of-50 us/call: funnel_896 4.5454 funnel_4864 27.7919 (refused 0) requant per token: 2.8783 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +2 cand :: requant best-of-50 us/call: funnel_896 0.8970 funnel_4864 4.9329 (refused 0) requant per token: 0.5283 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 base :: requant best-of-50 us/call: funnel_896 5.1775 funnel_4864 30.1768 (refused 0) requant per token: 3.1720 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 cand_sse2 :: requant best-of-50 us/call: funnel_896 4.5594 funnel_4864 27.5217 (refused 0) requant per token: 2.8615 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +3 cand :: requant best-of-50 us/call: funnel_896 1.0252 funnel_4864 5.4262 (refused 0) requant per token: 0.5886 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 base :: requant best-of-50 us/call: funnel_896 4.8747 funnel_4864 30.7106 (refused 0) requant per token: 3.1520 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 cand_sse2 :: requant best-of-50 us/call: funnel_896 4.3216 funnel_4864 26.5323 (refused 0) requant per token: 2.7444 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +4 cand :: requant best-of-50 us/call: funnel_896 1.0268 funnel_4864 5.6436 (refused 0) requant per token: 0.6045 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 base :: requant best-of-50 us/call: funnel_896 5.2363 funnel_4864 31.5347 (refused 0) requant per token: 3.2811 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 cand_sse2 :: requant best-of-50 us/call: funnel_896 4.6727 funnel_4864 27.9207 (refused 0) requant per token: 2.9121 ms at 24 layers x (8 x 896 + 3 x 4864) + embed +5 cand :: requant best-of-50 us/call: funnel_896 0.9925 funnel_4864 5.5962 (refused 0) requant per token: 0.5945 ms at 24 layers x (8 x 896 + 3 x 4864) + embed diff --git a/docs/attention-rowsites/s3/bench.md b/docs/attention-rowsites/s3/bench.md new file mode 100644 index 00000000..50ce298d --- /dev/null +++ b/docs/attention-rowsites/s3/bench.md @@ -0,0 +1,59 @@ +# S3 bench: what the requant row leaf saves + +These are reports, not gates (plan §8 10.1). The host is the shared 4-vCPU cloud Xeon (AVX2, AVX-512BW) with GCC 13.3 -O3. +`tools/sslm_sites_bench.cpp` gains a `requant` mode. One object was linked three ways: + +- against the base library, the S2 head, which runs v1.9.0's element loop in the funnel; +- against S3's library with auto dispatch, which runs the AVX-512 row leaf here; +- against S3's `libsuperslm_avx2_forced.a`, which runs the AVX2 row leaf. + +Each reading is a best-of-R inside one process. The builds run alternately, and the medians are taken over rounds. The raw +output is in `bench-requant.txt` and `bench-forward.txt`. + +## Method 1 (the plan's): the funnel call, scaled to per-token call counts + +`requant` mode times one whole `RequantChainChecked` call: max-abs, preflight, the element loop that S3 replaces, and the +scale fold. It uses the 0.5B's two funnel widths, 896 and 4,864, over 16 rows whose max-abs spans 2^16 to 2^31. The per-token +figure is 24 layers × (8 calls at 896 + 3 at 4,864), plus the embed's one call at 896 (G26). Prefill and decode make the same +calls per token. Only the element loop differs between builds, so the difference is the loop's saving. + +The runs used best of 50 and 9 interleaved rounds. The saving is the median of the paired differences; the range is the minimum +and maximum over the rounds. + +| Reading | Base (v1.9.0 loop) | S3 AVX-512 | Saved (range) | × | S3 AVX2 | Saved (range) | × | Plan §0 / §4.3 estimate | +|---|---|---|---|---|---|---|---|---| +| µs per call, width 896 | 5.219 | 1.009 | 4.15 (3.84–4.28) | 5.2 | 1.234 | 3.90 (3.46–4.08) | 4.2 | loop 2.4 → 1.1 (AVX2), 0.5 (AVX-512) | +| µs per call, width 4,864 | 31.459 | 5.608 | 25.8 (23.6–26.1) | 5.6 | 6.700 | 24.8 (22.5–24.9) | 4.7 | loop 19.8 → 6.1 (AVX2), 2.4 (AVX-512) | +| **ms per token** (prefill and decode) | 3.270 | 0.599 | **2.66** (2.52–2.69) | 5.5 | 0.720 | **2.53** (2.30–2.57) | 4.5 | **1.22** (AVX2), about 1.6 (AVX-512) | + +**Against the estimate.** The saving is about twice the estimate: 2.53 ms/token on AVX2 against 1.22 (207%), and 2.66 on +AVX-512 against about 1.6. The lanes are not faster than the spike found. The difference is the base: at width 4,864 the +v1.9.0 loop's saving alone is 24.8 µs on AVX2, where the spike measured the whole loop at 19.8 µs. The plan's per-token +arithmetic reproduces the estimate from the spike's figures, 24 × (3 × 13.7 + 8 × 1.3) µs = 1.24 ms. So the error is in the +spike's base per-element cost, not in the lanes. + +- **Part of the gap is a per-element call, but only a small part.** In the base, the funnel reaches `RequantTokenCodeWide` + through a relocated out-of-line call per element (it lives in `intmath.cpp`, and the build has no LTO). The same loop built + inside `intmath.cpp` (the forced-SSE2 library, where `RequantRowWide` runs it) measures 27.5 µs at 4,864 against the base's + 31.3, and 2.86 against 3.24 ms/token (supplementary rows in `bench-requant.txt`). That is about 12%. +- **The rest is the element code's own cost on this host**, which is higher than the spike's reading. +- **AVX2 is within 10–20% of AVX-512** (1.23 against 1.01 µs at 896; 6.70 against 5.61 at 4,864). + +## Method 2: the whole forward on the reduced-layer artifacts, scaled to 24 layers + +This is `prefill` and `decode` mode on the 0.5B-width synthetics, with p05_l1 at sha256 f0fd4886…6ed3. It used best of 3 and +7 interleaved rounds, and compares base (auto) with S3 (auto). + +| Reading | Base | S3 | Paired saving, median (Q1, Q3) | Per layer | × 24 | Method 1 | +|---|---|---|---|---|---|---| +| Prefill, p05_l1, T = 128, ms per prompt token | 0.518 | 0.394 | 0.121 (0.115, 0.138) | 0.121 | 2.9 | 2.66 | +| Prefill, p05_l1, T = 512 | 0.583 | 0.454 | 0.123 (0.106, 0.155) | 0.123 | 3.0 | 2.66 | +| Decode, p05_l1, context 300, 32 steps, ms per step | 1.460 | 1.270 | 0.219 (0.111, 0.405) | 0.219 | 5.3 | 2.66 | +| Decode, p05_l2, context 300, 32 steps | 5.561 | 5.357 | 0.517 (−0.098, 0.753) | 0.259 | 6.2 | 2.66 | + +The two prefill readings agree with method 1: 2.9 and 3.0 against 2.66, where the per-layer figure also carries the embed's +call once per token rather than once per 24 layers. The decode readings are too wide to use: their quartiles span 3–4× on the +shared host. Method 1's figures are the ones reported. + +These are per-layer engine figures on synthetic weights. They say nothing about any consumer's end-to-end speed, and they were +not measured on the project's reference hardware (the box's B2, §8 10.2, is box-only). diff --git a/docs/attention-rowsites/s3/blob-protocol.txt b/docs/attention-rowsites/s3/blob-protocol.txt new file mode 100644 index 00000000..76e6f4e0 --- /dev/null +++ b/docs/attention-rowsites/s3/blob-protocol.txt @@ -0,0 +1,55 @@ +# S3 save-blob protocol (plan 6.4, as S1 and S2 ran it): the sslm_bench_prefill tool (tools/t2147_chunk_batched_pins.cpp) as built by +# CMake (GCC 13.3, Release). Base = the S1 head's build (the v1.9.0 funnel element loop, and v1.9.0 prob-V), unchanged from S2's run. +# Candidates = S3's implementation commit: pins_s3 (auto dispatch, the AVX-512 row leaf here) and pins_s3_avx2 +# (sslm_bench_prefill_avx2_forced, the AVX2 row leaf). Each row prefills the prompt at the chunk budget, decodes 32 greedy tokens and +# dumps the SSB5 blob; EQUAL = blobs byte-equal (cmp) and decoded tokens equal. Artifacts: the S1 synthetic set (wide_l8, p05_l1 +# sha256 f0fd4886...6ed3, p05_l2) and the in-tree 8-layer fixture. +# +# Result: 46 of 46 rows EQUAL (32 auto, 14 forced AVX2); on the 29 rows S2 also ran, every blob hash equals S2's record. +wide_l8 ids:8 chunk_budget=1 +32 decode, candidate pins_s3: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode, candidate pins_s3: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode, candidate pins_s3: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate pins_s3: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=1 +32 decode, candidate pins_s3: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode, candidate pins_s3: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode, candidate pins_s3: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate pins_s3: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate pins_s3: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate pins_s3: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate pins_s3: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate pins_s3: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate pins_s3: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=1 +32 decode, candidate pins_s3: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode, candidate pins_s3: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode, candidate pins_s3: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate pins_s3: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate pins_s3: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate pins_s3: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate pins_s3: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate pins_s3: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate pins_s3: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:24:128 chunk_budget=1 +32 decode, candidate pins_s3: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=8 +32 decode, candidate pins_s3: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=24 +32 decode, candidate pins_s3: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:8:128 chunk_budget=1 +32 decode, candidate pins_s3: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +intree ids:8:128 chunk_budget=8 +32 decode, candidate pins_s3: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate pins_s3_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate pins_s3_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate pins_s3_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate pins_s3_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate pins_s3_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate pins_s3_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +intree ids:24:128 chunk_budget=1 +32 decode, candidate pins_s3_avx2: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=8 +32 decode, candidate pins_s3_avx2: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +DONE diff --git a/docs/attention-rowsites/s3/coverage.txt b/docs/attention-rowsites/s3/coverage.txt new file mode 100644 index 00000000..87308d83 --- /dev/null +++ b/docs/attention-rowsites/s3/coverage.txt @@ -0,0 +1,33 @@ +# S3 branch coverage (plan §3.4, cell 11.6). INDICATIVE ONLY, as S2's: a floor is pinned only from the hosted branch-coverage leg's own +# recorded measurement, and this series is delivered as patches, not pushed, so that leg has not run on it. These are local replicas +# of the leg's commands (clang-18, llvm-cov/llvm-profdata 18, RelWithDebInfo, -fprofile-instr-generate -fcoverage-mapping; the five +# binaries, merged, exported over src/*.cpp include/superslm/*.h), on this host (AVX-512BW), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. +# All five binaries exit 0 on both trees. +# +# Base = the S2 head; candidate = S3's implementation commit plus the evidence commit's rows. Branches covered / total. +# +# (1) All five profiles (the leg's union), this host (the auto and forced AVX-512 binaries run AVX-512): +# base src/intmath.cpp 153/174 87.93103448275862 src/matmul.cpp 286/312 91.66666666666666 +# candidate src/intmath.cpp 167/188 88.82978723404256 src/matmul.cpp 286/312 91.66666666666666 +# check_branch_coverage_floors.py: OK on both (19 files at or above their pinned floor). S3 adds 14 branches to intmath.cpp and +# covers all 14; its uncovered lines are exactly the base's 14, moved by S3's insertions (+27 above the new code, +151 below it). +# +# (2) Approximating the hosted runner, which has no AVX-512: only the sse2-forced, avx2-forced and scalar-forced-digest profiles merged +# (there, the auto binary dispatches AVX2 and the forced AVX-512 binary faults at its first AVX-512 instruction): +# base src/intmath.cpp 153/174 87.93103448275862 src/matmul.cpp 203/286 70.97902097902097 +# candidate src/intmath.cpp 163/188 86.70212765957447 src/matmul.cpp 203/286 70.97902097902097 +# check_branch_coverage_floors.py on the candidate's projection: FAILED, src/intmath.cpp 86.70% < floor 87.93% and src/matmul.cpp +# 70.98% < floor 72.22% (the latter is S2's, unchanged by S3, and recorded in S2's coverage.txt). +# So on a runner without AVX-512, S3 takes intmath.cpp about 1.2 points below its floor. The four branches it loses there are the +# AVX-512 body's lane loop (line 637, both sides) and the dispatcher's AVX-512 arm (lines 660/664); every one needs AVX-512 to run. +# +# Uncovered S3 branches (lines of src/intmath.cpp at the implementation commit): +# With all five profiles: none. +# Without AVX-512 profiles: 637 (RequantRowAvx512's loop) and 660/664 (RequantRowWide's kAvx512 arm). +# No reachable S3 branch is uncovered, so no cell is added (plan §3.4 step 1). +# +# Plan §3.4 steps 2 and 3: allowlist lines for the above are added in tools/ci/branch_coverage_allowlist.txt, in the file's existing +# reviewer-note form (its checker's pytest: 6 passed). The floors in tools/ci/branch_coverage_floors.json are NOT re-pinned: step 3 +# copies the leg's recorded value, which does not exist until the series runs in CI. The allowlist does not exclude lines from the +# measured percentage, so if the hosted runner lacks AVX-512 the leg will fail on intmath.cpp (as it already would on matmul.cpp +# from S2) until the floor is lowered; that is the owner's one-line call at code review (G27), as for S2. diff --git a/docs/attention-rowsites/s3/fp-scan.txt b/docs/attention-rowsites/s3/fp-scan.txt new file mode 100644 index 00000000..7e06c2e8 --- /dev/null +++ b/docs/attention-rowsites/s3/fp-scan.txt @@ -0,0 +1,116 @@ +# S3 fp-free scan (plan §3.4; tests/ci/scan_build_output.py --build-dir --target superslm --isa x86-64), allow-lists unchanged. +# +# Clang 18.1 build at the implementation commit: PASS (488 symbols ACCEPT, 0 REJECT). intmath.cpp.o is clean. +# GCC 13.3 build at the implementation commit: FAIL, one symbol, TiledGemmAvx512 in matmul.cpp.o. It is inherited, not S3's: +# the S2 head's own GCC build rejects the same symbol (555 ACCEPT, 1 REJECT); main fixes it in 90e48de. With 90e48de's +# src/matmul.cpp change applied temporarily (not part of this series), the GCC scan PASSES (559 ACCEPT, 0 REJECT). intmath.cpp.o is +# clean on GCC either way. +# +# How the AVX2 body got past Clang, without widening the allow-list (the rejects met on the way, each located with +# check_fp_free_scan._x86_check_a over the scan's own capstone decode of RequantRowAvx2): +# 1. A 64-bit select (the clamp by vpblendvb on a vpcmpgtq mask) and the xor/sub abs and sign idioms: Clang lowers them to +# vblendvpd and vxorpd, FP-domain encodings. An and/andnot spelling of the same select was lowered to vblendvpd too. +# Fix: |x|, the clamp and the sign are done in 32-bit lanes (vpabsd, vpminud with the high dword folded into bit 7, vpsignd). +# 2. vbroadcasti128: the byte-pick constant had identical 128-bit halves, and Clang loaded it with a 128-bit broadcast, which is +# not on the vetted move list. Fix: the high half's never-read bytes differ, so the constant is a plain vmovdqu load. +# +# Outputs follow (Clang, GCC, GCC with 90e48de applied, the S2 head's GCC build), with the non-gating check-(C)-only symbol lists +# elided (they are diagnostics only, and unchanged by S3 apart from the new symbols' own entries). + +######## scan-clang +Scanning 17 object member(s) of archive build-clang/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (30 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (52 symbol(s), format=elf) + clean model.cpp.o (77 symbol(s), format=elf) + clean intmath.cpp.o (27 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (23 symbol(s), format=elf) + clean proof_manifest.cpp.o (21 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (15 symbol(s), format=elf) + clean forward_sites.cpp.o (49 symbol(s), format=elf) + clean decode_digest.cpp.o (3 symbol(s), format=elf) + clean sslm_abi.cpp.o (125 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (26 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (6 symbol(s), format=elf) + +Totals: 17 object(s); 488 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 284 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-gcc +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (28 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + REJECT matmul.cpp.o (1 symbol(s), format=elf) + _ZN8superslm12_GLOBAL__N_115TiledGemmAvx512EPKsmPKammmmmPlPa + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (58 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) + +Totals: 17 object(s); 558 symbol(s) ACCEPT, 1 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 373 symbol(s) reject under check (C) alone (non-gating diagnostic) +FAIL: the scan did not come back clean. + +######## scan-gcc-fix +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (28 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (24 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (58 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) + +Totals: 17 object(s); 559 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 374 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-gcc-base +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (25 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + REJECT matmul.cpp.o (1 symbol(s), format=elf) + _ZN8superslm12_GLOBAL__N_115TiledGemmAvx512EPKsmPKammmmmPlPa + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (58 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) + +Totals: 17 object(s); 555 symbol(s) ACCEPT, 1 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 373 symbol(s) reject under check (C) alone (non-gating diagnostic) +FAIL: the scan did not come back clean. diff --git a/docs/attention-rowsites/s3/golden.txt b/docs/attention-rowsites/s3/golden.txt new file mode 100644 index 00000000..f3089046 --- /dev/null +++ b/docs/attention-rowsites/s3/golden.txt @@ -0,0 +1,9 @@ +# S3 golden pin provenance (plan §3.3 evidence 3, cell 6.3) +# tools/gen_attn_rowsite_golden.cpp (one hash per slice) compiled with GCC 13.3 -O2 against the v1.9.0 tag (d870d27) include/ +# and its Release libsuperslm.a (the recipe in the generator's header): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +S2 prob-V golden: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed over 30100 values +S3 requant-row golden: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 over 3567018 values +# Re-running the v1.9.0 build with the header path as argument wrote tests/attn_rowsite_golden_pin.h; S1's and S2's hashes and +# value counts are unchanged, and S3's constants are added beside them. The S3 set drives RequantChainChecked (a v1.9.0 +# signature) over tests/support/rowsite_cases.h's RunRequantRowCases. diff --git a/docs/attention-rowsites/s3/leaf-plant.txt b/docs/attention-rowsites/s3/leaf-plant.txt new file mode 100644 index 00000000..cd129704 --- /dev/null +++ b/docs/attention-rowsites/s3/leaf-plant.txt @@ -0,0 +1,7 @@ +== 11.4 plant: a call to RequantRowWide appended to src/forward/forward_sites.cpp +check_no_forward_leaf_calls.py: FAILED + - src/forward/forward_sites.cpp:3206: names banned leaf 'RequantRowWide' outside the funnel's own file +exit 1 +== unplanted +check_no_forward_leaf_calls.py: OK -- 2 forward-composition file(s) scanned, zero banned-leaf calls outside the allowlist, and the door count holds at 3 +exit 0 diff --git a/docs/attention-rowsites/s3/linkage-plant.txt b/docs/attention-rowsites/s3/linkage-plant.txt new file mode 100644 index 00000000..c1c3a0b3 --- /dev/null +++ b/docs/attention-rowsites/s3/linkage-plant.txt @@ -0,0 +1,31 @@ +== 11.3 plant: RequantRowAvx2 moved out of the anonymous namespace (external linkage), auto library +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 2 population symbols, record x0 +check_tiled_matmul_linkage: FAIL + build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 'superslm::RequantRowAvx2(long const*, unsigned long, long, int, signed char*)' is not local (nm type 'T') +exit 1 +== unplanted, the six objects of the CI job +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o: 3 population symbols, record x1 +build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o: 5 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 2 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o: 2 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o: 2 population symbols, record x0 + TiledGemmAvx2: symbol in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledGemmAvx512: symbol in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledMicroAvx2: inlined into superslm::TiledGemmAvx2 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledMicroAvx512: inlined into superslm::TiledGemmAvx512 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledPackPanel16: inlined into superslm::TiledGemmAvx2 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledTranspose8x8Epi16: inlined into superslm::TiledGemmAvx2 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + TiledWidenActivations: inlined into superslm::detail::GemmInt8AccumulateCols in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + RunTiledGemm: inlined into superslm::detail::GemmInt8AccumulateCols in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbVAccumulateIntoAvx2: symbol in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbVAccumulateIntoAvx512: symbol in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbVBlockAvx2: inlined into superslm::ProbVAccumulateIntoAvx2 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbVBlockAvx512: symbol in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbVTail16Avx512: inlined into superslm::ProbVAccumulateIntoAvx512 in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + ProbQ15AccumulateInto: inlined into superslm::GemmProbQ15Accumulate in build/CMakeFiles/superslm.dir/src/matmul.cpp.o + RequantRowAvx2: symbol in build/CMakeFiles/superslm.dir/src/intmath.cpp.o + RequantRowAvx512: symbol in build/CMakeFiles/superslm.dir/src/intmath.cpp.o +check_tiled_matmul_linkage: OK +exit 0 diff --git a/docs/attention-rowsites/s3/mutants.txt b/docs/attention-rowsites/s3/mutants.txt new file mode 100644 index 00000000..c1f5e5aa --- /dev/null +++ b/docs/attention-rowsites/s3/mutants.txt @@ -0,0 +1,199 @@ +# S3 mutation evidence (plan §9; GCC 13.3 Release; this host has AVX-512BW, so the auto binary dispatches AVX-512), run at the final +# implementation commit. Each mutant is applied to a synced copy of that commit by its script (mutation-scripts/.py, run from +# the repository root). superslm_tests (auto), superslm_tests_avx2_forced and superslm_tests_avx512_forced are then rebuilt and run +# from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT set (p05_l1, sha256 f0fd4886...6ed3). 'none' is the unmutated +# control. The raw log follows the summary; it shows at most four distinct failing lines per run (paths shortened). +# +# Arithmetic and loop mutants are applied to each tier's body separately (§9 preamble). Each AVX2-body mutant must die on the forced +# AVX2 binary (and on auto on a runner without AVX-512, where auto dispatches AVX2); it survives on the AVX-512 binaries here, which +# never run that body. Each AVX-512-body mutant must die on forced AVX-512 and on auto, and it survives on forced AVX2. The +# dispatcher's tail and the funnel's call site are shared by every tier, so those mutants die on all three binaries. +# 'all_avx512_runs_avx2_body' can die only where the AVX-512 kernel is selected (auto here, and forced AVX-512); it is output-exact +# (both bodies are exact), so only the path assertion sees it. +# +# Summary (K = killed with that many failures, s = survives by construction, see above): +# mutant auto(AVX-512) forced AVX2 forced AVX-512 killing cells +# s3_round_2e_avx2 (§9) s K 58 s output: the attn-rowsites cells (46-104 failures each) and the +# s3_round_2e_avx512 (§9) K 53 s K 53 pre-existing funnel witnesses in test_main (T-1254 extremes, +# s3_shift_e31_avx2 (§9) s K 123 s the §4.1 LUT/i-exp code pins), which run the funnel and so +# s3_shift_e31_avx512 (§9) K 91 s K 91 now run the row leaf +# s3_carry_dropped_avx2 (§9) s K 55 s +# s3_carry_dropped_avx512 (§9) K 53 s K 53 +# s3_loop_bound_avx2 (§9) s K abort s sentinel (4.S3): "n 1, d' 1: 3 of the 32 sentinel bytes around the +# s3_loop_bound_avx512 (§9) K abort s K abort row changed" (7 for AVX-512). In the suite's own order an earlier +# test_main cell (TestApplyWeightScaleFoldC24...AgainstTheRealFunnel, +# a stack row) is reached first and glibc aborts on "stack smashing +# detected"; the sentinel kill was shown with the attn-rowsites cells +# run first (a scratch-only reorder of main), see the end of this file +# s3_never_entered (§9) K 6 K 6 K 6 counter: 11.1(d) prefill +0 (want 1,536) and decode +0 (want 384), +# the funnel call-site cell, and the 4.S3 path totals +# all_avx512_runs_avx2_body K 10 s K 10 counter (path): requant_row_avx2 moves where avx512 is expected +# extras (not in §9): +# s3_rhi_dropped_avx2 s K 111 s output: r = 2^32 rows (4.S3 corner premises, 7.S3, grid) and the +# s3_rhi_dropped_avx512 K 79 s K 79 test_main oracle V landing (code 0 for 127) +# s3_tail_dropped K 52 K 69 K 52 output: n = 5, 9 rows (4.S3 grid), 6.3 S1/S2/S3 golden +# s3_counter_deleted_avx2 s K 10 s counter (path) +# s3_counter_deleted_avx512 K 10 s K 10 counter (path) +# withdrawn in rev 3 (§9 preamble), run only to confirm they are equivalent; each survives everywhere, as the plan predicts: +# wd_clamp_dropped_avx2 s s s (the AVX2 clamp is now the unsigned 32-bit minimum; dropping it +# wd_clamp_dropped_avx512 s s s keeps the low byte of a magnitude that never exceeds 127 in contract) +# wd_h_arith_shift_avx512 s s s (AVX2 has no 64-bit arithmetic shift, so there is no AVX2 twin) +# +# Result: every §9 S3 mutant is killed on every binary that runs the tier it mutates, with each tier's body mutated separately. +# +# Raw log: +== none on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== none on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures|superslm tests: 109744 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +16359 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +16359 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +1 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +== all_avx512_runs_avx2_body on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures|superslm tests: 109702 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +16359 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +16359 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +1 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +== s3_carry_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_carry_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 6797 checks, 46 failures|superslm tests: 32286 checks, 55 failures| +FAIL tests/test_main.cpp:8517: out_codes[j] == c.expected_codes[j] — positive_extreme: RequantChainChecked out_codes[1] == 63, want 64 (T-1254 witness, matches _requant_row_int64/intmath.requant_token_code) +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -64, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-64) must differ from codes_iexp[60] (-64) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1886465,-1886465), want (-1889041,- +== s3_carry_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_carry_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22988 checks, 46 failures|superslm tests: 48519 checks, 53 failures| +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -64, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-64) must differ from codes_iexp[60] (-64) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1886465,-1886465), want (-1889041,- +FAIL tests/test_main.cpp:15445: m[1] == b[1] - 756240 && m[3] == b[3] - 756240 — cell 3b: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -756240 -- baseline (-1886465,-1886465), want (-264 +== s3_carry_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_carry_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22988 checks, 46 failures|superslm tests: 48477 checks, 53 failures| +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -64, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-64) must differ from codes_iexp[60] (-64) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1886465,-1886465), want (-1889041,- +FAIL tests/test_main.cpp:15445: m[1] == b[1] - 756240 && m[3] == b[3] - 756240 — cell 3b: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -756240 -- baseline (-1886465,-1886465), want (-264 +== s3_counter_deleted_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_counter_deleted_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures|superslm tests: 109702 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +0 requant_row_avx512 +0; want +16359/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +0 requant_row_avx512 +0; want +16359/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +1/+0 (kernel: AVX2) +== s3_counter_deleted_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_counter_deleted_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures|superslm tests: 109744 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +== s3_counter_deleted_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_counter_deleted_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures|superslm tests: 109702 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +== s3_loop_bound_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_loop_bound_avx2 on superslm_tests_avx2_forced: exit 134; +== s3_loop_bound_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_loop_bound_avx512 on superslm_tests: exit 134; +== s3_loop_bound_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_loop_bound_avx512 on superslm_tests_avx512_forced: exit 134; +== s3_never_entered on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 6 failures|superslm tests: 109744 checks, 6 failures| +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) prefill requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1536 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) decode requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+384 (kernel: AVX-512) +== s3_never_entered on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 6 failures|superslm tests: 109702 checks, 6 failures| +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +1/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) prefill requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +1536/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) decode requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +384/+0 (kernel: AVX2) +== s3_never_entered on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 6 failures|superslm tests: 109702 checks, 6 failures| +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) prefill requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1536 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) decode requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+384 (kernel: AVX-512) +== s3_rhi_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_rhi_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 6828 checks, 100 failures|superslm tests: 32296 checks, 111 failures| +FAIL tests/test_main.cpp:8517: out_codes[j] == c.expected_codes[j] — positive_extreme: RequantChainChecked out_codes[0] == 0, want 127 (T-1254 witness, matches _requant_row_int64/intmath.requant_token_code) +FAIL tests/test_main.cpp:16369: oracle_v_landed[2] == INT8_C(127) && oracle_v_landed[3] == INT8_C(-127) — oracle: this fixture's own real landed V code for head 1's two dims == {0,0}, want {127,-127} (ClampRopeCode's saturating clamp on t +FAIL tests/test_main.cpp:16383: oracle_ctx_h1[0] == INT64_C(4161536) && oracle_ctx_h1[1] == INT64_C(-4161536) — oracle: GemmProbQ15Accumulate(probs={2^15}, values={127,-127}) == {0,0}, want {4161536,-4161536} (32768*127 per dim, the real +FAIL tests/test_main.cpp:16395: expected_fold_h1_dim0 == INT64_C(2080768) && expected_fold_h1_dim1 == INT64_C(-2080768) — oracle: ApplyWeightScaleFold({4161536,-4161536}, non-identity ratio=0.5) == {0,0}, want {2080768,-2080768} (the inde +== s3_rhi_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_rhi_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22908 checks, 76 failures|superslm tests: 48430 checks, 79 failures| +FAIL tests/test_slm19x_schema_damped_greedy.cpp:397: st == SSLM_OK -- decode step 9 status 4 +FAIL tests/test_attn_rowsites.cpp:585: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: fcf3f641b5b8d1c832f4423fbda38c72884c1922bd2738c66243a9cdff33bbd1 ove +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 sentinel pass (n 8, d' 1, r 4294967296, s 30): 5 of 8 codes differ from RequantTokenCodeWide, first at 0 (x 1: 0 vs 127) +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 exact-size pass (n 8, d' 1, r 4294967296, s 30): 3 of 8 codes differ from RequantTokenCodeWide, first at 0 (x 1: 0 vs 127) +== s3_rhi_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_rhi_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22908 checks, 76 failures|superslm tests: 48388 checks, 79 failures| +FAIL tests/test_slm19x_schema_damped_greedy.cpp:397: st == SSLM_OK -- decode step 9 status 4 +FAIL tests/test_attn_rowsites.cpp:585: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: fcf3f641b5b8d1c832f4423fbda38c72884c1922bd2738c66243a9cdff33bbd1 ove +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 sentinel pass (n 8, d' 1, r 4294967296, s 30): 5 of 8 codes differ from RequantTokenCodeWide, first at 0 (x 1: 0 vs 127) +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 exact-size pass (n 8, d' 1, r 4294967296, s 30): 3 of 8 codes differ from RequantTokenCodeWide, first at 0 (x 1: 0 vs 127) +== s3_round_2e_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_round_2e_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 6789 checks, 47 failures|superslm tests: 32278 checks, 58 failures| +FAIL tests/test_main.cpp:8517: out_codes[j] == c.expected_codes[j] — positive_extreme: RequantChainChecked out_codes[2] == -1, want 0 (T-1254 witness, matches _requant_row_int64/intmath.requant_token_code) +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -65, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-65) must differ from codes_iexp[60] (-65) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1919322,-1919322), want (-1921898,- +== s3_round_2e_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_round_2e_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22916 checks, 46 failures|superslm tests: 48447 checks, 53 failures| +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -65, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-65) must differ from codes_iexp[60] (-65) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1919322,-1919322), want (-1921898,- +FAIL tests/test_main.cpp:15445: m[1] == b[1] - 756240 && m[3] == b[3] - 756240 — cell 3b: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -756240 -- baseline (-1919322,-1919322), want (-267 +== s3_round_2e_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_round_2e_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22916 checks, 46 failures|superslm tests: 48405 checks, 53 failures| +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -65, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-65) must differ from codes_iexp[60] (-65) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15350: m[1] == b[1] - 2576 && m[3] == b[3] - 2576 — cell 3a: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -2576 -- baseline (-1919322,-1919322), want (-1921898,- +FAIL tests/test_main.cpp:15445: m[1] == b[1] - 756240 && m[3] == b[3] - 756240 — cell 3b: heads 0,1's own dim-1 element (KV head 0's mutated row) did not move by the measured exact delta -756240 -- baseline (-1919322,-1919322), want (-267 +== s3_shift_e31_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== s3_shift_e31_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 6748 checks, 104 failures|superslm tests: 32237 checks, 123 failures| +FAIL tests/test_main.cpp:8517: out_codes[j] == c.expected_codes[j] — positive_extreme: RequantChainChecked out_codes[0] == 63, want 127 (T-1254 witness, matches _requant_row_int64/intmath.requant_token_code) +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -32, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -32, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-32) must differ from codes_iexp[60] (-32) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +== s3_shift_e31_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_shift_e31_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22908 checks, 79 failures|superslm tests: 48439 checks, 91 failures| +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -32, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -32, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-32) must differ from codes_iexp[60] (-32) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15346: m[0] == b[0] && m[2] == b[2] — cell 3a: heads 0,1's own dim-0 element (KV head 0's OTHER row, unmutated) moved -- want byte-identical to baseline (2736027,2736027), got (2735936,2735936) +== s3_shift_e31_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== s3_shift_e31_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 22908 checks, 79 failures|superslm tests: 48397 checks, 91 failures| +FAIL tests/test_main.cpp:13093: codes_iexp[60] == INT8_C(-65) — the i-exp-sigmoid witness's own requantized code at index 60 == -32, want -65 -- this pins the executed divergence `mlp_act_iexp_mutation_proof.py` found (LUT: -64, i-exp-sig +FAIL tests/test_main.cpp:13121: codes_lut[60] == INT8_C(-64) — the LUT-based row's own requantized code at index 60 == -32, want -64 (§4.1's own citation of the excluded construction, executed against this fixture) +FAIL tests/test_main.cpp:13125: codes_lut[60] != codes_iexp[60] — codes_lut[60] (-32) must differ from codes_iexp[60] (-32) -- this is the executed divergence that makes F-S3-1 unable to recur silently: an implementation computing i-exp-s +FAIL tests/test_main.cpp:15346: m[0] == b[0] && m[2] == b[2] — cell 3a: heads 0,1's own dim-0 element (KV head 0's OTHER row, unmutated) moved -- want byte-identical to baseline (2736027,2736027), got (2735936,2735936) +== s3_tail_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 30988 checks, 52 failures|superslm tests: 56519 checks, 52 failures| +FAIL tests/test_attn_rowsites.cpp:585: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: c891cb46692e31ae4745dc83dde29c87f067973ae093795402bbc91b49aadcd8 ove +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 sentinel pass (n 9, d' 1, r 4294967296, s 30): 1 of 9 codes differ from RequantTokenCodeWide, first at 8 (x -1: 90 vs -127) +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 exact-size pass (n 9, d' 1, r 4294967296, s 30): 1 of 9 codes differ from RequantTokenCodeWide, first at 8 (x 1: 90 vs 127) +== s3_tail_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 10788 checks, 69 failures|superslm tests: 36277 checks, 69 failures| +FAIL tests/test_attn_rowsites.cpp:585: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: e956af91b49bab963a3b3cae7b0df0d0f7c6e6deafaf39956ab0eb8301fd1bfe ove +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 sentinel pass (n 5, d' 1, r 4294967296, s 30): 1 of 5 codes differ from RequantTokenCodeWide, first at 4 (x 1: 90 vs 127) +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 exact-size pass (n 5, d' 1, r 4294967296, s 30): 1 of 5 codes differ from RequantTokenCodeWide, first at 4 (x -1: 90 vs -127) +FAIL tests/test_attn_rowsites.cpp:1123: st == SslmForwardStatus::Ok && out == want -- S3 funnel n=5: status Ok, codes differ from the element loop's +== s3_tail_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3): 30988 checks, 52 failures|superslm tests: 56477 checks, 52 failures| +FAIL tests/test_attn_rowsites.cpp:585: hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && values == superslm_test::kAttnRowsiteS1GoldenValues -- 6.3 S1 golden: c891cb46692e31ae4745dc83dde29c87f067973ae093795402bbc91b49aadcd8 ove +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 sentinel pass (n 9, d' 1, r 4294967296, s 30): 1 of 9 codes differ from RequantTokenCodeWide, first at 8 (x -1: 90 vs -127) +FAIL tests/test_attn_rowsites.cpp:966: bad == 0 -- 4.S3 exact-size pass (n 9, d' 1, r 4294967296, s 30): 1 of 9 codes differ from RequantTokenCodeWide, first at 8 (x 1: 90 vs 127) +== wd_clamp_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== wd_clamp_dropped_avx2 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== wd_clamp_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== wd_clamp_dropped_avx512 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== wd_clamp_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== wd_clamp_dropped_avx512 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== wd_h_arith_shift_avx512 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| +== wd_h_arith_shift_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +== wd_h_arith_shift_avx512 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| +ALL-DONE + +# The loop-bound mutants with the attn-rowsites cells run first (scratch-only edit to main, not committed): +== s3_loop_bound_avx2, attn-rowsites cells run first, on superslm_tests_avx2_forced: exit 134 +FAIL tests/test_attn_rowsites.cpp:976: clobbered == 0 -- 4.S3 sentinel pass (n 1, d' 1): 3 of the 32 sentinel bytes around the row changed +ATTN-FIRST 35310 checks 41 failures +*** stack smashing detected ***: terminated +== s3_loop_bound_avx512, attn-rowsites cells run first, on superslm_tests_avx512_forced: exit 134 +FAIL tests/test_attn_rowsites.cpp:976: clobbered == 0 -- 4.S3 sentinel pass (n 1, d' 1): 7 of the 32 sentinel bytes around the row changed +ATTN-FIRST 35310 checks 45 failures +*** stack smashing detected ***: terminated diff --git a/docs/attention-rowsites/s3/mutation-scripts/all_avx512_runs_avx2_body.py b/docs/attention-rowsites/s3/mutation-scripts/all_avx512_runs_avx2_body.py new file mode 100644 index 00000000..45b4ffe4 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/all_avx512_runs_avx2_body.py @@ -0,0 +1,11 @@ +# Dispatch: the AVX-512 kernel runs the AVX2 body. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('i = RequantRowAvx512(x, n, r, s, out);', 'i = RequantRowAvx2(x, n, r, s, out); /* MUTANT */') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx2.py new file mode 100644 index 00000000..47284534 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx2.py @@ -0,0 +1,11 @@ +# S3 AVX2 body only: the carry term (127L + 2^(e-1)) >> 32 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_add_epi64(_mm256_mul_epu32(h, c127), carry)', '_mm256_add_epi64(_mm256_mul_epu32(h, c127), _mm256_setzero_si256()) /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx512.py new file mode 100644 index 00000000..fe3a3f86 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_carry_dropped_avx512.py @@ -0,0 +1,11 @@ +# S3 AVX-512 body only: the carry term (127L + 2^(e-1)) >> 32 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_add_epi64(_mm512_mul_epu32(h, c127), carry)', '_mm512_add_epi64(_mm512_mul_epu32(h, c127), _mm512_setzero_si512()) /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx2.py new file mode 100644 index 00000000..26528d99 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx2.py @@ -0,0 +1,11 @@ +# (extra) S3 AVX2 body only: its requant_row increment deleted. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('superslm_test::g_requant_row_avx2.fetch_add(1, std::memory_order_relaxed);', '/* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx512.py new file mode 100644 index 00000000..64ac4a99 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_counter_deleted_avx512.py @@ -0,0 +1,11 @@ +# (extra) S3 AVX-512 body only: its requant_row increment deleted. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('superslm_test::g_requant_row_avx512.fetch_add(1, std::memory_order_relaxed);', '/* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx2.py new file mode 100644 index 00000000..6cc38998 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx2.py @@ -0,0 +1,11 @@ +# S3 AVX2 body only: vector loop bound n instead of n - lanes + 1 (writes past the row). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('for (; i + 4 <= n; i += 4)', 'for (; i < n; i += 4) /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx512.py new file mode 100644 index 00000000..0d88edb6 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_loop_bound_avx512.py @@ -0,0 +1,11 @@ +# S3 AVX-512 body only: vector loop bound n instead of n - lanes + 1 (writes past the row). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('for (; i + 8 <= n; i += 8)', 'for (; i < n; i += 8) /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_never_entered.py b/docs/attention-rowsites/s3/mutation-scripts/s3_never_entered.py new file mode 100644 index 00000000..c53f1af7 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_never_entered.py @@ -0,0 +1,11 @@ +# S3 call site: the funnel keeps its element loop (the row leaf is never entered). +p='src/forward/checked_chain_funnel.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('RequantRowWide(wide_row, n, preflight.reciprocal, preflight.normalized.s, out_codes);', 'for (size_t i = 0; i < n; ++i) out_codes[i] = RequantTokenCodeWide(wide_row[i], preflight.reciprocal, preflight.normalized.s); /* MUTANT */') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx2.py new file mode 100644 index 00000000..e44758de --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx2.py @@ -0,0 +1,11 @@ +# (extra) S3 AVX2 body only: P formed from r's low half alone (wrong only where r = 2^32). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_add_epi64(_mm256_mul_epu32(ax, r_lo), _mm256_slli_epi64(_mm256_mul_epu32(ax, r_hi), 32))', '_mm256_mul_epu32(ax, r_lo) /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx512.py new file mode 100644 index 00000000..11ded241 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_rhi_dropped_avx512.py @@ -0,0 +1,11 @@ +# (extra) S3 AVX-512 body only: P formed from r's low half alone (wrong only where r = 2^32). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_add_epi64(_mm512_mul_epu32(ax, r_lo), _mm512_slli_epi64(_mm512_mul_epu32(ax, r_hi), 32))', '_mm512_mul_epu32(ax, r_lo) /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx2.py new file mode 100644 index 00000000..f7124b46 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx2.py @@ -0,0 +1,11 @@ +# S3 AVX2 body only: rounding term 2^e instead of 2^(e-1). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('uint64_t{1} << (e - 1)', 'uint64_t{1} << e /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx512.py new file mode 100644 index 00000000..b5b27364 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_round_2e_avx512.py @@ -0,0 +1,11 @@ +# S3 AVX-512 body only: rounding term 2^e instead of 2^(e-1). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('uint64_t{1} << (e - 1)', 'uint64_t{1} << e /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx2.py new file mode 100644 index 00000000..fecd7e13 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx2.py @@ -0,0 +1,11 @@ +# S3 AVX2 body only: final shift e - 31 instead of e - 32. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm_cvtsi32_si128(e - 32)', '_mm_cvtsi32_si128(e - 31) /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx512.py new file mode 100644 index 00000000..34b66526 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_shift_e31_avx512.py @@ -0,0 +1,11 @@ +# S3 AVX-512 body only: final shift e - 31 instead of e - 32. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm_cvtsi32_si128(e - 32)', '_mm_cvtsi32_si128(e - 31) /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/s3_tail_dropped.py b/docs/attention-rowsites/s3/mutation-scripts/s3_tail_dropped.py new file mode 100644 index 00000000..b1ae110c --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/s3_tail_dropped.py @@ -0,0 +1,11 @@ +# (extra) S3 dispatcher: the scalar tail after the lanes is skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('for (; i < n; ++i) out[i] = RequantTokenCodeWide(x[i], r, s); // the tail, or the whole row', 'if (i == 0) for (; i < n; ++i) out[i] = RequantTokenCodeWide(x[i], r, s); /* MUTANT */') +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx2.py b/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx2.py new file mode 100644 index 00000000..7a7308c2 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx2.py @@ -0,0 +1,11 @@ +# (withdrawn, §5.3/§9: run to confirm it is equivalent) S3 AVX2 body only: the clamp at 127 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_min_epu32(_mm256_or_si256(mag, big), c127);', 'mag; /* MUTANT */', start="size_t RequantRowAvx2(", end="SUPERSLM_INTMATH_AVX512_TARGET") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx512.py new file mode 100644 index 00000000..8734371d --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/wd_clamp_dropped_avx512.py @@ -0,0 +1,11 @@ +# (withdrawn, §5.3/§9: run to confirm it is equivalent) S3 AVX-512 body only: the clamp at 127 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_min_epu64(_mm512_srl_epi64(_mm512_add_epi64(_mm512_mul_epu32(h, c127), carry), shift), c127)', '_mm512_srl_epi64(_mm512_add_epi64(_mm512_mul_epu32(h, c127), carry), shift) /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/mutation-scripts/wd_h_arith_shift_avx512.py b/docs/attention-rowsites/s3/mutation-scripts/wd_h_arith_shift_avx512.py new file mode 100644 index 00000000..3ceb50c5 --- /dev/null +++ b/docs/attention-rowsites/s3/mutation-scripts/wd_h_arith_shift_avx512.py @@ -0,0 +1,11 @@ +# (withdrawn, §5.3/§9: run to confirm it is equivalent) S3 AVX-512 body only: H by an arithmetic shift. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_srli_epi64(p, 32);', '_mm512_srai_epi64(p, 32); /* MUTANT */', start="size_t RequantRowAvx512(", end="} // namespace") +open(p,'w').write(s) diff --git a/docs/attention-rowsites/s3/recipe-link.txt b/docs/attention-rowsites/s3/recipe-link.txt new file mode 100644 index 00000000..16c8b3c6 --- /dev/null +++ b/docs/attention-rowsites/s3/recipe-link.txt @@ -0,0 +1,36 @@ +# Plan §3.4 / cell 11.5, GCC 13.3 on the cloud host: the four hand-written MSVC recipes of G21 after S3 (s3/link_recipes.sh in +# the progress file). Box step B0 runs the same recipes once with MSVC. +== without src/matmul.cpp +build_cert.bat: + FAILED cert_intmath: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); +tools/build_inspect.bat line 1: + FAILED sslm_inspect: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); +tools/build_inspect.bat line 2: + FAILED tok_verify: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); +tests/t2296-fp-free-open-red-suite/build_link_red.bat (five cells; each cell includes , so the cell + files themselves do not compile on Linux. What S3 changes is the engine half of the cl line, so each cell's + engine source set is linked as a closed set: g++ -shared -fPIC -Wl,--no-undefined, which fails on any + symbol the set references and does not define. The cells' own references into it (Clz64, AntiLm*) are + unchanged by S3 and were satisfied by intmath.cpp (+ damped_greedy_antilm.cpp) before it): + FAILED dim4_shape_red.engine.so: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); + FAILED dim6_determinism_red.engine.so: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); + FAILED dim7_contract_red.engine.so: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); + FAILED dim11_guard_red.engine.so: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); + FAILED dim7_capacity_red.engine.so: undefined reference to `superslm::detail::ActiveGemmTier();undefined reference to `superslm::detail::DispatchSitesKernel(superslm::detail::GemmTier); +== with src/matmul.cpp +build_cert.bat: + LINKED cert_intmath +tools/build_inspect.bat line 1: + LINKED sslm_inspect +tools/build_inspect.bat line 2: + LINKED tok_verify +tests/t2296-fp-free-open-red-suite/build_link_red.bat (five cells; each cell includes , so the cell + files themselves do not compile on Linux. What S3 changes is the engine half of the cl line, so each cell's + engine source set is linked as a closed set: g++ -shared -fPIC -Wl,--no-undefined, which fails on any + symbol the set references and does not define. The cells' own references into it (Clz64, AntiLm*) are + unchanged by S3 and were satisfied by intmath.cpp (+ damped_greedy_antilm.cpp) before it): + LINKED dim4_shape_red.engine.so + LINKED dim6_determinism_red.engine.so + LINKED dim7_contract_red.engine.so + LINKED dim11_guard_red.engine.so + LINKED dim7_capacity_red.engine.so diff --git a/docs/attention-rowsites/s3/red-sslm_axis_digest.txt b/docs/attention-rowsites/s3/red-sslm_axis_digest.txt new file mode 100644 index 00000000..aa7363c5 --- /dev/null +++ b/docs/attention-rowsites/s3/red-sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/red-sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s3/red-sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..e2b14fbf --- /dev/null +++ b/docs/attention-rowsites/s3/red-sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/red-sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s3/red-sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..024362ef --- /dev/null +++ b/docs/attention-rowsites/s3/red-sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/red-sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s3/red-sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..a2bf679a --- /dev/null +++ b/docs/attention-rowsites/s3/red-sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/red-sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s3/red-sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..4edecdd8 --- /dev/null +++ b/docs/attention-rowsites/s3/red-sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/red-suites.txt b/docs/attention-rowsites/s3/red-suites.txt new file mode 100644 index 00000000..5723f2fe --- /dev/null +++ b/docs/attention-rowsites/s3/red-suites.txt @@ -0,0 +1,81 @@ +# S3 red run (plan §4 red-first): GCC 13.3 Release, the red commit (cells, counters declared, RequantRowWide a +# stub that runs the element loop, the funnel unchanged). Suites run from the repository root with +# SUPERSLM_ATTN_ROWSITES_ARTIFACT= (11.1(d)), each binary with its own TMPDIR. +# Every value assertion passes on every binary; the failures are the path assertions (requant_row never moves). +== superslm_tests (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures +superslm tests: 109744 checks, 10 failures + FAIL lines: 10 +== superslm_tests_sse2_forced (red): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 195 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109686 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures +superslm tests: 109702 checks, 10 failures + FAIL lines: 10 +== superslm_tests_avx512_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 16359 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 10 failures +superslm tests: 109702 checks, 10 failures + FAIL lines: 10 +== sslm_axis_digest: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 + +# The failing assertions on the auto binary (AVX-512 here); forced AVX2 and forced AVX-512 fail the same ten, forced SSE2 none: +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 sentinel pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 sentinel pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1094: path_bad == 0 -- 4.S3 exact-size pass: 16359 of 16359 row-leaf calls moved the requant_row counters wrongly (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 4.S3 exact-size pass: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+16359 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=0: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=896: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- S3 funnel call n=4864: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) prefill requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1536 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:913: ok -- 11.1(d) decode requant_row: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+384 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp (interleaved with the artifact preflight line): S3 funnel call n=5: requant_row_avx2 +0 requant_row_avx512 +0; want +0/+1 (kernel: AVX-512) diff --git a/docs/attention-rowsites/s3/sanitizers.txt b/docs/attention-rowsites/s3/sanitizers.txt new file mode 100644 index 00000000..dff53871 --- /dev/null +++ b/docs/attention-rowsites/s3/sanitizers.txt @@ -0,0 +1,8 @@ +# S3 sanitizer runs at the implementation commit (GCC 13.3, RelWithDebInfo; ASan+UBSan with -fno-sanitize-recover=undefined, and TSan, as the CI legs +# configure them), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. 'reports' counts AddressSanitizer errors, UBSan runtime errors and +# ThreadSanitizer warnings in the run's output. The 4.S3 exact-size pass (rows allocated to exactly n bytes, no sentinel) is the +# cell these legs exist for (plan F10, G22): an over-long vector store lands outside the allocation. +== build-asan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109702 checks, 0 failures| reports: 0 +== build-tsan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures|superslm tests: 109744 checks, 0 failures| reports: 0 diff --git a/docs/attention-rowsites/s3/sslm_axis_digest.txt b/docs/attention-rowsites/s3/sslm_axis_digest.txt new file mode 100644 index 00000000..aa7363c5 --- /dev/null +++ b/docs/attention-rowsites/s3/sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s3/sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..e2b14fbf --- /dev/null +++ b/docs/attention-rowsites/s3/sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s3/sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..024362ef --- /dev/null +++ b/docs/attention-rowsites/s3/sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s3/sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..a2bf679a --- /dev/null +++ b/docs/attention-rowsites/s3/sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s3/sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..4edecdd8 --- /dev/null +++ b/docs/attention-rowsites/s3/sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s3/suites-clang.txt b/docs/attention-rowsites/s3/suites-clang.txt new file mode 100644 index 00000000..4bea46dc --- /dev/null +++ b/docs/attention-rowsites/s3/suites-clang.txt @@ -0,0 +1,77 @@ +# The same runs on Clang 18.1 (Release). +== superslm_tests (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109744 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 195 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109686 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109702 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109702 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +DONE-clang diff --git a/docs/attention-rowsites/s3/suites.txt b/docs/attention-rowsites/s3/suites.txt new file mode 100644 index 00000000..ef143d09 --- /dev/null +++ b/docs/attention-rowsites/s3/suites.txt @@ -0,0 +1,77 @@ +# S3 suites and digests at the implementation commit, GCC 13.3 Release, this host (AVX-512BW, so auto dispatches AVX-512). The four suite binaries run from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1 and separate TMPDIRs; then the five digest legs. Paths shortened. +== superslm_tests (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109744 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 195 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109686 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109702 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +tiled GEMM cells (plan slice 1): 211 checks, 0 failures +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites 11.1(d): prefill and decode windows driven on p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3): 84213 checks, 0 failures +superslm tests: 109702 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL 4892b5be694c0501899be342cf0cf9386f8db7d0fda65f1323ee8b3815903902 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed values=30100 +DONE-gcc diff --git a/docs/attention-rowsites/s4/bench-softmax.txt b/docs/attention-rowsites/s4/bench-softmax.txt new file mode 100644 index 00000000..fcbd4be3 --- /dev/null +++ b/docs/attention-rowsites/s4/bench-softmax.txt @@ -0,0 +1,28 @@ +# sslm_sites_bench softmax --repeat=30, 9 interleaved rounds (round, build :: output). base = S3 head (v1.9.0 softmax body); cand = S4 auto (AVX-512 body); cand_avx2 = S4 linked against libsuperslm_avx2_forced.a (AVX2 body). +1 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0261 w=128 0.9204 w=301 2.2370 w=512 3.7889 w=601 4.4465 w=1024 7.5642 softmax prefill T=128: 0.1598 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6497 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.3219 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7460 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.4876 ms/token at 24 layers x 14 heads softmax refused rows: 0 +1 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0916 w=128 0.2912 w=301 0.7221 w=512 0.8946 w=601 1.1663 w=1024 2.4957 softmax prefill T=128: 0.0657 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1681 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.4352 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.1945 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3768 ms/token at 24 layers x 14 heads softmax refused rows: 0 +1 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0626 w=128 0.3157 w=301 0.6163 w=512 1.0176 w=601 1.1878 w=1024 2.0853 softmax prefill T=128: 0.0637 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.2278 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.4628 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2567 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4336 ms/token at 24 layers x 14 heads softmax refused rows: 0 +2 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0256 w=128 0.9205 w=301 2.2487 w=512 3.8045 w=601 4.2354 w=1024 7.0764 softmax prefill T=128: 0.1511 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6117 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.2604 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7514 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.4160 ms/token at 24 layers x 14 heads softmax refused rows: 0 +2 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0753 w=128 0.2141 w=301 0.5326 w=512 0.8440 w=601 1.1199 w=1024 1.8218 softmax prefill T=128: 0.0598 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.2206 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.4547 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2491 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3911 ms/token at 24 layers x 14 heads softmax refused rows: 0 +2 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0578 w=128 0.2763 w=301 0.6527 w=512 1.0638 w=601 1.2585 w=1024 2.1098 softmax prefill T=128: 0.0590 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1866 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3489 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2194 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4226 ms/token at 24 layers x 14 heads softmax refused rows: 0 +3 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0269 w=128 0.9206 w=301 2.2401 w=512 3.7942 w=601 4.4536 w=1024 7.6890 softmax prefill T=128: 0.1598 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6605 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.3260 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7514 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.4913 ms/token at 24 layers x 14 heads softmax refused rows: 0 +3 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0853 w=128 0.2374 w=301 0.5942 w=512 0.9164 w=601 1.1367 w=1024 1.8825 softmax prefill T=128: 0.0595 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1777 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3613 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2002 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3840 ms/token at 24 layers x 14 heads softmax refused rows: 0 +3 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0587 w=128 0.2765 w=301 0.6534 w=512 1.0614 w=601 1.2597 w=1024 2.1329 softmax prefill T=128: 0.0582 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1867 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3894 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2195 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4234 ms/token at 24 layers x 14 heads softmax refused rows: 0 +4 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0258 w=128 0.9215 w=301 2.2386 w=512 3.7849 w=601 4.4492 w=1024 7.0641 softmax prefill T=128: 0.1466 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.5960 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.1784 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.6773 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.3371 ms/token at 24 layers x 14 heads softmax refused rows: 0 +4 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0827 w=128 0.2343 w=301 0.5835 w=512 0.9004 w=601 1.1121 w=1024 1.8097 softmax prefill T=128: 0.0581 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1701 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3572 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2308 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3899 ms/token at 24 layers x 14 heads softmax refused rows: 0 +4 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0587 w=128 0.2765 w=301 0.6534 w=512 1.0613 w=601 1.2583 w=1024 2.1123 softmax prefill T=128: 0.0584 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1864 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3797 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2198 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4235 ms/token at 24 layers x 14 heads softmax refused rows: 0 +5 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0261 w=128 0.9204 w=301 2.2377 w=512 3.8199 w=601 4.4727 w=1024 7.3534 softmax prefill T=128: 0.1547 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6157 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.1752 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.6840 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.3778 ms/token at 24 layers x 14 heads softmax refused rows: 0 +5 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0826 w=128 0.2297 w=301 0.5749 w=512 0.8865 w=601 1.1093 w=1024 1.7954 softmax prefill T=128: 0.0577 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1674 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3485 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.1969 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3812 ms/token at 24 layers x 14 heads softmax refused rows: 0 +5 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0561 w=128 0.2647 w=301 0.6156 w=512 1.0015 w=601 1.2600 w=1024 2.1063 softmax prefill T=128: 0.0582 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1915 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3902 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2194 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4232 ms/token at 24 layers x 14 heads softmax refused rows: 0 +6 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0261 w=128 0.9207 w=301 2.2348 w=512 3.7977 w=601 4.4579 w=1024 7.5536 softmax prefill T=128: 0.1597 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6510 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.3074 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7473 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.4972 ms/token at 24 layers x 14 heads softmax refused rows: 0 +6 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0855 w=128 0.2424 w=301 0.5789 w=512 0.8910 w=601 1.1021 w=1024 1.8106 softmax prefill T=128: 0.0595 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1749 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3726 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2030 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3813 ms/token at 24 layers x 14 heads softmax refused rows: 0 +6 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0578 w=128 0.2762 w=301 0.6528 w=512 1.0618 w=601 1.2584 w=1024 2.1229 softmax prefill T=128: 0.0582 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1865 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3801 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2196 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4233 ms/token at 24 layers x 14 heads softmax refused rows: 0 +7 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0264 w=128 0.9204 w=301 2.2303 w=512 3.7535 w=601 4.4399 w=1024 7.1724 softmax prefill T=128: 0.1575 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6575 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.6586 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7537 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.4994 ms/token at 24 layers x 14 heads softmax refused rows: 0 +7 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0860 w=128 0.2368 w=301 0.6060 w=512 0.9212 w=601 1.1414 w=1024 1.8559 softmax prefill T=128: 0.0595 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1724 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3642 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2029 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4035 ms/token at 24 layers x 14 heads softmax refused rows: 0 +7 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0587 w=128 0.2770 w=301 0.6538 w=512 1.0693 w=601 1.2993 w=1024 2.1134 softmax prefill T=128: 0.0582 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1875 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3970 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2194 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4228 ms/token at 24 layers x 14 heads softmax refused rows: 0 +8 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0267 w=128 0.9226 w=301 2.2849 w=512 3.8535 w=601 4.4448 w=1024 7.6105 softmax prefill T=128: 0.1603 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.6582 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.2722 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.7039 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.5019 ms/token at 24 layers x 14 heads softmax refused rows: 0 +8 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0992 w=128 0.2820 w=301 0.7708 w=512 1.1890 w=601 1.5038 w=1024 2.3881 softmax prefill T=128: 0.0719 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.2219 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3628 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2015 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4696 ms/token at 24 layers x 14 heads softmax refused rows: 0 +8 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0648 w=128 0.3010 w=301 0.7455 w=512 1.2571 w=601 1.5030 w=1024 2.5511 softmax prefill T=128: 0.0749 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.1942 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3906 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2194 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4248 ms/token at 24 layers x 14 heads softmax refused rows: 0 +9 base :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0280 w=128 0.9225 w=301 2.2816 w=512 3.8158 w=601 4.4661 w=1024 7.6076 softmax prefill T=128: 0.1843 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.7927 ms/token at 24 layers x 14 heads softmax prefill T=1024: 1.6171 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.8787 ms/token at 24 layers x 14 heads softmax decode ctx=600: 1.7701 ms/token at 24 layers x 14 heads softmax refused rows: 0 +9 cand :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0874 w=128 0.2429 w=301 0.5954 w=512 0.9111 w=601 1.1902 w=1024 1.9291 softmax prefill T=128: 0.0645 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.2098 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3630 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2023 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.3760 ms/token at 24 layers x 14 heads softmax refused rows: 0 +9 cand_avx2 :: softmax constants: 16 triples, q_ln2 381..914 softmax best-of-30 us/call: w=1 0.0587 w=128 0.2726 w=301 0.7959 w=512 1.3379 w=601 1.2600 w=1024 2.1537 softmax prefill T=128: 0.0762 ms/token at 24 layers x 14 heads softmax prefill T=512: 0.2369 ms/token at 24 layers x 14 heads softmax prefill T=1024: 0.3975 ms/token at 24 layers x 14 heads softmax decode ctx=300: 0.2193 ms/token at 24 layers x 14 heads softmax decode ctx=600: 0.4248 ms/token at 24 layers x 14 heads softmax refused rows: 0 diff --git a/docs/attention-rowsites/s4/bench.md b/docs/attention-rowsites/s4/bench.md new file mode 100644 index 00000000..6ba2a0e2 --- /dev/null +++ b/docs/attention-rowsites/s4/bench.md @@ -0,0 +1,54 @@ +# S4 bench: what the guarded softmax saves + +These are reports, not gates (plan §8 10.1). The host is the shared 4-vCPU cloud Xeon (AVX2, AVX-512BW) with GCC 13.3 -O3. +`tools/sslm_sites_bench.cpp` gains a `softmax` mode. One object was linked three ways: + +- against the base library, the S3 head, which runs v1.9.0's softmax body; +- against S4's library with auto dispatch, which runs the AVX-512 body here; +- against S4's `libsuperslm_avx2_forced.a`, which runs the AVX2 body. That library also carries the test counters (one relaxed + atomic increment per row), as S3's AVX2 reading did. + +Each reading is a best-of-30 inside one process. The builds ran alternately for 9 rounds. The saving is the median of the paired +differences; the range is the minimum and maximum over the rounds. The raw output is in `bench-softmax.txt`. + +## Method: the softmax call, at the forward's widths and constants + +`softmax` mode times `SoftmaxRowQ15` on rows with the forward's own constants: `IExpScaleConstants` with the format-30 +coefficients the forward passes, kept when q_ln2 falls in [347, 944]. That is the range cell 11.1(d)'s data term measured on the +0.5B-width synthetic (`softmax-data-terms.txt`). The scores are spread over about 16·q_ln2. Every row is inside the guard, as all +2,240 rows in 11.1(d)'s windows are. The per-token figure is §6's: 24 layers × 14 query heads, one call per head per token. +Prefill of T tokens sums the calls at widths 1..T and divides by T. Decode at context C is one call at width C + 1. + +| Reading | Base (v1.9.0) | S4 AVX-512 | Saved (range) | × | S4 AVX2 | Saved (range) | × | Plan §0 (AVX2) | +|---|---|---|---|---|---|---|---|---| +| µs per call, width 1 | 0.026 | 0.086 | −0.059 | 0.31 | 0.059 | −0.032 | 0.44 | | +| µs per call, width 128 | 0.921 | 0.237 | 0.683 (0.63–0.71) | 3.9 | 0.277 | 0.644 (0.60–0.66) | 3.3 | | +| µs per call, width 512 | 3.798 | 0.900 | 2.894 (2.66–2.96) | 4.2 | 1.062 | 2.733 (2.48–2.82) | 3.6 | | +| µs per call, width 1,024 | 7.554 | 1.856 | 5.317 (5.07–5.81) | 4.1 | 2.113 | 5.247 (4.95–5.56) | 3.6 | | +| **Prefill T = 128**, ms/token | 0.160 | 0.060 | **0.097** (0.088–0.120) | 2.7 | 0.058 | **0.097** (0.085–0.108) | 2.7 | **0.15** | +| **Prefill T = 512** | 0.651 | 0.175 | **0.476** (0.39–0.58) | 3.7 | 0.188 | **0.464** (0.41–0.56) | 3.5 | **0.54** | +| **Prefill T = 1,024** | 1.307 | 0.363 | **0.909** (0.81–1.29) | 3.6 | 0.390 | **0.912** (0.79–1.26) | 3.4 | **1.15** | +| **Decode, context 300**, ms/token | 0.747 | 0.202 | **0.544** (0.45–0.68) | 3.7 | 0.219 | **0.528** (0.46–0.66) | 3.4 | **0.53** | +| Decode, context 600 | 1.491 | 0.384 | 1.096 (0.95–1.39) | 3.9 | 0.423 | 1.068 (0.91–1.35) | 3.5 | 1.07 (§6) | + +**Against the estimate.** Decode matches: 0.53 ms/token on AVX2 at context 300, as estimated, and 1.07 at 600, as §6 estimates. +Prefill falls short: 0.10 / 0.46 / 0.91 against 0.15 / 0.54 / 1.15, that is 65%, 86% and 79%. + +- **§0's prefill and decode figures do not use the same per-element saving.** Per token at full depth, prefill at T touches + 336·(T + 1)/2 elements and decode at context 300 touches 336·301. The estimates then imply 6.9 ns saved per element at T = 128, + 6.3 at 512 and 6.7 at 1,024, but 5.2 in decode. Measured here, the AVX2 body saves about 5.1–5.3 ns per element on long rows + (base 7.4 ns, S4 2.1 ns). That agrees with the decode estimate and sits below the prefill ones. The spike's 7.2–9.0 → 2.7 ns + per element gives 4.5–6.3, so the prefill figures look taken from the top of that range or above it. +- **Short rows pay a fixed cost per row.** It is the guard's pass over the scores and two 64-bit divides (the z reciprocal and + the p reciprocal). A width-1 row is 0.03 µs slower than v1.9.0 on AVX2 and 0.06 µs on AVX-512. Prefill at T = 128 averages + width 64, so the fixed cost takes a larger share there. This is why T = 128 shows the largest shortfall. The total cost of the + short-row slowdown is included in the prefill figures above; no prefill length measured here is slower than the base. +- **AVX-512 is barely ahead of AVX2** (at most 10–15% per call; equal in prefill). The per-row work outside the vector loops + (the guard's scalar max pass and the two divides) does not get wider. + +No whole-forward reading is reported for S4. On the reduced-layer artifacts the per-layer saving is about 0.02 ms per token (0.46 +/ 24). That is below the forward bench's run-to-run spread on this host (S3's `bench.md` method 2: quartile widths of 0.03–0.3 +ms). + +These are engine figures on synthetic constants from the forward's measured range. They say nothing about any consumer's +end-to-end speed, and they were not measured on the project's reference hardware (the box's B2, §8 10.2, is box-only). diff --git a/docs/attention-rowsites/s4/blob-protocol.txt b/docs/attention-rowsites/s4/blob-protocol.txt new file mode 100644 index 00000000..88f86117 --- /dev/null +++ b/docs/attention-rowsites/s4/blob-protocol.txt @@ -0,0 +1,54 @@ +# S4 save-blob protocol (plan 6.4, as S1-S3 ran it): the sslm_bench_prefill tool (tools/t2147_chunk_batched_pins.cpp) as built by +# CMake (GCC 13.3, Release). Base = the S1 head's build (v1.9.0 softmax, prob-V and funnel loop), unchanged from S2's and S3's runs. +# Candidates = S4's implementation commit: pins_s4 (auto dispatch, the AVX-512 softmax body here) and pins_s4_avx2 +# (sslm_bench_prefill_avx2_forced, the AVX2 body). Each row prefills the prompt at the chunk budget, decodes 32 greedy tokens and +# dumps the SSB5 blob; EQUAL = blobs byte-equal (cmp) and decoded tokens equal. Artifacts: the S1 synthetic set (wide_l8, p05_l1 +# sha256 f0fd4886...6ed3, p05_l2) and the in-tree 8-layer fixture. +# +# Result: 46 of 46 rows EQUAL (32 auto, 14 forced AVX2); every row's blob hash and token hash equals S3's record row for row. +wide_l8 ids:8 chunk_budget=1 +32 decode, candidate pins_s4: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode, candidate pins_s4: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode, candidate pins_s4: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base 5f8a4bafc1a4789a cand 5f8a4bafc1a4789a; decoded tokens base 391ba224 cand 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base 69d606622b873076 cand 69d606622b873076; decoded tokens base 1dec854e cand 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate pins_s4: blob base cdcbfbe1fdcc65b1 cand cdcbfbe1fdcc65b1; decoded tokens base 6894a696 cand 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=1 +32 decode, candidate pins_s4: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode, candidate pins_s4: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode, candidate pins_s4: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate pins_s4: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate pins_s4: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate pins_s4: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate pins_s4: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate pins_s4: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate pins_s4: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=1 +32 decode, candidate pins_s4: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode, candidate pins_s4: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode, candidate pins_s4: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate pins_s4: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate pins_s4: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate pins_s4: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate pins_s4: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate pins_s4: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate pins_s4: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +intree ids:24:128 chunk_budget=1 +32 decode, candidate pins_s4: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=8 +32 decode, candidate pins_s4: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=24 +32 decode, candidate pins_s4: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:8:128 chunk_budget=1 +32 decode, candidate pins_s4: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +intree ids:8:128 chunk_budget=8 +32 decode, candidate pins_s4: blob base 02718152a6f8941d cand 02718152a6f8941d; decoded tokens base 9ab82cad cand 9ab82cad; EQUAL (16576 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate pins_s4_avx2: blob base 48efd7f2de021c02 cand 48efd7f2de021c02; decoded tokens base 1f43a9c7 cand 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate pins_s4_avx2: blob base 7dcb7a5fb76a6276 cand 7dcb7a5fb76a6276; decoded tokens base a0edc532 cand a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate pins_s4_avx2: blob base 4722fb9a0f133c70 cand 4722fb9a0f133c70; decoded tokens base 681400be cand 681400be; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate pins_s4_avx2: blob base 9a18bb27908136eb cand 9a18bb27908136eb; decoded tokens base 8435904c cand 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate pins_s4_avx2: blob base 77d1a3589a0c813f cand 77d1a3589a0c813f; decoded tokens base 75387074 cand 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate pins_s4_avx2: blob base 904ffb07ded91feb cand 904ffb07ded91feb; decoded tokens base e7799ea5 cand e7799ea5; EQUAL (4194720 bytes) +intree ids:24:128 chunk_budget=1 +32 decode, candidate pins_s4_avx2: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) +intree ids:24:128 chunk_budget=8 +32 decode, candidate pins_s4_avx2: blob base 4dd83f12817436f7 cand 4dd83f12817436f7; decoded tokens base 5707bd0c cand 5707bd0c; EQUAL (16576 bytes) diff --git a/docs/attention-rowsites/s4/coverage.txt b/docs/attention-rowsites/s4/coverage.txt new file mode 100644 index 00000000..8d7422c3 --- /dev/null +++ b/docs/attention-rowsites/s4/coverage.txt @@ -0,0 +1,35 @@ +# S4 branch coverage (plan §3.4, cell 11.6). INDICATIVE ONLY, as S2's and S3's: a floor is pinned only from the hosted +# branch-coverage leg's own recorded measurement, and this series is delivered as patches, not pushed, so that leg has not run on +# it. These are local replicas of the leg's commands (clang-18, llvm-cov/llvm-profdata 18, RelWithDebInfo, -fprofile-instr-generate +# -fcoverage-mapping; the five binaries, merged, exported over src/*.cpp include/superslm/*.h), on this host (AVX-512BW), +# SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. All five binaries exit 0. +# +# Base = the S3 head (its numbers are S3's candidate numbers, docs/attention-rowsites/s3/coverage.txt); candidate = S4's +# implementation commit. Branches covered / total. +# +# (1) All five profiles (the leg's union), this host (the auto and forced AVX-512 binaries run AVX-512): +# base src/intmath.cpp 167/188 88.83% src/matmul.cpp 286/312 91.67% +# candidate src/intmath.cpp 221/242 91.32% src/matmul.cpp 286/312 91.67% +# check_branch_coverage_floors.py: OK (19 files at or above their pinned floor). S4 adds 54 branches to intmath.cpp and covers +# all 54; the 21 uncovered branch sides are the base's, moved by S4's insertions. +# +# (2) Approximating the hosted runner, which has no AVX-512: only the sse2-forced, avx2-forced and scalar-forced-digest profiles +# merged (there, the auto binary dispatches AVX2 and the forced AVX-512 binary faults at its first AVX-512 instruction): +# base src/intmath.cpp 163/188 86.70% src/matmul.cpp 203/286 70.98% +# candidate src/intmath.cpp 202/242 83.47% src/matmul.cpp 203/286 70.98% +# check_branch_coverage_floors.py on the candidate's projection: FAILED, src/intmath.cpp 83.47% < floor 87.93% (S3 already +# projected 86.70%) and src/matmul.cpp 70.98% < floor 72.22% (S2's, unchanged). +# So on a runner without AVX-512, S4 takes intmath.cpp a further 3.2 points below its floor. The 15 branch sides it loses there +# are all AVX-512-only: SoftmaxRowAvx512's loops, tail tests and tail copies (lines 1204, 1212, 1214, 1218, 1231, 1232, 1236) +# and the dispatcher's kAvx2-or-kAvx512 choices (1268, 1275). +# +# Uncovered S4 branches (lines of src/intmath.cpp at the implementation commit): +# With all five profiles: none. +# Without AVX-512 profiles: the nine lines above. +# No reachable S4 branch is uncovered, so no cell is added (plan §3.4 step 1). The z estimate's downward correction is +# branch-free in both bodies, so it adds no branch. +# +# Plan §3.4 steps 2 and 3: allowlist notes for the above are added in tools/ci/branch_coverage_allowlist.txt, in the file's existing +# reviewer-note form (its checker's pytest: 6 passed). The floors in tools/ci/branch_coverage_floors.json are NOT re-pinned. If the +# hosted runner lacks AVX-512, the leg fails on intmath.cpp (as it already would from S3, and on matmul.cpp from S2) until the +# floor is lowered; that is the owner's one-line call at code review (G27). diff --git a/docs/attention-rowsites/s4/fp-scan.txt b/docs/attention-rowsites/s4/fp-scan.txt new file mode 100644 index 00000000..3b701fd6 --- /dev/null +++ b/docs/attention-rowsites/s4/fp-scan.txt @@ -0,0 +1,90 @@ +# S4 fp-free scan (plan §3.4; tests/ci/scan_build_output.py --build-dir --target superslm --isa x86-64), allow-lists unchanged. +# +# Clang 18.1 build at the implementation commit: PASS (490 symbols ACCEPT, 0 REJECT). intmath.cpp.o is clean +# (SoftmaxRowAvx2 and SoftmaxRowAvx512 are local 't' symbols in it; the helpers inline). +# GCC 13.3 build at the implementation commit: FAIL, one symbol, TiledGemmAvx512 in matmul.cpp.o, inherited from the base +# (as in S3; main fixes it in 90e48de). intmath.cpp.o is clean. With 90e48de's src/matmul.cpp change applied temporarily +# (not part of this series): PASS (561 ACCEPT, 0 REJECT). +# +# The rejects met on the way to a clean intmath.cpp.o, without widening the allow-list: +# 1. GCC, SoftmaxRowAvx512: the tail pads were first written as per-lane `i < rest ? src[i] : pad` loops; GCC vectorised +# them with an AVX-512 mask-register compare (vpcmpnleuq into k1), which the scan rejects. Fix: the pad buffer is +# initialised with the pad value and the live elements are memcpy'd over it. +# 2. Clang, by construction (S3's lesson): no 64-bit select anywhere. The AVX2 masks from vpcmpgtq are folded by +# add/sub/and; the clip is vpminud on the low dword with the high dword folded into bit 31. No vblendvpd, no vxorpd. +# +# Summaries follow (Clang, GCC, GCC with 90e48de applied); the per-object lines are verbatim, the non-gating check-(C) +# listings are elided. + +######## scan-clang +Scanning 17 object member(s) of archive build-clang/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (30 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (52 symbol(s), format=elf) + clean model.cpp.o (77 symbol(s), format=elf) + clean intmath.cpp.o (29 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (23 symbol(s), format=elf) + clean proof_manifest.cpp.o (21 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (15 symbol(s), format=elf) + clean forward_sites.cpp.o (49 symbol(s), format=elf) + clean decode_digest.cpp.o (3 symbol(s), format=elf) + clean sslm_abi.cpp.o (125 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (26 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (6 symbol(s), format=elf) + +Totals: 17 object(s); 490 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 284 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-gcc +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + REJECT matmul.cpp.o (1 symbol(s), format=elf) + _ZN8superslm12_GLOBAL__N_115TiledGemmAvx512EPKsmPKammmmmPlPa + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (58 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) + +Totals: 17 object(s); 560 symbol(s) ACCEPT, 1 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 375 symbol(s) reject under check (C) alone (non-gating diagnostic) +FAIL: the scan did not come back clean. + +######## scan-gcc-90e48de +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (24 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (58 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) + +Totals: 17 object(s); 561 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 376 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). diff --git a/docs/attention-rowsites/s4/golden.txt b/docs/attention-rowsites/s4/golden.txt new file mode 100644 index 00000000..4a5eb8e0 --- /dev/null +++ b/docs/attention-rowsites/s4/golden.txt @@ -0,0 +1,12 @@ +# S4 golden pin provenance (plan §3.3 evidence 3, cell 6.3) +# tools/gen_attn_rowsite_golden.cpp (one hash per slice) compiled with GCC 13.3 -O2 against the v1.9.0 tag (d870d27) include/ +# and its Release libsuperslm.a (the recipe in the generator's header): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +S2 prob-V golden: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed over 30100 values +S3 requant-row golden: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 over 3567018 values +S4 softmax golden: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 over 268078 values +# Re-running the v1.9.0 build with the header path as argument wrote tests/attn_rowsite_golden_pin.h; S1's, S2's and S3's +# hashes and value counts are unchanged, and S4's constants are added beside them. The S4 set drives SoftmaxRowQ15 (a v1.9.0 +# signature) over tests/support/attention_cases.h's RunSoftmaxCases: 3,123 calls (per call its width, bool and output row). +# The set's rows are chosen by the header's own test-side guard copy and estimate replica (integer-only), so building the +# set needs nothing from the library under test; the generator's build against v1.9.0 is the proof. diff --git a/docs/attention-rowsites/s4/linkage-plant.txt b/docs/attention-rowsites/s4/linkage-plant.txt new file mode 100644 index 00000000..b7670566 --- /dev/null +++ b/docs/attention-rowsites/s4/linkage-plant.txt @@ -0,0 +1,23 @@ +# S4 cell 11.3 vitality: the linkage checker over all six objects (auto and both forced builds' matmul.cpp.o and intmath.cpp.o). +# A first plant (SoftmaxRowAvx2 moved out of the anonymous namespace) stays local: its parameter type SoftmaxFastRow is +# in the anonymous namespace, so the function has internal linkage anyway, and the checker correctly reports OK. The plant +# below is an external target-attributed function in the population (SoftmaxExpAvx2Planted). +## plant +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o: 3 population symbols, record x1 +build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o: 5 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 5 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +check_tiled_matmul_linkage: FAIL + build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 'superslm::SoftmaxExpAvx2Planted(long long __vector(4))' is not local (nm type 'T') +exit 1 +## restored (the implementation commit) +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o: 3 population symbols, record x1 +build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o: 5 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +check_tiled_matmul_linkage: OK +exit 0 diff --git a/docs/attention-rowsites/s4/mutants.txt b/docs/attention-rowsites/s4/mutants.txt new file mode 100644 index 00000000..85caffb0 --- /dev/null +++ b/docs/attention-rowsites/s4/mutants.txt @@ -0,0 +1,383 @@ +# S4 mutation evidence (plan §9; GCC 13.3 Release; this host has AVX-512BW, so the auto binary dispatches AVX-512), run at the +# implementation commit. Each mutant is applied to a synced copy of that commit by its script (mutation-scripts/.py, run from +# the repository root). superslm_tests (auto), superslm_tests_avx2_forced and superslm_tests_avx512_forced are then rebuilt and run +# from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT set (p05_l1, sha256 f0fd4886...6ed3), the three at once, each with +# its own TMPDIR. 'none' is the unmutated control. The raw log follows; it shows at most four distinct failing lines per run. +# +# The correction mutants are applied to each tier's body separately (§9 preamble): each AVX2-body mutant must die on forced AVX2 +# (and on auto on a runner without AVX-512) and survives here on the two AVX-512 binaries, which never run that body; each +# AVX-512-body mutant must die on forced AVX-512 and on auto and survives on forced AVX2. Guard, dispatcher and counter mutants sit in +# code both tiers share and die on all three. The z downward correction owes no mutant (§5.4 step 3; with the integer estimate it +# still cannot fire, see the progress file's deviation 1). +# +# Summary (K n = killed, n failures in the attn-rowsites cells; K trap = killed by SIGFPE, located below; s = survives by +# construction, as above): +# mutant auto(AVX-512) forced AVX2 forced AVX-512 §9 row / killing cell +# none s s s +# all_avx512_runs_avx2_body K 2813 s K 2813 +# all_fallback_increment_deleted K 312 K 312 K 312 +# all_fast_increment_before_guard K 312 K 312 K 312 +# s4_always_fallback K 2813 K 2813 K 2813 +# s4_m_ge1_dropped K trap K trap K trap +# s4_m_ge2 K 3 K 3 K 3 +# s4_m_le_dropped K 243 K 243 K 243 +# s4_m_lt K 2 K 2 K 2 +# s4_pdown_skipped_avx2 s K 94 s +# s4_pdown_skipped_avx512 K 94 s K 94 +# s4_pup_skipped_avx2 s K 393 s +# s4_pup_skipped_avx512 K 393 s K 393 +# s4_qc_ge0_dropped K trap K trap K trap +# s4_qc_ge1 K 29 K 29 K 29 +# s4_qln2_ge1_dropped K trap K trap K trap +# s4_qln2_ge2 K 186 K 186 K 186 +# s4_ratio_dropped K 3 K 3 K 3 +# s4_ratio_le2qb K 33 K 33 K 33 +# s4_score_dropped K 2 K 2 K 2 +# s4_score_lt K 1 K 1 K 1 +# s4_true_outside_guard K 313 K 313 K 313 +# s4_width_2p15 K 1 K 1 K 1 +# s4_width_lt K 4 K 4 K 4 +# s4_zup_skipped_avx2 s K 307 s +# s4_zup_skipped_avx512 K 307 s K 307 +# x_clip_dropped_avx2 s K 10 s +# x_clip_dropped_avx512 K 10 s K 10 +# x_clip_high_dword_avx2 s K 2 s +# x_tail_total_dropped_avx2 s K trap s +# x_tail_total_dropped_avx512 K trap s K trap +# +# Which cell kills each, against §9's column (from the raw log): +# q_ln2 >= 1 dropped: SIGFPE (q_ln2 = 0 reaches MakeSoftmaxFastRow's 2^kz / q_ln2), in TestS4Grid, the 4.S4 grid's outside-guard +# constants. §9 allows "a trap". q_c >= 0 dropped and M >= 1 dropped: SIGFPE in the body's reciprocal (a row total of 0), also in +# the 4.S4 grid, which runs before the 2.S4 rows §9 names. Both are stronger than §9's "bool and counter"; the 2.S4 rows are +# not reached in the suite's order. +# q_ln2 >= 2 (186), q_c >= 1 (29), M >= 2 (3: 4.S4 "inside: M = 1", "total = 1"), M < 2^47 (2: "M = 2^47, width 2^14"), +# ratio <= 2 q_b (33), width < 2^14 (4: "grid, width 2^14"), score < 2^61 (1: "scores of exactly +-2^61"): counter, fast expected. +# M <= 2^47 dropped (243), ratio dropped (3: 2.S4 "q_ln2 = 2 q_b + 2", bool 1 against v1.9.0's 0, probabilities differ, and the +# golden): bool, output and counter. The ratio mutant also turns test_main's pre-existing well-formedness cell red. +# width <= 2^15 (1: 2.S4 "width 2^14 + 1") and score conjunct dropped (1: 2.S4 "one score 2^61 + 1"): output-equivalent, as §9 +# says; counter, fallback expected. +# z up, p up, p down skipped, per body: output (4.S4's correction rows, the grid and the golden). Counted by label on the forced +# AVX2 binary (a rerun of the three AVX2-body mutants): z up fails all 24 z-up rows, 280 grid rows, 2 inside corners and the +# golden; p up all 24 p-up rows, 367 grid rows and the golden; p down all 48 steered rows (24 per constant shape), 44 grid +# rows, 1 inside corner and the golden. So the p downward correction also fires on ordinary grid rows under the integer +# estimate; under the plan's double it fired on about 2^-38 of reachable rows (§5.4 step 4). +# return true outside the guard (313): bool (2.S4, every bool row, the off-ratio witness) plus four pre-existing test_main cells. +# always fall back (2,813), fallback increment deleted (312), fast increment before the guard (312), AVX-512 runs the AVX2 body +# (2,813, on the two binaries that select AVX-512): counter. For always-fall-back, §9's named cell was confirmed by a rerun +# on forced AVX2: "11.1(d) prefill softmax: softmax_fast_avx2 +0 softmax_fallback_avx2 +1792 ... want +1792/+0" and +# "11.1(d) decode softmax: ... +0 ... +448 ...; want +448/+0". +# Extras (not in §9): the clip dropped (10 per body), the AVX2 clip's high-dword fold dropped (2), the tail's e left out of +# the total (SIGFPE on a width-1..3 or 1..7 row whose tail is the whole row, in test_main's shipped-primitives oracle). +# +# Harness note. In the parallel run, s4_pup_skipped_avx2 showed 2 failures on the auto and forced AVX-512 binaries, both in +# test_main's adapter end-to-end cell (tests/test_main.cpp:24099), which writes a fixed cwd-relative file +# (t2102_sslm_generate_e2e_fixture.sslm.tmp): three binaries sharing one working directory collided on it. Neither binary runs the +# AVX2 body. Rerun alone, both are green (mutants-traps below). The summary above counts only the attn-rowsites cells. +# +# Trap locations (gdb batch backtraces on the forced binary) and the serial rerun: +# == s4_m_ge1_dropped on superslm_tests_avx2_forced (gdb): +# #0 0x000055555569b7c3 in superslm::(anonymous namespace)::SoftmaxRowAvx2(long const*, unsigned long, superslm::(anonymous namespace)::SoftmaxFastRow const&, long*) () +# #1 0x000055555569d541 in superslm::SoftmaxRowQ15(long const*, unsigned long, long, long, long, long*) () +# #2 0x000055555566cdea in (anonymous namespace)::TestS4Grid()::{lambda(superslm_attention_cases::SmCase const&)#1}::operator()(superslm_attention_cases::SmCase const&) const () +# #3 0x000055555566f17c in (anonymous namespace)::TestS4Grid() () +# #4 0x00005555556755ff in RunAttnRowsiteCells(int&, int&) () +# #5 0x000055555557bf17 in main () +# == s4_qc_ge0_dropped on superslm_tests_avx2_forced (gdb): +# #0 0x000055555569b7c3 in superslm::(anonymous namespace)::SoftmaxRowAvx2(long const*, unsigned long, superslm::(anonymous namespace)::SoftmaxFastRow const&, long*) () +# #1 0x000055555569d54d in superslm::SoftmaxRowQ15(long const*, unsigned long, long, long, long, long*) () +# #2 0x000055555566cdea in (anonymous namespace)::TestS4Grid()::{lambda(superslm_attention_cases::SmCase const&)#1}::operator()(superslm_attention_cases::SmCase const&) const () +# #3 0x000055555566f0f2 in (anonymous namespace)::TestS4Grid() () +# #4 0x00005555556755ff in RunAttnRowsiteCells(int&, int&) () +# #5 0x000055555557bf17 in main () +# == s4_qln2_ge1_dropped on superslm_tests_avx2_forced (gdb): +# #0 0x000055555569d508 in superslm::SoftmaxRowQ15(long const*, unsigned long, long, long, long, long*) () +# #1 0x000055555566cdea in (anonymous namespace)::TestS4Grid()::{lambda(superslm_attention_cases::SmCase const&)#1}::operator()(superslm_attention_cases::SmCase const&) const () +# #2 0x000055555566df59 in (anonymous namespace)::TestS4Grid() () +# #3 0x00005555556755ff in RunAttnRowsiteCells(int&, int&) () +# #4 0x000055555557bf17 in main () +# == x_tail_total_dropped_avx2 on superslm_tests_avx2_forced (gdb): +# #0 0x000055555569ba83 in superslm::(anonymous namespace)::SoftmaxRowAvx2(long const*, unsigned long, superslm::(anonymous namespace)::SoftmaxFastRow const&, long*) () +# #1 0x000055555569d5b5 in superslm::SoftmaxRowQ15(long const*, unsigned long, long, long, long, long*) () +# #2 0x0000555555585e62 in TestSoftmaxRowQ15AgainstComposedShippedPrimitivesOracle() () +# #3 0x000055555557bb27 in main () +# == x_tail_total_dropped_avx512 on superslm_tests_avx512_forced (gdb): +# #0 0x000055555569b106 in superslm::(anonymous namespace)::SoftmaxRowAvx512(long const*, unsigned long, superslm::(anonymous namespace)::SoftmaxFastRow const&, long*) () +# #1 0x000055555569d47b in superslm::SoftmaxRowQ15(long const*, unsigned long, long, long, long, long*) () +# #2 0x0000555555585e62 in TestSoftmaxRowQ15AgainstComposedShippedPrimitivesOracle() () +# #3 0x000055555557bb27 in main () +# == s4_pup_skipped_avx2 on superslm_tests, run alone: exit 0 attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +# == s4_pup_skipped_avx2 on superslm_tests_avx512_forced, run alone: exit 0 attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| + +######## raw +== none on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== none on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2813 failures|superslm tests: 116129 checks, 2813 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +== all_avx512_runs_avx2_body on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2813 failures|superslm tests: 116087 checks, 2813 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +== all_fallback_increment_deleted on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116129 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== all_fallback_increment_deleted on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116087 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== all_fallback_increment_deleted on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116087 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== all_fast_increment_before_guard on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116129 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +== all_fast_increment_before_guard on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116087 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== all_fast_increment_before_guard on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 312 failures|superslm tests: 116087 checks, 312 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 0 (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_always_fallback on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2813 failures|superslm tests: 116129 checks, 2813 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_always_fallback on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2813 failures|superslm tests: 116087 checks, 2813 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +== s4_always_fallback on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2813 failures|superslm tests: 116087 checks, 2813 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_m_ge1_dropped on superslm_tests: exit 136; +== s4_m_ge1_dropped on superslm_tests_avx2_forced: exit 136; +== s4_m_ge1_dropped on superslm_tests_avx512_forced: exit 136; +== s4_m_ge2 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116129 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_m_ge2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116087 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +== s4_m_ge2 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116087 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_m_le_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 243 failures|superslm tests: 116129 checks, 243 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 12509543, q_b 24418208, q_c 312536884071128): bool 1 (v1.9.0 0); 1 of 1 probabilities differ, first at 0 (32769 vs 0) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 M = 2^47 + 1 (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_m_le_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 243 failures|superslm tests: 116087 checks, 243 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 12509543, q_b 24418208, q_c 312536884071128): bool 1 (v1.9.0 0); 1 of 1 probabilities differ, first at 0 (32769 vs 0) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 2, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 M = 2^47 + 1 (width 2, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== s4_m_le_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 243 failures|superslm tests: 116087 checks, 243 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 12509543, q_b 24418208, q_c 312536884071128): bool 1 (v1.9.0 0); 1 of 1 probabilities differ, first at 0 (32769 vs 0) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 M = 2^47 + 1 (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_m_lt on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116129 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 1 (width 6, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_m_lt on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 1 (width 6, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +== s4_m_lt on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 1 (width 6, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_pdown_skipped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== s4_pdown_skipped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 94 failures|superslm tests: 116087 checks, 94 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 3552184, q_b 6933744, q_c 25200511793588): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 1 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpp:1278:preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fo +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 3c92a388916d2316f2b62733a4c4a309ed20657739e88045b615b66fce4f6af5 ov +== s4_pdown_skipped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== s4_pdown_skipped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 94 failures|superslm tests: 116129 checks, 94 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 3552184, q_b 6933744, q_c 25200511793588): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 1 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpppreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fold tri +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 3c92a388916d2316f2b62733a4c4a309ed20657739e88045b615b66fce4f6af5 ov +== s4_pdown_skipped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== s4_pdown_skipped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 94 failures|superslm tests: 116087 checks, 94 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 3552184, q_b 6933744, q_c 25200511793588): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 1 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpppreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fold tri +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 3c92a388916d2316f2b62733a4c4a309ed20657739e88045b615b66fce4f6af5 ov +== s4_pup_skipped_avx2 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 2 failures| +FAIL tests/test_main.cpp:24099: adapter_status == AdapterLoadStatus::Ok — the --adapter code path's own LoadAdapterArtifact call must succeed: got FileReadFailed (could not read adapter file: t2102_sslm_generate_e2e_fixture.sslm.tmp) +FAIL tests/test_main.cpp:24125: seq.hidden_codes[0] != base_only.hidden_codes[0] || seq.hidden_codes[1] != base_only.hidden_codes[1] — the --adapter flag's own end-to-end sequence must produce output that differs from base-only -- got ide +== s4_pup_skipped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 393 failures|superslm tests: 116087 checks, 393 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 2033099, q_b 3968542, q_c 8255355406617): bool 1 (v1.9.0 1); 1 of 1 probabilities differ, first at 0 (32767 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 28396c888f805d78ba0e8289f43512c9d7f1bf99211cc79d66e776f80e8aa78a ov +FAIL tests/test_attn_rowsites.cpp:717: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 11.1(d) prefill prob-V: pv_fast_avx2 +1785 pv_fallback_avx2 +7 pv_fast_avx512 +0 pv_fallback_avx512 +0; want +== s4_pup_skipped_avx2 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_main.cpp:24099: adapter_status == AdapterLoadStatus::Ok — the --adapter code path's own LoadAdapterArtifact call must succeed: got ArtifactRejected (adapter model rejected: ArtifactRejected -- artifact rejected (NullData): +FAIL tests/test_main.cpp:24125: seq.hidden_codes[0] != base_only.hidden_codes[0] || seq.hidden_codes[1] != base_only.hidden_codes[1] — the --adapter flag's own end-to-end sequence must produce output that differs from base-only -- got ide +== s4_pup_skipped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 393 failures|superslm tests: 116129 checks, 393 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 2033099, q_b 3968542, q_c 8255355406617): bool 1 (v1.9.0 1); 1 of 1 probabilities differ, first at 0 (32767 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 28396c888f805d78ba0e8289f43512c9d7f1bf99211cc79d66e776f80e8aa78a ov +FAIL tests/test_attn_rowsites.cpp:717: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 11.1(d) prefill prob-V: pv_fast_avx2 +0 pv_fallback_avx2 +0 pv_fast_avx512 +1785 pv_fallback_avx512 +7; want +== s4_pup_skipped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== s4_pup_skipped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 393 failures|superslm tests: 116087 checks, 393 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 1, q_ln2 2033099, q_b 3968542, q_c 8255355406617): bool 1 (v1.9.0 1); 1 of 1 probabilities differ, first at 0 (32767 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 28396c888f805d78ba0e8289f43512c9d7f1bf99211cc79d66e776f80e8aa78a ov +FAIL tests/test_attn_rowsites.cpp:717: d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5 -- 11.1(d) prefill prob-V: pv_fast_avx2 +0 pv_fallback_avx2 +0 pv_fast_avx512 +1785 pv_fallback_avx512 +7; want +== s4_qc_ge0_dropped on superslm_tests: exit 136; +== s4_qc_ge0_dropped on superslm_tests_avx2_forced: exit 136; +== s4_qc_ge0_dropped on superslm_tests_avx512_forced: exit 136; +== s4_qc_ge1 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 29 failures|superslm tests: 116129 checks, 29 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_c = 0 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_qc_ge1 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 29 failures|superslm tests: 116087 checks, 29 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_c = 0 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +== s4_qc_ge1 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 29 failures|superslm tests: 116087 checks, 29 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 1 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_c = 0 (width 4, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +== s4_qln2_ge1_dropped on superslm_tests: exit 136; +== s4_qln2_ge1_dropped on superslm_tests_avx2_forced: exit 136; +== s4_qln2_ge1_dropped on superslm_tests_avx512_forced: exit 136; +== s4_qln2_ge2 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 186 failures|superslm tests: 116129 checks, 186 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^6preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst +== s4_qln2_ge2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 186 failures|superslm tests: 116087 checks, 186 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; w +== s4_qln2_ge2 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 186 failures|superslm tests: 116087 checks, 186 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 1,024 (width 1024, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^6preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst +== s4_ratio_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116129 checks, 4 failures| +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, q_ln2 10, q_b 4, q_c 0): bool 1 (v1.9.0 0); 2 of 2 probabilities differ, first at 0 (12787 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kerne +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 8d3b69830515db6df4be13cfb9805bc326d5f9e0b7f8bba159eed0bd399490ee ov +== s4_ratio_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116087 checks, 4 failures| +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, q_ln2 10, q_b 4, q_c 0): bool 1 (v1.9.0 0); 2 of 2 probabilities differ, first at 0 (12787 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kerne +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 8d3b69830515db6df4be13cfb9805bc326d5f9e0b7f8bba159eed0bd399490ee ov +== s4_ratio_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3 failures|superslm tests: 116087 checks, 4 failures| +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, q_ln2 10, q_b 4, q_c 0): bool 1 (v1.9.0 0); 2 of 2 probabilities differ, first at 0 (12787 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10) (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kerne +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 8d3b69830515db6df4be13cfb9805bc326d5f9e0b7f8bba159eed0bd399490ee ov +== s4_ratio_le2qb on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 33 failures|superslm tests: 116129 checks, 33 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 2 q_b + 1, realistic spread (width 301, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +== s4_ratio_le2qb on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 33 failures|superslm tests: 116087 checks, 33 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 2 q_b + 1, realistic spread (width 301, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2 +== s4_ratio_le2qb on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 33 failures|superslm tests: 116087 checks, 33 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 total = 1, p = 2^15 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: q_ln2 = 2 q_b + 1, realistic spread (width 301, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +== s4_score_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116129 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 one score 2^61 + 1 (width 64, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_score_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 one score 2^61 + 1 (width 64, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== s4_score_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 one score 2^61 + 1 (width 64, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_score_lt on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116129 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +== s4_score_lt on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116087 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2 +== s4_score_lt on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116087 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: scores of exactly +-2^61 (width 3, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX- +== s4_true_outside_guard on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 313 failures|superslm tests: 116129 checks, 317 failures| +FAIL tests/test_main.cpp:11552: !well_formed — SoftmaxRowQ15(width=3, q_ln2=0, q_b=0, q_c=0) returned true (well-formed), want false -- every element's IExpConstruct(q=0, q_ln2=0, ...) == kBadQLn2 (q_ln2 == 0 has no valid decomposition), +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_main.cpp:12016: well_formed == false — SoftmaxRowQ15(q_b=0, q_c=1152921504606846976, q_ln2=3000000001, width=1) reported well_formed=true for a row whose only element's real value (1152921504606846976) equals M (1152921504 +FAIL tests/test_main.cpp:15716: result == SslmForwardStatus::SoftmaxKernelRefusedAfterGateAccepted — RunLayerLoop(derived q_ln2 collapsed to 0, gate accepts) status == Ok, want SoftmaxKernelRefusedAfterGateAccepted +== s4_true_outside_guard on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 313 failures|superslm tests: 116087 checks, 317 failures| +FAIL tests/test_main.cpp:11552: !well_formed — SoftmaxRowQ15(width=3, q_ln2=0, q_b=0, q_c=0) returned true (well-formed), want false -- every element's IExpConstruct(q=0, q_ln2=0, ...) == kBadQLn2 (q_ln2 == 0 has no valid decomposition), +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_main.cpp:12016: well_formed == false — SoftmaxRowQ15(q_b=0, q_c=1152921504606846976, q_ln2=3000000001, width=1) reported well_formed=true for a row whose only element's real value (1152921504606846976) equals M (1152921504 +FAIL tests/test_main.cpp:15716: result == SslmForwardStatus::SoftmaxKernelRefusedAfterGateAccepted — RunLayerLoop(derived q_ln2 collapsed to 0, gate accepts) status == Ok, want SoftmaxKernelRefusedAfterGateAccepted +== s4_true_outside_guard on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 313 failures|superslm tests: 116087 checks, 317 failures| +FAIL tests/test_main.cpp:11552: !well_formed — SoftmaxRowQ15(width=3, q_ln2=0, q_b=0, q_c=0) returned true (well-formed), want false -- every element's IExpConstruct(q=0, q_ln2=0, ...) == kBadQLn2 (q_ln2 == 0 has no valid decomposition), +FAIL tests/test_main.cpp:11892: !well_formed — SoftmaxRowQ15(q_b=10, q_c=0, q_ln2=3000000001) reported well_formed=true for a row whose real evaluated element at k=1 (8999999940000000100) exceeds M (100) -- the kernel's own per-element ce +FAIL tests/test_main.cpp:12016: well_formed == false — SoftmaxRowQ15(q_b=0, q_c=1152921504606846976, q_ln2=3000000001, width=1) reported well_formed=true for a row whose only element's real value (1152921504606846976) equals M (1152921504 +FAIL tests/test_main.cpp:15716: result == SslmForwardStatus::SoftmaxKernelRefusedAfterGateAccepted — RunLayerLoop(derived q_ln2 collapsed to 0, gate accepts) status == Ok, want SoftmaxKernelRefusedAfterGateAccepted +== s4_width_2p15 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116129 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 width 2^14 + 1 (width 16385, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_width_2p15 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116087 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 width 2^14 + 1 (width 16385, guard copy: fallback): softmax_fast_avx2 +1 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== s4_width_2p15 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 1 failures|superslm tests: 116087 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 2.S4 width 2^14 + 1 (width 16385, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +1 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== s4_width_lt on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 4 failures|superslm tests: 116129 checks, 4 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 2^14 (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +== s4_width_lt on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 4 failures|superslm tests: 116087 checks, 4 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 2^14 (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +1 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; w +== s4_width_lt on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 4 failures|superslm tests: 116087 checks, 4 failures| +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, width 2^14 (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61) (width 16384, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +1; w +== s4_zup_skipped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== s4_zup_skipped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 307 failures|superslm tests: 116087 checks, 307 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 1916242, q_b 3740440, q_c 7333633156155): bool 1 (v1.9.0 1); 2 of 2 probabilities differ, first at 0 (32514 vs 32513) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 68f50980d4203d0cb87c42b0e1c760e18ff09656cdff1c7a91db60635c989f3d ov +== s4_zup_skipped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== s4_zup_skipped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 307 failures|superslm tests: 116129 checks, 307 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 1916242, q_b 3740440, q_c 7333633156155): bool 1 (v1.9.0 1); 2 of 2 probabilities differ, first at 0 (32514 vs 32513) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 68f50980d4203d0cb87c42b0e1c760e18ff09656cdff1c7a91db60635c989f3d ov +== s4_zup_skipped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== s4_zup_skipped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 307 failures|superslm tests: 116087 checks, 307 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid (width 2, q_ln2 1916242, q_b 3740440, q_c 7333633156155): bool 1 (v1.9.0 1); 2 of 2 probabilities differ, first at 0 (32514 vs 32513) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 68f50980d4203d0cb87c42b0e1c760e18ff09656cdff1c7a91db60635c989f3d ov +== x_clip_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== x_clip_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 10 failures|superslm tests: 116087 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid, aliased (width 2, q_ln2 117415, q_b 229191, q_c 27533976130): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 0 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 2fe70b1f375b217cf33f67475a90450409b557bd02e9a834fa9aa0a0e7952e56 ov +== x_clip_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== x_clip_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 10 failures|superslm tests: 116129 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid, aliased (width 2, q_ln2 117415, q_b 229191, q_c 27533976130): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 0 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 2fe70b1f375b217cf33f67475a90450409b557bd02e9a834fa9aa0a0e7952e56 ov +== x_clip_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== x_clip_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 10 failures|superslm tests: 116087 checks, 10 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 grid, aliased (width 2, q_ln2 117415, q_b 229191, q_c 27533976130): bool 1 (v1.9.0 1); 1 of 2 probabilities differ, first at 0 (32768 vs 32767) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 2fe70b1f375b217cf33f67475a90450409b557bd02e9a834fa9aa0a0e7952e56 ov +== x_clip_high_dword_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== x_clip_high_dword_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 2 failures|superslm tests: 116087 checks, 2 failures| +FAIL tests/test_attn_rowsites.cpp:1278: ok == want_ok && bad == 0 -- 4.S4 inside: scores of exactly +-2^61 (width 3, q_ln2 9, q_b 4, q_c 0): bool 1 (v1.9.0 1); 3 of 3 probabilities differ, first at 0 (10922 vs 32768) +FAIL tests/test_attn_rowsites.cpp:1434: hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && values == superslm_test::kAttnRowsiteS4GoldenValues -- 6.3 S4 golden: 5271e2068a9ffe99e9a96672bee7d4bd0a51a464ebc3e2211bc9ab18f6ee1354 ov +== x_clip_high_dword_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== x_tail_total_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| +== x_tail_total_dropped_avx2 on superslm_tests_avx2_forced: exit 136; +== x_tail_total_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== x_tail_total_dropped_avx512 on superslm_tests: exit 136; +== x_tail_total_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| +== x_tail_total_dropped_avx512 on superslm_tests_avx512_forced: exit 136; diff --git a/docs/attention-rowsites/s4/mutation-scripts/all_avx512_runs_avx2_body.py b/docs/attention-rowsites/s4/mutation-scripts/all_avx512_runs_avx2_body.py new file mode 100644 index 00000000..38cab925 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/all_avx512_runs_avx2_body.py @@ -0,0 +1,11 @@ +# Dispatch: the AVX-512 kernel runs the AVX2 body. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('SoftmaxRowAvx512(scores, width, row, out_probs);', 'SoftmaxRowAvx2(scores, width, row, out_probs); /* MUTANT */', 1, start='bool SoftmaxRowQ15(', end='// C32 (§5.2') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/all_fallback_increment_deleted.py b/docs/attention-rowsites/s4/mutation-scripts/all_fallback_increment_deleted.py new file mode 100644 index 00000000..20b93f1f --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/all_fallback_increment_deleted.py @@ -0,0 +1,14 @@ +# Dispatcher: the fallback increment deleted. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('''\t\t\t(kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_softmax_fallback_avx2 +\t\t\t : superslm_test::g_softmax_fallback_avx512) +\t\t\t .fetch_add(1, std::memory_order_relaxed); +''', '\t\t\t/* MUTANT */\n', 1, start='bool SoftmaxRowQ15(', end='// C32 (§5.2') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/all_fast_increment_before_guard.py b/docs/attention-rowsites/s4/mutation-scripts/all_fast_increment_before_guard.py new file mode 100644 index 00000000..3fd23211 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/all_fast_increment_before_guard.py @@ -0,0 +1,18 @@ +# The fast increment moved from the bodies to before the guard. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\tsuperslm_test::g_softmax_fast_avx2.fetch_add(1, std::memory_order_relaxed);\n', '\t/* MUTANT */\n') +sub('\tsuperslm_test::g_softmax_fast_avx512.fetch_add(1, std::memory_order_relaxed);\n', '\t/* MUTANT */\n') +sub('\t\t\tint64_t peak = 0;\n', '''#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT +\t\t\t(kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_softmax_fast_avx2 : superslm_test::g_softmax_fast_avx512) +\t\t\t .fetch_add(1, std::memory_order_relaxed); /* MUTANT */ +#endif +\t\t\tint64_t peak = 0; +''', 1, start='bool SoftmaxRowQ15(', end='// C32 (§5.2') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_always_fallback.py b/docs/attention-rowsites/s4/mutation-scripts/s4_always_fallback.py new file mode 100644 index 00000000..408c803a --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_always_fallback.py @@ -0,0 +1,11 @@ +# Dispatcher: always fall back. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (SoftmaxFastGuard(scores', 'if (false /* MUTANT */ && SoftmaxFastGuard(scores', 1, start='bool SoftmaxRowQ15(', end='// C32 (§5.2') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge1_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge1_dropped.py new file mode 100644 index 00000000..cee4a743 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge1_dropped.py @@ -0,0 +1,11 @@ +# Guard: M >= 1 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (!SGe(m128, SFromI64(1)) || !SGe(', 'if (/* MUTANT */ !SGe(', 1, start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge2.py b/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge2.py new file mode 100644 index 00000000..ca50f6f6 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_m_ge2.py @@ -0,0 +1,11 @@ +# Guard: M >= 1 -> >= 2. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('!SGe(m128, SFromI64(1))', '!SGe(m128, SFromI64(2)) /* MUTANT */', 1, start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_m_le_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_m_le_dropped.py new file mode 100644 index 00000000..97f75f71 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_m_le_dropped.py @@ -0,0 +1,11 @@ +# Guard: M <= 2^47 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' || !SGe(SFromI64(kSoftmaxRowMaxSafeExponent), m128)', ' /* MUTANT */', 1, start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_m_lt.py b/docs/attention-rowsites/s4/mutation-scripts/s4_m_lt.py new file mode 100644 index 00000000..8d234beb --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_m_lt.py @@ -0,0 +1,11 @@ +# Guard: M <= 2^47 -> < 2^47. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('SFromI64(kSoftmaxRowMaxSafeExponent), m128', 'SFromI64(kSoftmaxRowMaxSafeExponent - 1), m128 /* MUTANT */', 1, start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx2.py new file mode 100644 index 00000000..2d1ae11e --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx2.py @@ -0,0 +1,11 @@ +# AVX2 body: p downward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m256i down = _mm256_cmpgt_epi64(prod, num);', 'const __m256i down = _mm256_setzero_si256(); /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx512.py b/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx512.py new file mode 100644 index 00000000..a6d35ab7 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_pdown_skipped_avx512.py @@ -0,0 +1,11 @@ +# AVX-512 body: p downward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m512i down = _mm512_srai_epi64(_mm512_sub_epi64(num, prod), 63);', 'const __m512i down = _mm512_setzero_si512(); /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx2.py new file mode 100644 index 00000000..000a5b70 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx2.py @@ -0,0 +1,11 @@ +# AVX2 body: p upward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('return _mm256_add_epi64(p, _mm256_add_epi64(no_up, c.one));', '(void)no_up; return p; /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx512.py b/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx512.py new file mode 100644 index 00000000..63f06c5b --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_pup_skipped_avx512.py @@ -0,0 +1,11 @@ +# AVX-512 body: p upward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('return _mm512_add_epi64(p, _mm512_add_epi64(no_up, c.one));', '(void)no_up; return p; /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge0_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge0_dropped.py new file mode 100644 index 00000000..3c84ad49 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge0_dropped.py @@ -0,0 +1,11 @@ +# Guard: q_c >= 0 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\tif (q_c < 0) return false;\n', '\t/* MUTANT */\n', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge1.py b/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge1.py new file mode 100644 index 00000000..a1a2c6d6 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_qc_ge1.py @@ -0,0 +1,11 @@ +# Guard: q_c >= 0 -> >= 1. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (q_c < 0) return false;', 'if (q_c < 1) return false; /* MUTANT */', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge1_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge1_dropped.py new file mode 100644 index 00000000..d5d43b62 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge1_dropped.py @@ -0,0 +1,11 @@ +# Guard: q_ln2 >= 1 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\tif (q_ln2 < 1) return false;\n', '\t/* MUTANT */\n', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge2.py b/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge2.py new file mode 100644 index 00000000..33fee022 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_qln2_ge2.py @@ -0,0 +1,11 @@ +# Guard: q_ln2 >= 1 -> >= 2. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (q_ln2 < 1) return false;', 'if (q_ln2 < 2) return false; /* MUTANT */', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_dropped.py new file mode 100644 index 00000000..337238a1 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_dropped.py @@ -0,0 +1,11 @@ +# Guard: q_ln2 <= 2 q_b + 1 dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\tif (q_ln2 > 2 * q_b + 1) return false;\n', '\t/* MUTANT */\n', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_le2qb.py b/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_le2qb.py new file mode 100644 index 00000000..58686802 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_ratio_le2qb.py @@ -0,0 +1,11 @@ +# Guard: q_ln2 <= 2 q_b + 1 -> <= 2 q_b. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (q_ln2 > 2 * q_b + 1) return false;', 'if (q_ln2 > 2 * q_b) return false; /* MUTANT */', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_score_dropped.py b/docs/attention-rowsites/s4/mutation-scripts/s4_score_dropped.py new file mode 100644 index 00000000..959cac14 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_score_dropped.py @@ -0,0 +1,11 @@ +# Guard: the score-magnitude conjunct dropped (output-equivalent). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\t\tif (v > kSoftmaxFastScoreLimit || v < -kSoftmaxFastScoreLimit) return false;\n', '\t\t/* MUTANT */\n', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_score_lt.py b/docs/attention-rowsites/s4/mutation-scripts/s4_score_lt.py new file mode 100644 index 00000000..e33d27fb --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_score_lt.py @@ -0,0 +1,11 @@ +# Guard: |score| <= 2^61 -> < 2^61. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (v > kSoftmaxFastScoreLimit || v < -kSoftmaxFastScoreLimit)', 'if (v >= kSoftmaxFastScoreLimit || v <= -kSoftmaxFastScoreLimit) /* MUTANT */', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_true_outside_guard.py b/docs/attention-rowsites/s4/mutation-scripts/s4_true_outside_guard.py new file mode 100644 index 00000000..c3bdefcb --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_true_outside_guard.py @@ -0,0 +1,16 @@ +# Dispatcher: return true outside the guard (the fallback skipped). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('''\t\t\t .fetch_add(1, std::memory_order_relaxed); +#endif +''', '''\t\t\t .fetch_add(1, std::memory_order_relaxed); +#endif +\t\t\treturn true; /* MUTANT */ +''', 1, start='bool SoftmaxRowQ15(', end='// C32 (§5.2') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_width_2p15.py b/docs/attention-rowsites/s4/mutation-scripts/s4_width_2p15.py new file mode 100644 index 00000000..fe02367f --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_width_2p15.py @@ -0,0 +1,11 @@ +# Guard: width <= 2^14 -> <= 2^15 (output-equivalent). +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('kSoftmaxFastMaxWidth = size_t{1} << 14;', 'kSoftmaxFastMaxWidth = size_t{1} << 15; /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_width_lt.py b/docs/attention-rowsites/s4/mutation-scripts/s4_width_lt.py new file mode 100644 index 00000000..d781ca18 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_width_lt.py @@ -0,0 +1,11 @@ +# Guard: width <= 2^14 -> < 2^14. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (width > kSoftmaxFastMaxWidth) return false;', 'if (width >= kSoftmaxFastMaxWidth) return false; /* MUTANT */', start='bool SoftmaxFastGuard(', end='struct SoftmaxFastRow') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx2.py new file mode 100644 index 00000000..19881755 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx2.py @@ -0,0 +1,11 @@ +# AVX2 body: z upward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m256i up = _mm256_cmpgt_epi64(r, c.q_ln2_m1);', 'const __m256i up = _mm256_setzero_si256(); /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx512.py b/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx512.py new file mode 100644 index 00000000..70a6ea05 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/s4_zup_skipped_avx512.py @@ -0,0 +1,11 @@ +# AVX-512 body: z upward correction skipped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m512i up = _mm512_srai_epi64(_mm512_sub_epi64(c.q_ln2_m1, r), 63);', 'const __m512i up = _mm512_setzero_si512(); /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx2.py new file mode 100644 index 00000000..e69e4370 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx2.py @@ -0,0 +1,11 @@ +# Extra. AVX2 body: the clip min(a, 30 q_ln2) dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m256i a = _mm256_min_epu32(_mm256_or_si256(a_raw, big), c.clip);', 'const __m256i a = a_raw; (void)big; /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx512.py b/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx512.py new file mode 100644 index 00000000..3caa4df1 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/x_clip_dropped_avx512.py @@ -0,0 +1,11 @@ +# Extra. AVX-512 body: the clip min(a, 30 q_ln2) dropped. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m512i a = _mm512_min_epu64(_mm512_sub_epi64(c.peak, s), c.clip);', 'const __m512i a = _mm512_sub_epi64(c.peak, s); /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/x_clip_high_dword_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/x_clip_high_dword_avx2.py new file mode 100644 index 00000000..1f0d4236 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/x_clip_high_dword_avx2.py @@ -0,0 +1,11 @@ +# Extra. AVX2 body: the high dword not folded into bit 31 before the 32-bit clip. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_min_epu32(_mm256_or_si256(a_raw, big), c.clip)', '_mm256_min_epu32(a_raw, c.clip); (void)big /* MUTANT */') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx2.py b/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx2.py new file mode 100644 index 00000000..9ee8d68d --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx2.py @@ -0,0 +1,11 @@ +# Extra. AVX2 body: the tail's e left out of the total. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\t\t\ttotal += buf[i];\n', '\t\t\t/* MUTANT */\n', 1, start='void SoftmaxRowAvx2(', end='SoftmaxAvx512Consts') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx512.py b/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx512.py new file mode 100644 index 00000000..8e7dca14 --- /dev/null +++ b/docs/attention-rowsites/s4/mutation-scripts/x_tail_total_dropped_avx512.py @@ -0,0 +1,11 @@ +# Extra. AVX-512 body: the tail's e left out of the total. +p='src/intmath.cpp'; s=open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\t\t\ttotal += buf[i];\n', '\t\t\t/* MUTANT */\n', 1, start='void SoftmaxRowAvx512(') +open(p,"w").write(s) diff --git a/docs/attention-rowsites/s4/probe_111d_s4.cpp b/docs/attention-rowsites/s4/probe_111d_s4.cpp new file mode 100644 index 00000000..d3471650 --- /dev/null +++ b/docs/attention-rowsites/s4/probe_111d_s4.cpp @@ -0,0 +1,242 @@ +// S4 re-derivation of 11.1(d)'s softmax data term (docs/attention-rowsites/s4/softmax-data-terms.txt). A copy of +// probes/rev3/probe_111d.cpp (plan of record's probe directory) with a --wrap shim; build: +// g++ -std=c++20 -O2 -I/include -I/tools probe_111d_s4.cpp /build/libsuperslm.a \ +// -Wl,--wrap=_ZN8superslm13SoftmaxRowQ15EPKlmlllPl -Wl,--wrap=_ZN8superslm21GemmProbQ15AccumulateEPKlPKammPl +// ./a.out direct +// Probe for plan rev 3, cell 11.1(d): counts, per window, what each §3.6 counter would count, +// on the shipped c3e0004 forward (hooks added only at function entries in a scratch copy). +// Mode "direct": RunLayerLoopChunkBatched over 128 tokens, then 32 x RunLayerLoop, one +// workspace, trace hook installed, no final norm, embeds inside the windows with the same hook. +// Mode "abi": sslm_prefill(128) then 32 x sslm_decode_step (greedy), as the strike describes. +#include +#include +#include +#include +#include +#include +#include + +#include "superslm/forward_sites.h" +#include "superslm/model.h" +#include "superslm/sslm_abi.h" +#include "superslm/trace_hook.h" +#include "sslm_marshal.h" + +using namespace superslm; +using superslm_marshal::LayerBacking; +using superslm_marshal::MarshalLayer; +using superslm_marshal::PreflightScanWscFolds; +using superslm_marshal::ReadCarriedScale; +using superslm_marshal::ReadFile; + +struct Counts { + uint64_t softmax = 0, softmax_guard_fail = 0, pv = 0, pv_fail = 0, pv_hd16_fail = 0; + uint64_t requant = 0, norm_taken = 0, norm_skip = 0, silu_taken = 0, silu_skip = 0, land_taken = 0, + land_skip = 0, chain_records = 0, kv_records = 0; + std::map by_site; +}; +static Counts g; + +extern "C" void sslm_probe_softmax(const int64_t* s, size_t w, int64_t q_ln2, int64_t q_b, int64_t q_c) { + if (w == 0) return; + ++g.softmax; + bool ok = w <= (size_t(1) << 14) && q_ln2 >= 1 && q_c >= 0; + if (ok) { + const __int128 M = (__int128)q_b * q_b + q_c; + ok = M >= 1 && M <= ((__int128)1 << 47) && q_ln2 <= 2 * q_b + 1; + } + if (ok) + for (size_t i = 0; i < w; ++i) + if (s[i] > (int64_t(1) << 61) || s[i] < -(int64_t(1) << 61)) ok = false; + if (!ok) ++g.softmax_guard_fail; +} +extern "C" void sslm_probe_pv(const int64_t* p, size_t w, size_t hd) { + ++g.pv; + if (hd % 16 != 0) ++g.pv_hd16_fail; + int64_t sum = 0; + bool ok = true; + for (size_t k = 0; k < w; ++k) { + if (p[k] < 0 || p[k] > 32767) ok = false; + sum += p[k]; + } + if (sum > 32768) ok = false; + if (!ok) { ++g.pv_fail; if (w > 1) { int64_t mx = 0; size_t at = 0, nz = 0; for (size_t k = 0; k < w; ++k) { if (p[k] > mx) { mx = p[k]; at = k; } if (p[k]) ++nz; } std::printf("pv fail: width=%zu sum=%lld max=%lld at=%zu nonzero=%zu (call %llu)\n", w, (long long)sum, (long long)mx, at, nz, (unsigned long long)g.pv); } } +} +extern "C" void sslm_probe_requant(size_t) { ++g.requant; } +extern "C" void sslm_probe_site(int kind, size_t n) { + const bool taken = n >= 512; + if (kind == 0) (taken ? g.norm_taken : g.norm_skip)++; + if (kind == 1) (taken ? g.silu_taken : g.silu_skip)++; + if (kind == 2) (taken ? g.land_taken : g.land_skip)++; +} + +// S4 re-derivation on the S3 head: SoftmaxRowQ15 and GemmProbQ15Accumulate reached through -Wl,--wrap from +// forward_sites.o (no source edit of the base), each row classified by the plan's guard copy above. +extern "C" { +bool __real__ZN8superslm13SoftmaxRowQ15EPKlmlllPl(const int64_t*, size_t, int64_t, int64_t, int64_t, int64_t*); +} +struct CStat { int64_t qln2_min = INT64_MAX, qln2_max = 0, qb_min = INT64_MAX, qb_max = 0, qc_min = INT64_MAX, qc_max = 0; + __int128 M_min = ((__int128)1) << 100, M_max = 0; int64_t spread_max = 0; uint64_t clipped = 0, elems = 0; }; +static CStat gc; +extern "C" bool __wrap__ZN8superslm13SoftmaxRowQ15EPKlmlllPl(const int64_t* s, size_t w, int64_t ql, int64_t qb, int64_t qc, int64_t* out) { + sslm_probe_softmax(s, w, ql, qb, qc); + if (w) { gc.qln2_min = std::min(gc.qln2_min, ql); gc.qln2_max = std::max(gc.qln2_max, ql); gc.qb_min = std::min(gc.qb_min, qb); + gc.qb_max = std::max(gc.qb_max, qb); gc.qc_min = std::min(gc.qc_min, qc); gc.qc_max = std::max(gc.qc_max, qc); + __int128 M = (__int128)qb * qb + qc; if (M < gc.M_min) gc.M_min = M; if (M > gc.M_max) gc.M_max = M; + int64_t mx = s[0], mn = s[0]; for (size_t k = 0; k < w; ++k) { mx = std::max(mx, s[k]); mn = std::min(mn, s[k]); } + gc.spread_max = std::max(gc.spread_max, mx - mn); + for (size_t k = 0; k < w; ++k) { ++gc.elems; if (mx - s[k] >= 30 * ql) ++gc.clipped; } } + return __real__ZN8superslm13SoftmaxRowQ15EPKlmlllPl(s, w, ql, qb, qc, out); +} +extern "C" { +void __real__ZN8superslm21GemmProbQ15AccumulateEPKlPKammPl(const int64_t*, const int8_t*, size_t, size_t, int64_t*); +} +extern "C" void __wrap__ZN8superslm21GemmProbQ15AccumulateEPKlPKammPl(const int64_t* p, const int8_t* v, size_t w, size_t hd, int64_t* o) { + sslm_probe_pv(p, w, hd); + __real__ZN8superslm21GemmProbQ15AccumulateEPKlPKammPl(p, v, w, hd, o); +} +static void PrintC() { std::printf("softmax constants over all rows: q_ln2 [%lld, %lld] q_b [%lld, %lld] q_c [%lld, %lld] M [%lld, %lld] " + "max score spread %lld; elements %llu, clipped (z = 30) %llu\n", (long long)gc.qln2_min, (long long)gc.qln2_max, (long long)gc.qb_min, + (long long)gc.qb_max, (long long)gc.qc_min, (long long)gc.qc_max, (long long)gc.M_min, (long long)gc.M_max, (long long)gc.spread_max, + (unsigned long long)gc.elems, (unsigned long long)gc.clipped); } + +static void Hook(const SslmChainTraceRecord* c, const SslmKvLandingTraceRecord* kv, void*) { + if (c) { + ++g.chain_records; + std::string s(c->site); + const size_t dot = s.rfind('.'); + std::string key = s; + if (s.rfind("layer", 0) == 0) key = s.substr(s.find('.') + 1); + ++g.by_site[key]; + } + if (kv) ++g.kv_records; +} +static void Print(const char* name, const Counts& a, const Counts& b) { + std::printf("[%s] softmax=%llu guard_fail=%llu pv=%llu pv_int16_fail=%llu pv_hd16_fail=%llu requant=%llu " + "chain_records=%llu kv_records=%llu norm_taken=%llu norm_skip=%llu silu_taken=%llu silu_skip=%llu " + "land_taken=%llu land_skip=%llu\n", + name, (unsigned long long)(b.softmax - a.softmax), + (unsigned long long)(b.softmax_guard_fail - a.softmax_guard_fail), (unsigned long long)(b.pv - a.pv), + (unsigned long long)(b.pv_fail - a.pv_fail), (unsigned long long)(b.pv_hd16_fail - a.pv_hd16_fail), + (unsigned long long)(b.requant - a.requant), (unsigned long long)(b.chain_records - a.chain_records), + (unsigned long long)(b.kv_records - a.kv_records), (unsigned long long)(b.norm_taken - a.norm_taken), + (unsigned long long)(b.norm_skip - a.norm_skip), (unsigned long long)(b.silu_taken - a.silu_taken), + (unsigned long long)(b.silu_skip - a.silu_skip), (unsigned long long)(b.land_taken - a.land_taken), + (unsigned long long)(b.land_skip - a.land_skip)); + std::printf("[%s] records by site:", name); + for (auto& [k, v] : b.by_site) { + auto it = a.by_site.find(k); + const uint64_t d = v - (it == a.by_site.end() ? 0 : it->second); + if (d) std::printf(" %s=%llu", k.c_str(), (unsigned long long)d); + } + std::printf("\n"); +} + +static int32_t Tok(size_t i) { return static_cast((i * 37 + 11) % 256); } + +int main(int argc, char** argv) { + if (argc < 3) return std::fprintf(stderr, "usage: probe direct|abi [T] [D]\n"), 2; + const size_t T = argc > 3 ? std::stoul(argv[3]) : 128, D = argc > 4 ? std::stoul(argv[4]) : 32; + std::vector bytes; + if (!ReadFile(argv[1], bytes)) return 1; + const std::string mode = argv[2]; + if (mode == "abi") { + sslm_model model = nullptr; + if (sslm_model_map(bytes.data(), bytes.size(), &model) != SSLM_OK) return 3; + const size_t kvb = sslm_kv_block_size(model), req = kvb + sslm_kv_pool_overhead_size(model, 1); + std::vector praw(req + SSLM_ABI_ALIGNMENT_BYTES); + void* pa = praw.data(); + size_t ps = praw.size(); + std::align(SSLM_ABI_ALIGNMENT_BYTES, req, pa, ps); + sslm_kv_pool pool = nullptr; + if (sslm_kv_pool_create(model, pa, req, 1, &pool) != SSLM_OK) return 4; + sslm_config cfg{}; + cfg.max_batch = 1; + cfg.max_chunk_budget = static_cast(T); + cfg.max_layer_budget = 1; + const size_t wsb = sslm_workspace_size(model, &cfg); + std::vector wraw(wsb + SSLM_ABI_ALIGNMENT_BYTES); + void* wa = wraw.data(); + size_t wsz = wraw.size(); + std::align(SSLM_ABI_ALIGNMENT_BYTES, wsb, wa, wsz); + sslm_workspace ws = nullptr; + if (sslm_workspace_create(model, &cfg, wa, wsb, &ws) != SSLM_OK) return 5; + sslm_seq seq = nullptr; + if (sslm_seq_create(model, &pool, &seq) != SSLM_OK) return 6; + std::vector toks(T); + for (size_t i = 0; i < T; ++i) toks[i] = Tok(i); + Counts c0 = g; + int32_t consumed = 0; + if (sslm_prefill(model, seq, toks.data(), (int32_t)T, (int32_t)T, SSLM_SPAN_PROMPT, ws, &consumed) != SSLM_OK || + consumed != (int32_t)T) + return 7; + Counts c1 = g; + sslm_decode_params params{}; + if (sslm_decode_params_init(model, SSLM_DECODE_MODE_GREEDY, 1, ¶ms) != SSLM_OK) return 8; + for (size_t i = 0; i < D; ++i) { + int32_t out = 0; + if (sslm_decode_step(model, &seq, 1, ¶ms, ws, &out) != SSLM_OK || out < 0) return 9; + } + Counts c2 = g; + Print("abi prefill", c0, c1); + Print("abi decode", c1, c2); + return 0; + } + SslmModelView model; + std::string error; + if (SslmModel::Load(bytes.data(), bytes.size(), model, &error) != SslmModelStatus::Ok) return 3; + const uint32_t L = model.config.num_hidden_layers; + const size_t hidden = model.config.hidden_size, head_dim = model.config.head_dim, + kv = model.config.num_key_value_heads, H = model.config.num_attention_heads; + const int64_t cap = model.config.context_cap; + std::printf("L=%u H=%zu KV=%zu head_dim=%zu hidden=%zu inter=%u cap=%lld\n", L, H, kv, head_dim, hidden, + model.config.intermediate_size, (long long)cap); + PreflightScanWscFolds(model); + std::vector backing(L); + std::vector layers(L); + for (uint32_t l = 0; l < L; ++l) + if (!MarshalLayer(model, l, (uint32_t)H, (uint32_t)kv, backing[l], layers[l], &error)) return 4; + const int8_t* embed = reinterpret_cast(model.weights.Tensor("embed")->data); + bool ok = true; + const CarriedScale embed_scale = ReadCarriedScale(model.composition_constants, "embed", &ok); + std::vector workspace(size_t(L) * size_t(cap) * kv * head_dim * 2); + SslmSetTraceHook(model.trace_hook, Hook, nullptr); + const OptionGKLandingMode k_mode = + model.option_g_fused_k_landing ? OptionGKLandingMode::kFused : OptionGKLandingMode::kLegacy; + + Counts c0 = g; + std::vector chunk(T * hidden); + std::vector scales(T); + for (size_t i = 0; i < T; ++i) + if (EmbedEntry(Tok(i), (int32_t)model.config.vocab_size, embed, hidden, embed_scale, chunk.data() + i * hidden, + &scales[i], "embed", i, &model.trace_hook) != SslmForwardStatus::Ok) + return 5; + SequenceLayerState seq; + std::vector hc(hidden); + seq.hidden_codes = hc.data(); + SslmForwardStatus st = RunLayerLoopChunkBatched( + chunk.data(), scales.data(), T, layers.data(), L, hidden, head_dim, kv, model.config.intermediate_size, cap, 0, + model.rope_tables, workspace.data(), workspace.size(), model.option_g_fused_k_landing, &seq.kv_saturation_count, + {}, &model.trace_hook, H * head_dim); + if (st != SslmForwardStatus::Ok) return std::fprintf(stderr, "chunk: %s\n", SslmForwardStatusName(st)), 6; + seq.context_length = (int64_t)T; + Counts c1 = g; + for (size_t i = 0; i < D; ++i) { + CarriedScale sc{}; + if (EmbedEntry(Tok(T + i), (int32_t)model.config.vocab_size, embed, hidden, embed_scale, hc.data(), &sc, "embed", + T + i, &model.trace_hook) != SslmForwardStatus::Ok) + return 7; + seq.hidden_scale = sc; + seq.layer_index = 0; + st = RunLayerLoop(seq, layers.data(), L, L, hidden, head_dim, kv, model.config.intermediate_size, cap, + model.rope_tables, workspace.data(), workspace.size(), k_mode, {}, T + i, &model.trace_hook, + H * head_dim); + if (st != SslmForwardStatus::Ok) return std::fprintf(stderr, "decode: %s\n", SslmForwardStatusName(st)), 8; + } + Counts c2 = g; + std::printf("context_length after decode = %lld\n", (long long)seq.context_length); + Print("direct prefill", c0, c1); + Print("direct decode", c1, c2); + PrintC(); + return 0; +} diff --git a/docs/attention-rowsites/s4/red-sslm_axis_digest.txt b/docs/attention-rowsites/s4/red-sslm_axis_digest.txt new file mode 100644 index 00000000..ea9fb14c --- /dev/null +++ b/docs/attention-rowsites/s4/red-sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/red-sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s4/red-sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..84fc4275 --- /dev/null +++ b/docs/attention-rowsites/s4/red-sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/red-sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s4/red-sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..9aa2a67c --- /dev/null +++ b/docs/attention-rowsites/s4/red-sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/red-sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s4/red-sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..1003bab6 --- /dev/null +++ b/docs/attention-rowsites/s4/red-sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/red-sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s4/red-sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..6dc4ae07 --- /dev/null +++ b/docs/attention-rowsites/s4/red-sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/red-suites.txt b/docs/attention-rowsites/s4/red-suites.txt new file mode 100644 index 00000000..8ac0cf32 --- /dev/null +++ b/docs/attention-rowsites/s4/red-suites.txt @@ -0,0 +1,102 @@ +# S4 red run (plan §4 red-first): GCC 13.3 Release, the red commit (cells, softmax counters declared and never +# incremented, SoftmaxRowQ15 unchanged). Suites run from the repository root with +# SUPERSLM_ATTN_ROWSITES_ARTIFACT= (11.1(d)), each binary with its own TMPDIR. +# Every value assertion passes on every binary (bool and probabilities against the v1.9.0 restatement, the S4 +# golden hash, the correction-row premises, the 2.S4 conjunct counts); the failures are the path assertions +# (softmax_fast/fallback never move): 3,125 on each binary whose selector picks a new kernel, 0 on forced SSE2. +== superslm_tests (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3125 failures +superslm tests: 116129 checks, 3125 failures + FAIL lines: 3125 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 4, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 5, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 7, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 8, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 9, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 15, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 16, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 17, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +== superslm_tests_sse2_forced (red): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116071 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3125 failures +superslm tests: 116087 checks, 3125 failures + FAIL lines: 3125 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 4, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 5, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 7, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 8, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 9, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 15, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 16, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 17, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +== superslm_tests_avx512_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 3125 failures +superslm tests: 116087 checks, 3125 failures + FAIL lines: 3125 +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 1, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid, aliased (width 1, guard copy: fast): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 2, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 3, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 4, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 5, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 7, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 8, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 9, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 15, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 16, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1215: ok -- 4.S4 grid (width 17, guard copy: fallback): softmax_fast_avx2 +0 softmax_fallback_avx2 +0 softmax_fast_avx512 +0 softmax_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-512) diff --git a/docs/attention-rowsites/s4/sanitizers.txt b/docs/attention-rowsites/s4/sanitizers.txt new file mode 100644 index 00000000..eb9f9e31 --- /dev/null +++ b/docs/attention-rowsites/s4/sanitizers.txt @@ -0,0 +1,8 @@ +# S4 sanitizer runs at the implementation commit (GCC 13.3, RelWithDebInfo; ASan+UBSan with -fno-sanitize-recover=all, and TSan, as the CI +# legs configure them), SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1, separate TMPDIRs. 'reports' counts AddressSanitizer errors, UBSan runtime errors +# and ThreadSanitizer warnings in the run's output. The S4 cells include rows of widths 1-17 and 2^14 whose padded tails are copied +# into stack buffers (never loaded past the row), and aliased rows (scores == out_probs). +== build-asan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116087 checks, 0 failures| reports: 0 +== build-tsan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures|superslm tests: 116129 checks, 0 failures| reports: 0 diff --git a/docs/attention-rowsites/s4/softmax-data-terms.txt b/docs/attention-rowsites/s4/softmax-data-terms.txt new file mode 100644 index 00000000..5301d2f7 --- /dev/null +++ b/docs/attention-rowsites/s4/softmax-data-terms.txt @@ -0,0 +1,19 @@ +# S4 cell 11.1(d) data term (rows outside §5.4's guard), re-derived on the S3 series head (the base for S4). +# Probe: probes/rev3/probe_111d.cpp's direct drive (128-token chunk, then 32 decode steps; pinned tokens; trace hook +# installed; no final norm), linked against the S3 head's unmodified Release libsuperslm.a with -Wl,--wrap on +# SoftmaxRowQ15 and GemmProbQ15Accumulate, so every row forward_sites.o hands either kernel is classified by the +# plan's own guard copies (probe_111d.cpp's sslm_probe_softmax / sslm_probe_pv) before the real kernel runs. The +# RmsNorm/SiLU/landing/requant hooks of probe_hooks.diff are not installed here (the base is not edited), so their +# columns read 0; the chain trace records and the prob-V column reproduce S2's pv-data-terms.txt exactly. +# GCC 13.3 -O2, artifact p05_l1 (sha256 f0fd4886...6ed3). Source: probe_111d_s4.cpp beside this file. +# +# Result: softmax guard_fail = 0 in both windows (the plan's measured value). The constant range printed last is the +# realistic range the S4 bench uses (tools/sslm_sites_bench.cpp, softmax mode). +L=1 H=14 KV=2 head_dim=64 hidden=896 inter=4864 cap=2048 +pv fail: width=3 sum=32768 max=32768 at=1 nonzero=1 (call 42) +context_length after decode = 160 +[direct prefill] softmax=1792 guard_fail=0 pv=1792 pv_int16_fail=15 pv_hd16_fail=0 requant=0 chain_records=1536 kv_records=0 norm_taken=0 norm_skip=0 silu_taken=0 silu_skip=0 land_taken=0 land_skip=0 +[direct prefill] records by site: attn_ctx=128 attn_norm=128 attn_residual=128 down_proj.requant=128 embed=128 gate_proj.requant=128 mlp_act=128 mlp_norm=128 mlp_residual=128 o_proj.requant=128 q_proj.requant=128 up_proj.requant=128 +[direct decode] softmax=448 guard_fail=0 pv=448 pv_int16_fail=0 pv_hd16_fail=0 requant=0 chain_records=384 kv_records=0 norm_taken=0 norm_skip=0 silu_taken=0 silu_skip=0 land_taken=0 land_skip=0 +[direct decode] records by site: attn_ctx=32 attn_norm=32 attn_residual=32 down_proj.requant=32 embed=32 gate_proj.requant=32 mlp_act=32 mlp_norm=32 mlp_residual=32 o_proj.requant=32 q_proj.requant=32 up_proj.requant=32 +softmax constants over all rows: q_ln2 [347, 944] q_b [677, 1843] q_c [240769, 1781891] M [699098, 5178540] max score spread 37018; elements 180320, clipped (z = 30) 1535 diff --git a/docs/attention-rowsites/s4/sslm_axis_digest.txt b/docs/attention-rowsites/s4/sslm_axis_digest.txt new file mode 100644 index 00000000..ea9fb14c --- /dev/null +++ b/docs/attention-rowsites/s4/sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s4/sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..84fc4275 --- /dev/null +++ b/docs/attention-rowsites/s4/sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s4/sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..9aa2a67c --- /dev/null +++ b/docs/attention-rowsites/s4/sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s4/sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..1003bab6 --- /dev/null +++ b/docs/attention-rowsites/s4/sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s4/sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..6dc4ae07 --- /dev/null +++ b/docs/attention-rowsites/s4/sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s4/suites-clang.txt b/docs/attention-rowsites/s4/suites-clang.txt new file mode 100644 index 00000000..5853b13d --- /dev/null +++ b/docs/attention-rowsites/s4/suites-clang.txt @@ -0,0 +1,77 @@ +# S4 suites and digests at the implementation commit, Clang 18.1 Release, this host (AVX-512BW, so auto dispatches AVX-512). The four suite binaries run from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1 and separate TMPDIRs; then the five digest legs. Every digest's GLOBAL and every section equal the red run's (commit 1). Paths shortened. +== superslm_tests (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116129 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116071 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116087 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (clang): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116087 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +DONE-clang diff --git a/docs/attention-rowsites/s4/suites.txt b/docs/attention-rowsites/s4/suites.txt new file mode 100644 index 00000000..1eef7f72 --- /dev/null +++ b/docs/attention-rowsites/s4/suites.txt @@ -0,0 +1,77 @@ +# S4 suites and digests at the implementation commit, GCC 13.3 Release, this host (AVX-512BW, so auto dispatches AVX-512). The four suite binaries run from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1 and separate TMPDIRs; then the five digest legs. Every digest's GLOBAL and every section equal the red run's (commit 1). Paths shortened. +== superslm_tests (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116129 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116071 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116087 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (gcc): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4): 90598 checks, 0 failures +superslm tests: 116087 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL bc7b1cbecd995216c2aaf069f43900d3df01922bd8ec97779fe2eb16fabdcca7 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention 6bb5971d0b8f4cd48b6f1d1fb1759207f45b6c4b193ab85e0db9533b3482e9fd values=298178 +DONE-gcc diff --git a/docs/attention-rowsites/s5/bench-forward.txt b/docs/attention-rowsites/s5/bench-forward.txt new file mode 100644 index 00000000..c428f118 --- /dev/null +++ b/docs/attention-rowsites/s5/bench-forward.txt @@ -0,0 +1,183 @@ +# S5 forward probe raw output (scratch program, not committed; see bench.md): tests/support/qk_attention_fixture.h re-parameterised to +# Qwen3-0.6B's layer geometry, one RunLayerLoopChunkBatched over T positions, best-of-5 (best-of-3 at T = 1,024) ms per token. +# p_base / p_base_avx2: linked against the S4 head (v1.9.0's per-key loop); p_s5 / p_s5_avx2: S5. 5 alternating rounds. +## round 1 p_base +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 99.254 ms total, 0.7754 ms/token +## round 1 p_s5 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 56.797 ms total, 0.4437 ms/token +## round 1 p_base_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 122.209 ms total, 0.9548 ms/token +## round 1 p_s5_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 68.271 ms total, 0.5334 ms/token +## round 1 p_base +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1015.312 ms total, 1.9830 ms/token +## round 1 p_s5 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 260.403 ms total, 0.5086 ms/token +## round 1 p_base_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1147.046 ms total, 2.2403 ms/token +## round 1 p_s5_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 308.940 ms total, 0.6034 ms/token +## round 1 p_base +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3580.264 ms total, 3.4964 ms/token +## round 1 p_s5 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 602.967 ms total, 0.5888 ms/token +## round 1 p_base_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3997.556 ms total, 3.9039 ms/token +## round 1 p_s5_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 896.149 ms total, 0.8751 ms/token +## round 2 p_base +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 98.017 ms total, 0.7658 ms/token +## round 2 p_s5 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 55.942 ms total, 0.4370 ms/token +## round 2 p_base_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 117.720 ms total, 0.9197 ms/token +## round 2 p_s5_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 67.672 ms total, 0.5287 ms/token +## round 2 p_base +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 976.078 ms total, 1.9064 ms/token +## round 2 p_s5 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 274.719 ms total, 0.5366 ms/token +## round 2 p_base_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1215.215 ms total, 2.3735 ms/token +## round 2 p_s5_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 366.443 ms total, 0.7157 ms/token +## round 2 p_base +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3736.478 ms total, 3.6489 ms/token +## round 2 p_s5 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 644.981 ms total, 0.6299 ms/token +## round 2 p_base_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 4939.236 ms total, 4.8235 ms/token +## round 2 p_s5_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 909.379 ms total, 0.8881 ms/token +## round 3 p_base +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 105.252 ms total, 0.8223 ms/token +## round 3 p_s5 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 58.858 ms total, 0.4598 ms/token +## round 3 p_base_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 128.980 ms total, 1.0077 ms/token +## round 3 p_s5_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 68.388 ms total, 0.5343 ms/token +## round 3 p_base +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1049.986 ms total, 2.0508 ms/token +## round 3 p_s5 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 274.696 ms total, 0.5365 ms/token +## round 3 p_base_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1256.159 ms total, 2.4534 ms/token +## round 3 p_s5_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 351.507 ms total, 0.6865 ms/token +## round 3 p_base +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3810.413 ms total, 3.7211 ms/token +## round 3 p_s5 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 637.161 ms total, 0.6222 ms/token +## round 3 p_base_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 4719.428 ms total, 4.6088 ms/token +## round 3 p_s5_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 853.664 ms total, 0.8337 ms/token +## round 4 p_base +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 100.319 ms total, 0.7837 ms/token +## round 4 p_s5 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 54.642 ms total, 0.4269 ms/token +## round 4 p_base_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 125.641 ms total, 0.9816 ms/token +## round 4 p_s5_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 69.959 ms total, 0.5466 ms/token +## round 4 p_base +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 968.884 ms total, 1.8924 ms/token +## round 4 p_s5 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 263.456 ms total, 0.5146 ms/token +## round 4 p_base_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1283.474 ms total, 2.5068 ms/token +## round 4 p_s5_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 325.704 ms total, 0.6361 ms/token +## round 4 p_base +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3679.239 ms total, 3.5930 ms/token +## round 4 p_s5 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 681.154 ms total, 0.6652 ms/token +## round 4 p_base_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 4638.634 ms total, 4.5299 ms/token +## round 4 p_s5_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 906.523 ms total, 0.8853 ms/token +## round 5 p_base +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 100.069 ms total, 0.7818 ms/token +## round 5 p_s5 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 54.906 ms total, 0.4290 ms/token +## round 5 p_base_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 118.042 ms total, 0.9222 ms/token +## round 5 p_s5_avx2 +T=128 status 0 sat 1104 out-clamped 132/131072 hash a3871fe255284c51 +T=128 best-of-5 67.169 ms total, 0.5248 ms/token +## round 5 p_base +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 946.019 ms total, 1.8477 ms/token +## round 5 p_s5 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 253.192 ms total, 0.4945 ms/token +## round 5 p_base_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 1157.399 ms total, 2.2605 ms/token +## round 5 p_s5_avx2 +T=512 status 0 sat 4397 out-clamped 541/524288 hash d2449513d1c2d5f3 +T=512 best-of-5 305.983 ms total, 0.5976 ms/token +## round 5 p_base +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 3989.500 ms total, 3.8960 ms/token +## round 5 p_s5 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 646.770 ms total, 0.6316 ms/token +## round 5 p_base_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 4493.212 ms total, 4.3879 ms/token +## round 5 p_s5_avx2 +T=1024 status 0 sat 8761 out-clamped 1076/1048576 hash 463c25ff1999c67e +T=1024 best-of-3 748.720 ms total, 0.7312 ms/token diff --git a/docs/attention-rowsites/s5/bench-q31.txt b/docs/attention-rowsites/s5/bench-q31.txt new file mode 100644 index 00000000..95295b4e --- /dev/null +++ b/docs/attention-rowsites/s5/bench-q31.txt @@ -0,0 +1,146 @@ +# S5 q31 bench raw output: sslm_sites_bench q31 --repeat=30, one object linked against S5's libsuperslm.a (auto: AVX-512 here) and +# libsuperslm_avx2_forced.a (AVX2), 9 alternating rounds. Each reading times v1.9.0's per-key loop and QkQ31ScoreRow in the same process. +## round 1 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 326.26 / 396.84 w=8 329.38 / 47.16 w=64 326.32 / 14.82 w=128 328.45 / 13.57 w=512 329.94 / 12.91 w=1024 332.01 / 11.18 +q31 prefill T=128: per-key 8.0694, row 0.4737 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.0277, row 1.6011 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 75.2667, row 2.7625 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.7423, row 1.7957 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 84.2643, row 3.6024 ms/token at 28 layers x 16 heads +## round 1 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 392.68 / 229.72 w=8 315.99 / 45.30 w=64 372.82 / 18.88 w=128 344.19 / 15.22 w=512 397.03 / 21.06 w=1024 396.55 / 13.41 +q31 prefill T=128: per-key 11.0291, row 0.4767 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 44.4661, row 1.6565 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 80.5315, row 2.8755 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 46.0421, row 1.8647 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 98.5998, row 3.4431 ms/token at 28 layers x 16 heads +## round 2 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 303.90 / 338.05 w=8 279.85 / 46.29 w=64 325.80 / 14.80 w=128 330.47 / 13.56 w=512 329.80 / 12.99 w=1024 334.45 / 12.98 +q31 prefill T=128: per-key 9.7077, row 0.4706 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.7635, row 1.6103 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 76.5913, row 3.1426 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.4528, row 1.7760 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 89.7932, row 3.7153 ms/token at 28 layers x 16 heads +## round 2 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 415.17 / 276.67 w=8 419.39 / 32.15 w=64 419.55 / 15.68 w=128 423.97 / 14.52 w=512 428.94 / 13.63 w=1024 431.62 / 13.69 +q31 prefill T=128: per-key 12.0750, row 0.4761 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 46.2645, row 1.6556 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 99.3200, row 3.0693 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 57.1269, row 1.8269 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 109.7217, row 3.7300 ms/token at 28 layers x 16 heads +## round 3 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 326.46 / 390.27 w=8 332.02 / 46.46 w=64 330.68 / 22.10 w=128 335.58 / 13.51 w=512 320.73 / 12.42 w=1024 302.61 / 13.02 +q31 prefill T=128: per-key 9.5853, row 0.4735 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.0623, row 1.5492 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 72.4387, row 2.9982 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 42.8600, row 1.7319 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 88.4927, row 3.5091 ms/token at 28 layers x 16 heads +## round 3 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 391.15 / 264.60 w=8 407.35 / 32.43 w=64 418.48 / 15.51 w=128 420.42 / 14.36 w=512 430.20 / 24.00 w=1024 386.18 / 13.13 +q31 prefill T=128: per-key 11.4776, row 0.4582 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 47.4031, row 1.6637 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 96.7927, row 3.2401 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 56.9778, row 1.8735 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 103.5179, row 3.7128 ms/token at 28 layers x 16 heads +## round 4 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 326.41 / 387.35 w=8 327.99 / 46.81 w=64 326.18 / 14.94 w=128 327.33 / 13.55 w=512 330.38 / 12.85 w=1024 295.59 / 12.82 +q31 prefill T=128: per-key 9.4723, row 0.4704 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 32.7738, row 1.5971 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 76.1368, row 2.6660 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 39.9142, row 1.7230 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 86.3541, row 3.5216 ms/token at 28 layers x 16 heads +## round 4 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 376.86 / 278.71 w=8 380.77 / 32.19 w=64 382.61 / 15.52 w=128 380.61 / 14.37 w=512 388.71 / 13.72 w=1024 397.05 / 13.45 +q31 prefill T=128: per-key 10.9857, row 0.4699 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 42.1664, row 1.6513 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 90.1787, row 3.0516 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 51.8100, row 1.8872 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 104.7925, row 3.7202 ms/token at 28 layers x 16 heads +## round 5 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 323.25 / 383.75 w=8 329.84 / 46.82 w=64 331.96 / 14.78 w=128 332.84 / 13.85 w=512 334.48 / 15.75 w=1024 332.03 / 12.74 +q31 prefill T=128: per-key 9.4615, row 0.4697 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 37.4959, row 1.6251 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 76.6467, row 3.1189 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.5528, row 1.7789 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 87.2510, row 3.5389 ms/token at 28 layers x 16 heads +## round 5 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 420.61 / 279.40 w=8 422.42 / 32.19 w=64 419.39 / 15.51 w=128 495.77 / 21.36 w=512 426.34 / 13.59 w=1024 408.17 / 12.80 +q31 prefill T=128: per-key 11.3966, row 0.4727 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 48.3093, row 1.7128 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 98.9763, row 3.0506 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 57.0493, row 3.1978 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 116.0369, row 3.7201 ms/token at 28 layers x 16 heads +## round 6 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 326.30 / 388.01 w=8 329.44 / 45.99 w=64 326.88 / 14.79 w=128 328.49 / 13.56 w=512 330.04 / 12.96 w=1024 292.73 / 12.41 +q31 prefill T=128: per-key 9.1493, row 0.4730 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 37.3557, row 1.5412 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 76.6483, row 3.1058 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.8730, row 1.7990 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 90.0001, row 3.5618 ms/token at 28 layers x 16 heads +## round 6 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 420.15 / 278.72 w=8 422.25 / 32.70 w=64 418.60 / 15.57 w=128 426.54 / 14.88 w=512 429.71 / 13.72 w=1024 429.79 / 13.58 +q31 prefill T=128: per-key 12.1585, row 0.4717 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 48.9211, row 1.6832 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 97.5835, row 3.3029 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 57.3826, row 1.8748 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 115.5109, row 3.7224 ms/token at 28 layers x 16 heads +## round 7 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 326.08 / 386.64 w=8 323.82 / 46.68 w=64 330.66 / 14.81 w=128 332.03 / 17.45 w=512 323.78 / 12.98 w=1024 333.63 / 12.80 +q31 prefill T=128: per-key 9.4527, row 0.4720 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.5266, row 1.6566 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 73.8011, row 3.6634 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.5790, row 1.7698 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 89.4716, row 3.5272 ms/token at 28 layers x 16 heads +## round 7 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 415.03 / 276.49 w=8 421.14 / 32.02 w=64 419.70 / 15.60 w=128 421.82 / 14.28 w=512 431.13 / 13.74 w=1024 428.84 / 13.49 +q31 prefill T=128: per-key 12.1026, row 0.4717 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 49.2593, row 1.6684 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 102.7592, row 3.3261 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 57.5080, row 1.9001 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 117.5635, row 3.7869 ms/token at 28 layers x 16 heads +## round 8 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 331.99 / 395.98 w=8 331.18 / 47.48 w=64 331.27 / 20.85 w=128 339.54 / 13.39 w=512 335.28 / 12.99 w=1024 334.08 / 12.79 +q31 prefill T=128: per-key 9.4923, row 0.4710 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.2692, row 1.6132 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 77.0920, row 3.0939 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 44.7173, row 1.7562 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 89.0059, row 3.5249 ms/token at 28 layers x 16 heads +## round 8 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 376.55 / 277.78 w=8 381.04 / 32.02 w=64 382.28 / 15.52 w=128 368.97 / 13.86 w=512 373.71 / 13.24 w=1024 365.65 / 13.48 +q31 prefill T=128: per-key 10.9391, row 0.4719 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 42.9701, row 1.6496 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 84.0872, row 3.2446 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 51.7042, row 1.8762 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 104.6316, row 4.7602 ms/token at 28 layers x 16 heads +## round 9 bench_auto +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 374.03 / 416.75 w=8 337.20 / 47.90 w=64 332.03 / 14.75 w=128 332.14 / 13.49 w=512 333.41 / 12.98 w=1024 340.81 / 13.10 +q31 prefill T=128: per-key 9.6696, row 0.4730 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 38.8312, row 1.6042 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 75.5400, row 3.1299 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 45.4145, row 1.7779 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 90.6434, row 4.0415 ms/token at 28 layers x 16 heads +## round 9 bench_avx2 +q31 head_dim 128, ratios in [2^29, 2^31]; row vs per-key mismatches: 0 +q31 best-of-30 ns per (head x key), per-key / row: w=1 462.00 / 345.93 w=8 472.28 / 42.38 w=64 467.56 / 18.43 w=128 420.03 / 14.44 w=512 430.60 / 13.59 w=1024 429.48 / 13.45 +q31 prefill T=128: per-key 12.1006, row 0.4730 ms/token at 28 layers x 16 heads +q31 prefill T=512: per-key 49.3608, row 1.6729 ms/token at 28 layers x 16 heads +q31 prefill T=1024: per-key 98.5593, row 3.2299 ms/token at 28 layers x 16 heads +q31 decode ctx=300: per-key 57.1847, row 1.8075 ms/token at 28 layers x 16 heads +q31 decode ctx=600: per-key 111.9797, row 3.7682 ms/token at 28 layers x 16 heads diff --git a/docs/attention-rowsites/s5/bench.md b/docs/attention-rowsites/s5/bench.md new file mode 100644 index 00000000..6d53e25c --- /dev/null +++ b/docs/attention-rowsites/s5/bench.md @@ -0,0 +1,67 @@ +# S5 bench: what the Q31 score row saves + +These are reports, not gates (plan §8 10.1). The host is the shared 4-vCPU cloud Xeon (AVX2, AVX-512BW) with GCC 13.3 -O3. +No real Qwen3 artifact is available here: a QK-norm artifact is still refused at map time (plan R4), so the real-artifact +run stays the box's B1. Two synthetic readings stand in for it. + +1. **Kernel**, the plan's own §0 method. `tools/sslm_sites_bench.cpp` gains a `q31` mode. It uses one query head at + head_dim 128 (Qwen3-0.6B's) over one KV head's key rows, with ratios in [2^29, 2^31] (inside the loader's range). In one + process it times the layer loops' v1.9.0 per-key `QkQ31Score` loop and `QkQ31ScoreRow`, alternating. Both paths run + the same tier, so their difference is S5's own effect. It checks equality first (0 mismatches in all 18 runs). + Per token at Qwen3-0.6B depth, 28 layers × 16 query heads make one call per head per token. Prefill of T tokens sums + the calls at widths 1..T and divides by T. Decode at context C is one call at width C + 1. +2. **Forward, one layer at Qwen3-0.6B width.** This is a scratch probe (`probe_q31_forward.cpp`, not built by CMake). It is + the in-tree QK-norm fixture (`tests/support/qk_attention_fixture.h`) re-parameterised to hidden 1,024, 16 query + heads over 8 KV heads, head_dim 128, intermediate 3,072 and context_cap 1,024. It uses the fixture's own constants, + except hidden gain 4,096 and q/k-norm gains in [2,048, 4,096]; with larger gains a later funnel is out of domain at + T = 1,024. It runs one `RunLayerLoopChunkBatched` over T positions and returns Ok at every T measured. It was linked + four ways: the S4 head (v1.9.0's per-key loop) and S5, each with auto dispatch (AVX-512) and forced AVX2. **All four + emit the same output codes and scales at T = 128, 512 and 1,024**, which is an extra bit-identity check at Qwen3 width. + +Each kernel reading is a best-of-30. The builds ran alternately for 9 rounds, and the saving is the median of the paired +differences, with the min–max range in brackets. Each forward reading is a best-of-5 (best-of-3 at T = 1,024) over 5 +alternating rounds. The raw output is in `bench-q31.txt` and `bench-forward.txt`. + +## Kernel: per (head × key), and per token at Qwen3-0.6B depth + +| Reading | AVX2: per-key | AVX2: row | Saved (range) | × | AVX-512: per-key | AVX-512: row | Saved | × | Plan §0 (AVX2) | +|---|---|---|---|---|---|---|---|---|---| +| ns per (head × key), width 1 | 415 | 278 | 139 (98–163) | 1.5 | 326 | 388 | −61 | 0.84 | | +| width 8 | 419 | 32.2 | 387 | 13 | 329 | 46.8 | 283 | 7.0 | | +| width 128 | 420 | 14.4 | 406 | 29 | 332 | 13.6 | 317 | 24 | 376–385 → 16.5–17.6 (spike, head_dim 128) | +| width 1,024 | 408 | 13.5 | 395 | 30 | 332 | 12.8 | 321 | 26 | | +| **Prefill T = 128**, ms/token | 11.48 | 0.47 | **11.02** (10.5–11.7) | 24 | 9.47 | 0.47 | **9.00** (7.6–9.2) | 20 | **10.9 → 0.5** | +| **Prefill T = 512** | 47.40 | 1.66 | **45.74** (40.5–47.7) | 28 | 38.06 | 1.60 | **36.51** | 24 | (43.6 → 2.0 by the same arithmetic) | +| **Prefill T = 1,024** | 97.58 | 3.23 | **94.28** (77.7–99.4) | 30 | 76.14 | 3.11 | **73.45** (69.4–74.0) | 25 | **87 → 3.9** | +| Decode, context 300, ms/token | 57.05 | 1.88 | 55.10 (44.2–55.6) | 30 | 44.58 | 1.78 | 42.81 | 25 | | +| Decode, context 600 | 109.72 | 3.72 | 105.99 | 30 | 89.01 | 3.54 | 85.48 | 25 | | + +## Forward: one layer at Qwen3-0.6B width, ms per prompt token + +| T | AVX2: base | AVX2: S5 | Saved (range) | × 28 layers | AVX-512: base | AVX-512: S5 | Saved (range) | × 28 layers | +|---|---|---|---|---|---|---|---|---| +| 128 | 0.955 | 0.533 | 0.421 (0.39–0.47) | 11.8 | 0.782 | 0.437 | 0.353 (0.33–0.36) | 9.9 | +| 512 | 2.373 | 0.636 | 1.663 (1.64–1.87) | 46.6 | 1.906 | 0.515 | 1.378 (1.35–1.51) | 38.6 | +| 1,024 | 4.530 | 0.875 | 3.657 (3.03–3.94) | 102 | 3.649 | 0.630 | 3.019 (2.91–3.26) | 84.5 | + +The "× 28 layers" column scales the one-layer saving to Qwen3-0.6B's depth. It is an extrapolation, not a 28-layer +reading. + +## Against the estimate + +- **The estimate holds on AVX2 and is exceeded at long prompts.** §0 has 10.9 → 0.5 at T = 128 and 87 → 3.9 at T = 1,024. + Measured on AVX2 here: 11.5 → 0.47 and 97.6 → 3.23. The row kernel costs 13.5–14.4 ns per (head × key) on long rows, + a little under the spike's 16.5–17.6. The per-key loop costs 408–420 ns, a little over the spike's 376–385. So the + saving (11.0 and 94.3 ms per token) is at or above §0's implied 10.4 and 83. The one-layer forward agrees with the + kernel: 0.42 and 3.66 ms per layer, which is 11.8 and 102 ms at 28 layers. +- **On AVX-512 the base is cheaper, so less is saved.** The shipped per-key AVX-512 tier costs about 330 ns per key. + The row kernel is barely faster than on AVX2 (12.8 vs 13.5 ns). It saves 9.0 / 36.5 / 73.4 ms per token. +- **A one-key row on AVX-512 is 0.06 µs slower than the per-key call.** The body packs a whole 16-key block (15 zero + rows) and builds the limbs for 128 channels, whatever the width. On AVX2 (8-key blocks) the one-key row is still + faster than the per-key AVX2 call. The cost is paid once per head for the first prompt token only, and it is included + in the prefill figures. +- **§6's whole-prefill figure (≈ 140 → ≈ 57 ms per token at T = 1,024) is not checked here.** It needs a 28-layer + forward on real weights, which is B1/B2 on the box. + +These are engine figures on synthetic weights. They say nothing about any consumer's end-to-end speed, and they were not +measured on the project's reference hardware (the box's B2, §8 10.2, is box-only). diff --git a/docs/attention-rowsites/s5/blob-protocol.txt b/docs/attention-rowsites/s5/blob-protocol.txt new file mode 100644 index 00000000..fb7e4ebc --- /dev/null +++ b/docs/attention-rowsites/s5/blob-protocol.txt @@ -0,0 +1,96 @@ +# S5 save-blob protocol (plan 6.4, as S1-S4 ran it): the sslm_bench_prefill tool (tools/t2147_chunk_batched_pins.cpp) as built by +# CMake (GCC 13.3, Release). Base = the S1 head's build (v1.9.0 attention), unchanged since S2's run. Candidates = S5's +# implementation commit: auto (sslm_bench_prefill, AVX-512 here) and avx2 (sslm_bench_prefill_avx2_forced). Each row prefills the +# prompt at the chunk budget, decodes 32 greedy tokens and dumps the SSB5 blob; EQUAL = blobs byte-equal and decoded tokens equal. +# Artifacts: the S1 synthetic set (wide_l8, p05_l1 sha256 f0fd4886...6ed3, p05_l2) and the in-tree 8-layer fixture. +# +# Result: 66 of 66 rows EQUAL (33 auto, 33 forced AVX2); every distinct row's blob hash and token hash equals S4's record row for +# row. The in-tree fixture's 128- and 512-token prompts exceed its context cap: the base refuses them (sslm_status 16), as in S1-S4, +# so those 12 rows are not compared and the fixture runs at 8 and 24 tokens instead. +# +# These artifacts are plain-path (no QK-norm): S5's kernel does not run in them, and the rows show that nothing else moved. +# The Q31 path's own equality evidence, since no QK-norm artifact loads on this host (plan R4): +# - cell 11.1(c), the QK-norm fixture: both layer loops (decode, chunk, decode with a sink) emit the v1.9.0 tag's stream, +# hash 336b8d41... over 14,384 values, on all eight suite binaries (GCC and Clang; auto, forced SSE2, AVX2, AVX-512); +# the fast counter moves on every position in both loops (4 per decode position, 96 in the chunk run). +# - the Qwen3-width forward probe (bench.md, reading 2): one layer at hidden 1,024, 16/8 heads, head_dim 128, over 128, 512 and +# 1,024 positions; the S4 head and S5, each auto and forced AVX2, emit the same output codes and scales at every T +# (hashes a3871fe255284c51, d2449513d1c2d5f3, 463c25ff1999c67e; bench-forward.txt). +intree ids:8 chunk_budget=1 +32 decode, candidate auto: blob base 02718152a6f8941d auto 02718152a6f8941d; decoded tokens base 9ab82cad auto 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=1 +32 decode, candidate auto: blob base 5f8a4bafc1a4789a auto 5f8a4bafc1a4789a; decoded tokens base 391ba224 auto 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode, candidate auto: blob base 69d606622b873076 auto 69d606622b873076; decoded tokens base 1dec854e auto 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode, candidate auto: blob base cdcbfbe1fdcc65b1 auto cdcbfbe1fdcc65b1; decoded tokens base 6894a696 auto 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 02718152a6f8941d auto 02718152a6f8941d; decoded tokens base 9ab82cad auto 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 5f8a4bafc1a4789a auto 5f8a4bafc1a4789a; decoded tokens base 391ba224 auto 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 69d606622b873076 auto 69d606622b873076; decoded tokens base 1dec854e auto 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base cdcbfbe1fdcc65b1 auto cdcbfbe1fdcc65b1; decoded tokens base 6894a696 auto 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 02718152a6f8941d auto 02718152a6f8941d; decoded tokens base 9ab82cad auto 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 5f8a4bafc1a4789a auto 5f8a4bafc1a4789a; decoded tokens base 391ba224 auto 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base 69d606622b873076 auto 69d606622b873076; decoded tokens base 1dec854e auto 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate auto: blob base cdcbfbe1fdcc65b1 auto cdcbfbe1fdcc65b1; decoded tokens base 6894a696 auto 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=1 +32 decode, candidate auto: blob base 4722fb9a0f133c70 auto 4722fb9a0f133c70; decoded tokens base 681400be auto 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode, candidate auto: blob base 48efd7f2de021c02 auto 48efd7f2de021c02; decoded tokens base 1f43a9c7 auto 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode, candidate auto: blob base 7dcb7a5fb76a6276 auto 7dcb7a5fb76a6276; decoded tokens base a0edc532 auto a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate auto: blob base 4722fb9a0f133c70 auto 4722fb9a0f133c70; decoded tokens base 681400be auto 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate auto: blob base 48efd7f2de021c02 auto 48efd7f2de021c02; decoded tokens base 1f43a9c7 auto 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate auto: blob base 7dcb7a5fb76a6276 auto 7dcb7a5fb76a6276; decoded tokens base a0edc532 auto a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate auto: blob base 4722fb9a0f133c70 auto 4722fb9a0f133c70; decoded tokens base 681400be auto 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate auto: blob base 48efd7f2de021c02 auto 48efd7f2de021c02; decoded tokens base 1f43a9c7 auto 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate auto: blob base 7dcb7a5fb76a6276 auto 7dcb7a5fb76a6276; decoded tokens base a0edc532 auto a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=1 +32 decode, candidate auto: blob base 904ffb07ded91feb auto 904ffb07ded91feb; decoded tokens base e7799ea5 auto e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode, candidate auto: blob base 9a18bb27908136eb auto 9a18bb27908136eb; decoded tokens base 8435904c auto 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode, candidate auto: blob base 77d1a3589a0c813f auto 77d1a3589a0c813f; decoded tokens base 75387074 auto 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate auto: blob base 904ffb07ded91feb auto 904ffb07ded91feb; decoded tokens base e7799ea5 auto e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate auto: blob base 9a18bb27908136eb auto 9a18bb27908136eb; decoded tokens base 8435904c auto 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate auto: blob base 77d1a3589a0c813f auto 77d1a3589a0c813f; decoded tokens base 75387074 auto 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate auto: blob base 904ffb07ded91feb auto 904ffb07ded91feb; decoded tokens base e7799ea5 auto e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate auto: blob base 9a18bb27908136eb auto 9a18bb27908136eb; decoded tokens base 8435904c auto 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate auto: blob base 77d1a3589a0c813f auto 77d1a3589a0c813f; decoded tokens base 75387074 auto 75387074; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=1 +32 decode, candidate avx2: blob base 02718152a6f8941d avx2 02718152a6f8941d; decoded tokens base 9ab82cad avx2 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=1 +32 decode, candidate avx2: blob base 5f8a4bafc1a4789a avx2 5f8a4bafc1a4789a; decoded tokens base 391ba224 avx2 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=1 +32 decode, candidate avx2: blob base 69d606622b873076 avx2 69d606622b873076; decoded tokens base 1dec854e avx2 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=1 +32 decode, candidate avx2: blob base cdcbfbe1fdcc65b1 avx2 cdcbfbe1fdcc65b1; decoded tokens base 6894a696 avx2 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 02718152a6f8941d avx2 02718152a6f8941d; decoded tokens base 9ab82cad avx2 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 5f8a4bafc1a4789a avx2 5f8a4bafc1a4789a; decoded tokens base 391ba224 avx2 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 69d606622b873076 avx2 69d606622b873076; decoded tokens base 1dec854e avx2 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base cdcbfbe1fdcc65b1 avx2 cdcbfbe1fdcc65b1; decoded tokens base 6894a696 avx2 6894a696; EQUAL (1049632 bytes) +intree ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 02718152a6f8941d avx2 02718152a6f8941d; decoded tokens base 9ab82cad avx2 9ab82cad; EQUAL (16576 bytes) +wide_l8 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 5f8a4bafc1a4789a avx2 5f8a4bafc1a4789a; decoded tokens base 391ba224 avx2 391ba224; EQUAL (4194720 bytes) +p05_l1 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base 69d606622b873076 avx2 69d606622b873076; decoded tokens base 1dec854e avx2 1dec854e; EQUAL (525344 bytes) +p05_l2 ids:8 chunk_budget=8 +32 decode, candidate avx2: blob base cdcbfbe1fdcc65b1 avx2 cdcbfbe1fdcc65b1; decoded tokens base 6894a696 avx2 6894a696; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=1 +32 decode, candidate avx2: blob base 4722fb9a0f133c70 avx2 4722fb9a0f133c70; decoded tokens base 681400be avx2 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=1 +32 decode, candidate avx2: blob base 48efd7f2de021c02 avx2 48efd7f2de021c02; decoded tokens base 1f43a9c7 avx2 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=1 +32 decode, candidate avx2: blob base 7dcb7a5fb76a6276 avx2 7dcb7a5fb76a6276; decoded tokens base a0edc532 avx2 a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=8 +32 decode, candidate avx2: blob base 4722fb9a0f133c70 avx2 4722fb9a0f133c70; decoded tokens base 681400be avx2 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=8 +32 decode, candidate avx2: blob base 48efd7f2de021c02 avx2 48efd7f2de021c02; decoded tokens base 1f43a9c7 avx2 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=8 +32 decode, candidate avx2: blob base 7dcb7a5fb76a6276 avx2 7dcb7a5fb76a6276; decoded tokens base a0edc532 avx2 a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:128 chunk_budget=128 +32 decode, candidate avx2: blob base 4722fb9a0f133c70 avx2 4722fb9a0f133c70; decoded tokens base 681400be avx2 681400be; EQUAL (4194720 bytes) +p05_l1 ids:128 chunk_budget=128 +32 decode, candidate avx2: blob base 48efd7f2de021c02 avx2 48efd7f2de021c02; decoded tokens base 1f43a9c7 avx2 1f43a9c7; EQUAL (525344 bytes) +p05_l2 ids:128 chunk_budget=128 +32 decode, candidate avx2: blob base 7dcb7a5fb76a6276 avx2 7dcb7a5fb76a6276; decoded tokens base a0edc532 avx2 a0edc532; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=1 +32 decode, candidate avx2: blob base 904ffb07ded91feb avx2 904ffb07ded91feb; decoded tokens base e7799ea5 avx2 e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=1 +32 decode, candidate avx2: blob base 9a18bb27908136eb avx2 9a18bb27908136eb; decoded tokens base 8435904c avx2 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=1 +32 decode, candidate avx2: blob base 77d1a3589a0c813f avx2 77d1a3589a0c813f; decoded tokens base 75387074 avx2 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=8 +32 decode, candidate avx2: blob base 904ffb07ded91feb avx2 904ffb07ded91feb; decoded tokens base e7799ea5 avx2 e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=8 +32 decode, candidate avx2: blob base 9a18bb27908136eb avx2 9a18bb27908136eb; decoded tokens base 8435904c avx2 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=8 +32 decode, candidate avx2: blob base 77d1a3589a0c813f avx2 77d1a3589a0c813f; decoded tokens base 75387074 avx2 75387074; EQUAL (1049632 bytes) +wide_l8 ids:512 chunk_budget=512 +32 decode, candidate avx2: blob base 904ffb07ded91feb avx2 904ffb07ded91feb; decoded tokens base e7799ea5 avx2 e7799ea5; EQUAL (4194720 bytes) +p05_l1 ids:512 chunk_budget=512 +32 decode, candidate avx2: blob base 9a18bb27908136eb avx2 9a18bb27908136eb; decoded tokens base 8435904c avx2 8435904c; EQUAL (525344 bytes) +p05_l2 ids:512 chunk_budget=512 +32 decode, candidate avx2: blob base 77d1a3589a0c813f avx2 77d1a3589a0c813f; decoded tokens base 75387074 avx2 75387074; EQUAL (1049632 bytes) +intree ids:24 chunk_budget=1 +32 decode, candidate auto: blob base 4dd83f12817436f7 auto 4dd83f12817436f7; decoded tokens base 5707bd0c auto 5707bd0c; EQUAL (16576 bytes) +intree ids:24 chunk_budget=8 +32 decode, candidate auto: blob base 4dd83f12817436f7 auto 4dd83f12817436f7; decoded tokens base 5707bd0c auto 5707bd0c; EQUAL (16576 bytes) +intree ids:24 chunk_budget=24 +32 decode, candidate auto: blob base 4dd83f12817436f7 auto 4dd83f12817436f7; decoded tokens base 5707bd0c auto 5707bd0c; EQUAL (16576 bytes) +intree ids:24 chunk_budget=1 +32 decode, candidate avx2: blob base 4dd83f12817436f7 avx2 4dd83f12817436f7; decoded tokens base 5707bd0c avx2 5707bd0c; EQUAL (16576 bytes) +intree ids:24 chunk_budget=8 +32 decode, candidate avx2: blob base 4dd83f12817436f7 avx2 4dd83f12817436f7; decoded tokens base 5707bd0c avx2 5707bd0c; EQUAL (16576 bytes) +intree ids:24 chunk_budget=24 +32 decode, candidate avx2: blob base 4dd83f12817436f7 avx2 4dd83f12817436f7; decoded tokens base 5707bd0c avx2 5707bd0c; EQUAL (16576 bytes) +intree ids:128 cb=1 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:128 cb=8 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:128 cb=128 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:512 cb=1 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) +intree ids:512 cb=8 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) +intree ids:512 cb=512 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) +intree ids:128 cb=1 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:128 cb=8 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:128 cb=128 base FAILED: prompt "ids:128:128" -> 128 real tokens (base refuses: context cap) +intree ids:512 cb=1 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) +intree ids:512 cb=8 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) +intree ids:512 cb=512 base FAILED: prompt "ids:512:128" -> 512 real tokens (base refuses: context cap) diff --git a/docs/attention-rowsites/s5/coverage.txt b/docs/attention-rowsites/s5/coverage.txt new file mode 100644 index 00000000..7378bc26 --- /dev/null +++ b/docs/attention-rowsites/s5/coverage.txt @@ -0,0 +1,36 @@ +# S5 branch coverage (plan §3.4, cell 11.6). INDICATIVE ONLY, as S2-S4's: a floor is pinned only from the hosted branch-coverage +# leg's own recorded measurement, and this series is delivered as patches, not pushed, so that leg has not run on it. These are +# local replicas of the leg's commands (clang-18, llvm-cov/llvm-profdata 18, RelWithDebInfo, -fprofile-instr-generate +# -fcoverage-mapping; the five binaries, merged, exported over src/*.cpp include/superslm/*.h), on this host (AVX-512BW), +# SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1. All five binaries exit 0. Branches covered / total. +# +# S5 edits neither src/intmath.cpp nor src/matmul.cpp, and src/forward/forward_sites.cpp is outside the leg's export glob +# (src/*.cpp does not reach src/forward/) and carries no floor (G27). So §3.4's per-slice step does not apply to S5 as the plan +# words it (it names intmath.cpp and matmul.cpp), and the leg's numbers cannot move. Measured to confirm, and forward_sites.cpp +# exported separately for the S5 lines: +# +# (1) All five profiles (the leg's union), this host: +# src/intmath.cpp 221/242 91.32% src/matmul.cpp 286/312 91.67% (S4's candidate numbers exactly) +# check_branch_coverage_floors.py: OK (19 files at or above their pinned floor). +# src/forward/forward_sites.cpp (separate export): 734/910 80.66%. S5's lines (614-840) and both call sites: no uncovered side. +# +# (2) Approximating the hosted runner, which has no AVX-512 (sse2-forced, avx2-forced and scalar-forced-digest profiles only): +# src/intmath.cpp 202/242 83.47% src/matmul.cpp 203/286 70.98% (S4's projection exactly: below the floors 87.93 / 72.22 +# already before S5, from S2-S4, as recorded there) +# src/forward/forward_sites.cpp: 716/906 79.03%. The S5 sides lost are all AVX-512-only: QkQ31RowAvx512's block, quad, limb and +# half loops and its tail count (lines 774, 775, 778, 781, 783, 790, 792) and the dispatcher's kAvx2-or-kAvx512 choices (819, +# 826). +# (check_branch_coverage_floors.py on this projection also reports "floor unrecorded for src/forward/forward_sites.cpp": that is +# this replica's separate export including the file, not the leg, whose glob never includes it.) +# +# Uncovered S5 branches, and the cell they produced (plan §3.4 step 1): +# The first replica (the implementation commit's suite) left two S5 sides untaken with all five profiles: line 654 +# (MakeQ31RowLimbs, `d < head_dim ? q * ratio : 0`) and line 699 (Q31PackKeyBlock's scalar tail, `d < head_dim ? key : 0`). +# Both are the channel pads of a head_dim that is not a multiple of 4, and 4.S5's in-guard head_dims {4, 8, 60, 64, 128, 132, +# 256, 512} are all multiples of 4. The evidence commit adds the "4.S5 channel tail" rows (head_dim 1, 2, 3, 5, 63, 66, 127, +# 130, 509, 511 x widths 1, 9, 17 x the three grid kinds, all inside the guard, not in the golden set); with them, (1) above. +# The mutant x_pads_nonzero (mutants.txt) dies on those 90 rows and on no other cell. +# Without AVX-512 profiles: the nine AVX-512-only lines above; no reachable S5 branch is uncovered. +# +# No allowlist note is added: tools/ci/branch_coverage_allowlist.txt documents branches of the measured files, and S5 adds none +# to them. The floors in tools/ci/branch_coverage_floors.json are NOT re-pinned. diff --git a/docs/attention-rowsites/s5/fixture-premise.txt b/docs/attention-rowsites/s5/fixture-premise.txt new file mode 100644 index 00000000..3596a3bb --- /dev/null +++ b/docs/attention-rowsites/s5/fixture-premise.txt @@ -0,0 +1,66 @@ +# Cell 11.1(c)'s premise on the base (plan §8: "Before any kernel lands, the test author confirms on the base that every +# step returns Ok and that the fixture's softmax rows sit inside §5.4's guard"). +# Base: the S4 head's Release libsuperslm.a (GCC 13.3; no S5 code). Probe: a scratch program that builds +# tests/support/qk_attention_fixture.h, runs (i) the decode loop, (ii) the chunk loop with a sink installed, (iii) the decode +# loop with a classifying sink, and prints each run's status and stream hash; the sink classifies every softmax row with +# attention_cases.h's TestSoftmaxGuard and every prob-V row with the int16 condition (maxp = the row's largest p). +# The same hash comes from the v1.9.0 tag (golden.txt). +# +# How the constants were chosen (sweeps on the same base, one exponent at a time; LandingRescale lands at +# x * m_a * r_t >> (62 - e_a + e_t), so a larger e_t is a coarser code): +# - norm gains: hidden 8192, q_norm/k_norm gains uniform in [4096, 8192]. At 16384 / [12000, 20000] the post-norm wide +# row passes the funnel's 2^31 preflight (ChainInputOutOfDomain). +# - gate site constant e = -52: at the canonical -30 the gate scale's exponent is +18, outside MlpActSite's runtime +# domain [-80, 8] (SiluCompositionScaleOutOfDomain); -52 lands it near -5. +# - K/V landing e = +12: at -30 every V code clamps to +-127 (128 of 128 per position), at +10 about 10 clamp, at +12 +# none (V spans about +-55), and the first K landing no longer clamps before k_norm renormalises it. +# - K channel landing e = -36 (source scale 2^0 x 2^-60): K codes span about +-88, none clamped. +# - softmax K-head constant e = -72: q_ln2 about 2,200-2,900, rows from diffuse (max p near 4,096) to peaked; -70 +# makes 8 of 96 rows one-hot; -74 drives a later funnel out of domain; -60 and above refuse in the kernel +# (SoftmaxKernelRefusedAfterGateAccepted), -90 fails the width gate. +== the fixture at its committed constants +loaded 1 +run 0: Ok, 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 over 14384 +run 1: Ok, 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 over 14384 + pos 23 head 0: q_ln2 2939 q_b 5736 q_c 17251337, maxp 13133, s[0..3] 3919 6457 6121 8083 + pos 23 head 1: q_ln2 2891 q_b 5644 q_c 16701103, maxp 23227, s[0..3] -15501 -2760 7929 -8226 + pos 23 head 2: q_ln2 2200 q_b 4295 q_c 9672348, maxp 9805, s[0..3] 6478 18229 2289 -8260 + pos 23 head 3: q_ln2 2586 q_b 5048 q_c 13360213, maxp 5697, s[0..3] 2838 10005 6297 882 +run 2: Ok, 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 over 14384 +rows 96 in-guard 96 onehot 4 pvfail 4 score range [-30628, 31140] + maxp bucket 4096: 4 + maxp bucket 8192: 16 + maxp bucket 12288: 23 + maxp bucket 16384: 11 + maxp bucket 20480: 18 + maxp bucket 24576: 10 + maxp bucket 28672: 10 + maxp bucket 32768: 4 +== positions 0-2 of the decode loop, funnel scales and K/V codes + layer0.q_norm tok 0 d' 1012648384 m_out 2025296768 e_out -1 codes [-127,84] + layer0.q_norm tok 0 d' 1257072509 m_out 1257072510 e_out 0 codes [-127,74] + layer0.q_norm tok 0 d' 1008091272 m_out 2016182544 e_out -1 codes [-127,101] + layer0.q_norm tok 0 d' 1031005680 m_out 2062011360 e_out -1 codes [-85,127] + layer0.attn_ctx tok 0 d' 1671168 m_out 1711276032 e_out -10 codes [-127,105] + layer0.gate_proj.requant tok 0 d' 204124 m_out 1259859707 e_out -4 codes [-87,127] +pos 0: Ok + K codes at pos 0: [-73,88], 0 at +-127 + V codes at pos 0: [-51,42], 0 at +-127 + layer0.q_norm tok 1 d' 988414560 m_out 1976829120 e_out -1 codes [-127,117] + layer0.q_norm tok 1 d' 980397864 m_out 1960795728 e_out -1 codes [-108,127] + layer0.q_norm tok 1 d' 896431188 m_out 1792862376 e_out -1 codes [-127,90] + layer0.q_norm tok 1 d' 1154604775 m_out 1154604776 e_out 0 codes [-127,124] + layer0.attn_ctx tok 1 d' 1539160 m_out 1576099840 e_out -10 codes [-127,96] + layer0.gate_proj.requant tok 1 d' 101303 m_out 1456053345 e_out -5 codes [-127,127] +pos 1: Ok + K codes at pos 1: [-82,53], 0 at +-127 + V codes at pos 1: [-55,36], 0 at +-127 + layer0.q_norm tok 2 d' 1320604920 m_out 1320604920 e_out 0 codes [-91,127] + layer0.q_norm tok 2 d' 1019596994 m_out 2039193988 e_out -1 codes [-127,117] + layer0.q_norm tok 2 d' 1103300968 m_out 1103300968 e_out 0 codes [-127,109] + layer0.q_norm tok 2 d' 1376739594 m_out 1376739594 e_out 0 codes [-65,127] + layer0.attn_ctx tok 2 d' 1785246 m_out 1828091904 e_out -10 codes [-127,95] + layer0.gate_proj.requant tok 2 d' 142864 m_out 1746824916 e_out -5 codes [-107,127] +pos 2: Ok + K codes at pos 2: [-59,86], 0 at +-127 + V codes at pos 2: [-43,50], 0 at +-127 diff --git a/docs/attention-rowsites/s5/fp-scan.txt b/docs/attention-rowsites/s5/fp-scan.txt new file mode 100644 index 00000000..40bf3477 --- /dev/null +++ b/docs/attention-rowsites/s5/fp-scan.txt @@ -0,0 +1,216 @@ +# S5 fp-free scan (plan §3.4; tests/ci/scan_build_output.py --build-dir --target --isa x86-64), allow-lists +# unchanged, at the implementation commit, on the auto library and both forced AVX libraries. +# +# Clang 18.1: PASS on all three (493 / 498 / 499 ACCEPT, 0 REJECT). forward_sites.cpp.o is clean. +# GCC 13.3: superslm and superslm_avx512_forced FAIL on one symbol each, TiledGemmAvx512 in matmul.cpp.o, inherited from +# the base (as in S3 and S4; main fixes it in 90e48de); superslm_avx2_forced PASS. forward_sites.cpp.o is clean in all +# three. With 90e48de's src/matmul.cpp change applied temporarily (not part of this series): PASS on all three +# (564 / 574 / 576 ACCEPT, 0 REJECT). +# +# No reject was met on the way: the bodies were written to S3/S4's rules from the start. No 64-bit compare or select +# anywhere (the rounding's "remainder > threshold" is the sign bit of threshold - remainder, floor is a biased logical +# shift on AVX2 and vpsraq on AVX-512), no mask register, no per-lane conditional loop, and the S5 code holds no +# float, double or FP intrinsic. The GCC bodies appear in the non-gating check-(C) listing (an external call edge: the +# memcpy of the block's results), like S3's and S4's bodies. +# +# Summaries follow; the per-object lines are verbatim, the check-(C) listings are elided. + +######## scan-build-clang-superslm +Scanning 17 object member(s) of archive build-clang/libsuperslm.a for target 'superslm' (isa=x86-64) + clean artifact.cpp.o (30 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (52 symbol(s), format=elf) + clean model.cpp.o (77 symbol(s), format=elf) + clean intmath.cpp.o (29 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (23 symbol(s), format=elf) + clean proof_manifest.cpp.o (21 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (15 symbol(s), format=elf) + clean forward_sites.cpp.o (52 symbol(s), format=elf) + clean decode_digest.cpp.o (3 symbol(s), format=elf) + clean sslm_abi.cpp.o (125 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (26 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (6 symbol(s), format=elf) +Totals: 17 object(s); 493 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 284 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build-clang-superslm_avx2_forced +Scanning 17 object member(s) of archive build-clang/libsuperslm_avx2_forced.a for target 'superslm_avx2_forced' (isa=x86-64) + clean artifact.cpp.o (31 symbol(s), format=elf) + clean sha256.cpp.o (9 symbol(s), format=elf) + clean tokenizer.cpp.o (53 symbol(s), format=elf) + clean model.cpp.o (78 symbol(s), format=elf) + clean intmath.cpp.o (29 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (20 symbol(s), format=elf) + clean proof_manifest.cpp.o (22 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (15 symbol(s), format=elf) + clean forward_sites.cpp.o (52 symbol(s), format=elf) + clean decode_digest.cpp.o (3 symbol(s), format=elf) + clean sslm_abi.cpp.o (128 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (26 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (6 symbol(s), format=elf) +Totals: 17 object(s); 498 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 290 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build-clang-superslm_avx512_forced +Scanning 17 object member(s) of archive build-clang/libsuperslm_avx512_forced.a for target 'superslm_avx512_forced' (isa=x86-64) + clean artifact.cpp.o (31 symbol(s), format=elf) + clean sha256.cpp.o (9 symbol(s), format=elf) + clean tokenizer.cpp.o (53 symbol(s), format=elf) + clean model.cpp.o (78 symbol(s), format=elf) + clean intmath.cpp.o (29 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (21 symbol(s), format=elf) + clean proof_manifest.cpp.o (22 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (15 symbol(s), format=elf) + clean forward_sites.cpp.o (52 symbol(s), format=elf) + clean decode_digest.cpp.o (3 symbol(s), format=elf) + clean sslm_abi.cpp.o (128 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (26 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (6 symbol(s), format=elf) +Totals: 17 object(s); 499 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 290 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build-superslm +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + REJECT matmul.cpp.o (1 symbol(s), format=elf) + _ZN8superslm12_GLOBAL__N_115TiledGemmAvx512EPKsmPKammmmmPlPa + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 563 symbol(s) ACCEPT, 1 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 378 symbol(s) reject under check (C) alone (non-gating diagnostic) +FAIL: the scan did not come back clean. + +######## scan-build-superslm_avx2_forced +Scanning 17 object member(s) of archive build/libsuperslm_avx2_forced.a for target 'superslm_avx2_forced' (isa=x86-64) + clean artifact.cpp.o (35 symbol(s), format=elf) + clean sha256.cpp.o (14 symbol(s), format=elf) + clean tokenizer.cpp.o (50 symbol(s), format=elf) + clean model.cpp.o (113 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (21 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (134 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 574 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 378 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build-superslm_avx512_forced +Scanning 17 object member(s) of archive build/libsuperslm_avx512_forced.a for target 'superslm_avx512_forced' (isa=x86-64) + clean artifact.cpp.o (35 symbol(s), format=elf) + clean sha256.cpp.o (14 symbol(s), format=elf) + clean tokenizer.cpp.o (50 symbol(s), format=elf) + clean model.cpp.o (113 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + REJECT matmul.cpp.o (1 symbol(s), format=elf) + _ZN8superslm12_GLOBAL__N_115TiledGemmAvx512EPKsmPKammmmmPlPa + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (134 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 575 symbol(s) ACCEPT, 1 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 379 symbol(s) reject under check (C) alone (non-gating diagnostic) +FAIL: the scan did not come back clean. + +######## scan-build90-superslm +Scanning 17 object member(s) of archive build/libsuperslm.a for target 'superslm' (isa=x86-64) + clean artifact.cpp.o (33 symbol(s), format=elf) + clean sha256.cpp.o (8 symbol(s), format=elf) + clean tokenizer.cpp.o (49 symbol(s), format=elf) + clean model.cpp.o (112 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (24 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (131 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 564 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 379 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build90-superslm_avx2_forced +Scanning 17 object member(s) of archive build/libsuperslm_avx2_forced.a for target 'superslm_avx2_forced' (isa=x86-64) + clean artifact.cpp.o (35 symbol(s), format=elf) + clean sha256.cpp.o (14 symbol(s), format=elf) + clean tokenizer.cpp.o (50 symbol(s), format=elf) + clean model.cpp.o (113 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (21 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (134 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 574 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 378 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). + +######## scan-build90-superslm_avx512_forced +Scanning 17 object member(s) of archive build/libsuperslm_avx512_forced.a for target 'superslm_avx512_forced' (isa=x86-64) + clean artifact.cpp.o (35 symbol(s), format=elf) + clean sha256.cpp.o (14 symbol(s), format=elf) + clean tokenizer.cpp.o (50 symbol(s), format=elf) + clean model.cpp.o (113 symbol(s), format=elf) + clean intmath.cpp.o (30 symbol(s), format=elf) + clean silu_lut.cpp.o (1 symbol(s), format=elf) + clean matmul.cpp.o (23 symbol(s), format=elf) + clean proof_manifest.cpp.o (47 symbol(s), format=elf) + clean trace_hook.cpp.o (4 symbol(s), format=elf) + clean checked_chain_funnel.cpp.o (14 symbol(s), format=elf) + clean forward_sites.cpp.o (61 symbol(s), format=elf) + clean decode_digest.cpp.o (7 symbol(s), format=elf) + clean sslm_abi.cpp.o (134 symbol(s), format=elf) + clean damped_greedy_antilm.cpp.o (20 symbol(s), format=elf) + clean damped_greedy_topk.cpp.o (13 symbol(s), format=elf) + clean damped_greedy_phaseD.cpp.o (8 symbol(s), format=elf) + clean damped_greedy_phaseD_loop.cpp.o (2 symbol(s), format=elf) +Totals: 17 object(s); 576 symbol(s) ACCEPT, 0 REJECT, 0 object(s) REFUSE (checks (A)/(B), gating); 380 symbol(s) reject under check (C) alone (non-gating diagnostic) +PASS: no floating-point arithmetic found in any object of this target (checks (A)/(B); check (C) is a non-gating diagnostic). diff --git a/docs/attention-rowsites/s5/golden.txt b/docs/attention-rowsites/s5/golden.txt new file mode 100644 index 00000000..bc2e6f76 --- /dev/null +++ b/docs/attention-rowsites/s5/golden.txt @@ -0,0 +1,17 @@ +# S5 golden pin provenance (plan §3.3 evidence 3, cell 6.3) +# tools/gen_attn_rowsite_golden.cpp (one hash per slice) compiled with GCC 13.3 -O2 against the v1.9.0 tag (d870d27) include/ +# and its Release libsuperslm.a (the recipe in the generator's header): +S1 row-table golden: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 over 634120 values +S2 prob-V golden: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed over 30100 values +S3 requant-row golden: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 over 3567018 values +S4 softmax golden: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 over 268078 values +S5 Q31-row golden: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 over 33618 values +S5 QK-norm fixture golden: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 over 14384 values (status 0) +# Re-running the v1.9.0 build with the header path as argument wrote tests/attn_rowsite_golden_pin.h; S1's to S4's hashes and +# value counts are unchanged, and S5's two are added beside them: +# - the Q31-row hash: tests/support/attention_cases.h's RunQ31Cases (258 calls; per call its width, head_dim and scores), +# driven through the v1.9.0 per-key QkQ31Score, since v1.9.0 has no row entry; the suite and the digest drive the same +# set through QkQ31ScoreRow. Every ratio in the set is in [1, 2^31] (§3.3). +# - the fixture hash: tests/support/qk_attention_fixture.h (cell 11.1(c)) through the v1.9.0 decode loop, 24 positions; +# per position the layer's 256 output codes and its scale (m, e), then the 8,192 K/V workspace bytes. +# The CMake target of the same name, built against this tree, prints the same six hashes (consistency, not provenance). diff --git a/docs/attention-rowsites/s5/linkage-plant.txt b/docs/attention-rowsites/s5/linkage-plant.txt new file mode 100644 index 00000000..d8dccbf7 --- /dev/null +++ b/docs/attention-rowsites/s5/linkage-plant.txt @@ -0,0 +1,30 @@ +# S5 cell 11.3 vitality: the linkage checker over all nine objects (auto and both forced builds' matmul.cpp.o, intmath.cpp.o and +# forward_sites.cpp.o). As in S4 a moved body would stay local (its parameter type Q31RowLimbs is in the anonymous namespace), +# so the plant is an external target-attributed function in the population (Q31RoundAvx2Planted). +## plant +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o: 3 population symbols, record x1 +build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o: 5 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm.dir/src/forward/forward_sites.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/forward/forward_sites.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/forward/forward_sites.cpp.o: 4 population symbols, record x0 +check_tiled_matmul_linkage: FAIL + build/CMakeFiles/superslm.dir/src/forward/forward_sites.cpp.o: 'superslm::Q31RoundAvx2Planted(long long __vector(4))' is not local (nm type 'T') + build/CMakeFiles/superslm_avx2_forced.dir/src/forward/forward_sites.cpp.o: 'superslm::Q31RoundAvx2Planted(long long __vector(4))' is not local (nm type 'T') + build/CMakeFiles/superslm_avx512_forced.dir/src/forward/forward_sites.cpp.o: 'superslm::Q31RoundAvx2Planted(long long __vector(4))' is not local (nm type 'T') +exit 1 +## restored (the implementation commit) +build/CMakeFiles/superslm.dir/src/matmul.cpp.o: 7 population symbols, record x1 +build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o: 3 population symbols, record x1 +build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o: 5 population symbols, record x1 +build/CMakeFiles/superslm.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o: 4 population symbols, record x0 +build/CMakeFiles/superslm.dir/src/forward/forward_sites.cpp.o: 3 population symbols, record x0 +build/CMakeFiles/superslm_avx2_forced.dir/src/forward/forward_sites.cpp.o: 3 population symbols, record x0 +build/CMakeFiles/superslm_avx512_forced.dir/src/forward/forward_sites.cpp.o: 3 population symbols, record x0 +check_tiled_matmul_linkage: OK +exit 0 diff --git a/docs/attention-rowsites/s5/mutants.txt b/docs/attention-rowsites/s5/mutants.txt new file mode 100644 index 00000000..3255e43b --- /dev/null +++ b/docs/attention-rowsites/s5/mutants.txt @@ -0,0 +1,309 @@ +# S5 mutation evidence (plan §9; GCC 13.3 Release; this host has AVX-512BW, so the auto binary dispatches AVX-512), run at the +# implementation commit. Each mutant is applied to a clone of that commit by its script (mutation-scripts/.py, run from the +# repository root). superslm_tests (auto), superslm_tests_avx2_forced and superslm_tests_avx512_forced are then rebuilt and run from +# the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT set (p05_l1, sha256 f0fd4886...6ed3), the three at once, each with its +# own TMPDIR. 'none' is the unmutated control. The raw log follows; it shows at most four distinct failing lines per run. +# +# Body mutants are applied to each tier's body separately (§9 preamble): an AVX2-body mutant must die on forced AVX2 (and on auto +# on a runner without AVX-512) and survives here on the two AVX-512 binaries, which never run that body; an AVX-512-body mutant must +# die on forced AVX-512 and on auto and survives on forced AVX2. Guard, limb, dispatcher, counter and call-site mutants sit in code +# both tiers share and die on all three. +# +# Summary (K n = killed, n failures in the attn-rowsites cells; s = survives): +# mutant auto(AVX-512) forced AVX2 forced AVX-512 §9 row / killing cell +# none s s s +# all_fallback_increment_deleted K 88 K 88 K 88 all: fallback increment deleted / 4.S5 grid, 2.S5, 7.S5b (counter) +# all_fast_increment_before_guard K 88 K 88 K 88 all: fast increment before the guard / same (counter) +# all_avx512_runs_avx2_body K 270 s K 270 all: AVX-512 dispatch runs the AVX2 body / every fast row (counter) +# s5_a2_logical_shift s s s S5: a2 by logical shift / EQUIVALENT (see below) +# s5_limbs_16bit K 223 K 223 K 223 S5: limbs of 16 bits / 4.S5 grid, 7.S5b-d (output) +# s5_hd_le520 K 63 K 63 K 63 S5: head_dim <= 520 / 4.S5 hd 513 (counter), 7.S5b hd 516 (output) +# s5_hd_le511 K 38 K 38 K 38 S5: head_dim <= 511 / 4.S5 and 7.S5b hd 512 (counter) +# s5_ties_up_avx2 s K 3 s S5: ties toward +inf, AVX2 body / 7.S5c (output), 6.3 +# s5_ties_up_avx512 K 3 s K 3 S5: ties toward +inf, AVX-512 body / 7.S5c (output), 6.3 +# s5_ratio_guard_dropped K 51 K 51 K 51 S5: ratio guard dropped / 2.S5 (output and counter) +# s5_ratio_lt_2p31 K 195 K 195 K 195 S5: ratio guard -> < 2^31 / 4.S5 ratio 2^31, 7.S5d, 11.1(c) (counter) +# s5_decode_per_key K 48 K 48 K 48 S5: decode loop keeps per-key loop / 11.1(c) runs (i), (iii) (counter) +# s5_chunk_per_key K 1 K 1 K 1 S5: chunk loop keeps per-key loop / 11.1(c) run (ii) (counter) +# x_always_fallback K 270 K 270 K 270 extra: always fall back (counter) +# x_a1_weight_avx2 s K 154 s extra: a1 recombined at 2^16, AVX2 body (output) +# x_a1_weight_avx512 K 154 s K 154 extra: a1 recombined at 2^16, AVX-512 body (output) +# x_pair_dropped_avx2 s K 229 s extra: pair sums not added, AVX2 body (output) +# x_pair_dropped_avx512 K 228 s K 228 extra: high pair sum not added, AVX-512 body (output) +# x_floor_unbiased_avx2 s K 217 s extra: floor by unbiased logical shift, AVX2 body (output) +# x_floor_logical_avx512 K 217 s K 217 extra: floor by logical shift, AVX-512 body (output) +# x_second_vector_dropped_avx2 s K 194 s extra: keys 4..7 of a block dropped, AVX2 body (output) +# x_second_vector_dropped_avx512 K 146 s K 146 extra: keys 8..15 of a block dropped, AVX-512 body (output) +# x_pack_tail_dropped K 97 K 97 K 97 extra: packer channel tail (head_dim % 16) dropped (output) +# +# All 11 killable §9 rows (3 all-slice, 8 S5; 12 scripts, the ties row once per tier) are killed on every binary that runs the +# code they mutate, and all 10 extras likewise. The killing cells are the ones §9 names: 7.S5b's hd 516 corner reads 65538 against 0 under "<= 520" (the int32 +# lane overflows); "<= 511" moves hd 512's grid and corner rows to the fallback counter; the ties mutants flip 7.S5c's -1/2 rows +# (and the S5 golden); "ratio guard dropped" differs on 2.S5's 2^48 rows and moves their counter; "< 2^31" moves every ratio-2^31 +# row, 7.S5d and 11.1(c) to the fallback counter; the call-site mutants leave 11.1(c)'s per-position q31_row deltas at zero, run (i) +# and (iii) for decode (48 positions x heads), run (ii) for chunk. +# +# "a2 by logical shift" SURVIVES on all three, by construction: it is equivalent in this kernel (plan error, recorded in the +# progress file). a2 is stored as an int16 lane (vpmaddwd's operand), and the low 16 bits of w >> 30 are bits 30..45 of w whichever +# shift forms it; the two shifts differ only in bits 34..63 of the int64 result, which the int16 store discards. |w| < 2^39 inside +# the guard, so the int16 value is a2 exactly in both cases. §9's killing input (negative q) cannot distinguish them; no input can. + +== none on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== none on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 270 failures|superslm tests: 117298 checks, 270 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 +== all_avx512_runs_avx2_body on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== all_avx512_runs_avx2_body on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 270 failures|superslm tests: 117256 checks, 270 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 +== all_fallback_increment_deleted on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117298 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+ +== all_fallback_increment_deleted on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117256 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+ +== all_fallback_increment_deleted on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117256 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+ +== all_fast_increment_before_guard on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117298 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+ +== all_fast_increment_before_guard on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117256 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+ +== all_fast_increment_before_guard on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 88 failures|superslm tests: 117256 checks, 88 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +1; want +0/+0/+ +== s5_a2_logical_shift on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== s5_a2_logical_shift on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== s5_a2_logical_shift on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== s5_chunk_per_key on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 1 failures|superslm tests: 117298 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (ii), 24 positions: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+96/+0 (kernel: AVX-512) +== s5_chunk_per_key on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 1 failures|superslm tests: 117256 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (ii), 24 positions: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +96/+0/+0/+0 (kernel: AVX2) +== s5_chunk_per_key on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 1 failures|superslm tests: 117256 checks, 1 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (ii), 24 positions: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+96/+0 (kernel: AVX-512) +== s5_decode_per_key on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 48 failures|superslm tests: 117298 checks, 48 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fold tripl +== s5_decode_per_key on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 48 failures|superslm tests: 117256 checks, 48 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +4/+0/+0/+0 (kernel: AVX2) +== s5_decode_per_key on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 48 failures|superslm tests: 117256 checks, 48 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpreflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fold tripl +== s5_hd_le511 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 38 failures|superslm tests: 117298 checks, 38 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-51 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 +== s5_hd_le511 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 38 failures|superslm tests: 117256 checks, 38 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 +== s5_hd_le511 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 38 failures|superslm tests: 117256 checks, 38 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-51 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 +== s5_hd_le520 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 63 failures|superslm tests: 117298 checks, 63 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (65538 vs 0) +== s5_hd_le520 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 63 failures|superslm tests: 117256 checks, 63 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (65538 vs 0) +== s5_hd_le520 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 63 failures|superslm tests: 117256 checks, 63 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kernel: AV +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 516, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (65538 vs 0) +== s5_limbs_16bit on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 223 failures|superslm tests: 117298 checks, 223 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13563 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (131074 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 8 of 24 scores differ from QkQ31ScoreScalarRef, first at key 0 (0 vs 1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (-1005 vs 5) +== s5_limbs_16bit on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 223 failures|superslm tests: 117256 checks, 223 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13563 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (131074 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 8 of 24 scores differ from QkQ31ScoreScalarRef, first at key 0 (0 vs 1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (-1005 vs 5) +== s5_limbs_16bit on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 223 failures|superslm tests: 117256 checks, 223 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13563 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (131074 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 8 of 24 scores differ from QkQ31ScoreScalarRef, first at key 0 (0 vs 1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (-1005 vs 5) +== s5_ratio_guard_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 51 failures|superslm tests: 117298 checks, 51 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17): 17 of 17 scores differ from this binary's per-key QkQ31Score, first at key 0 (13533 vs 13193) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (ker +FAIL preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (Layer +== s5_ratio_guard_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 51 failures|superslm tests: 117256 checks, 51 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17): 17 of 17 scores differ from this binary's per-key QkQ31Score, first at key 0 (13533 vs 13193) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +1 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (ker +== s5_ratio_guard_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 51 failures|superslm tests: 117256 checks, 51 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17): 17 of 17 scores differ from this binary's per-key QkQ31Score, first at key 0 (13533 vs 13193) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +1 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (ker +FAIL preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (Layer +== s5_ratio_lt_2p31 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 195 failures|superslm tests: 117298 checks, 195 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +4; want +0/+0/+4/+0 (kernel: AVX-512) +== s5_ratio_lt_2p31 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 195 failures|superslm tests: 117256 checks, 195 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +4 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +4/+0/+0/+0 (kernel: AVX2) +== s5_ratio_lt_2p31 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 195 failures|superslm tests: 117256 checks, 195 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +4; want +0/+0/+4/+0 (kernel: AVX-512) +== s5_ties_up_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== s5_ties_up_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 3 failures|superslm tests: 117256 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 4 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (0 vs -1) +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: 81e8140d208c1bbafb4ef5570256959287d0bddfc8850e47d222286cdacbfb27 ov +== s5_ties_up_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== s5_ties_up_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 3 failures|superslm tests: 117298 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 4 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (0 vs -1) +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: 81e8140d208c1bbafb4ef5570256959287d0bddfc8850e47d222286cdacbfb27 ov +== s5_ties_up_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== s5_ties_up_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 3 failures|superslm tests: 117256 checks, 3 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 4 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (0 vs -1) +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: 81e8140d208c1bbafb4ef5570256959287d0bddfc8850e47d222286cdacbfb27 ov +== x_a1_weight_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== x_a1_weight_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 154 failures|superslm tests: 117256 checks, 154 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13341 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-32767 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 20 of 24 scores differ from QkQ31ScoreScalarRef, first at key 2 (3 vs 2) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (257 vs 5) +== x_a1_weight_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_a1_weight_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 154 failures|superslm tests: 117298 checks, 154 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13341 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-32767 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 20 of 24 scores differ from QkQ31ScoreScalarRef, first at key 2 (3 vs 2) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (257 vs 5) +== x_a1_weight_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_a1_weight_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 154 failures|superslm tests: 117256 checks, 154 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-13341 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-32767 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 20 of 24 scores differ from QkQ31ScoreScalarRef, first at key 2 (3 vs 2) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (257 vs 5) +== x_always_fallback on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 270 failures|superslm tests: 117298 checks, 270 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 +== x_always_fallback on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 270 failures|superslm tests: 117256 checks, 270 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +1 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 +== x_always_fallback on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 270 failures|superslm tests: 117256 checks, 270 failures| +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-5 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +1; want +0/+0/+1/+0 +== x_floor_logical_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91687 checks, 217 failures|superslm tests: 117218 checks, 217 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589921217 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589934592 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 12 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589934591 vs -1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 4 of 9 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589902722 vs -31870) +== x_floor_logical_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_floor_logical_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91687 checks, 217 failures|superslm tests: 117176 checks, 217 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589921217 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589934592 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 12 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589934591 vs -1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 4 of 9 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589902722 vs -31870) +== x_floor_unbiased_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== x_floor_unbiased_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91687 checks, 217 failures|superslm tests: 117176 checks, 217 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589921217 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (8589934592 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 12 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589934591 vs -1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 4 of 9 scores differ from QkQ31ScoreScalarRef, first at key 1 (8589902722 vs -31870) +== x_floor_unbiased_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_pack_tail_dropped on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 97 failures|superslm tests: 117298 checks, 97 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-579091 vs -13375) +FAIL preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: 0fef792befd4af01fdc462c656d5951db450b4e3f4b82c769d6965bfec78bb10 ov +== x_pack_tail_dropped on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 97 failures|superslm tests: 117256 checks, 97 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (0 vs -13375) +FAIL preflight: 1/1 layers carry a non-degenerate (per-output-channel) WSC1 fold tensor on at least one of q/k/v/o/gate/up/down_proj; worst case 4864 rows in a single tensor (LayerWeights now carries one fold triple per +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: a161360192ea2c0c1d8e4b26d120a92d0c3ec2dd06b948cd553df1673b0f6136 ov +== x_pack_tail_dropped on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 97 failures|superslm tests: 117256 checks, 97 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (0 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1686: hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && values == superslm_test::kAttnRowsiteS5GoldenValues -- 6.3 S5 golden: c17f2f9ff8e4e5a264068a0d528b4c4bed4a7bd0e9c908738cbbbc5703890f12 ov +== x_pair_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== x_pair_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91743 checks, 229 failures|superslm tests: 117232 checks, 229 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-2939 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-16384 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 16 of 24 scores differ from QkQ31ScoreScalarRef, first at key 1 (3 vs -1) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (32132 vs 5) +== x_pair_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_pair_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91751 checks, 228 failures|superslm tests: 117282 checks, 228 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-2939 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-16384 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 8 of 24 scores differ from QkQ31ScoreScalarRef, first at key 12 (1 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (32132 vs 5) +== x_pair_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_pair_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91751 checks, 228 failures|superslm tests: 117240 checks, 228 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-2939 vs -13375) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (-16384 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 8 of 24 scores differ from QkQ31ScoreScalarRef, first at key 12 (1 vs 0) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 9 of 9 scores differ from QkQ31ScoreScalarRef, first at key 0 (32132 vs 5) +== x_second_vector_dropped_avx2 on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| +== x_second_vector_dropped_avx2 on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 194 failures|superslm tests: 117256 checks, 194 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 7): 3 of 7 scores differ from QkQ31ScoreScalarRef, first at key 4 (0 vs 1931) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 17): 8 of 17 scores differ from QkQ31ScoreScalarRef, first at key 4 (0 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 9 of 24 scores differ from QkQ31ScoreScalarRef, first at key 4 (0 vs 3) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 4 of 9 scores differ from QkQ31ScoreScalarRef, first at key 4 (0 vs 5) +== x_second_vector_dropped_avx2 on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_second_vector_dropped_avx512 on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 146 failures|superslm tests: 117298 checks, 146 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 9): 1 of 9 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs 3389) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 17): 8 of 17 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 5 of 24 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -64) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 1 of 9 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -32380) +== x_second_vector_dropped_avx512 on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| +== x_second_vector_dropped_avx512 on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 146 failures|superslm tests: 117256 checks, 146 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 grid, uniform (head_dim 4, width 9): 1 of 9 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs 3389) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5b margin corner, w = 2^30 - 1, key fill (head_dim 512, width 17): 8 of 17 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -32768) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5c ties, q = +1 (head_dim 64, width 24): 5 of 24 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -64) +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 7.S5d inside corner (head_dim 64, width 9): 1 of 9 scores differ from QkQ31ScoreScalarRef, first at key 8 (0 vs -32380) +ALL-DONE + +######## Added after the coverage replica (the 4.S5 channel-tail cell, plan §3.4 step 1), same harness, the suite with that cell: +# x_pads_nonzero K 90 K 90 K 90 extra: both channel pads nonzero (output; 4.S5 channel tail only) +# Either pad alone is output-equivalent (a padded channel multiplies a zero from the other side), so the mutant sets both. All 90 +# failures are channel-tail rows: no other cell reaches a head_dim % 4 != 0 row inside the guard, so the suite before that cell +# could not kill it. +== none on superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures|superslm tests: 117479 checks, 0 failures| +== none on superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures|superslm tests: 117437 checks, 0 failures| +== none on superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures|superslm tests: 117437 checks, 0 failures| +== x_pads_nonzero on superslm_tests: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 90 failures|superslm tests: 117479 checks, 90 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 channel tail (head_dim 1, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (194598 vs -474) +== x_pads_nonzero on superslm_tests_avx2_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 90 failures|superslm tests: 117437 checks, 90 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 channel tail (head_dim 1, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (194598 vs -474) +== x_pads_nonzero on superslm_tests_avx512_forced: exit 1; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 90 failures|superslm tests: 117437 checks, 90 failures| +FAIL tests/test_attn_rowsites.cpp:1538: bad == 0 -- 4.S5 channel tail (head_dim 1, width 1): 1 of 1 scores differ from QkQ31ScoreScalarRef, first at key 0 (194598 vs -474) +ALL-DONE diff --git a/docs/attention-rowsites/s5/mutation-scripts/all_avx512_runs_avx2_body.py b/docs/attention-rowsites/s5/mutation-scripts/all_avx512_runs_avx2_body.py new file mode 100644 index 00000000..80631e97 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/all_avx512_runs_avx2_body.py @@ -0,0 +1,11 @@ +# Dispatch: the AVX-512 kernel runs the AVX2 body. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' QkQ31RowAvx512(keys, head_dim, width, limbs, out);', ' QkQ31RowAvx2(keys, head_dim, width, limbs, out); /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/all_fallback_increment_deleted.py b/docs/attention-rowsites/s5/mutation-scripts/all_fallback_increment_deleted.py new file mode 100644 index 00000000..fe20ab2d --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/all_fallback_increment_deleted.py @@ -0,0 +1,14 @@ +# All slices: the fallback increment deleted. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(''' (kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_q31_row_fallback_avx2 + : superslm_test::g_q31_row_fallback_avx512) + .fetch_add(1, std::memory_order_relaxed); +''', ' /* MUTANT */\n') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/all_fast_increment_before_guard.py b/docs/attention-rowsites/s5/mutation-scripts/all_fast_increment_before_guard.py new file mode 100644 index 00000000..2243bcd6 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/all_fast_increment_before_guard.py @@ -0,0 +1,19 @@ +# All slices: the fast increment moved from the bodies to before the guard. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('\tsuperslm_test::g_q31_row_fast_avx2.fetch_add(1, std::memory_order_relaxed);\n', '\t/* MUTANT */\n') +sub('\tsuperslm_test::g_q31_row_fast_avx512.fetch_add(1, std::memory_order_relaxed);\n', '\t/* MUTANT */\n') +sub(''' if (kernel != detail::SitesKernel::kShipped) { + if (Q31RowFastPathAdmits(''', ''' if (kernel != detail::SitesKernel::kShipped) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + (kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_q31_row_fast_avx2 : superslm_test::g_q31_row_fast_avx512) + .fetch_add(1, std::memory_order_relaxed); /* MUTANT */ +#endif + if (Q31RowFastPathAdmits(''') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_a2_logical_shift.py b/docs/attention-rowsites/s5/mutation-scripts/s5_a2_logical_shift.py new file mode 100644 index 00000000..9d1fd2eb --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_a2_logical_shift.py @@ -0,0 +1,11 @@ +# S5: a2 by logical shift. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('w >> 30};', 'static_cast(static_cast(w) >> 30)}; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_chunk_per_key.py b/docs/attention-rowsites/s5/mutation-scripts/s5_chunk_per_key.py new file mode 100644 index 00000000..14f9b0f6 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_chunk_per_key.py @@ -0,0 +1,14 @@ +# S5: the chunk loop keeps its per-key loop. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(''' QkQ31ScoreRow(q_rot.data() + h * head_dim, k_rows_base, ratio, head_dim, width, + scores.data());''', ''' for (size_t row = 0; row < width; ++row) /* MUTANT */ + scores[row] = QkQ31Score(q_rot.data() + h * head_dim, + k_rows_base + row * head_dim, ratio, head_dim);''', 1, start='SslmForwardStatus RunLayerLoopChunkBatched(') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_decode_per_key.py b/docs/attention-rowsites/s5/mutation-scripts/s5_decode_per_key.py new file mode 100644 index 00000000..38a8b3a7 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_decode_per_key.py @@ -0,0 +1,13 @@ +# S5: the decode loop keeps its per-key QkQ31Score loop. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' QkQ31ScoreRow(q_rot.data() + h * head_dim, k_rows_base, ratio, head_dim, width, scores.data());', ''' for (size_t row = 0; row < width; ++row) /* MUTANT */ + scores[row] = QkQ31Score(q_rot.data() + h * head_dim, + k_rows_base + row * head_dim, ratio, head_dim);''', 1, start='static SslmForwardStatus RunLayerLoopImpl(') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le511.py b/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le511.py new file mode 100644 index 00000000..b1ed438d --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le511.py @@ -0,0 +1,11 @@ +# S5: head_dim <= 512 -> <= 511. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('if (head_dim > kQ31RowMaxHeadDim) return false;', 'if (head_dim > 511) return false; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le520.py b/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le520.py new file mode 100644 index 00000000..51f31044 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_hd_le520.py @@ -0,0 +1,11 @@ +# S5: head_dim <= 512 -> <= 520 (the constant, so the pack and limb buffers grow with it and only the guard moves). +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('constexpr size_t kQ31RowMaxHeadDim = 512;', 'constexpr size_t kQ31RowMaxHeadDim = 520; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_limbs_16bit.py b/docs/attention-rowsites/s5/mutation-scripts/s5_limbs_16bit.py new file mode 100644 index 00000000..d5fb75bf --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_limbs_16bit.py @@ -0,0 +1,13 @@ +# S5: limbs of 16 bits (a0 < 2^16): a0 = w & 0xFFFF, a1 = (w >> 16) & 0xFFFF, a2 = w >> 32, recombined at 2^32 and 2^16 in both bodies. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const int64_t a[3] = {w & 0x7FFF, (w >> 15) & 0x7FFF, w >> 30};', 'const int64_t a[3] = {w & 0xFFFF, (w >> 16) & 0xFFFF, w >> 32}; /* MUTANT */') +sub('_mm256_slli_epi64(w[2], 30), _mm256_slli_epi64(w[1], 15)', '_mm256_slli_epi64(w[2], 32), _mm256_slli_epi64(w[1], 16) /* MUTANT */') +sub('_mm512_slli_epi64(w[2], 30), _mm512_slli_epi64(w[1], 15)', '_mm512_slli_epi64(w[2], 32), _mm512_slli_epi64(w[1], 16) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_guard_dropped.py b/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_guard_dropped.py new file mode 100644 index 00000000..9a52a0f9 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_guard_dropped.py @@ -0,0 +1,11 @@ +# S5: the ratio guard dropped. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' if (ratio[d] < 0 || ratio[d] > INT64_C(0xFFFFFFFF)) return false;\n', ' (void)ratio[d]; /* MUTANT */\n') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_lt_2p31.py b/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_lt_2p31.py new file mode 100644 index 00000000..c3ab3ab8 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_ratio_lt_2p31.py @@ -0,0 +1,11 @@ +# S5: the ratio guard -> < 2^31. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('ratio[d] > INT64_C(0xFFFFFFFF)', 'ratio[d] > INT64_C(0x7FFFFFFF) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx2.py b/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx2.py new file mode 100644 index 00000000..2659946d --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx2.py @@ -0,0 +1,11 @@ +# S5: ties toward +inf, AVX2 body (the threshold loses its +1 for negative x). +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_add_epi64(_mm256_set1_epi64x(INT64_C(0x3FFFFFFF)), _mm256_srli_epi64(x, 63))', '_mm256_set1_epi64x(INT64_C(0x3FFFFFFF)) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx512.py b/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx512.py new file mode 100644 index 00000000..16e72ae5 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/s5_ties_up_avx512.py @@ -0,0 +1,11 @@ +# S5: ties toward +inf, AVX-512 body. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_add_epi64(_mm512_set1_epi64(INT64_C(0x3FFFFFFF)), _mm512_srli_epi64(x, 63))', '_mm512_set1_epi64(INT64_C(0x3FFFFFFF)) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx2.py b/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx2.py new file mode 100644 index 00000000..2209cf5b --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx2.py @@ -0,0 +1,11 @@ +# Extra, AVX2 body: a1 recombined at 2^16. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_slli_epi64(w[1], 15)', '_mm256_slli_epi64(w[1], 16) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx512.py b/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx512.py new file mode 100644 index 00000000..60928be3 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_a1_weight_avx512.py @@ -0,0 +1,11 @@ +# Extra, AVX-512 body: a1 recombined at 2^16. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm512_slli_epi64(w[1], 15)', '_mm512_slli_epi64(w[1], 16) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_always_fallback.py b/docs/attention-rowsites/s5/mutation-scripts/x_always_fallback.py new file mode 100644 index 00000000..c12a39a1 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_always_fallback.py @@ -0,0 +1,11 @@ +# Extra: always fall back. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' if (Q31RowFastPathAdmits(ratio, head_dim)) {', ' if (false && Q31RowFastPathAdmits(ratio, head_dim)) { /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_floor_logical_avx512.py b/docs/attention-rowsites/s5/mutation-scripts/x_floor_logical_avx512.py new file mode 100644 index 00000000..0d5c33d3 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_floor_logical_avx512.py @@ -0,0 +1,11 @@ +# Extra, AVX-512 body: floor by a logical shift. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m512i floor = _mm512_srai_epi64(x, 31);', 'const __m512i floor = _mm512_srli_epi64(x, 31); /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_floor_unbiased_avx2.py b/docs/attention-rowsites/s5/mutation-scripts/x_floor_unbiased_avx2.py new file mode 100644 index 00000000..ede0ce9b --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_floor_unbiased_avx2.py @@ -0,0 +1,12 @@ +# Extra, AVX2 body: floor by a logical shift without the bias (wrong for negative x). +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_set1_epi64x(INT64_C(1) << 62)', '_mm256_setzero_si256() /* MUTANT */', 1, start='inline __m256i Q31RoundAvx2(', end='void QkQ31RowAvx2(') +sub('_mm256_set1_epi64x(INT64_C(1) << 31))', '_mm256_setzero_si256())', 1, start='inline __m256i Q31RoundAvx2(', end='void QkQ31RowAvx2(') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_pack_tail_dropped.py b/docs/attention-rowsites/s5/mutation-scripts/x_pack_tail_dropped.py new file mode 100644 index 00000000..cf66d2e6 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_pack_tail_dropped.py @@ -0,0 +1,11 @@ +# Extra, shared packer: the scalar channel tail (head_dim % 16) not packed (the stale pack is used). +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub(' for (size_t d = full16; d < nq * 4; ++d)', ' for (size_t d = nq * 4; d < nq * 4; ++d) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_pads_nonzero.py b/docs/attention-rowsites/s5/mutation-scripts/x_pads_nonzero.py new file mode 100644 index 00000000..c0504b37 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_pads_nonzero.py @@ -0,0 +1,11 @@ +# Extra (coverage vitality, shared code): both channel pads nonzero -- the limbs' padded channels carry w = 2^40 and the +# key pack's padded channels 127. Either pad alone is output-equivalent (a padded channel multiplies a zero from the +# other side); together they add 127 * 2^40 per padded channel. Only rows with head_dim % 4 != 0 reach them. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new): + global s + assert s.count(old) == 1, old + s = s.replace(old, new) +sub('const int64_t w = d < head_dim ? static_cast(q[d]) * ratio[d] : 0;', 'const int64_t w = d < head_dim ? static_cast(q[d]) * ratio[d] : (INT64_C(1) << 40); /* MUTANT */') +sub('base[(d / 4) * stride + t * 4 + d % 4] = d < head_dim ? rows[t][d] : int16_t{0};', 'base[(d / 4) * stride + t * 4 + d % 4] = d < head_dim ? rows[t][d] : int16_t{127}; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx2.py b/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx2.py new file mode 100644 index 00000000..c24a2295 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx2.py @@ -0,0 +1,11 @@ +# Extra, AVX2 body: the pair sums not added (hadd -> unpacklo, only one pair per key). +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('_mm256_hadd_epi32(acc[l][0], acc[l][1])', '_mm256_unpacklo_epi32(acc[l][0], acc[l][1]) /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx512.py b/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx512.py new file mode 100644 index 00000000..1a19c6eb --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_pair_dropped_avx512.py @@ -0,0 +1,11 @@ +# Extra, AVX-512 body: the high pair sum not added. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('const __m512i pair = _mm512_add_epi32(acc[l][h], _mm512_srli_epi64(acc[l][h], 32));', 'const __m512i pair = acc[l][h]; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx2.py b/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx2.py new file mode 100644 index 00000000..c910d984 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx2.py @@ -0,0 +1,11 @@ +# Extra, AVX2 body: the second vector of each block (keys 4..7) not accumulated. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('acc[l][1] = _mm256_add_epi32(acc[l][1], _mm256_madd_epi16(k1, b));', '(void)k1; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx512.py b/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx512.py new file mode 100644 index 00000000..24cbcea4 --- /dev/null +++ b/docs/attention-rowsites/s5/mutation-scripts/x_second_vector_dropped_avx512.py @@ -0,0 +1,11 @@ +# Extra, AVX-512 body: the second vector of each block (keys 8..15) not accumulated. +p = "src/forward/forward_sites.cpp"; s = open(p).read() +def sub(old, new, count=1, start=None, end=None): + global s + i = s.index(start) if start else 0 + j = s.index(end, i) if end else len(s) + region = s[i:j] + assert region.count(old) == count, (old, region.count(old)) + s = s[:i] + region.replace(old, new) + s[j:] +sub('acc[l][1] = _mm512_add_epi32(acc[l][1], _mm512_madd_epi16(k1, b));', '(void)k1; /* MUTANT */') +open(p, "w").write(s) diff --git a/docs/attention-rowsites/s5/probe_q31_forward.cpp b/docs/attention-rowsites/s5/probe_q31_forward.cpp new file mode 100644 index 00000000..03a128c8 --- /dev/null +++ b/docs/attention-rowsites/s5/probe_q31_forward.cpp @@ -0,0 +1,63 @@ +// S5 forward probe (bench.md, reading 2). Not built by CMake. fx.h is tests/support/qk_attention_fixture.h with: +// sed -e 's/kHidden = 256/kHidden = 1024/; s/kHeads = 4;/kHeads = 16;/; s/kKvHeads = 2;/kKvHeads = 8;/; +// s/kHeadDim = 64;/kHeadDim = 128;/; s/kInter = 256;/kInter = 3072;/; s/kCap = 32;/kCap = 1024;/; +// s/kPositions = 24;/kPositions = 1024;/; s/superslm_qk_fixture/q31fwd/g; s#"\.\./sslm_#"sslm_#' +// -e 's/hidden_gain.assign(kHidden, 8192);/hidden_gain.assign(kHidden, HG);/; s/rng.InRange(4096, 8192));/rng.InRange(QKLO, QKHI));/g' +// (plus the guard rename and the other constants made overridable, all left at the fixture's values). +// Built with: g++ -O3 -DNDEBUG -std=gnu++20 -ffp-contract=off -DHG=4096 -DQKLO=2048 -DQKHI=4096 -I/include -Itests +// and linked against the S4 head's or S5's libsuperslm.a / libsuperslm_avx2_forced.a. Usage: probe . +#include +#include +#include +#ifndef KVE +#define KVE 12 +#endif +#ifndef KCE +#define KCE -36 +#endif +#ifndef SME +#define SME -72 +#endif +#ifndef GSE +#define GSE -52 +#endif +#ifndef KWE +#define KWE -60 +#endif +#ifndef ACTE +#define ACTE INT64_C(-96) +#endif +#ifndef HG +#define HG 8192 +#endif +#ifndef QKLO +#define QKLO 4096 +#define QKHI 8192 +#endif +#include "fx.h" +using namespace q31fwd; +int main(int argc, char** argv) { + const size_t T = argc > 1 ? std::atoi(argv[1]) : 128; + const int R = argc > 2 ? std::atoi(argv[2]) : 1; + static QkAttentionFixture f; + if (!f.loaded) { std::printf("load failed: %s\n", f.load_error.c_str()); return 2; } + double best = 1e300; int st = 0; uint64_t h = 1469598103934665603ULL; + for (int r = 0; r < R; ++r) { + std::vector workspace(kWorkspaceBytes, 0); + std::vector chunk(f.hidden_in.begin(), f.hidden_in.begin() + T * kHidden); + std::vector scales(f.scale_in.begin(), f.scale_in.begin() + T); + uint64_t sat = 0; + const auto t0 = std::chrono::steady_clock::now(); + st = static_cast(superslm::RunLayerLoopChunkBatched(chunk.data(), scales.data(), T, &f.layer, 1, kHidden, + kHeadDim, kKvHeads, kInter, kCap, 0, f.view.rope_tables, workspace.data(), workspace.size(), false, &sat, {}, + nullptr, kHeads * kHeadDim)); + const auto t1 = std::chrono::steady_clock::now(); + best = std::min(best, std::chrono::duration(t1 - t0).count()); + if (r == 0) { for (auto c : chunk) h = (h ^ static_cast(c)) * 1099511628211ULL; + for (auto s : scales) h = (h ^ static_cast(s.m) ^ (static_cast(s.e) << 40)) * 1099511628211ULL; + int clamp = 0; for (auto c : chunk) clamp += (c == 127 || c == -127 || c == -128); + std::printf("T=%zu status %d sat %llu out-clamped %d/%zu hash %016llx\n", T, st, (unsigned long long)sat, clamp, chunk.size(), (unsigned long long)h); } + } + std::printf("T=%zu best-of-%d %.3f ms total, %.4f ms/token\n", T, R, best, best / T); + return st; +} diff --git a/docs/attention-rowsites/s5/red-sslm_axis_digest.txt b/docs/attention-rowsites/s5/red-sslm_axis_digest.txt new file mode 100644 index 00000000..048dd275 --- /dev/null +++ b/docs/attention-rowsites/s5/red-sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/red-sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s5/red-sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..a6076d9f --- /dev/null +++ b/docs/attention-rowsites/s5/red-sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/red-sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s5/red-sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..705394bc --- /dev/null +++ b/docs/attention-rowsites/s5/red-sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/red-sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s5/red-sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..e439375c --- /dev/null +++ b/docs/attention-rowsites/s5/red-sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/red-sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s5/red-sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..a99c7d17 --- /dev/null +++ b/docs/attention-rowsites/s5/red-sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/red-suites.txt b/docs/attention-rowsites/s5/red-suites.txt new file mode 100644 index 00000000..311adbe4 --- /dev/null +++ b/docs/attention-rowsites/s5/red-suites.txt @@ -0,0 +1,119 @@ +# S5 red run (plan §4 red-first): GCC 13.3 Release, the red commit (cells, q31_row counters declared and never +# incremented, QkQ31ScoreRow declared with a stub that runs the per-key QkQ31Score loop, both layer loops unchanged). +# Suites run from the repository root with SUPERSLM_ATTN_ROWSITES_ARTIFACT= (11.1(d)), each binary with +# its own TMPDIR. +# Every value assertion passes on every binary (scores against QkQ31ScoreScalarRef, or the same binary's per-key +# QkQ31Score for 2.S5; the S5 golden hash; the 7.S5b margin and 7.S5c tie premises; the 2.S5 conjunct counts; the +# 11.1(c) fixture hash on all three runs, its premise and its softmax and prob-V totals; 11.1(d)'s q31_row rows at 0). +# The failures are the q31_row path assertions (nothing increments them): 358 on each binary whose selector picks a +# new kernel, 0 on forced SSE2. +== superslm_tests (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 358 failures +superslm tests: 117298 checks, 358 failures + FAIL lines: 358 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: A +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (ker +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want + +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = +1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = -1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 width 0 (head_dim 64, width 0, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-51 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/ +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 head_dim 513 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kern +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (i) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +== superslm_tests_sse2_forced (red): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures +superslm tests: 117240 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 358 failures +superslm tests: 117256 checks, 358 failures + FAIL lines: 358 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: A +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (ker +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want + +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = +1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = -1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 width 0 (head_dim 64, width 0, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +1/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/ +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 head_dim 513 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+1/+0/+0 (kern +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +4/+0/+0/+0 (kernel: AVX2) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (i) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +4/+0/+0/+0 (kernel: AVX2) +== superslm_tests_avx512_forced (red): exit 1 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 358 failures +superslm tests: 117256 checks, 358 failures + FAIL lines: 358 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, uniform (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: A +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, int8 extremes (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (ker +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 grid, ratio 2^31 (head_dim 4, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5b margin corner, w = -1, key fill (head_dim 512, width 1, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want + +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = +1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5c ties, q = -1 (head_dim 64, width 24, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 7.S5d inside corner (head_dim 64, width 9, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 4.S5 width 0 (head_dim 64, width 0, guard copy: fast): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+1/+0 (kernel: AVX-51 +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 ratio outside [0, 2^32) (head_dim 63, width 17, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/ +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 2.S5 head_dim 513 (head_dim 513, width 1, guard copy: fallback): q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+0/+1 (kern +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (iii) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +FAIL tests/test_attn_rowsites.cpp:1507: ok -- 11.1(c) run (i) position 0: q31_row_fast_avx2 +0 q31_row_fallback_avx2 +0 q31_row_fast_avx512 +0 q31_row_fallback_avx512 +0; want +0/+0/+4/+0 (kernel: AVX-512) +== sslm_axis_digest: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +DONE-red diff --git a/docs/attention-rowsites/s5/sanitizers.txt b/docs/attention-rowsites/s5/sanitizers.txt new file mode 100644 index 00000000..aecb8505 --- /dev/null +++ b/docs/attention-rowsites/s5/sanitizers.txt @@ -0,0 +1,14 @@ +# S5 sanitizer runs (GCC 13.3, RelWithDebInfo; ASan+UBSan with -fno-sanitize-recover=all, and TSan, as the CI legs configure them), +# SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1, separate TMPDIRs. 'reports' counts AddressSanitizer errors, UBSan runtime errors and +# ThreadSanitizer warnings in the run's output. The S5 cells include widths 1-17 and 1,024 (partial key blocks packed with zero +# rows into the stack buffer), head_dim 512 at the int32 margin, and, in the evidence commit's suite, head_dims 1-3 and other +# non-multiples of 4 and 16 (the pack's scalar tail). ASan+UBSan was run on both suites; TSan on the implementation commit's. +## the implementation commit's suite +== build-asan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| reports: 0 +== build-asan/superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117256 checks, 0 failures| reports: 0 +== build-tsan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91767 checks, 0 failures|superslm tests: 117298 checks, 0 failures| reports: 0 +## the evidence commit's suite (with the 4.S5 channel-tail rows) +== build-asan/superslm_tests_avx512_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures reports: 0 +== build-asan/superslm_tests: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures reports: 0 +== build-asan/superslm_tests_avx2_forced: exit 0; attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures reports: 0 diff --git a/docs/attention-rowsites/s5/sslm_axis_digest.txt b/docs/attention-rowsites/s5/sslm_axis_digest.txt new file mode 100644 index 00000000..048dd275 --- /dev/null +++ b/docs/attention-rowsites/s5/sslm_axis_digest.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/sslm_axis_digest_avx2_forced.txt b/docs/attention-rowsites/s5/sslm_axis_digest_avx2_forced.txt new file mode 100644 index 00000000..a6076d9f --- /dev/null +++ b/docs/attention-rowsites/s5/sslm_axis_digest_avx2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX2-forced) +# int64_digits: 64 +# gemm tier: AVX2; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/sslm_axis_digest_avx512_forced.txt b/docs/attention-rowsites/s5/sslm_axis_digest_avx512_forced.txt new file mode 100644 index 00000000..705394bc --- /dev/null +++ b/docs/attention-rowsites/s5/sslm_axis_digest_avx512_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul AVX512-forced) +# int64_digits: 64 +# gemm tier: AVX-512; tiled at M >= 8: yes +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/sslm_axis_digest_scalar_forced.txt b/docs/attention-rowsites/s5/sslm_axis_digest_scalar_forced.txt new file mode 100644 index 00000000..e439375c --- /dev/null +++ b/docs/attention-rowsites/s5/sslm_axis_digest_scalar_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul dispatch runtime-selected: SSE2/AVX2/AVX-512) +# int64_digits: 64 +# gemm tier: scalar; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/sslm_axis_digest_sse2_forced.txt b/docs/attention-rowsites/s5/sslm_axis_digest_sse2_forced.txt new file mode 100644 index 00000000..a99c7d17 --- /dev/null +++ b/docs/attention-rowsites/s5/sslm_axis_digest_sse2_forced.txt @@ -0,0 +1,19 @@ +# compiler: gcc 13.3.0 +# cplusplus: 202002 +# ndebug: 1 +# arch: x86_64 (matmul SSE2-forced) +# int64_digits: 64 +# gemm tier: SSE2; tiled at M >= 8: no +sha256 105c497cff5677e065332e1b6c81aca753bd248f17848686906c4be44dbe9608 values=6496 +c1c2c3_requant 971380367417462803dd256379c766443d7e3e74cdf0e3c63545f72e01737e66 values=54193 +c19c22_dynamic_scale 5ea870a875d9dfb5766d03b983742a5e69a125e976ae2202860b934cd7038242 values=115117 +c4c6_isqrt e78a2cfb60bc393c8ea64866d5c03e50b6baea196e9fd8a92c96677c6c34d814 values=32594 +c7c9_iexp 66cf1fa41fea0b98da8bc05a5d91988391c214aefa8cac87610534d189a08699 values=44588 +c11c13_rope c874e071c1dca2b435efd3d3a25d130607e8077c9664afd9cb4f73c70e036e07 values=60000 +c10_silu_lut 7e7951dab1a2a26a4c52d41968ee78895d1b3a79cd30ddc98c50ddf6aa0c6c43 values=122510 +c17_matmul ee456f50d00f6811f5bb0ecd72355258a0ece1a19575505fcd4e97458c7a2ba3 values=6865 +c17_matmul_tiled aac2f53a87b85ffc881ae2d694701f75771fd01f9c4ce7f373ed4aed98498803 values=9600 +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +local_invariant_failures 0 diff --git a/docs/attention-rowsites/s5/suites-clang.txt b/docs/attention-rowsites/s5/suites-clang.txt new file mode 100644 index 00000000..b5cae049 --- /dev/null +++ b/docs/attention-rowsites/s5/suites-clang.txt @@ -0,0 +1,85 @@ +# S5 suites at the evidence commit, Clang 18.1 Release, as suites.txt. +== superslm_tests (clang2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117479 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (clang2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117421 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (clang2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117437 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (clang2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117437 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +DONE-clang2 diff --git a/docs/attention-rowsites/s5/suites.txt b/docs/attention-rowsites/s5/suites.txt new file mode 100644 index 00000000..910517c0 --- /dev/null +++ b/docs/attention-rowsites/s5/suites.txt @@ -0,0 +1,87 @@ +# S5 suites at the evidence commit (the implementation plus the 4.S5 channel-tail rows), GCC 13.3 Release, run from the repository +# root, SUPERSLM_ATTN_ROWSITES_ARTIFACT = p05_l1, one TMPDIR per binary; then the five digest legs. The implementation commit's +# own run (before the channel-tail rows) was identical apart from the check counts: 0 failures everywhere, the same hashes and digests. +== superslm_tests (green2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117479 checks, 0 failures + FAIL lines: 0 +== superslm_tests_sse2_forced (green2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 1, kernel v1.9.0 code (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117421 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx2_forced (green2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 2, kernel AVX2 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117437 checks, 0 failures + FAIL lines: 0 +== superslm_tests_avx512_forced (green2): exit 0 +S2.4 SiLU-LUT golden hash: 587576aba105a73a74b0dc75763259fb3e24ba170977caaf511440513b1fa5c6 (10200 inputs, 40800 bytes) +S2.5 matmul golden hash: 932478a449091dacf9210e69c5961d3ab6e2915d2fc783e0d19d35f53dd5d9c9 (13 cases, 44244 bytes) +tiled GEMM golden hash: b7c5b06c1ebfa23be0e40ced8e7e409d7a87e8f15ce879d78284d0f99a16710d (41 cases, 127600 bytes) +attn-rowsites S1 golden hash: 8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8 (634120 values) +attn-rowsites S2: tier 3, kernel AVX-512 (switch 0, msvc 0), prob-V counters read +attn-rowsites S2 golden hash: b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed (30100 values) +attn-rowsites S3 4.S3 sentinel pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 4.S3 exact-size pass: 16359 row-leaf calls, 0 with a wrong code or fence, 0 with a wrong path +attn-rowsites S3 golden hash: 3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9 (3567018 values) +attn-rowsites S4 golden hash: 2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9 (268078 values) +attn-rowsites S5 golden hash: daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5 (33618 values) +attn-rowsites 11.1(c) fixture hash: 336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779 (14384 values); 96 rows, softmax 96/0, prob-V 92/4 (fast/fallback by the guard copies) +attn-rowsites 11.1(d): prefill and decode windows driven on /p05_l1.sslm +attn-rowsites cells (plan slices S1, S2, S3, S4, S5): 91948 checks, 0 failures +superslm tests: 117437 checks, 0 failures + FAIL lines: 0 +== sslm_axis_digest: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_scalar_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_sse2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_avx2_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +== sslm_axis_digest_avx512_forced: exit 0 GLOBAL f740f8338668182724b900ddd078be3cf714b28f4004b504496de4f4b4ae5acc +c_rowsites d02721801c8296897bb30f4aaaed4f53d5f06600daee528c090701f8e2bd8e27 values=4201138 +c32_attention ddbdb76eea70a0e0fc1cfb90461331741a389afaa82880c7a88602158c0ac96d values=331796 +DONE-green2 diff --git a/docs/platform-support.md b/docs/platform-support.md index 36568726..6d7b39bf 100644 --- a/docs/platform-support.md +++ b/docs/platform-support.md @@ -119,6 +119,135 @@ best of 3:** 0.31 s with the SHA extensions against 3.25 s for 1.9.0's portable hash. The portable path itself is also about 1.7x faster than in 1.9.0 (1.95 s). +### Attention prob·V on int16 multiply-add (unreleased) + +On the AVX2 and AVX-512 tiers, `GemmProbQ15Accumulate` (the attention +probability row times the value rows, per head) multiplies pairs of keys +with `vpmaddwd` into 32-bit lanes and widens each lane to 64 bits. The +probability pairs are formed in registers. The fast path is taken only when +the head dimension is a multiple of 16 and the row passes the int16 +condition: every p in [0, 32767], and sum at most 2^15. Under that condition +no lane can exceed 128 x 2^15 = 2^22, so every output is the exact sum +v1.9.0's loop computes. Every other row takes the v1.9.0 loop, as do the +scalar and SSE2 tiers. On real prompts that fallback is the one-hot rows: +every width-1 row, and rarely a wider one. + +| Build | Fast path | Status | +|---|---|---| +| GCC / Clang, AVX2 tier | on | Bit-identity: full suite forced AVX2, the prob·V golden pinned from 1.9.0, cross-tier digest, save-blob equality against 1.9.0 | +| GCC / Clang, AVX-512 tier | on | Same evidence, forced AVX-512 and auto dispatch | +| MSVC / clang-cl, AVX2 tier | on | Built by the forced Windows legs; not yet executed on Windows | +| MSVC / clang-cl, AVX-512 tier | **off** (`SUPERSLM_SITES_AVX512_MSVC=0`) | Held on the 1.9.0 loop until an MSVC AVX-512 build has executed the fast path; the forced AVX-512 Windows legs build with it on | + +**Measured, engine level, same host as above (best of 50, 9 interleaved +rounds):** at head dimension 64 one call is 17-21x faster than 1.9.0's +loop at 128 to 1,024 keys on AVX-512, and 16-19x on AVX2. For example, at +601 keys it takes 24.8 µs on 1.9.0, 1.18 µs on AVX-512 and 1.32 µs on AVX2. +Scaled to Qwen2.5-0.5B depth (24 layers x 14 heads), the step saves about +0.82 / 3.4 / 6.7 ms per prompt token at 128 / 512 / 1,024 tokens, and +3.9 / 7.9 ms per decode token at context 300 / 600. A one-layer forward +agrees at 512 tokens and in decode. These are engine figures on synthetic +weights, not a consumer's end-to-end speed +(`docs/attention-rowsites/s2/bench.md`). + +### Requantization in 64-bit lanes (unreleased) + +Every checked-chain funnel call (`RequantChainChecked`: the projections, +norms, activation and residuals) ends by converting a row of 64-bit +accumulators to int8 codes. On the AVX2 and AVX-512 tiers that conversion +now runs in 4 or 8 unsigned 64-bit lanes (`RequantRowWide`). It computes +the element code's exact identity: the product |x| x r splits into 32-bit +halves, is rounded and shifted, then clamped at 127 and given back its sign. +Every intermediate stays exact up to the funnel's largest input, so every +code equals v1.9.0's per-element `RequantTokenCodeWide`. There is no +runtime guard; the funnel's own preflight is the contract. The last +n mod 4 (or 8) elements, and every element on the scalar and SSE2 tiers, +run the v1.9.0 code. + +| Build | Lanes | Status | +|---|---|---| +| GCC / Clang, AVX2 tier | on | Bit-identity: full suite forced AVX2, the requant golden pinned from 1.9.0, cross-tier digest, save-blob equality against 1.9.0 | +| GCC / Clang, AVX-512 tier | on | Same evidence, forced AVX-512 and auto dispatch | +| MSVC / clang-cl, AVX2 tier | on | Built by the forced Windows legs; not yet executed on Windows | +| MSVC / clang-cl, AVX-512 tier | **off** (`SUPERSLM_SITES_AVX512_MSVC=0`) | The same switch as prob·V above | + +**Measured, engine level, same host as above (best of 50, 9 interleaved +rounds):** one funnel call at width 4,864 takes 31.5 µs on 1.9.0, 5.6 µs +on AVX-512 and 6.7 µs on AVX2; at 896, 5.2, 1.0 and 1.2 µs. At +Qwen2.5-0.5B depth (24 layers x 11 funnel calls, plus the embed) that saves +about 2.7 ms per token on AVX-512 and 2.5 on AVX2, prefill and decode alike. +These are engine figures on synthetic weights, not a consumer's end-to-end +speed (`docs/attention-rowsites/s3/bench.md`). + +### Guarded softmax rows (unreleased) + +Attention's softmax row (`SoftmaxRowQ15`) turns a row of scores into Q15 +probabilities. On the AVX2 and AVX-512 tiers it first checks a row guard: +width at most 2^14, q_ln2 >= 1, q_c >= 0, M = q_b^2 + q_c in [1, 2^47] +(formed in 128 bits, as the 1.9.0 body forms it), q_ln2 <= 2 q_b + 1, and +every score within 2^61. Inside the guard it computes the row 4 or 8 +elements at a time. Each element's quotient by q_ln2 and each +probability's divide by the row total is an integer reciprocal estimate, +corrected exactly by one integer comparison each way, so every probability +and the returned bool equal v1.9.0's. Outside the guard the 1.9.0 body +runs unchanged. The estimates are integer arithmetic, so the library +stays floating-point-free. + +| Build | Fast path | Status | +|---|---|---| +| GCC / Clang, AVX2 tier | on | Bit-identity: full suite forced AVX2, the softmax golden pinned from 1.9.0, cross-tier digest, save-blob equality against 1.9.0 | +| GCC / Clang, AVX-512 tier | on | Same evidence, forced AVX-512 and auto dispatch | +| MSVC / clang-cl, AVX2 tier | on | Built by the forced Windows legs; not yet executed on Windows | +| MSVC / clang-cl, AVX-512 tier | **off** (`SUPERSLM_SITES_AVX512_MSVC=0`) | The same switch as prob·V above | + +**Measured, engine level, same host as above (best of 30, 9 interleaved +rounds):** a row of 512 keys takes 3.8 µs on 1.9.0, 1.06 µs on AVX2 and +0.90 µs on AVX-512. At Qwen2.5-0.5B depth (24 layers x 14 heads) that saves +about 0.10 / 0.46 / 0.91 ms per prompt token at 128 / 512 / 1,024 tokens +on AVX2, and 0.53 ms per decode token at context 300. A one-key row is +about 0.03 µs slower (the guard and two reciprocal divides per row). +These are engine figures on synthetic weights, not a consumer's end-to-end +speed (`docs/attention-rowsites/s4/bench.md`). + +### Q31 attention score rows (unreleased) + +QK-norm models (the Qwen3 path) score each key with a Q31 product, +`RoundingDivideByPOT(sum_d q_d * k_d * ratio_d, 31)`. `QkQ31ScoreRow` +computes every key's score for one query head in one call, and both layer +loops (prefill and decode) call it. On the AVX2 and AVX-512 tiers it first +checks a guard: head_dim at most 512 and every ratio in [0, 2^32). Inside the +guard, each channel's w = q * ratio is split exactly into three pieces, +w = a2 * 2^30 + a1 * 2^15 + a0, with a0 and a1 in [0, 32767] and a2 inside +int16. Each piece's sum over the channels is a 16-bit multiply-add into +int32 lanes. At head_dim 512 that sum stays inside int32 by 65,535, so the +guard is load-bearing. The three sums recombine exactly in int64, and the +rounding is vectorised with ties away from zero. Every score equals v1.9.0's +per-key `QkQ31Score`. Outside the guard, the per-key loop runs unchanged. It +is integer arithmetic only. + +| Build | Fast path | Status | +|---|---|---| +| GCC / Clang, AVX2 tier | on | Bit-identity: full suite forced AVX2, the Q31 golden and the QK-norm fixture pinned from 1.9.0, cross-tier digest, save-blob equality against 1.9.0 | +| GCC / Clang, AVX-512 tier | on | Same evidence, forced AVX-512 and auto dispatch | +| MSVC / clang-cl, AVX2 tier | on | Built by the forced Windows legs; not yet executed on Windows | +| MSVC / clang-cl, AVX-512 tier | **off** (`SUPERSLM_SITES_AVX512_MSVC=0`) | The same switch as prob·V above | + +No real Qwen3 artifact has run the new kernel yet: a QK-norm artifact is +still refused at map time on this host. The evidence is a QK-norm fixture +that drives both layer loops, pinned to v1.9.0, plus a one-layer forward +at Qwen3-0.6B width whose outputs match the base. + +**Measured, engine level, same host as above (best of 30, 9 interleaved +rounds, head_dim 128):** one score costs about 410 ns per head and key on +1.9.0's AVX2 path and 13.5 ns on the AVX2 row (330 and 12.8 ns on AVX-512). +At Qwen3-0.6B depth (28 layers x 16 heads) that saves about 11 / 46 / 94 ms +per prompt token at 128 / 512 / 1,024 tokens on AVX2, and 55 ms per decode +token at context 300. A one-layer forward at the same width saves 0.42 / +1.66 / 3.66 ms per layer and token. A one-key row on AVX-512 is about +0.06 µs slower (it packs a whole 16-key block). These are engine figures +on synthetic weights, not a consumer's end-to-end speed +(`docs/attention-rowsites/s5/bench.md`). + ### Damped-greedy decoding The 1.2 candidate's opt-in decoder was confirmed on Windows x64 through the diff --git a/include/superslm/forward_sites.h b/include/superslm/forward_sites.h index 2d118c81..18b68694 100644 --- a/include/superslm/forward_sites.h +++ b/include/superslm/forward_sites.h @@ -1266,6 +1266,16 @@ int64_t QkQ31ScoreForTier(const int8_t* q, const int8_t* k, const int64_t* ratio size_t head_dim, QkQ31ScoreTier tier); int64_t QkQ31Score(const int8_t* q, const int8_t* k, const int64_t* ratio_q31, size_t head_dim); +// Attention and per-row sites plan (rev 3.1), slice S5 (§4.5, §5.5): every key's Q31 score for one query +// head, the internal entry both layer loops call in place of their per-key QkQ31Score loops. `keys` holds +// `width` rows of `head_dim` int8 codes, row j at keys + j * head_dim (one KV head's K store, or any run of +// it); out[j] receives QkQ31Score(q, keys + j * head_dim, ratio_q31, head_dim) for every j in [0, width). +// On the AVX2 and AVX-512 tiers, with head_dim <= 512 and every ratio in [0, 2^32), the row runs the §5.5 +// limb kernel (w = q * ratio split into three 16-bit pieces, int16 multiply-add over blocked, packed keys); +// otherwise it loops over QkQ31Score exactly as the loops did. Not exported (C5). +void QkQ31ScoreRow(const int8_t* q, const int8_t* keys, const int64_t* ratio_q31, size_t head_dim, size_t width, + int64_t* out); + // --- S3.6: the head and the greedy decode loop (SuperSLM_S3a_WalkingSkeleton_ // Plan.md §11 S3.6; §9.1; master plan §6.4; C16). This is // the "host-facing entry point" LayerWeights' own header comment above names diff --git a/include/superslm/gpu_layer_loop_guards.def b/include/superslm/gpu_layer_loop_guards.def index c016912a..23430469 100644 --- a/include/superslm/gpu_layer_loop_guards.def +++ b/include/superslm/gpu_layer_loop_guards.def @@ -104,10 +104,10 @@ // CORRECTED 2026-09-23 (1.7.0): two `#include` lines (, ) added at the top of // forward_sites.cpp, above this function's own guard ladder, shifted every citation below by +2 // lines, uniformly. Each citation was re-read at its new line and holds its rejecting `return`. -SSLM_GPU_LAYER_LOOP_GUARD(LayerBudgetZero, InvalidLayerBudget, "forward_sites.cpp:1686") -SSLM_GPU_LAYER_LOOP_GUARD(ContextCapNonPositive, InvalidContextCap, "forward_sites.cpp:1696") -SSLM_GPU_LAYER_LOOP_GUARD(HeadDimGeometryMismatch, HeadDimGeometryMismatch, "forward_sites.cpp:1721") -SSLM_GPU_LAYER_LOOP_GUARD(KvHeadGeometryMismatch, KvHeadGeometryMismatch, "forward_sites.cpp:1732") +SSLM_GPU_LAYER_LOOP_GUARD(LayerBudgetZero, InvalidLayerBudget, "forward_sites.cpp:2017") +SSLM_GPU_LAYER_LOOP_GUARD(ContextCapNonPositive, InvalidContextCap, "forward_sites.cpp:2027") +SSLM_GPU_LAYER_LOOP_GUARD(HeadDimGeometryMismatch, HeadDimGeometryMismatch, "forward_sites.cpp:2052") +SSLM_GPU_LAYER_LOOP_GUARD(KvHeadGeometryMismatch, KvHeadGeometryMismatch, "forward_sites.cpp:2063") // Guard 5 is one CPU check with two distinct rejecting outcomes: an // overflow anywhere in the factor-by-factor KV-size product returns // InvalidContextCap; a null or undersized workspace (checked only once the @@ -115,10 +115,10 @@ SSLM_GPU_LAYER_LOOP_GUARD(KvHeadGeometryMismatch, KvHeadGeometryMismatch, "forwa // Table entries stay one row per source guard, matching the reconfirmation // review's own 9-row enumeration exactly, so `kCount` counts what a CPU // reader counts, not what a naive per-`return`-statement scan would. -SSLM_GPU_LAYER_LOOP_GUARD(WorkspaceSizeOrOverflow, WorkspaceTooSmall, "forward_sites.cpp:1758-1763 (InvalidContextCap on overflow, WorkspaceTooSmall on null/undersized)") -SSLM_GPU_LAYER_LOOP_GUARD(HiddenCodesNull, InvalidHiddenCodes, "forward_sites.cpp:1855") -SSLM_GPU_LAYER_LOOP_GUARD(SequenceAlreadyComplete, SequenceAlreadyComplete, "forward_sites.cpp:1873") -SSLM_GPU_LAYER_LOOP_GUARD(PositionOverCap, PositionOverCap, "forward_sites.cpp:1908") -SSLM_GPU_LAYER_LOOP_GUARD(KvCapacityExhausted, KvCapacityExhausted, "forward_sites.cpp:1910") +SSLM_GPU_LAYER_LOOP_GUARD(WorkspaceSizeOrOverflow, WorkspaceTooSmall, "forward_sites.cpp:2089-2094 (InvalidContextCap on overflow, WorkspaceTooSmall on null/undersized)") +SSLM_GPU_LAYER_LOOP_GUARD(HiddenCodesNull, InvalidHiddenCodes, "forward_sites.cpp:2186") +SSLM_GPU_LAYER_LOOP_GUARD(SequenceAlreadyComplete, SequenceAlreadyComplete, "forward_sites.cpp:2204") +SSLM_GPU_LAYER_LOOP_GUARD(PositionOverCap, PositionOverCap, "forward_sites.cpp:2239") +SSLM_GPU_LAYER_LOOP_GUARD(KvCapacityExhausted, KvCapacityExhausted, "forward_sites.cpp:2241") #undef SSLM_GPU_LAYER_LOOP_GUARD diff --git a/include/superslm/intmath.h b/include/superslm/intmath.h index 97b8e5da..bdd5e73b 100644 --- a/include/superslm/intmath.h +++ b/include/superslm/intmath.h @@ -253,6 +253,24 @@ void RowBoundsWide(const int64_t* x, size_t n, int64_t* out_max, int64_t* out_mi // DynamicScaleReciprocal/NormalizeScale, whose own contracts bound them. int8_t RequantTokenCodeWide(int64_t x_i, int64_t r, int s); +// Attention and per-row sites plan (rev 3.1), slice S3 (§4.3, §5.3) — the requant element loop +// as one row leaf: out[i] = RequantTokenCodeWide(x[i], r, s) for every i in [0, n), byte for byte. +// A funnel leaf like RequantTokenCodeWide: only the checked chain funnel may call it (the +// forward-leaf check lists it). +// +// **Contract: the funnel's.** Every |x[i]| <= d' <= 2^31, with r = DynamicScaleReciprocal(Dn) and +// s from NormalizeScale(d') of that same d' (so 1 <= r <= 2^32 and s in [-1, 30]); `x` holds n +// int64 values and `out` has room for exactly n codes. n == 0 reads and writes nothing. Nothing +// outside [out, out + n) is written. +// +// On the AVX2 and AVX-512BW tiers the row runs in 4 or 8 unsigned 64-bit lanes by the identity +// floor((254·P + 2^e) / 2^(e+1)) = (127·H + ((127·L + 2^(e-1)) >> 32)) >> (e - 32), where +// P = |x|·r = H·2^32 + L and e = 62 - s, followed by the same clamp and sign restore; the last +// n mod 4 (or 8) elements run RequantTokenCodeWide. P reaches exactly 2^63 at the contract's corner +// (|x| = d' = 2^31, r = 2^32), which the unsigned lane holds exactly. The scalar and SSE2 tiers +// (and an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off) run the element loop. Internal; not in the C API. +void RequantRowWide(const int64_t* x, size_t n, int64_t r, int s, int8_t* out); + // --- §6.3 nonlinear scalar primitives (i-sqrt C4/C5/C6, i-exp C7/C8/C9) ------- // // The reproducible-path integer cores only. The float-taking offline derivations diff --git a/include/superslm/matmul.h b/include/superslm/matmul.h index 533ba705..6ff1edcf 100644 --- a/include/superslm/matmul.h +++ b/include/superslm/matmul.h @@ -120,6 +120,21 @@ GemmPath DispatchGemmPath(GemmTier tier, size_t num_tokens); // The tier this process's GEMM dispatches on (see GemmTier). GemmTier ActiveGemmTier(); +// Attention and per-row sites plan (rev 3.1, §3.2, cell 11.2): which body the attention kernels of +// slices S2-S6 run. kShipped is the v1.9.0 code; kAvx2 and kAvx512 are the new SIMD bodies. +enum class SitesKernel : int { kShipped = 0, kAvx2 = 1, kAvx512 = 2 }; + +// The pure selector, compiled into every build and testable with any arguments on any runner: the +// new kernels run only on the AVX2 and AVX-512 tiers, and on the AVX-512 tier of an MSVC or clang-cl +// build (`is_msvc_build`) only when `msvc_avx512_switch` is nonzero (§3.2: SUPERSLM_SITES_AVX512_MSVC, +// default 0, independent of the tiled GEMM's switch). +SitesKernel SelectSitesKernel(GemmTier tier, int msvc_avx512_switch, bool is_msvc_build); + +// The call-site wiring: SelectSitesKernel with this build's own switch value and compiler identity. +// Every S2-S6 dispatch decides through this function and nothing else. SitesKernel and both selectors +// are internal C++ declarations like the GemmPath ones above: not exported, not part of the C API. +SitesKernel DispatchSitesKernel(GemmTier tier); + } // namespace detail // C17 -- narrow one accumulator row to int32 AFTER a conversion-time proof (design §4, @@ -190,6 +205,13 @@ int DetectBestDotRowTierForCpu(); // exact int64 products are exactly associative and commutative, so any // traversal order must produce the bit-identical `out_ctx`. // +// Attention and per-row sites plan, slice S2 (§4.2, §5.2): on the AVX2 and AVX-512BW tiers a call +// whose head_dim is a multiple of 16 and whose row passes the int16 condition (every p in +// [0, 32767] and Sum p <= 2^15, checked by the function itself in one pass) accumulates p_k*v_k[d] + +// p_{k+1}*v_{k+1}[d] with vpmaddwd into one int32 lane per output dimension, then widens to int64. +// Every lane's running sum is bounded by 128 * Sum p <= 2^22, so each output equals the int64 sum +// exactly; any other row takes the shipped loop. No allocation, no new status, same contract. +// // Caller ensures (contract, not runtime-checked -- the same convention as // GemmInt8AccumulateRow above): `probs` has `width` elements; `values` has // `width * head_dim` elements, row-major (`values[k*head_dim + d]` is key diff --git a/src/forward/checked_chain_funnel.cpp b/src/forward/checked_chain_funnel.cpp index c2ac1707..2d70ca2b 100644 --- a/src/forward/checked_chain_funnel.cpp +++ b/src/forward/checked_chain_funnel.cpp @@ -4,7 +4,7 @@ // second limb). This file is the funnel's own translation unit: the only place in // the whole S3a forward composition permitted to call MaxAbsReduceWide/ // RowBoundsWide/NormalizeScale/DynamicScaleReciprocal/RequantTokenCodeWide/ -// NarrowAccumulatorToI32 directly (§7.3's CI source check enforces this +// RequantRowWide/NarrowAccumulatorToI32 directly (§7.3's CI source check enforces this // structurally on every other forward TU). // // C28's derived-operand pair predicate (CheckRoundingDivideByPotExponentDomain) is @@ -411,11 +411,12 @@ ChainResult RequantChainChecked(const int64_t* wide_row, size_t n, if (preflight_result.status != SslmForwardStatus::Ok) return preflight_result; // Step 6: RequantTokenCodeWide per element, directly on the int64 row — never - // narrowed to int32 first (T-1254's fold). - for (size_t i = 0; i < n; ++i) { - out_codes[i] = RequantTokenCodeWide(wide_row[i], preflight.reciprocal, - preflight.normalized.s); - } + // narrowed to int32 first (T-1254's fold). Attention and per-row sites plan, + // slice S3: the element loop is the row leaf RequantRowWide (intmath.h), which + // writes exactly RequantTokenCodeWide's code for every element, in 64-bit lanes + // on the AVX2 and AVX-512 tiers. Its contract is this funnel's: the preflight + // above has bounded every |x_i| <= d' <= 2^31 and derived r and s from that d'. + RequantRowWide(wide_row, n, preflight.reciprocal, preflight.normalized.s, out_codes); *out_scale = preflight.output_scale; // §11 S3.1a's instrumentation seam (trace_hook.h), attached to this diff --git a/src/forward/forward_sites.cpp b/src/forward/forward_sites.cpp index 1902cb3d..64fab5a3 100644 --- a/src/forward/forward_sites.cpp +++ b/src/forward/forward_sites.cpp @@ -22,7 +22,9 @@ #include #include +#include #include +#include #include #include @@ -36,6 +38,12 @@ #include #endif +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT +// The test-only dispatch-instrument seam (src/matmul.cpp's convention): the per-row sites' row-table +// path counters (attention and per-row sites plan, §3.6). Never defined for the production library. +#include "support/matmul_dispatch_instrument.h" +#endif + #if defined(__clang__) || (defined(__GNUC__) && !defined(_MSC_VER)) #define SUPERSLM_QK_AVX2_TARGET __attribute__((target("avx2"))) #define SUPERSLM_QK_AVX512_TARGET __attribute__((target("avx512f,avx512bw"))) @@ -419,6 +427,56 @@ inline int64_t ComposedExponent(int64_t e_a, int64_t e_t, int64_t target_normali return k; } +// Attention and per-row sites plan (rev 3.1), slice S1 (§4.1, §5.1): the per-row tables. +// RmsNormSite's divide, MlpActSite's sigmoid and ResidualReconcileSite's landing rescale are each a +// pure function of one int8 code plus constants fixed for the whole row, so a row of n elements +// evaluates it at most 256 distinct ways. At n >= kRowTableMinWidth the site evaluates it once per +// code into a stack table, with exactly the arguments its per-element loop passes, and the loop reads +// the table: table[code] IS the value the loop would compute, so the output is bit-identical at every +// width. SiLU and landing tables cover [-127, 127] only, the funnel's output range (intmath.cpp's +// clamp): a -128 code, which a direct caller can pass, is evaluated directly, so no table entry is +// ever computed at an argument the v1.9.0 loop would not have evaluated for a valid input. The norm +// table covers every int8, since the shipped loop divides every element. +// +// The threshold is a speed constant only (tables lose below about 256 elements on this plan's +// measurement; 512 is the plan's pinned value). Forced scalar keeps the v1.9.0 per-element loops, +// so that build's digest leg is the normative reference axis (§3.3); every other build, MSVC and +// arm64 included, takes the tables. +#if defined(SUPERSLM_FORCE_SCALAR_MATMUL) +constexpr bool kRowTablesOn = false; +#else +constexpr bool kRowTablesOn = true; +#endif +constexpr size_t kRowTableMinWidth = 512; + +inline bool RowTableTaken(size_t n) { return kRowTablesOn && n >= kRowTableMinWidth; } + +enum class RowTableSite { kNorm, kSilu, kLanding }; + +// The §3.6 row-table counters: one increment per site call, after the table decision, on the side it +// took. Compiles to nothing outside the instrument seam. +inline void CountRowTableDecision(RowTableSite site, bool taken) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + switch (site) { + case RowTableSite::kNorm: + (taken ? superslm_test::g_rowtable_norm_taken : superslm_test::g_rowtable_norm_skipped) + .fetch_add(1, std::memory_order_relaxed); + break; + case RowTableSite::kSilu: + (taken ? superslm_test::g_rowtable_silu_taken : superslm_test::g_rowtable_silu_skipped) + .fetch_add(1, std::memory_order_relaxed); + break; + case RowTableSite::kLanding: + (taken ? superslm_test::g_rowtable_landing_taken : superslm_test::g_rowtable_landing_skipped) + .fetch_add(1, std::memory_order_relaxed); + break; + } +#else + (void)site; + (void)taken; +#endif +} + } // namespace int64_t FloorDivI64(int64_t a, int64_t b) { @@ -551,6 +609,230 @@ int64_t QkQ31Score(const int8_t* q, const int8_t* k, const int64_t* ratio, size_ #endif } +namespace { + +// ---- Attention and per-row sites plan (rev 3.1), slice S5 (§4.5, §5.5): the Q31 score row -------------------- +// +// score_j = RoundingDivideByPOT(Sum_d q_d * k_jd * ratio_d, 31). Per channel, w_d = q_d * ratio_d (an exact int64 +// product, |w| < 2^39 for ratio in [0, 2^32)) is split into three 16-bit pieces, +// w = a2 * 2^30 + a1 * 2^15 + a0, a0 = w & 0x7FFF, a1 = (w >> 15) & 0x7FFF, a2 = w >> 30 (arithmetic), +// exact for every int64 w; a0, a1 are in [0, 32767] and a2 in [-512, 511]. So Sum_d k_d * w_d = +// 2^30 Sum k a2 + 2^15 Sum k a1 + Sum k a0, and each limb sum is an int16 x int16 multiply-add (vpmaddwd) into +// int32 lanes. Per key and limb the lane sum is at most head_dim * 128 * 32,767, which at head_dim 512 is +// 2,147,418,112: inside int32 by 65,535, so the head_dim <= 512 guard is load-bearing (cell 7.S5b sits on it). +// The three limb sums recombine exactly in int64 (below 2^56), and the rounding is RoundingDivideByPOT's own +// (ties away from zero), vectorised. Outside the guard (head_dim > 512, or a ratio outside [0, 2^32)) the row +// is the per-key QkQ31Score loop the layer loops ran, the same binary's v1.9.0 code (§3.3, §5.5). +// +// Keys are packed per block into a stack buffer (§3.5): widened to int16 and transposed so each vector holds +// one 4-channel quad of 4 (AVX2) or 8 (AVX-512) keys, 8 or 16 keys per block, at most 16 x 512 x 2 bytes. Every +// lane is then one key's pair sum, and no horizontal reduction runs per key. + +constexpr size_t kQ31RowMaxHeadDim = 512; + +// The guard (§4.5): head_dim <= 512 (the int32 margin and the stack pack) and every ratio in [0, 2^32) (the +// range in which w keeps a2 inside int16 however it is formed, §5.5). +inline bool Q31RowFastPathAdmits(const int64_t* ratio, size_t head_dim) { + if (head_dim > kQ31RowMaxHeadDim) return false; + for (size_t d = 0; d < head_dim; ++d) + if (ratio[d] < 0 || ratio[d] > INT64_C(0xFFFFFFFF)) return false; + return true; +} + +// The query head's three limbs, four channels (one quad) per int64, channels past head_dim zero. +struct Q31RowLimbs { + int64_t quad[3][kQ31RowMaxHeadDim / 4]; + size_t nq; +}; + +inline void MakeQ31RowLimbs(const int8_t* q, const int64_t* ratio, size_t head_dim, Q31RowLimbs* limbs) { + limbs->nq = (head_dim + 3) / 4; + for (size_t i = 0; i < limbs->nq; ++i) { + uint64_t packed[3] = {0, 0, 0}; + for (size_t t = 0; t < 4; ++t) { + const size_t d = 4 * i + t; + const int64_t w = d < head_dim ? static_cast(q[d]) * ratio[d] : 0; + const int64_t a[3] = {w & 0x7FFF, (w >> 15) & 0x7FFF, w >> 30}; + for (int l = 0; l < 3; ++l) packed[l] |= (static_cast(a[l]) & 0xFFFFu) << (16 * t); + } + for (int l = 0; l < 3; ++l) limbs->quad[l][i] = static_cast(packed[l]); + } +} + +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 +// Sixteen channels of four key rows, widened to int16 (baseline SSE2, shared by both tiers): quad qq of the +// four keys is written as [k0 k1] at dst + qq * stride and [k2 k3] at dst + qq * stride + 8. +inline void Q31PackQuads4x16(const int8_t* const rows[4], size_t c, int16_t* dst, size_t stride) { + __m128i lo[4], hi[4]; + for (int t = 0; t < 4; ++t) { + const __m128i x = _mm_loadu_si128(reinterpret_cast(rows[t] + c)); + lo[t] = _mm_srai_epi16(_mm_unpacklo_epi8(x, x), 8); + hi[t] = _mm_srai_epi16(_mm_unpackhi_epi8(x, x), 8); + } + const __m128i* half[2] = {lo, hi}; + for (int h = 0; h < 2; ++h) { + const __m128i* r = half[h]; + int16_t* d0 = dst + (2 * h) * stride; + int16_t* d1 = dst + (2 * h + 1) * stride; + _mm_storeu_si128(reinterpret_cast<__m128i*>(d0), _mm_unpacklo_epi64(r[0], r[1])); + _mm_storeu_si128(reinterpret_cast<__m128i*>(d0 + 8), _mm_unpacklo_epi64(r[2], r[3])); + _mm_storeu_si128(reinterpret_cast<__m128i*>(d1), _mm_unpackhi_epi64(r[0], r[1])); + _mm_storeu_si128(reinterpret_cast<__m128i*>(d1 + 8), _mm_unpackhi_epi64(r[2], r[3])); + } +} + +// One block of keys into the pack: `n` live rows of head_dim from `keys`, rows n .. keys_per_block - 1 zero. +// Layout: keys_per_vector (kpv) keys share a vector; the vector of block row b's quad i starts at +// pw + ((b / kpv) * nq + i) * kpv * 4, with key b % kpv's four channels at offset (b % kpv) * 4. +inline void Q31PackKeyBlock(const int8_t* keys, size_t n, size_t head_dim, size_t nq, size_t kpv, + size_t keys_per_block, int16_t* pw) { + static const int8_t kZeroRow[kQ31RowMaxHeadDim] = {}; + const size_t stride = kpv * 4; + const size_t full16 = head_dim / 16 * 16; + for (size_t g = 0; g < keys_per_block; g += 4) { + const int8_t* rows[4]; + for (size_t t = 0; t < 4; ++t) rows[t] = g + t < n ? keys + (g + t) * head_dim : kZeroRow; + int16_t* base = pw + (g / kpv) * nq * stride + (g % kpv) * 4; + for (size_t c = 0; c < full16; c += 16) Q31PackQuads4x16(rows, c, base + (c / 4) * stride, stride); + for (size_t d = full16; d < nq * 4; ++d) + for (size_t t = 0; t < 4; ++t) + base[(d / 4) * stride + t * 4 + d % 4] = d < head_dim ? rows[t][d] : int16_t{0}; + } +} + +// RoundingDivideByPOT(x, 31) per 64-bit lane, |x| < 2^56, without a 64-bit compare or select (which Clang lowers +// to FP-domain blends the fp-free scan rejects): floor(x / 2^31) by biasing into the unsigned range and a +// logical shift; the remainder and the tie threshold (2^30 - 1, plus 1 for negative x) as in the scalar rule; +// "remainder > threshold" as the sign bit of threshold - remainder. +SUPERSLM_QK_AVX2_TARGET +inline __m256i Q31RoundAvx2(__m256i x) { + const __m256i floor = _mm256_sub_epi64(_mm256_srli_epi64(_mm256_add_epi64(x, _mm256_set1_epi64x(INT64_C(1) << 62)), 31), + _mm256_set1_epi64x(INT64_C(1) << 31)); + const __m256i rem = _mm256_and_si256(x, _mm256_set1_epi64x(INT64_C(0x7FFFFFFF))); + const __m256i thr = _mm256_add_epi64(_mm256_set1_epi64x(INT64_C(0x3FFFFFFF)), _mm256_srli_epi64(x, 63)); + return _mm256_add_epi64(floor, _mm256_srli_epi64(_mm256_sub_epi64(thr, rem), 63)); +} + +// The AVX2 body: blocks of 8 keys, two vectors of 4 keys; per quad, three broadcast limb quads. +SUPERSLM_QK_AVX2_TARGET +void QkQ31RowAvx2(const int8_t* keys, size_t head_dim, size_t width, const Q31RowLimbs& limbs, int64_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_q31_row_fast_avx2.fetch_add(1, std::memory_order_relaxed); +#endif + alignas(32) int16_t pw[2 * (kQ31RowMaxHeadDim / 4) * 16]; // 8 KiB + const size_t nq = limbs.nq; + for (size_t j = 0; j < width; j += 8) { + const size_t n = width - j < 8 ? width - j : 8; + Q31PackKeyBlock(keys + j * head_dim, n, head_dim, nq, 4, 8, pw); + __m256i acc[3][2]; + for (int l = 0; l < 3; ++l) acc[l][0] = acc[l][1] = _mm256_setzero_si256(); + const __m256i* p0 = reinterpret_cast(pw); + const __m256i* p1 = p0 + nq; + for (size_t i = 0; i < nq; ++i) { + const __m256i k0 = _mm256_load_si256(p0 + i), k1 = _mm256_load_si256(p1 + i); + for (int l = 0; l < 3; ++l) { + const __m256i b = _mm256_set1_epi64x(limbs.quad[l][i]); + acc[l][0] = _mm256_add_epi32(acc[l][0], _mm256_madd_epi16(k0, b)); + acc[l][1] = _mm256_add_epi32(acc[l][1], _mm256_madd_epi16(k1, b)); + } + } + // Lanes [k0 k0 k1 k1 | k2 k2 k3 k3] and the same for k4..k7: add the pairs, then put the keys in order. + __m256i s[3]; + for (int l = 0; l < 3; ++l) + s[l] = _mm256_permute4x64_epi64(_mm256_hadd_epi32(acc[l][0], acc[l][1]), 0xD8); + alignas(32) int64_t tmp[8]; + for (int h = 0; h < 2; ++h) { + __m256i w[3]; + for (int l = 0; l < 3; ++l) + w[l] = _mm256_cvtepi32_epi64(h ? _mm256_extracti128_si256(s[l], 1) : _mm256_castsi256_si128(s[l])); + const __m256i x = + _mm256_add_epi64(_mm256_add_epi64(_mm256_slli_epi64(w[2], 30), _mm256_slli_epi64(w[1], 15)), w[0]); + _mm256_store_si256(reinterpret_cast<__m256i*>(tmp + 4 * h), Q31RoundAvx2(x)); + } + std::memcpy(out + j, tmp, n * sizeof(int64_t)); + } +} + +SUPERSLM_QK_AVX512_TARGET +inline __m512i Q31RoundAvx512(__m512i x) { + const __m512i floor = _mm512_srai_epi64(x, 31); + const __m512i rem = _mm512_and_si512(x, _mm512_set1_epi64(INT64_C(0x7FFFFFFF))); + const __m512i thr = _mm512_add_epi64(_mm512_set1_epi64(INT64_C(0x3FFFFFFF)), _mm512_srli_epi64(x, 63)); + return _mm512_add_epi64(floor, _mm512_srli_epi64(_mm512_sub_epi64(thr, rem), 63)); +} + +// The AVX-512BW body (F and BW instructions only, no mask register): blocks of 16 keys, two vectors of 8 keys. +// Each 64-bit lane is one key's two pair sums; adding the high dword into the low one and sign-extending it +// gives that key's int32 limb sum as an int64, in key order, with no shuffle. +SUPERSLM_QK_AVX512_TARGET +void QkQ31RowAvx512(const int8_t* keys, size_t head_dim, size_t width, const Q31RowLimbs& limbs, int64_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_q31_row_fast_avx512.fetch_add(1, std::memory_order_relaxed); +#endif + alignas(64) int16_t pw[2 * (kQ31RowMaxHeadDim / 4) * 32]; // 16 KiB + const size_t nq = limbs.nq; + for (size_t j = 0; j < width; j += 16) { + const size_t n = width - j < 16 ? width - j : 16; + Q31PackKeyBlock(keys + j * head_dim, n, head_dim, nq, 8, 16, pw); + __m512i acc[3][2]; + for (int l = 0; l < 3; ++l) acc[l][0] = acc[l][1] = _mm512_setzero_si512(); + const int16_t* p0 = pw; + const int16_t* p1 = pw + nq * 32; + for (size_t i = 0; i < nq; ++i) { + const __m512i k0 = _mm512_load_si512(p0 + i * 32), k1 = _mm512_load_si512(p1 + i * 32); + for (int l = 0; l < 3; ++l) { + const __m512i b = _mm512_set1_epi64(limbs.quad[l][i]); + acc[l][0] = _mm512_add_epi32(acc[l][0], _mm512_madd_epi16(k0, b)); + acc[l][1] = _mm512_add_epi32(acc[l][1], _mm512_madd_epi16(k1, b)); + } + } + alignas(64) int64_t tmp[16]; + for (int h = 0; h < 2; ++h) { + __m512i w[3]; + for (int l = 0; l < 3; ++l) { + const __m512i pair = _mm512_add_epi32(acc[l][h], _mm512_srli_epi64(acc[l][h], 32)); + w[l] = _mm512_srai_epi64(_mm512_slli_epi64(pair, 32), 32); + } + const __m512i x = + _mm512_add_epi64(_mm512_add_epi64(_mm512_slli_epi64(w[2], 30), _mm512_slli_epi64(w[1], 15)), w[0]); + _mm512_store_si512(reinterpret_cast(tmp + 8 * h), Q31RoundAvx512(x)); + } + std::memcpy(out + j, tmp, n * sizeof(int64_t)); + } +} +#endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 + +} // namespace + +void QkQ31ScoreRow(const int8_t* q, const int8_t* keys, const int64_t* ratio, size_t head_dim, size_t width, + int64_t* out) { +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + // The selector (§3.2, cell 11.2) picks the new kernels on AVX2 and AVX-512 only (v1.9.0 code on the scalar and + // SSE2 tiers and on an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off). Inside the guard the + // tier's body writes the row (its fast counter moves there); outside it the per-key loop below runs and the + // tier's fallback counter moves (§3.6). + const detail::SitesKernel kernel = detail::DispatchSitesKernel(detail::ActiveGemmTier()); + if (kernel != detail::SitesKernel::kShipped) { + if (Q31RowFastPathAdmits(ratio, head_dim)) { + Q31RowLimbs limbs; + MakeQ31RowLimbs(q, ratio, head_dim, &limbs); + if (kernel == detail::SitesKernel::kAvx2) + QkQ31RowAvx2(keys, head_dim, width, limbs, out); + else + QkQ31RowAvx512(keys, head_dim, width, limbs, out); + return; + } +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + (kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_q31_row_fallback_avx2 + : superslm_test::g_q31_row_fallback_avx512) + .fetch_add(1, std::memory_order_relaxed); +#endif + } +#endif + // The per-key loop both layer loops ran (v1.9.0). + for (size_t j = 0; j < width; ++j) out[j] = QkQ31Score(q, keys + j * head_dim, ratio, head_dim); +} + SslmForwardStatus RmsNormSite(const int8_t* h, const int32_t* g, size_t hidden_size, CarriedScale /*incoming_scale*/, CarriedScale site_constant, int8_t* out_codes, CarriedScale* out_scale, @@ -585,9 +867,20 @@ SslmForwardStatus RmsNormSite(const int8_t* h, const int32_t* g, size_t hidden_s wide_fallback.assign(hidden_size, 0); wide = wide_fallback.data(); } - for (size_t i = 0; i < hidden_size; ++i) { - const int64_t hi = static_cast(h[i]); - wide[i] = FloorDivI64(hi << (2 * kNormFracBits), root) * static_cast(g[i]); + const bool use_table = RowTableTaken(hidden_size); + CountRowTableDecision(RowTableSite::kNorm, use_table); + if (use_table) { + // Plan S1: FloorDivI64(c << 2*NORM_FRAC_BITS, root) for every int8 c, then one lookup per element. + int64_t divided[256]; + for (int c = -128; c <= 127; ++c) + divided[c + 128] = FloorDivI64(static_cast(c) << (2 * kNormFracBits), root); + for (size_t i = 0; i < hidden_size; ++i) + wide[i] = divided[static_cast(h[i]) + 128] * static_cast(g[i]); + } else { + for (size_t i = 0; i < hidden_size; ++i) { + const int64_t hi = static_cast(h[i]); + wide[i] = FloorDivI64(hi << (2 * kNormFracBits), root) * static_cast(g[i]); + } } // §11 S3.1a (D-SLM362): `site`/`token_index`/`trace_hook_state` are @@ -996,7 +1289,21 @@ SslmForwardStatus MlpActSite(const int8_t* gate_code, CarriedScale gate_scale, const int gate_e = static_cast(gate_scale.e); std::vector wide(n); - for (size_t i = 0; i < n; ++i) { + const bool use_table = RowTableTaken(n); + CountRowTableDecision(RowTableSite::kSilu, use_table); + if (use_table) { + // Plan S1: step 2's value at every code the funnel emits, with step 2's own arguments; a -128 + // code (reachable only from a direct caller) is evaluated directly, never from the table. + int32_t sigmoid[255]; + for (int c = -127; c <= 127; ++c) + sigmoid[c + 127] = SiluSigmoidQ15(sigmoid_lut_table, static_cast(c), gate_scale.m, gate_e); + for (size_t i = 0; i < n; ++i) { + const int8_t code = gate_code[i]; + const int32_t sig = code == INT8_MIN ? SiluSigmoidQ15(sigmoid_lut_table, code, gate_scale.m, gate_e) + : sigmoid[static_cast(code) + 127]; + wide[i] = static_cast(code) * static_cast(sig) * static_cast(up_code[i]); + } + } else for (size_t i = 0; i < n; ++i) { // the v1.9.0 per-element loop, unchanged // Step 2: C10's fixed-point LUT construction (silu_lut.h), never the // i-exp-sigmoid construction F-S3-1 found the reference computing // before S3.0's reconciliation. Substituting i-exp-sigmoid here @@ -1073,6 +1380,10 @@ SslmForwardStatus ResidualReconcileSite(const int8_t* branch_code, CarriedScale branch_selected = d >= 0 ? (branch_magnitude << d) < stream_magnitude : branch_magnitude < (stream_magnitude << -d); } + // Plan S1: the table decision is per call (counted once), the table itself per candidate, since + // each candidate lands a different row at different constants. + const bool use_table = RowTableTaken(hidden_size); + CountRowTableDecision(RowTableSite::kLanding, use_table); struct Candidate { SslmForwardStatus status = SslmForwardStatus::Ok; CarriedScale scale{}; @@ -1086,11 +1397,31 @@ SslmForwardStatus ResidualReconcileSite(const int8_t* branch_code, CarriedScale const int8_t* other_code = select_branch ? stream_code : branch_code; const auto reciprocal = CarriedScaleNormalizedReciprocal(magnitude(candidate.scale.m)); candidate.wide.resize(hidden_size); + // Plan S1: (value, flag) per code in [-127, 127], with the loop's own arguments and null + // counters. The loop below keeps its per-element order, its first-flag return and its sign + // and overflow checks; the flag is read for each element present, never OR-ed over the table + // (a code absent from the row must not refuse it). -128 is landed directly. + int64_t landed_table[255]; + bool exceeded_table[255]; + if (use_table) { + for (int c = -127; c <= 127; ++c) { + bool exceeded = false; + landed_table[c + 127] = LandingRescale(c, other_scale.m, reciprocal.r, other_scale.e, + candidate.scale.e, nullptr, &exceeded, nullptr, reciprocal.s); + exceeded_table[c + 127] = exceeded; + } + } for (size_t i = 0; i < hidden_size; ++i) { bool magnitude_exceeded = false; - int64_t landed = LandingRescale(static_cast(other_code[i]), other_scale.m, - reciprocal.r, other_scale.e, candidate.scale.e, nullptr, - &magnitude_exceeded, nullptr, reciprocal.s); + int64_t landed; + if (use_table && other_code[i] != INT8_MIN) { + landed = landed_table[static_cast(other_code[i]) + 127]; + magnitude_exceeded = exceeded_table[static_cast(other_code[i]) + 127]; + } else { + landed = LandingRescale(static_cast(other_code[i]), other_scale.m, + reciprocal.r, other_scale.e, candidate.scale.e, nullptr, + &magnitude_exceeded, nullptr, reciprocal.s); + } if (magnitude_exceeded || (candidate.scale.m < 0 && landed == INT64_MIN)) { candidate.status = SslmForwardStatus::ResidualReconciliationMagnitudeOutOfDomain; return candidate; @@ -2165,10 +2496,9 @@ static SslmForwardStatus RunLayerLoopImpl(SequenceLayerState& seq, const LayerWe const int8_t* const k_rows_base = KeyRow(workspace, l, context_cap, num_key_value_heads, head_dim, kv_head, 0); if (direct_qk) { + // Slice S5 (§4.5): one score-row call per query head, in place of the per-key loop. const int64_t* ratio = lw.k_channel_ratio + kv_head * head_dim; - for (size_t row = 0; row < width; ++row) - scores[row] = QkQ31Score(q_rot.data() + h * head_dim, - k_rows_base + row * head_dim, ratio, head_dim); + QkQ31ScoreRow(q_rot.data() + h * head_dim, k_rows_base, ratio, head_dim, width, scores.data()); } else { GemmInt8AccumulateRow(q_rot.data() + h * head_dim, k_rows_base, head_dim, width, scores.data()); @@ -2647,10 +2977,10 @@ SslmForwardStatus RunLayerLoopChunkBatched(int8_t* hidden_codes_chunk, CarriedSc const int8_t* const k_rows_base = KeyRow(workspace, l, context_cap, num_key_value_heads, head_dim, kv_head, 0); if (direct_qk) { + // Slice S5 (§4.5): one score-row call per query head, in place of the per-key loop. const int64_t* ratio = lw.k_channel_ratio + kv_head * head_dim; - for (size_t row = 0; row < width; ++row) - scores[row] = QkQ31Score(q_rot.data() + h * head_dim, - k_rows_base + row * head_dim, ratio, head_dim); + QkQ31ScoreRow(q_rot.data() + h * head_dim, k_rows_base, ratio, head_dim, width, + scores.data()); } else { GemmInt8AccumulateRow(q_rot.data() + h * head_dim, k_rows_base, head_dim, width, scores.data()); diff --git a/src/intmath.cpp b/src/intmath.cpp index 4c5cf687..609ea4fd 100644 --- a/src/intmath.cpp +++ b/src/intmath.cpp @@ -20,10 +20,37 @@ #include #include #include +#include #include #include #include "superslm/checked_chain_funnel.h" // kSoftmaxRowMaxSafeExponent (D-SLM409, plan Sec14.1) +#include "superslm/matmul.h" // detail::ActiveGemmTier / DispatchSitesKernel (RequantRowWide's tier dispatch) + +// Attention and per-row sites plan (rev 3.1), slice S3: this file's first target-attributed functions, +// RequantRowAvx2 and RequantRowAvx512 below. Per function, never translation-unit-wide (§3.2; the +// isolation checker's prose and the linkage checker's population name them), exactly as src/matmul.cpp +// does: GCC and Clang need the target enabled per function for an AVX2/AVX-512 intrinsic to compile, and +// a TU-wide flag would let the auto-vectorizer use those instructions anywhere in this file. MSVC gates +// no intrinsic by /arch; clang-cl does, and needs the attribute. The AVX-512 target is F and BW only (C9). +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 +#include +#include +#endif + +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT +// The test-only dispatch-instrument seam (src/matmul.cpp's convention): the requant row's per-tier path +// counters (attention and per-row sites plan, §3.6). Never defined for the production library. +#include "support/matmul_dispatch_instrument.h" +#endif + +#if defined(__clang__) || (defined(__GNUC__) && !defined(_MSC_VER)) +#define SUPERSLM_INTMATH_AVX2_TARGET __attribute__((target("avx2"))) +#define SUPERSLM_INTMATH_AVX512_TARGET __attribute__((target("avx512f,avx512bw"))) +#else +#define SUPERSLM_INTMATH_AVX2_TARGET +#define SUPERSLM_INTMATH_AVX512_TARGET +#endif namespace superslm { namespace { @@ -520,6 +547,130 @@ int8_t RequantTokenCodeWide(int64_t x_i, int64_t r, int s) { return static_cast(x_i < 0 ? -q : q); } +// ---- Attention and per-row sites plan, slice S3: the requant row leaf (§4.3, §5.3) --------------------- +// +// The identity (§5.3). With 0 <= |x| <= 2^31, 1 <= r <= 2^32, e = 62 - s in [32, 63] and +// P = |x|·r = H·2^32 + L (H = P >> 32, L = P mod 2^32): +// floor((254·P + 2^e) / 2^(e+1)) = floor((127·P + 2^(e-1)) / 2^e) +// = (127·H + ((127·L + 2^(e-1)) >> 32)) >> (e - 32), +// because 127·P + 2^(e-1) = 127·H·2^32 + X with X = 127·L + 2^(e-1), and nested floor division by +// positive integers (2^32, then 2^(e-32)) equals the single floor. The clamp at 127 and the sign restore +// are RequantTokenCodeWide's. Every intermediate fits an unsigned 64-bit lane: P <= 2^63 (exactly 2^63 at +// the contract's corner |x| = d' = 2^31, r = 2^32, which is why the lane is unsigned and H is taken with +// a logical shift), 127·H <= 127·2^31 < 2^38, 127·L + 2^(e-1) < 2^39 + 2^62. P is formed from the 32-bit +// halves of r (r itself can be 2^32): |x|·r_lo + ((|x|·r_hi) << 32), each a 32x32 -> 64 product. +// Operand dispositions: |x| <= d' <= 2^31 is guarded by the funnel's preflight (C29), r and s are +// canonical from that d' (NormalizeScale, DynamicScaleReciprocal); the funnel is the only caller (the +// forward-leaf check), so the leaf needs no runtime guard and has no fallback. + +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 +namespace { + +// AVX2 body: four elements per step, as many whole steps as the row holds; returns how many elements it +// wrote (the caller runs the rest through RequantTokenCodeWide). Counts its own entry (§3.6), so a +// dispatch that reached the wrong tier's body moves the wrong counter. +// +// |x|, the clamp and the sign restore are done in 32-bit lanes (vpabsd, vpminud, vpsignd) rather than as +// 64-bit selects: Clang lowers a 64-bit select (vpblendvb on a vpcmpgtq mask, or its and/andnot/xor +// spellings) to vblendvpd and vxorpd, FP-domain instructions the fp-free scan rejects. Each is exact here: +// |x| <= 2^31, so the low dword of vpabsd is |x| read as unsigned (x = -2^31 gives 0x80000000), and +// vpmuludq reads only low dwords; the magnitude's low dword is the whole magnitude unless its high dword +// is nonzero, which is folded into bit 7 before the unsigned 32-bit minimum; and x's high dword is 0 for +// x >= 0 and all-ones for x < 0, so OR-ing 1 into it gives the +-1 that vpsignd applies. +SUPERSLM_INTMATH_AVX2_TARGET size_t RequantRowAvx2(const int64_t* x, size_t n, int64_t r, int s, int8_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_requant_row_avx2.fetch_add(1, std::memory_order_relaxed); +#endif + const int e = 62 - s; // in [32, 63] + const __m256i r_lo = _mm256_set1_epi64x(static_cast(static_cast(r) & 0xFFFFFFFFu)); + const __m256i r_hi = _mm256_set1_epi64x(static_cast(static_cast(r) >> 32)); + const __m256i round = _mm256_set1_epi64x(static_cast(uint64_t{1} << (e - 1))); + const __m256i low32 = _mm256_set1_epi64x(INT64_C(0xFFFFFFFF)); + const __m256i c127 = _mm256_set1_epi64x(127); // low dword 127, high dword 0 + const __m256i c128 = _mm256_set1_epi32(128); + const __m256i one = _mm256_set1_epi32(1); + const __m256i zero = _mm256_setzero_si256(); + const __m128i shift = _mm_cvtsi32_si128(e - 32); + // Byte 0 and byte 8 of each 128-bit half (each 64-bit lane's low byte) to bytes 0 and 1 of that half. Bytes + // 2-15 are never read (the unpack below takes word 0 of each half); the upper half fills them with 1, not + // -1, only so the two halves differ: identical halves let Clang load the constant with vbroadcasti128, + // which the fp-free scan does not admit. + const __m256i pick = _mm256_setr_epi8(0, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1); + size_t i = 0; + for (; i + 4 <= n; i += 4) { + const __m256i xv = _mm256_loadu_si256(reinterpret_cast(x + i)); + const __m256i ax = _mm256_abs_epi32(xv); // low dword: |x| <= 2^31, unsigned + const __m256i p = _mm256_add_epi64(_mm256_mul_epu32(ax, r_lo), _mm256_slli_epi64(_mm256_mul_epu32(ax, r_hi), 32)); + const __m256i h = _mm256_srli_epi64(p, 32); // logical: P can be 2^63 + const __m256i l = _mm256_and_si256(p, low32); + const __m256i carry = _mm256_srli_epi64(_mm256_add_epi64(_mm256_mul_epu32(l, c127), round), 32); + const __m256i mag = _mm256_srl_epi64(_mm256_add_epi64(_mm256_mul_epu32(h, c127), carry), shift); // < 2^39 + // The clamp at 127: a magnitude with a nonzero high dword is >= 2^32, so it sets bit 7 of the low dword; + // then the unsigned 32-bit minimum with 127 (the high dword becomes min(high, 0) = 0). + const __m256i big = _mm256_andnot_si256(_mm256_cmpeq_epi32(_mm256_shuffle_epi32(mag, 0xF5), zero), c128); + const __m256i clamped = _mm256_min_epu32(_mm256_or_si256(mag, big), c127); + // The sign: +1 or -1 from x's high dword. + const __m256i q = _mm256_sign_epi32(clamped, _mm256_or_si256(_mm256_shuffle_epi32(xv, 0xF5), one)); + const __m256i b = _mm256_shuffle_epi8(q, pick); + const __m128i four = _mm_unpacklo_epi16(_mm256_castsi256_si128(b), _mm256_extracti128_si256(b, 1)); + const int32_t packed = _mm_cvtsi128_si32(four); + std::memcpy(out + i, &packed, 4); + } + return i; +} + +// AVX-512BW body, the same construction in eight lanes, F and BW instructions only (C9): |x| by vpabsq, +// the sign by vpsraq, the clamp by vpminuq, the narrowing by vpmovqb, and no mask register anywhere. +SUPERSLM_INTMATH_AVX512_TARGET size_t RequantRowAvx512(const int64_t* x, size_t n, int64_t r, int s, int8_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_requant_row_avx512.fetch_add(1, std::memory_order_relaxed); +#endif + const int e = 62 - s; // in [32, 63] + const __m512i r_lo = _mm512_set1_epi64(static_cast(static_cast(r) & 0xFFFFFFFFu)); + const __m512i r_hi = _mm512_set1_epi64(static_cast(static_cast(r) >> 32)); + const __m512i round = _mm512_set1_epi64(static_cast(uint64_t{1} << (e - 1))); + const __m512i low32 = _mm512_set1_epi64(INT64_C(0xFFFFFFFF)); + const __m512i c127 = _mm512_set1_epi64(127); + const __m128i shift = _mm_cvtsi32_si128(e - 32); + size_t i = 0; + for (; i + 8 <= n; i += 8) { + const __m512i xv = _mm512_loadu_si512(x + i); + const __m512i neg = _mm512_srai_epi64(xv, 63); // all-ones where x < 0 + const __m512i ax = _mm512_abs_epi64(xv); // |x| <= 2^31 + const __m512i p = _mm512_add_epi64(_mm512_mul_epu32(ax, r_lo), _mm512_slli_epi64(_mm512_mul_epu32(ax, r_hi), 32)); + const __m512i h = _mm512_srli_epi64(p, 32); // logical: P can be 2^63 + const __m512i l = _mm512_and_si512(p, low32); + const __m512i carry = _mm512_srli_epi64(_mm512_add_epi64(_mm512_mul_epu32(l, c127), round), 32); + const __m512i mag = _mm512_min_epu64(_mm512_srl_epi64(_mm512_add_epi64(_mm512_mul_epu32(h, c127), carry), shift), c127); + const __m512i q = _mm512_sub_epi64(_mm512_xor_si512(mag, neg), neg); + _mm_storel_epi64(reinterpret_cast<__m128i*>(out + i), _mm512_cvtepi64_epi8(q)); + } + return i; +} + +} // namespace +#endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 + +void RequantRowWide(const int64_t* x, size_t n, int64_t r, int s, int8_t* out) { + size_t i = 0; +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + // The selector (§3.2, cell 11.2): the lanes on AVX2 and AVX-512; v1.9.0's element loop on the scalar and + // SSE2 tiers and on an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off. + switch (detail::DispatchSitesKernel(detail::ActiveGemmTier())) { + case detail::SitesKernel::kAvx2: + i = RequantRowAvx2(x, n, r, s, out); + break; + case detail::SitesKernel::kAvx512: + i = RequantRowAvx512(x, n, r, s, out); + break; + case detail::SitesKernel::kShipped: + break; + } +#endif + for (; i < n; ++i) out[i] = RequantTokenCodeWide(x[i], r, s); // the tail, or the whole row +} + // --- §6.3 nonlinear scalar primitives (i-sqrt C4/C5/C6, i-exp C7/C8/C9) -------- namespace { @@ -824,6 +975,271 @@ RopePair RopeApplyPair(int32_t x, int32_t y, int32_t cos_q30, int32_t sin_q30) { // --- §6.2/§5.2 C32 softmax row (arm A) ---------------------------------------- +// Attention and per-row sites plan (rev 3.1), slice S4 (§4.4, §5.4): the guarded fast path. Inside the row guard +// the fast path returns the shipped body's `true` and writes the same probabilities, integer for integer; outside +// it the shipped body below runs unchanged. +// +// The guard, checked once per row in dependency order (each step makes the next one's arithmetic defined): +// 1. width in [1, 2^14] (width >= 1 is the caller's early return); +// 2. q_ln2 >= 1; +// 3. q_c >= 0; +// 4. M = q_b^2 + q_c, formed in 128 bits exactly as the shipped body forms it, in [1, 2^47] +// (kSoftmaxRowMaxSafeExponent); +// 5. q_ln2 <= 2 q_b + 1 (safe to form: step 4 bounds |q_b| < 2^23.5); +// 6. every score within +-2^61. +// Implied (§5.4): q_b in [0, 2^23.5], q_c in [0, 2^47], q_ln2 in [1, 2^24.5 + 1]. So after the max shift every +// element's IExpConstruct is kOk with |base| <= q_b, its value is in [0, M], and the row is `true`. +// +// The arithmetic is §5.4's estimate-then-correct, with INTEGER estimates: the library is floating-point-free (the +// fp-free scan gates it), and §5.4's own argument needs only an estimate within one of the floor, because the +// exact integer corrections decide. Per element, a = min(max - s, 30 q_ln2) (the clip), in [0, 30 q_ln2]: +// z: estimate (a * inv_z) >> kz, inv_z = floor(2^kz / q_ln2) <= 2^31, kz = 30 + bit_width(q_ln2); a * inv_z < +// 2^61. A floored reciprocal never overestimates, and the error is below a / 2^kz < 1, so the estimate is the +// floor or one below it: the upward correction (r >= q_ln2) is live, the downward one (r < 0) cannot fire and +// is kept as defensive code (§5.4 step 3). r = a - z q_ln2 is then q_p's negation; base = q_b - r; +// e = (base^2 + q_c) >> z, base^2 < 2^47 from a signed 32x32 multiply. +// p: estimate (e * R) >> 47, R = round(2^62 / denom), denom = Sum e in [M, 2^14 M] (the max element's e is M). +// e * R <= 2^62 + 2^46 < 2^63, and |error| <= e / 2^48 <= 1/2, so the estimate is the floor or one either +// side: both corrections are live (§5.4 step 4). Every product is below 2^63: p * denom <= e 2^15 + denom <= +// 2^62 + 2^61 (the width guard is what bounds denom). +// Each pass reads element k before writing element k, so scores == out_probs stays correct (cell 4.S4, aliased). +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 +namespace { + +constexpr size_t kSoftmaxFastMaxWidth = size_t{1} << 14; +constexpr int64_t kSoftmaxFastScoreLimit = int64_t{1} << 61; + +// The row guard (steps 1-6 above) for a row of width >= 1; on success, the row's maximum. +bool SoftmaxFastGuard(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, int64_t q_c, int64_t* row_max) { + if (width > kSoftmaxFastMaxWidth) return false; + if (q_ln2 < 1) return false; + if (q_c < 0) return false; + const S128 m128 = SAdd(SMul(q_b, q_b), SFromI64(q_c)); + if (!SGe(m128, SFromI64(1)) || !SGe(SFromI64(kSoftmaxRowMaxSafeExponent), m128)) return false; + if (q_ln2 > 2 * q_b + 1) return false; + int64_t peak = scores[0]; + for (size_t k = 0; k < width; ++k) { + const int64_t v = scores[k]; + if (v > kSoftmaxFastScoreLimit || v < -kSoftmaxFastScoreLimit) return false; + if (v > peak) peak = v; + } + *row_max = peak; + return true; +} + +// The row constants both bodies broadcast. +struct SoftmaxFastRow { + int64_t peak, q_ln2, q_b, q_c; + uint64_t inv_z; // floor(2^kz / q_ln2), <= 2^31 + int kz; // 30 + bit_width(q_ln2) +}; + +SoftmaxFastRow MakeSoftmaxFastRow(int64_t peak, int64_t q_ln2, int64_t q_b, int64_t q_c) { + const int kz = 30 + static_cast(std::bit_width(static_cast(q_ln2))); + return SoftmaxFastRow{peak, q_ln2, q_b, q_c, (uint64_t{1} << kz) / static_cast(q_ln2), kz}; +} + +// R = round(2^62 / denom), denom >= 1. +uint64_t SoftmaxProbReciprocal(int64_t denom) { + const uint64_t d = static_cast(denom); + return ((uint64_t{1} << 62) + (d >> 1)) / d; +} + +// AVX2: four elements per step. Masks are 64-bit compares (vpcmpgtq) folded in by add, sub and and, never by a +// select; the clip is an unsigned 32-bit minimum with the high dword folded into bit 31 (Clang lowers 64-bit +// selects to vblendvpd, which the fp-free scan rejects; see RequantRowAvx2). +struct SoftmaxAvx2Consts { + __m256i peak, clip, inv_z, q_ln2, q_ln2_m1, q_b, q_c, bit31; + __m128i kz; +}; + +SUPERSLM_INTMATH_AVX2_TARGET inline __m256i SoftmaxExpAvx2(__m256i s, const SoftmaxAvx2Consts& c) { + const __m256i zero = _mm256_setzero_si256(); + const __m256i a_raw = _mm256_sub_epi64(c.peak, s); // in [0, 2^62] + const __m256i big = _mm256_andnot_si256(_mm256_cmpeq_epi32(_mm256_shuffle_epi32(a_raw, 0xF5), zero), c.bit31); + const __m256i a = _mm256_min_epu32(_mm256_or_si256(a_raw, big), c.clip); // min(a, 30 q_ln2); high dword 0 + __m256i z = _mm256_srl_epi64(_mm256_mul_epu32(a, c.inv_z), c.kz); // floor or floor - 1 + __m256i r = _mm256_sub_epi64(a, _mm256_mul_epu32(z, c.q_ln2)); // in [0, 2 q_ln2) + const __m256i up = _mm256_cmpgt_epi64(r, c.q_ln2_m1); // r >= q_ln2: z was one low + z = _mm256_sub_epi64(z, up); + r = _mm256_sub_epi64(r, _mm256_and_si256(up, c.q_ln2)); + const __m256i down = _mm256_cmpgt_epi64(zero, r); // defensive: cannot fire (floored reciprocal) + z = _mm256_add_epi64(z, down); + r = _mm256_add_epi64(r, _mm256_and_si256(down, c.q_ln2)); + const __m256i base = _mm256_sub_epi64(c.q_b, r); // |base| <= q_b < 2^24 + const __m256i v = _mm256_add_epi64(_mm256_mul_epi32(base, base), c.q_c); + return _mm256_srlv_epi64(v, z); +} + +struct SoftmaxProbAvx2Consts { + __m256i r_lo, r_hi, d_lo, d_hi, d, one; +}; + +SUPERSLM_INTMATH_AVX2_TARGET inline __m256i SoftmaxProbAvx2(__m256i e, const SoftmaxProbAvx2Consts& c) { + const __m256i num = _mm256_slli_epi64(e, 15); + const __m256i cross = _mm256_add_epi64(_mm256_mul_epu32(e, c.r_hi), _mm256_mul_epu32(_mm256_srli_epi64(e, 32), c.r_lo)); + const __m256i er = _mm256_add_epi64(_mm256_mul_epu32(e, c.r_lo), _mm256_slli_epi64(cross, 32)); // e R < 2^63 + __m256i p = _mm256_srli_epi64(er, 47); + __m256i prod = _mm256_add_epi64(_mm256_mul_epu32(p, c.d_lo), _mm256_slli_epi64(_mm256_mul_epu32(p, c.d_hi), 32)); + const __m256i down = _mm256_cmpgt_epi64(prod, num); // estimate one high + p = _mm256_add_epi64(p, down); + prod = _mm256_sub_epi64(prod, _mm256_and_si256(down, c.d)); + const __m256i no_up = _mm256_cmpgt_epi64(_mm256_add_epi64(prod, c.d), num); // -1 unless the estimate was one low + return _mm256_add_epi64(p, _mm256_add_epi64(no_up, c.one)); +} + +SUPERSLM_INTMATH_AVX2_TARGET void SoftmaxRowAvx2(const int64_t* scores, size_t width, const SoftmaxFastRow& row, + int64_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_softmax_fast_avx2.fetch_add(1, std::memory_order_relaxed); +#endif + const SoftmaxAvx2Consts c{_mm256_set1_epi64x(row.peak), + _mm256_set1_epi64x(30 * row.q_ln2), + _mm256_set1_epi64x(static_cast(row.inv_z)), + _mm256_set1_epi64x(row.q_ln2), + _mm256_set1_epi64x(row.q_ln2 - 1), + _mm256_set1_epi64x(row.q_b), + _mm256_set1_epi64x(row.q_c), + _mm256_set1_epi32(INT32_MIN), + _mm_cvtsi32_si128(row.kz)}; + // Pass 2: e per element, written over out_probs, and the total. + __m256i acc = _mm256_setzero_si256(); + size_t k = 0; + for (; k + 4 <= width; k += 4) { + const __m256i e = SoftmaxExpAvx2(_mm256_loadu_si256(reinterpret_cast(scores + k)), c); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(out + k), e); + acc = _mm256_add_epi64(acc, e); + } + alignas(32) int64_t lanes[4]; + _mm256_store_si256(reinterpret_cast<__m256i*>(lanes), acc); + int64_t total = lanes[0] + lanes[1] + lanes[2] + lanes[3]; + const size_t rest = width - k; + if (rest > 0) { // the last 1-3 elements through the same lanes, padded with the maximum (read, never kept) + alignas(32) int64_t buf[4] = {row.peak, row.peak, row.peak, row.peak}; + std::memcpy(buf, scores + k, rest * sizeof(int64_t)); + _mm256_store_si256(reinterpret_cast<__m256i*>(buf), + SoftmaxExpAvx2(_mm256_load_si256(reinterpret_cast(buf)), c)); + for (size_t i = 0; i < rest; ++i) { + out[k + i] = buf[i]; + total += buf[i]; + } + } + // Pass 3: p = floor(e 2^15 / total) in place. total >= M >= 1. + const uint64_t r = SoftmaxProbReciprocal(total); + const uint64_t d = static_cast(total); + const SoftmaxProbAvx2Consts pc{_mm256_set1_epi64x(static_cast(r & 0xFFFFFFFFu)), + _mm256_set1_epi64x(static_cast(r >> 32)), + _mm256_set1_epi64x(static_cast(d & 0xFFFFFFFFu)), + _mm256_set1_epi64x(static_cast(d >> 32)), + _mm256_set1_epi64x(total), + _mm256_set1_epi64x(1)}; + for (k = 0; k + 4 <= width; k += 4) { + const __m256i e = _mm256_loadu_si256(reinterpret_cast(out + k)); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(out + k), SoftmaxProbAvx2(e, pc)); + } + if (rest > 0) { // padded with e = 0 (read, never kept) + alignas(32) int64_t buf[4] = {0, 0, 0, 0}; + std::memcpy(buf, out + k, rest * sizeof(int64_t)); + _mm256_store_si256(reinterpret_cast<__m256i*>(buf), + SoftmaxProbAvx2(_mm256_load_si256(reinterpret_cast(buf)), pc)); + for (size_t i = 0; i < rest; ++i) out[k + i] = buf[i]; + } +} + +// AVX-512BW: eight elements per step, F and BW instructions only (C9), no mask register: the clip is vpminuq, +// and every decision is a sign mask from vpsraq by 63 of an exact difference. +struct SoftmaxAvx512Consts { + __m512i peak, clip, inv_z, q_ln2, q_ln2_m1, q_b, q_c; + __m128i kz; +}; + +SUPERSLM_INTMATH_AVX512_TARGET inline __m512i SoftmaxExpAvx512(__m512i s, const SoftmaxAvx512Consts& c) { + const __m512i a = _mm512_min_epu64(_mm512_sub_epi64(c.peak, s), c.clip); // min(a, 30 q_ln2) + __m512i z = _mm512_srl_epi64(_mm512_mul_epu32(a, c.inv_z), c.kz); // floor or floor - 1 + __m512i r = _mm512_sub_epi64(a, _mm512_mul_epu32(z, c.q_ln2)); // in [0, 2 q_ln2) + const __m512i up = _mm512_srai_epi64(_mm512_sub_epi64(c.q_ln2_m1, r), 63); // r >= q_ln2: z was one low + z = _mm512_sub_epi64(z, up); + r = _mm512_sub_epi64(r, _mm512_and_si512(up, c.q_ln2)); + const __m512i down = _mm512_srai_epi64(r, 63); // defensive: cannot fire (floored reciprocal) + z = _mm512_add_epi64(z, down); + r = _mm512_add_epi64(r, _mm512_and_si512(down, c.q_ln2)); + const __m512i base = _mm512_sub_epi64(c.q_b, r); // |base| <= q_b < 2^24 + const __m512i v = _mm512_add_epi64(_mm512_mul_epi32(base, base), c.q_c); + return _mm512_srlv_epi64(v, z); +} + +struct SoftmaxProbAvx512Consts { + __m512i r_lo, r_hi, d_lo, d_hi, d, one; +}; + +SUPERSLM_INTMATH_AVX512_TARGET inline __m512i SoftmaxProbAvx512(__m512i e, const SoftmaxProbAvx512Consts& c) { + const __m512i num = _mm512_slli_epi64(e, 15); + const __m512i cross = + _mm512_add_epi64(_mm512_mul_epu32(e, c.r_hi), _mm512_mul_epu32(_mm512_srli_epi64(e, 32), c.r_lo)); + const __m512i er = _mm512_add_epi64(_mm512_mul_epu32(e, c.r_lo), _mm512_slli_epi64(cross, 32)); // e R < 2^63 + __m512i p = _mm512_srli_epi64(er, 47); + __m512i prod = _mm512_add_epi64(_mm512_mul_epu32(p, c.d_lo), _mm512_slli_epi64(_mm512_mul_epu32(p, c.d_hi), 32)); + const __m512i down = _mm512_srai_epi64(_mm512_sub_epi64(num, prod), 63); // estimate one high + p = _mm512_add_epi64(p, down); + prod = _mm512_sub_epi64(prod, _mm512_and_si512(down, c.d)); + const __m512i no_up = _mm512_srai_epi64(_mm512_sub_epi64(num, _mm512_add_epi64(prod, c.d)), 63); + return _mm512_add_epi64(p, _mm512_add_epi64(no_up, c.one)); +} + +SUPERSLM_INTMATH_AVX512_TARGET void SoftmaxRowAvx512(const int64_t* scores, size_t width, const SoftmaxFastRow& row, + int64_t* out) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_softmax_fast_avx512.fetch_add(1, std::memory_order_relaxed); +#endif + const SoftmaxAvx512Consts c{_mm512_set1_epi64(row.peak), + _mm512_set1_epi64(30 * row.q_ln2), + _mm512_set1_epi64(static_cast(row.inv_z)), + _mm512_set1_epi64(row.q_ln2), + _mm512_set1_epi64(row.q_ln2 - 1), + _mm512_set1_epi64(row.q_b), + _mm512_set1_epi64(row.q_c), + _mm_cvtsi32_si128(row.kz)}; + __m512i acc = _mm512_setzero_si512(); + size_t k = 0; + for (; k + 8 <= width; k += 8) { + const __m512i e = SoftmaxExpAvx512(_mm512_loadu_si512(scores + k), c); + _mm512_storeu_si512(out + k, e); + acc = _mm512_add_epi64(acc, e); + } + alignas(64) int64_t lanes[8]; + _mm512_store_si512(lanes, acc); + int64_t total = 0; + for (int i = 0; i < 8; ++i) total += lanes[i]; + const size_t rest = width - k; + if (rest > 0) { // the last 1-7 elements through the same lanes, padded with the maximum (read, never kept) + alignas(64) int64_t buf[8] = {row.peak, row.peak, row.peak, row.peak, row.peak, row.peak, row.peak, row.peak}; + std::memcpy(buf, scores + k, rest * sizeof(int64_t)); + _mm512_store_si512(buf, SoftmaxExpAvx512(_mm512_load_si512(buf), c)); + for (size_t i = 0; i < rest; ++i) { + out[k + i] = buf[i]; + total += buf[i]; + } + } + const uint64_t r = SoftmaxProbReciprocal(total); + const uint64_t d = static_cast(total); + const SoftmaxProbAvx512Consts pc{_mm512_set1_epi64(static_cast(r & 0xFFFFFFFFu)), + _mm512_set1_epi64(static_cast(r >> 32)), + _mm512_set1_epi64(static_cast(d & 0xFFFFFFFFu)), + _mm512_set1_epi64(static_cast(d >> 32)), + _mm512_set1_epi64(total), + _mm512_set1_epi64(1)}; + for (k = 0; k + 8 <= width; k += 8) _mm512_storeu_si512(out + k, SoftmaxProbAvx512(_mm512_loadu_si512(out + k), pc)); + if (rest > 0) { // padded with e = 0 (read, never kept) + alignas(64) int64_t buf[8] = {0, 0, 0, 0, 0, 0, 0, 0}; + std::memcpy(buf, out + k, rest * sizeof(int64_t)); + _mm512_store_si512(buf, SoftmaxProbAvx512(_mm512_load_si512(buf), pc)); + for (size_t i = 0; i < rest; ++i) out[k + i] = buf[i]; + } +} + +} // namespace +#endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 + bool SoftmaxRowQ15(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, int64_t q_c, int64_t* out_probs) { // D-SLM497: `width == 0` is guarded here, before `scores` or `out_probs` @@ -838,6 +1254,31 @@ bool SoftmaxRowQ15(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q if (width == 0) { return true; } +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + // Slice S4 (above): the selector (§3.2, cell 11.2) picks the new kernels on AVX2 and AVX-512 only (v1.9.0 code + // on the scalar and SSE2 tiers and on an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off). Inside + // the row guard the tier's body writes the row and the row is true; outside it the shipped body below runs, + // and the tier's fallback counter moves (§3.6). The fast counter moves inside each body. + { + const detail::SitesKernel kernel = detail::DispatchSitesKernel(detail::ActiveGemmTier()); + if (kernel != detail::SitesKernel::kShipped) { + int64_t peak = 0; + if (SoftmaxFastGuard(scores, width, q_ln2, q_b, q_c, &peak)) { + const SoftmaxFastRow row = MakeSoftmaxFastRow(peak, q_ln2, q_b, q_c); + if (kernel == detail::SitesKernel::kAvx2) + SoftmaxRowAvx2(scores, width, row, out_probs); + else + SoftmaxRowAvx512(scores, width, row, out_probs); + return true; + } +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + (kernel == detail::SitesKernel::kAvx2 ? superslm_test::g_softmax_fallback_avx2 + : superslm_test::g_softmax_fallback_avx512) + .fetch_add(1, std::memory_order_relaxed); +#endif + } + } +#endif // C32 (§5.2, §11 S3.3 §6.2 step 5): ShiftByMax -> per-element // IExpConstruct/IExpEvaluate -> sum -> Q15 divide. The caller gates this // kernel with CheckSoftmaxRowWidthDomain(q_b, q_c, width) before calling diff --git a/src/matmul.cpp b/src/matmul.cpp index e6ddc0f3..f0485956 100644 --- a/src/matmul.cpp +++ b/src/matmul.cpp @@ -697,6 +697,247 @@ inline void RunTiledGemm(detail::GemmTier tier, const int16_t* a16, size_t kp, c #endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 +// ---- Attention and per-row sites plan, slice S2: prob·V on int16 multiply-add (§4.2, §5.2) ---------- +// +// Plan §3.2: in MSVC and clang-cl builds the AVX-512 tier keeps the v1.9.0 attention code until an MSVC +// AVX-512 build has executed the new kernels. The forced AVX-512 Windows legs build with +// SUPERSLM_SITES_AVX512_MSVC=1; the default flips in a follow-up release once one such run exists. A +// separate switch from the tiled GEMM's, so one closure's executed run cannot enable the other's code. +#if defined(SUPERSLM_SITES_AVX512_MSVC) +constexpr int kSitesAvx512MsvcSwitch = SUPERSLM_SITES_AVX512_MSVC; +#else +constexpr int kSitesAvx512MsvcSwitch = 0; +#endif + +// The shipped construction (v1.9.0's loop body), as an accumulate-into: out_ctx[d] += Sum_k p_k * v_k[d], +// exact int64. GemmProbQ15Accumulate zeroes out_ctx and then runs this or a SIMD body; the paged-KV +// plan's accumulate-into entry (no zeroing) shares the same core (§10). +inline void ProbQ15AccumulateIntoScalar(const int64_t* probs, const int8_t* values, size_t width, size_t head_dim, + int64_t* out_ctx) { + for (size_t k = 0; k < width; ++k) { + const int64_t p = probs[k]; + const int8_t* row = values + k * head_dim; + for (size_t d = 0; d < head_dim; ++d) { + out_ctx[d] += p * static_cast(row[d]); + } + } +} + +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + +// The int16 condition (§4.2, §5.2), checked in one pass: every p in [0, 32767] and Sum p <= 2^15. Inside +// it every p fits an int16 lane and every int32 lane's running sum is bounded by 128 * Sum p <= 2^22, so +// the vpmaddwd bodies below equal the int64 loop exactly. The first out-of-range p returns at once, so +// the running sum of in-range values cannot overflow for any width. +constexpr int64_t kProbVMaxP = 32767; +constexpr int64_t kProbVMaxSum = INT64_C(1) << 15; + +inline bool ProbVInt16Condition(const int64_t* probs, size_t width) { + int64_t sum = 0; + for (size_t k = 0; k < width; ++k) { + const int64_t p = probs[k]; + if (p < 0 || p > kProbVMaxP) return false; + sum += p; + } + return sum <= kProbVMaxSum; +} + +// The guard (§4.2): head_dim a multiple of 16 (one 16-dimension unit per 16-byte value load) and the +// int16 condition. +inline bool ProbVFastPathAdmits(const int64_t* probs, size_t width, size_t head_dim) { + return head_dim % 16 == 0 && ProbVInt16Condition(probs, width); +} + +// One probability pair as the 32-bit value vpmaddwd multiplies each (v_k[d], v_{k+1}[d]) pair by: p_k in +// the low 16 bits, p_{k+1} in the high 16 bits. Both are in [0, 32767] here (the guard). +inline int32_t ProbVPair(int64_t p_lo, int64_t p_hi) { + return static_cast(static_cast(p_lo) | (static_cast(p_hi) << 16)); +} + +// AVX2 body over NB 16-dimension units starting at d0: per key pair, the two value rows' 16 bytes are +// interleaved (v_k[d], v_{k+1}[d]), widened to int16 and multiplied by the broadcast pair with vpmaddwd, +// one int32 lane per output dimension; an odd last key is paired with a zero row and p = 0. The lanes +// are then added, widened, into out_ctx. +template +SUPERSLM_AVX2_TARGET inline void ProbVBlockAvx2(const int64_t* probs, const int8_t* values, size_t width, + size_t head_dim, size_t d0, int64_t* out_ctx) { + __m256i acc[NB][2]; + for (int u = 0; u < NB; ++u) acc[u][0] = acc[u][1] = _mm256_setzero_si256(); + size_t k = 0; + for (; k + 2 <= width; k += 2) { + const __m256i pair = _mm256_set1_epi32(ProbVPair(probs[k], probs[k + 1])); + const int8_t* r0 = values + k * head_dim + d0; + const int8_t* r1 = r0 + head_dim; + for (int u = 0; u < NB; ++u) { + const __m128i a = _mm_loadu_si128(reinterpret_cast(r0 + 16 * u)); + const __m128i b = _mm_loadu_si128(reinterpret_cast(r1 + 16 * u)); + acc[u][0] = _mm256_add_epi32(acc[u][0], _mm256_madd_epi16(_mm256_cvtepi8_epi16(_mm_unpacklo_epi8(a, b)), pair)); + acc[u][1] = _mm256_add_epi32(acc[u][1], _mm256_madd_epi16(_mm256_cvtepi8_epi16(_mm_unpackhi_epi8(a, b)), pair)); + } + } + if (k < width) { // the unpaired last key + const __m256i pair = _mm256_set1_epi32(ProbVPair(probs[k], 0)); + const int8_t* r0 = values + k * head_dim + d0; + const __m128i zero = _mm_setzero_si128(); + for (int u = 0; u < NB; ++u) { + const __m128i a = _mm_loadu_si128(reinterpret_cast(r0 + 16 * u)); + acc[u][0] = _mm256_add_epi32(acc[u][0], _mm256_madd_epi16(_mm256_cvtepi8_epi16(_mm_unpacklo_epi8(a, zero)), pair)); + acc[u][1] = _mm256_add_epi32(acc[u][1], _mm256_madd_epi16(_mm256_cvtepi8_epi16(_mm_unpackhi_epi8(a, zero)), pair)); + } + } + for (int u = 0; u < NB; ++u) + for (int h = 0; h < 2; ++h) { + int64_t* o = out_ctx + d0 + 16 * u + 8 * h; + const __m256i lo = _mm256_cvtepi32_epi64(_mm256_castsi256_si128(acc[u][h])); + const __m256i hi = _mm256_cvtepi32_epi64(_mm256_extracti128_si256(acc[u][h], 1)); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(o), + _mm256_add_epi64(_mm256_loadu_si256(reinterpret_cast(o)), lo)); + _mm256_storeu_si256(reinterpret_cast<__m256i*>(o + 4), + _mm256_add_epi64(_mm256_loadu_si256(reinterpret_cast(o + 4)), hi)); + } +} + +// The AVX2 accumulate-into, over every 16-dimension unit, four units (64 dimensions) per pass. Counts +// its own fast-path entry (§3.6), so a dispatch that reached the wrong tier's body moves the wrong counter. +SUPERSLM_AVX2_TARGET inline void ProbVAccumulateIntoAvx2(const int64_t* probs, const int8_t* values, size_t width, + size_t head_dim, int64_t* out_ctx) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_pv_fast_avx2.fetch_add(1, std::memory_order_relaxed); +#endif + size_t d0 = 0; + for (; d0 + 64 <= head_dim; d0 += 64) ProbVBlockAvx2<4>(probs, values, width, head_dim, d0, out_ctx); + switch ((head_dim - d0) / 16) { + case 3: ProbVBlockAvx2<3>(probs, values, width, head_dim, d0, out_ctx); break; + case 2: ProbVBlockAvx2<2>(probs, values, width, head_dim, d0, out_ctx); break; + case 1: ProbVBlockAvx2<1>(probs, values, width, head_dim, d0, out_ctx); break; + default: break; + } +} + +// AVX-512BW body, same construction in 512-bit registers, F and BW instructions only (C9). A 32-dimension +// unit: the two value rows' 32 bytes are interleaved in-lane (vpunpck[lh]bw on ymm), so the low result +// holds dimensions 0-7 and 16-23 and the high one 8-15 and 24-31; each is widened by vpmovsxbw to 32 +// int16 and multiplied by the broadcast pair with vpmaddwd into 16 int32 lanes. The stores put each +// 8-lane half back at its own dimensions, so no cross-lane permute is needed. NB units per pass. +template +SUPERSLM_AVX512_TARGET inline void ProbVBlockAvx512(const int64_t* probs, const int8_t* values, size_t width, + size_t head_dim, size_t d0, int64_t* out_ctx) { + __m512i acc[NB][2]; + for (int u = 0; u < NB; ++u) acc[u][0] = acc[u][1] = _mm512_setzero_si512(); + size_t k = 0; + for (; k + 2 <= width; k += 2) { + const __m512i pair = _mm512_set1_epi32(ProbVPair(probs[k], probs[k + 1])); + const int8_t* r0 = values + k * head_dim + d0; + const int8_t* r1 = r0 + head_dim; + for (int u = 0; u < NB; ++u) { + const __m256i a = _mm256_loadu_si256(reinterpret_cast(r0 + 32 * u)); + const __m256i b = _mm256_loadu_si256(reinterpret_cast(r1 + 32 * u)); + acc[u][0] = _mm512_add_epi32(acc[u][0], _mm512_madd_epi16(_mm512_cvtepi8_epi16(_mm256_unpacklo_epi8(a, b)), pair)); + acc[u][1] = _mm512_add_epi32(acc[u][1], _mm512_madd_epi16(_mm512_cvtepi8_epi16(_mm256_unpackhi_epi8(a, b)), pair)); + } + } + if (k < width) { // the unpaired last key + const __m512i pair = _mm512_set1_epi32(ProbVPair(probs[k], 0)); + const int8_t* r0 = values + k * head_dim + d0; + const __m256i zero = _mm256_setzero_si256(); + for (int u = 0; u < NB; ++u) { + const __m256i a = _mm256_loadu_si256(reinterpret_cast(r0 + 32 * u)); + acc[u][0] = _mm512_add_epi32(acc[u][0], _mm512_madd_epi16(_mm512_cvtepi8_epi16(_mm256_unpacklo_epi8(a, zero)), pair)); + acc[u][1] = _mm512_add_epi32(acc[u][1], _mm512_madd_epi16(_mm512_cvtepi8_epi16(_mm256_unpackhi_epi8(a, zero)), pair)); + } + } + for (int u = 0; u < NB; ++u) + for (int h = 0; h < 2; ++h) { + // Lanes 0-7 hold dimensions 8h .. 8h + 7 of the unit, lanes 8-15 hold 16 + 8h .. 16 + 8h + 7. + int64_t* o = out_ctx + d0 + 32 * u + 8 * h; + const __m512i lo = _mm512_cvtepi32_epi64(_mm512_castsi512_si256(acc[u][h])); + const __m512i hi = _mm512_cvtepi32_epi64(_mm512_extracti64x4_epi64(acc[u][h], 1)); + _mm512_storeu_si512(o, _mm512_add_epi64(_mm512_loadu_si512(o), lo)); + _mm512_storeu_si512(o + 16, _mm512_add_epi64(_mm512_loadu_si512(o + 16), hi)); + } +} + +// The AVX-512 16-dimension tail unit (head_dim % 32 == 16): the 16-byte rows interleaved and joined into +// one ymm, widened to 32 int16, one vpmaddwd into 16 int32 lanes in dimension order. +SUPERSLM_AVX512_TARGET inline void ProbVTail16Avx512(const int64_t* probs, const int8_t* values, size_t width, + size_t head_dim, size_t d0, int64_t* out_ctx) { + __m512i acc = _mm512_setzero_si512(); + size_t k = 0; + for (; k + 2 <= width; k += 2) { + const __m512i pair = _mm512_set1_epi32(ProbVPair(probs[k], probs[k + 1])); + const int8_t* r0 = values + k * head_dim + d0; + const __m128i a = _mm_loadu_si128(reinterpret_cast(r0)); + const __m128i b = _mm_loadu_si128(reinterpret_cast(r0 + head_dim)); + const __m256i ab = _mm256_inserti128_si256(_mm256_castsi128_si256(_mm_unpacklo_epi8(a, b)), + _mm_unpackhi_epi8(a, b), 1); + acc = _mm512_add_epi32(acc, _mm512_madd_epi16(_mm512_cvtepi8_epi16(ab), pair)); + } + if (k < width) { // the unpaired last key + const __m512i pair = _mm512_set1_epi32(ProbVPair(probs[k], 0)); + const __m128i a = _mm_loadu_si128(reinterpret_cast(values + k * head_dim + d0)); + const __m128i zero = _mm_setzero_si128(); + const __m256i ab = _mm256_inserti128_si256(_mm256_castsi128_si256(_mm_unpacklo_epi8(a, zero)), + _mm_unpackhi_epi8(a, zero), 1); + acc = _mm512_add_epi32(acc, _mm512_madd_epi16(_mm512_cvtepi8_epi16(ab), pair)); + } + int64_t* o = out_ctx + d0; + const __m512i lo = _mm512_cvtepi32_epi64(_mm512_castsi512_si256(acc)); + const __m512i hi = _mm512_cvtepi32_epi64(_mm512_extracti64x4_epi64(acc, 1)); + _mm512_storeu_si512(o, _mm512_add_epi64(_mm512_loadu_si512(o), lo)); + _mm512_storeu_si512(o + 8, _mm512_add_epi64(_mm512_loadu_si512(o + 8), hi)); +} + +SUPERSLM_AVX512_TARGET inline void ProbVAccumulateIntoAvx512(const int64_t* probs, const int8_t* values, + size_t width, size_t head_dim, int64_t* out_ctx) { +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_pv_fast_avx512.fetch_add(1, std::memory_order_relaxed); +#endif + size_t d0 = 0; + for (; d0 + 128 <= head_dim; d0 += 128) ProbVBlockAvx512<4>(probs, values, width, head_dim, d0, out_ctx); + switch ((head_dim - d0) / 32) { + case 3: ProbVBlockAvx512<3>(probs, values, width, head_dim, d0, out_ctx); d0 += 96; break; + case 2: ProbVBlockAvx512<2>(probs, values, width, head_dim, d0, out_ctx); d0 += 64; break; + case 1: ProbVBlockAvx512<1>(probs, values, width, head_dim, d0, out_ctx); d0 += 32; break; + default: break; + } + if (d0 < head_dim) ProbVTail16Avx512(probs, values, width, head_dim, d0, out_ctx); // head_dim % 32 == 16 +} + +#endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 + +// The tiered accumulate-into core (§4.2, §10): the SIMD body when the selected kernel is AVX2 or AVX-512 +// and the guard admits the row, else the shipped loop. Each path counter moves once per call, after the +// guard has decided, on the branch it names (§3.6); on the scalar and SSE2 tiers, and on an MSVC build's +// AVX-512 tier with its switch off, none moves. +inline void ProbQ15AccumulateInto(const int64_t* probs, const int8_t* values, size_t width, size_t head_dim, + int64_t* out_ctx) { +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + switch (detail::DispatchSitesKernel(detail::ActiveGemmTier())) { + case detail::SitesKernel::kAvx2: + if (ProbVFastPathAdmits(probs, width, head_dim)) { + ProbVAccumulateIntoAvx2(probs, values, width, head_dim, out_ctx); + return; + } +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_pv_fallback_avx2.fetch_add(1, std::memory_order_relaxed); +#endif + break; + case detail::SitesKernel::kAvx512: + if (ProbVFastPathAdmits(probs, width, head_dim)) { + ProbVAccumulateIntoAvx512(probs, values, width, head_dim, out_ctx); + return; + } +#ifdef SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT + superslm_test::g_pv_fallback_avx512.fetch_add(1, std::memory_order_relaxed); +#endif + break; + case detail::SitesKernel::kShipped: + break; + } +#endif + ProbQ15AccumulateIntoScalar(probs, values, width, head_dim, out_ctx); +} + } // namespace int64_t DotRowScalarRef(const int8_t* activations, const int8_t* weights, size_t in_channels) { @@ -816,6 +1057,24 @@ GemmPath DispatchGemmPath(GemmTier tier, size_t num_tokens) { return SelectGemmPath(tier, num_tokens, kTiledAvx512MsvcSwitch, kIsMsvcBuild); } +SitesKernel SelectSitesKernel(GemmTier tier, int msvc_avx512_switch, bool is_msvc_build) { + switch (tier) { + case GemmTier::kAvx2: + return SitesKernel::kAvx2; + case GemmTier::kAvx512: + // Plan §3.2: off in MSVC and clang-cl builds until executed there. + return (is_msvc_build && msvc_avx512_switch == 0) ? SitesKernel::kShipped : SitesKernel::kAvx512; + case GemmTier::kScalar: + case GemmTier::kSse2: + break; + } + return SitesKernel::kShipped; +} + +SitesKernel DispatchSitesKernel(GemmTier tier) { + return SelectSitesKernel(tier, kSitesAvx512MsvcSwitch, kIsMsvcBuild); +} + GemmTier ActiveGemmTier() { #if defined(SUPERSLM_FORCE_SCALAR_MATMUL) return GemmTier::kScalar; @@ -920,17 +1179,12 @@ void GemmProbQ15Accumulate(const int64_t* probs, const int8_t* values, size_t wi // out_ctx[d] = Sum_k probs[k] * values[k*head_dim + d]. Exact int64 // accumulation, no saturation, no rounding (F-S3-6's derived bound: // |Sum_k p_k*v_k| <= 2^15*127 < 2^22, independent of context length -- - // far inside int64, so no intermediate can overflow). + // far inside int64, so no intermediate can overflow). Attention and per-row + // sites plan, slice S2: zero, then the tiered accumulate-into core. for (size_t d = 0; d < head_dim; ++d) { out_ctx[d] = 0; } - for (size_t k = 0; k < width; ++k) { - const int64_t p = probs[k]; - const int8_t* row = values + k * head_dim; - for (size_t d = 0; d < head_dim; ++d) { - out_ctx[d] += p * static_cast(row[d]); - } - } + ProbQ15AccumulateInto(probs, values, width, head_dim, out_ctx); } } // namespace superslm diff --git a/tests/attn_rowsite_golden_pin.h b/tests/attn_rowsite_golden_pin.h new file mode 100644 index 00000000..f92c622b --- /dev/null +++ b/tests/attn_rowsite_golden_pin.h @@ -0,0 +1,47 @@ +// GENERATED FILE. Do not hand-edit. +// +// Produced by tools/gen_attn_rowsite_golden.cpp built against the v1.9.0 tag's library (the +// normative per-element code), over tests/support/rowsite_cases.h's and +// tests/support/attention_cases.h's fixed input sets, one hash per slice. Attention and per-row +// sites plan, §3.3 evidence 3, coverage cell 6.3. Re-running the generator against v1.9.0 must +// reproduce this file byte-for-byte. +#ifndef SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H +#define SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H + +#include + +namespace superslm_test { + +// Slice S1: RmsNormSite, MlpActSite and ResidualReconcileSite over RunRowTableCases. +inline constexpr const char* kAttnRowsiteS1GoldenHash = + "8836d5eb32a4badb492a8bcdf11e00222ad59a1e4b98013a3b8cb0c059d98ec8"; +inline constexpr uint64_t kAttnRowsiteS1GoldenValues = 634120ULL; + +// Slice S2: GemmProbQ15Accumulate over RunProbVCases. +inline constexpr const char* kAttnRowsiteS2GoldenHash = + "b0d1a6cd065347e799e5bb9857ce5db1f51ff351c8d4edde22896f11974506ed"; +inline constexpr uint64_t kAttnRowsiteS2GoldenValues = 30100ULL; + +// Slice S3: RequantChainChecked's element loop over RunRequantRowCases. +inline constexpr const char* kAttnRowsiteS3GoldenHash = + "3e3abed7c746191e8745c89ad38019076eff290aa7f4ffb57fb51c4527fdb3b9"; +inline constexpr uint64_t kAttnRowsiteS3GoldenValues = 3567018ULL; + +// Slice S4: SoftmaxRowQ15 over RunSoftmaxCases. +inline constexpr const char* kAttnRowsiteS4GoldenHash = + "2e47ea3c27774db43d9c952972325a5c19d901ba6871f0bd124c8c874f6a55d9"; +inline constexpr uint64_t kAttnRowsiteS4GoldenValues = 268078ULL; + +// Slice S5: QkQ31Score per key over RunQ31Cases. +inline constexpr const char* kAttnRowsiteS5GoldenHash = + "daea9a39c4df72b9431140446ee511cb9f60101646d83fbc2ae22d4b3d4faaa5"; +inline constexpr uint64_t kAttnRowsiteS5GoldenValues = 33618ULL; + +// Slice S5, cell 11.1(c): the QK-norm fixture's forward through the decode loop. +inline constexpr const char* kAttnRowsiteS5FixtureGoldenHash = + "336b8d417d078cdf91c0cd714e557df752e952eeb29b79ac594348e3be085779"; +inline constexpr uint64_t kAttnRowsiteS5FixtureGoldenValues = 14384ULL; + +} // namespace superslm_test + +#endif // SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H diff --git a/tests/ci/check_no_forward_leaf_calls.py b/tests/ci/check_no_forward_leaf_calls.py index 5f030cfe..c9bc3570 100644 --- a/tests/ci/check_no_forward_leaf_calls.py +++ b/tests/ci/check_no_forward_leaf_calls.py @@ -98,7 +98,10 @@ _INCLUDE_DIR = os.path.join(_REPO_ROOT, "include") # The eight funnel leaves named at Sec7.3, verbatim: the forward-leaf caller-ensures -# set the funnel exists to keep off every call site but its own. +# set the funnel exists to keep off every call site but its own. Plus a ninth: the +# attention and per-row sites plan's slice S3 (rev 3.1, Sec3.4, cell 11.4) moves the +# funnel's element loop into the row leaf RequantRowWide (intmath.h), whose contract +# is the funnel's preflight, so only the funnel may call it. BANNED_LEAVES = ( "MaxAbsReduce", "MaxAbsReduceWide", @@ -107,6 +110,7 @@ "DynamicScaleReciprocal", "RequantTokenCode", "RequantTokenCodeWide", + "RequantRowWide", "NarrowAccumulatorToI32", ) @@ -700,7 +704,7 @@ def check_door_count( expected: tuple[str, ...] = _EXPECTED_DOOR_FUNCTIONS, ) -> list[str]: """Significant 9 (Poirot e4b398c review, T-1357/D-SLM433): `scan_files` above - holds every OTHER forward TU off the eight banned leaves; nothing holds the + holds every OTHER forward TU off the nine banned leaves; nothing holds the DOOR COUNT itself inside the funnel's own file. Asserts the exact, named set of functions in `funnel_path` that forward a banned leaf to an outside caller equals `expected` -- a second door opened alongside diff --git a/tests/ci/test_check_no_forward_leaf_calls.py b/tests/ci/test_check_no_forward_leaf_calls.py index ef65ff6c..d536f0e4 100644 --- a/tests/ci/test_check_no_forward_leaf_calls.py +++ b/tests/ci/test_check_no_forward_leaf_calls.py @@ -10,7 +10,7 @@ `finally`). WHY SCRATCH FIXTURES, NOT THE REAL FORWARD DIRECTORY. The mechanism cells below -(the eight banned-leaf rule-coverage cells, the input-coverage cell, and the +(the nine banned-leaf rule-coverage cells, the input-coverage cell, and the allowlist-control cell) drive the check against constructed scratch directories standing in for "a forward TU," never against the real src/forward/ tree -- exactly as check_no_pow_operator.py's own self-test @@ -78,7 +78,7 @@ def _write(tmpdir: str, rel_path: str, content: str) -> str: return abs_path -# --- Rule coverage: every one of the eight banned leaves is individually detected. --- +# --- Rule coverage: every one of the nine banned leaves is individually detected. --- def test_each_banned_leaf_is_individually_detected(): diff --git a/tests/support/attention_cases.h b/tests/support/attention_cases.h new file mode 100644 index 00000000..5d0d1e33 --- /dev/null +++ b/tests/support/attention_cases.h @@ -0,0 +1,655 @@ +// Attention and per-row sites plan, slice S2 (prob·V on int16 multiply-add): the fixed input set that +// the digest section `c32_attention`, the golden-pin generator (tools/gen_attn_rowsite_golden.cpp) and +// the suite's golden and grid cells all run through GemmProbQ15Accumulate. Header-only, so it needs no +// build entry: the digest and the generator include it by relative path, the suite through `tests/`. +// Later slices (S4, S5, S6) append their own entries to the same section and their own hashes. +// +// Everything here calls only GemmProbQ15Accumulate, SoftmaxRowQ15 and IExpScaleConstants, whose +// signatures are unchanged since v1.9.0, so the generator can be built against the v1.9.0 tag's library +// and the pin takes no input from the code it grades (plan §3.3 evidence 3). +// +// The set is plan §8 4.S2's grid: head_dim {4, 8, 12, 16, 60, 64, 100, 128, 132, 256, 512} x width +// {1, 2, 3, 7, 8, 9, 63, 64, 65, 1,024, 4,097} with realistic and peaked rows (both sides of +// head_dim % 16, odd widths for the unpaired last key), plus the int16 condition's corners (p = 32,767 +// with Sum p = 2^15 exactly and 2^15 + 1, one-hot rows at width 1 and inside a wider row, Sum p = 2^15 +// over many keys with every value -128 or 127, the lane bound's corners), width 0, and 2.S2's four +// hostile rows, each failing exactly one conjunct. Each call contributes width, head_dim and its whole +// output row (poisoned before the call, so the zeroing contract is digested too). +#ifndef SUPERSLM_TESTS_SUPPORT_ATTENTION_CASES_H +#define SUPERSLM_TESTS_SUPPORT_ATTENTION_CASES_H + +#include +#include +#include +#include + +#include "superslm/intmath.h" +#include "superslm/matmul.h" + +namespace superslm_attention_cases { + +// splitmix64, integer-only (the same generator as rowsite_cases.h, restated so this header stands alone). +struct Rng { + uint64_t s; + explicit Rng(uint64_t seed) : s(seed) {} + uint64_t Next() { + s += 0x9e3779b97f4a7c15ULL; + uint64_t z = s; + z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL; + z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL; + return z ^ (z >> 31); + } + int64_t InRange(int64_t lo, int64_t hi) { + const uint64_t span = static_cast(hi - lo) + 1ULL; + return lo + static_cast(Next() % span); + } +}; + +inline constexpr size_t kPvHeadDims[] = {4, 8, 12, 16, 60, 64, 100, 128, 132, 256, 512}; +inline constexpr size_t kPvWidths[] = {1, 2, 3, 7, 8, 9, 63, 64, 65, 1024, 4097}; +inline constexpr int64_t kPvPoison = INT64_C(0x5A5A5A5A5A5A5A5A); + +// One prob·V call: its row, its values and a label. +struct PvCase { + const char* label; + size_t width; + size_t head_dim; + std::vector probs; // exactly `width` elements + std::vector values; // exactly width * head_dim elements +}; + +// A realistic row: random weights e_k, p_k = floor(e_k * 2^15 / Sum e), as the softmax forms them +// (Sum p <= 2^15 by construction). `spread` sets how peaked it is: weights are 2^(random in [0, spread]). +inline std::vector RealisticRow(size_t width, int spread, Rng& rng) { + std::vector e(width), p(width); + int64_t total = 0; + for (size_t k = 0; k < width; ++k) { + const int sh = static_cast(rng.InRange(0, spread)); + e[k] = (INT64_C(1) << sh) + rng.InRange(0, (INT64_C(1) << sh) - 1); + total += e[k]; + } + for (size_t k = 0; k < width; ++k) p[k] = total == 0 ? 0 : (e[k] << 15) / total; + return p; +} + +inline std::vector RandomValues(size_t n, Rng& rng) { + std::vector v(n); + for (auto& x : v) x = static_cast(rng.InRange(-128, 127)); + return v; +} + +// Calls fn(const PvCase&) for every case of the set, in a fixed order. +template +void ForEachProbVCase(Fn&& fn) { + Rng rng(0x5332505653455431ULL); // "S2PVSET1" + // The grid: a realistic row and a peaked row at every (head_dim, width). + for (size_t hd : kPvHeadDims) + for (size_t w : kPvWidths) + for (int spread : {4, 24}) { + PvCase c{spread == 4 ? "4.S2 grid, realistic row" : "4.S2 grid, peaked row", w, hd, + RealisticRow(w, spread, rng), RandomValues(w * hd, rng)}; + fn(c); + } + + const auto values_of = [&rng](size_t n, int mode) { + std::vector v(n); + for (auto& x : v) x = mode == 0 ? static_cast(rng.InRange(-128, 127)) : static_cast(mode); + return v; + }; + // The int16 condition's inside and outside corners (4.S2, 7.S2). + for (size_t hd : {size_t{16}, size_t{64}, size_t{60}}) { + fn(PvCase{"4.S2 corner p = 32,767 and 1 (Sum p = 2^15)", 2, hd, {32767, 1}, values_of(2 * hd, 0)}); + fn(PvCase{"4.S2 corner p = 32,767 and 2 (Sum p = 2^15 + 1)", 2, hd, {32767, 2}, values_of(2 * hd, 0)}); + fn(PvCase{"4.S2 corner width-1 one-hot (p = 2^15)", 1, hd, {32768}, values_of(hd, 0)}); + fn(PvCase{"4.S2 corner one-hot inside a width-3 row", 3, hd, {0, 32768, 0}, values_of(3 * hd, 0)}); + fn(PvCase{"4.S2 corner width 0", 0, hd, {}, {}}); + } + // Sum p = 2^15 exactly over many keys, every value at an extreme: each lane reaches 128 * 2^15 = 2^22 + // (all -128) or 127 * 2^15 (all 127), the §5.2 lane bound's corner, at an odd width. + for (int mode : {-128, 127}) { + std::vector p(1025, 0); + for (size_t k = 0; k < 1024; ++k) p[k] = 32; // 1,024 x 32 = 2^15; the 1,025th key is 0 + fn(PvCase{mode < 0 ? "4.S2 corner Sum p = 2^15, every v = -128" : "4.S2 corner Sum p = 2^15, every v = 127", + 1025, 64, p, values_of(1025 * 64, mode)}); + } + { + std::vector p(3, 0); + p[2] = 32767; // the unpaired last key carries the maximum + fn(PvCase{"4.S2 corner odd last key p = 32,767", 3, 64, p, values_of(3 * 64, -128)}); + } + + // 2.S2: each row fails exactly one conjunct of the guard (head_dim 64). + { + std::vector p = RealisticRow(64, 8, rng); + for (auto& x : p) x /= 2; // keeps Sum p far from 2^15 whatever key 17 holds + p[17] = -32769; // only p >= 0 fails; the int16 pack would truncate it + fn(PvCase{"2.S2 p = -32,769 at one key", 64, 64, p, values_of(64 * 64, 0)}); + } + { + std::vector p(1024); + std::vector v(1024 * 64); + for (size_t k = 0; k < 1024; ++k) { + p[k] = (k % 2 == 0) ? 32767 : -32767; // Sum p = 0: only p >= 0 fails + for (size_t d = 0; d < 64; ++d) v[k * 64 + d] = (k % 2 == 0) ? 127 : -127; // every product positive + } + fn(PvCase{"2.S2 alternating p = +-32,767, v sign-matched, W = 1,024", 1024, 64, p, v}); + } + fn(PvCase{"2.S2 width-1 one-hot p = 32,768", 1, 64, {32768}, values_of(64, 0)}); + fn(PvCase{"2.S2 W = 1,024, p = 32,767, v = 127", 1024, 64, std::vector(1024, 32767), + values_of(1024 * 64, 127)}); +} + +// The whole S2 set through GemmProbQ15Accumulate: per call, width, head_dim and every output value. +template +void RunProbVCases(Emit& emit) { + ForEachProbVCase([&emit](const PvCase& c) { + std::vector out(c.head_dim, kPvPoison); + superslm::GemmProbQ15Accumulate(c.probs.data(), c.values.data(), c.width, c.head_dim, out.data()); + emit(static_cast(c.width)); + emit(static_cast(c.head_dim)); + for (int64_t x : out) emit(x); + }); +} + +// ==== Slice S4: the guarded softmax (plan §4.4, §5.4, §8 2.S4 and 4.S4) ============================= +// +// The S4 set runs SoftmaxRowQ15 over plan §8 4.S4's grid and corners and 2.S4's hostile rows. Its rows +// are chosen by two test-side copies written from the plan, never by the build under test: +// +// * TestSoftmaxGuard, §5.4's row guard in dependency order, each conjunct reported separately so a 2.S4 +// row can be shown to fail exactly one; +// * SoftmaxEstimateReplica, the fast path's estimate-and-correct arithmetic element by element, which +// reports which corrections a row needs, so the correction rows are chosen, not hoped for. +// +// The estimates are integer, not IEEE double (docs/attention-rowsites-s4-progress.md, deviation 1: the +// library is floating-point-free and the fp-free scan gates it). §5.4's own floating-point bullet is what +// carries over: the estimates only need to land within one of the floor, because exactness comes from the +// exact integer corrections. Per element, after the max shift (a = min(max - s, 30 q_ln2) >= 0): +// +// z estimate: (a * inv_z) >> kz, inv_z = floor(2^kz / q_ln2), kz = 30 + bit_width(q_ln2). inv_z is at +// most 2^31, a * inv_z < 2^61, and the estimate never exceeds floor(a / q_ln2) (a floored +// reciprocal) and is at most one below it. So only the UPWARD z correction can fire, exactly as +// §5.4 step 3 argues for the double estimate; the downward one is kept as defensive code. +// p estimate: (e * R) >> 47, R = round(2^62 / denom) = (2^62 + floor(denom / 2)) / denom. e <= 2^47 and +// e <= denom give e * R <= 2^62 + 2^46, and |error| <= e / 2^48 <= 1/2, so the estimate lands on the +// floor or one either side of it: BOTH p corrections are live, as §5.4 step 4 has them. +// +// Integer-only, so this header builds anywhere the suite does (no __int128: the MSVC legs build it). + +// §5.4's guard, one flag per conjunct. `scores` is read only when width is in range. +struct SoftmaxGuard { + bool width_ok = true, q_ln2_ge1 = true, q_c_ge0 = true, m_ge1 = true, m_le = true, ratio = true, scores_ok = true; + bool Fast() const { return width_ok && q_ln2_ge1 && q_c_ge0 && m_ge1 && m_le && ratio && scores_ok; } + int Failing() const { return !width_ok + !q_ln2_ge1 + !q_c_ge0 + !m_ge1 + !m_le + !ratio + !scores_ok; } +}; + +inline constexpr size_t kSmMaxWidth = size_t{1} << 14; +inline constexpr int64_t kSmMaxM = int64_t{1} << 47; +inline constexpr int64_t kSmScoreLimit = int64_t{1} << 61; + +inline SoftmaxGuard TestSoftmaxGuard(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, int64_t q_c) { + SoftmaxGuard g; + g.width_ok = width >= 1 && width <= kSmMaxWidth; + g.q_ln2_ge1 = q_ln2 >= 1; + g.q_c_ge0 = q_c >= 0; + // M = q_b^2 + q_c, judged exactly without 128-bit arithmetic: |q_b| > 2^32 makes q_b^2 > 2^64 > 2^47 + |q_c|. + const uint64_t ab = q_b < 0 ? 0 - static_cast(q_b) : static_cast(q_b); + if (ab > (uint64_t{1} << 32)) { + g.m_le = false; + } else { + const uint64_t sq = ab == (uint64_t{1} << 32) ? UINT64_MAX : ab * ab; // 2^64 saturates; still > 2^47 + 2^63 + if (q_c >= 0) { + g.m_ge1 = sq > 0 || q_c >= 1; + g.m_le = sq <= static_cast(kSmMaxM) && sq + static_cast(q_c) <= static_cast(kSmMaxM); + } else { + const uint64_t mag = 0 - static_cast(q_c); // |q_c| <= 2^63 + g.m_ge1 = sq > mag; // M >= 1 + g.m_le = sq <= mag || sq - mag <= static_cast(kSmMaxM); + } + } + // q_ln2 <= 2 q_b + 1, exact: q_b is an int64, so 2 q_b + 1 is formed from its halves. + if (q_b >= 0) { + g.ratio = q_ln2 <= q_b || q_b == INT64_MAX || q_ln2 - q_b <= q_b + 1; + } else { + g.ratio = q_ln2 < 0 && q_ln2 - q_b <= q_b + 1; // q_ln2 - q_b cannot overflow for q_ln2 < 0 <= -q_b + } + if (g.width_ok) + for (size_t k = 0; k < width; ++k) + if (scores[k] > kSmScoreLimit || scores[k] < -kSmScoreLimit) g.scores_ok = false; + return g; +} + +// Which corrections the fast path's estimates need on one row, element counts. +struct SoftmaxCorrections { + long long z_up = 0, z_down = 0, p_up = 0, p_down = 0; +}; + +inline int BitWidth64(uint64_t x) { + int b = 0; + while (x != 0) { + ++b; + x >>= 1; + } + return b; +} + +// The replica of the fast path (above), for a row the guard copy admits: fills `out` (width values) with +// the probabilities and `c` with the number of elements on which each correction fires. `skip` names one +// correction the replica leaves out (0 none, 1 z up, 2 p up, 3 p down), which is how a test states what +// the §9 "correction skipped" mutant would compute. Returns false, touching nothing, outside the guard. +inline bool SoftmaxEstimateReplica(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, int64_t q_c, + int64_t* out, SoftmaxCorrections* c, int skip = 0) { + if (!TestSoftmaxGuard(scores, width, q_ln2, q_b, q_c).Fast()) return false; + int64_t mx = scores[0]; + for (size_t k = 1; k < width; ++k) + if (scores[k] > mx) mx = scores[k]; + const int kz = 30 + BitWidth64(static_cast(q_ln2)); + const uint64_t inv_z = (uint64_t{1} << kz) / static_cast(q_ln2); + const uint64_t clip = 30 * static_cast(q_ln2); + std::vector e(width); + int64_t total = 0; + for (size_t k = 0; k < width; ++k) { + uint64_t a = static_cast(mx - scores[k]); // <= 2^62 + if (a > clip) a = clip; + int64_t z = static_cast((a * inv_z) >> kz); + int64_t r = static_cast(a) - z * q_ln2; + if (r >= q_ln2) { + ++c->z_up; + if (skip != 1) { + ++z; + r -= q_ln2; + } + } + if (r < 0) { + ++c->z_down; + --z; + r += q_ln2; + } + const int64_t base = q_b - r; // q_p + q_b with q_p = -r + e[k] = (base * base + q_c) >> z; + total += e[k]; + } + const uint64_t denom = static_cast(total); + const uint64_t R = ((uint64_t{1} << 62) + (denom >> 1)) / denom; + for (size_t k = 0; k < width; ++k) { + const uint64_t num = static_cast(e[k]) << 15; + uint64_t p = (static_cast(e[k]) * R) >> 47; + uint64_t prod = p * denom; + if (prod > num) { + ++c->p_down; + if (skip != 3) { + --p; + prod -= denom; + } + } + if (prod + denom <= num) { + ++c->p_up; + if (skip != 2) ++p; + } + out[k] = static_cast(p); + } + return true; +} + +// Integer square root, floor, for 0 <= x < 2^62. +inline int64_t ISqrtFloor(int64_t x) { + uint64_t lo = 0, hi = uint64_t{1} << 31; + while (lo < hi) { + const uint64_t mid = (lo + hi + 1) / 2; + if (mid * mid <= static_cast(x)) + lo = mid; + else + hi = mid - 1; + } + return static_cast(lo); +} + +// One SoftmaxRowQ15 call of the set. +struct SmCase { + const char* label; + int64_t q_ln2, q_b, q_c; + std::vector scores; // the row, width = size() + bool aliased; // scores == out_probs +}; + +inline constexpr int64_t kSmPoison = INT64_C(0x3C3C3C3C3C3C3C3C); + +// The realistic constant triples: IExpScaleConstants over the forward's scale range, C30's pinned +// coefficients (the production call site's), mantissas in [2^30, 2^31) and exponents -52..-31 (inside the +// guard; the 0.5B-width forward's rows sit at about -41 to -39), plus -55 and -53 (M above 2^47: outside) +// and -30 (q_ln2 = 0: outside). +inline std::vector> SmRealisticConstants(Rng& rng, int per_exponent) { + std::vector> out; + for (int e = -55; e <= -30; ++e) { + if (e == -54) continue; + for (int t = 0; t < per_exponent; ++t) { + const int64_t m = (INT64_C(1) << 30) + rng.InRange(0, (INT64_C(1) << 30) - 1); + int64_t a = 0, b = 0, c = 0; + if (superslm::IExpScaleConstants(m, e, superslm::kIExpLn2Q, 30, superslm::kIExpBQ, 30, superslm::kIExpCaQ, 30, + &a, &b, &c) == superslm::IExpScaleDomain::kOk) + out.push_back({a, b, c}); + } + } + return out; +} + +// A score row of `width` with `kind`: 0 realistic (a spread of 2^0..2^20, int8 dot products at head_dim 64 +// reach 2^20), 1 one dominant score, 2 clip-heavy (most scores past 30 q_ln2 below the max: z = 30), 3 exact +// multiples of q_ln2 below the max (z at every value 0..30). +inline std::vector SmScoreRow(size_t width, int kind, int64_t q_ln2, Rng& rng) { + std::vector s(width); + const int bits = static_cast(rng.InRange(0, 20)); + for (auto& x : s) { + switch (kind) { + case 0: x = rng.InRange(-(INT64_C(1) << bits), INT64_C(1) << bits); break; + case 1: x = rng.InRange(-40 * q_ln2, -q_ln2); break; + case 2: x = rng.InRange(-64 * q_ln2, 0); break; + default: x = -q_ln2 * rng.InRange(0, 31); break; + } + } + if (kind == 1 && width > 0) s[rng.Next() % width] = rng.InRange(0, 1000); + return s; +} + +// The steered generator for rows the p DOWNWARD correction needs (plan §5.4 step 4, §8 4.S4, after +// probes/rev3/probe_pdown.cpp): the denominator first, then a row summing to it. The probe steered the +// double estimate; this one steers the integer estimate above, whose error comes from R's rounding. The +// estimate for e = M is floor(M R / 2^47). For a reciprocal R = k, take N = floor(M k / 2^47) and +// denom = ceil(M 2^15 / N) + j for small j: the true quotient M 2^15 / denom is then just below N, so the +// estimate overshoots whenever denom still rounds to R = k (which holds when M k / 2^47 sits less than about +// M / 2^48 above N, so about half of all k). Each (k, j) is kept only if the overshoot is confirmed. The row +// is floor(denom / M) scores at the maximum (each e = M) plus fine elements whose e values sum to the exact +// remainder: z = 0 elements with e = base^2 + q_c (base from an integer square root) while the remainder is +// >= 2^18, then one z = 29 element whose e is what is left. From each seed k0 the generator tries k0, k0 + 1, +// ... and keeps at most `per_k` rows; widths run from about 2^15 / k0. +inline void SmSteeredPDownRows(int64_t q_ln2, int64_t q_b, int64_t q_c, const std::vector& seeds, int per_k, + const char* label, std::vector& out) { + const int64_t M = q_b * q_b + q_c; // <= 2^47 + const auto elem = [&](int64_t j, int64_t z) { return -(z * q_ln2 + (q_b - j)); }; // r = q_b - j, base = j + for (int64_t k0 : seeds) { + int kept = 0; + for (int64_t k = k0; k < k0 + 32 && kept < per_k; ++k) { + const int64_t N = static_cast((static_cast(M) * static_cast(k)) >> 47); + if (N < 1) continue; + for (int64_t j = 0; j < 8 && kept < per_k; ++j) { + const int64_t D = (M * 32768 + N - 1) / N + j; // M 2^15 <= 2^62 + const int64_t n = D / M; + if (n < 1 || static_cast(n) > kSmMaxWidth) continue; + const uint64_t R = ((uint64_t{1} << 62) + (static_cast(D) >> 1)) / static_cast(D); + const uint64_t p_est = (static_cast(M) * R) >> 47; + if (!(p_est * static_cast(D) > (static_cast(M) << 15))) continue; + int64_t rem = D - n * M; + std::vector s(static_cast(n), 0); + bool ok = true; + while (rem >= (INT64_C(1) << 18)) { + int64_t b = ISqrtFloor(rem - q_c > 0 ? rem - q_c : 0); + if (b > q_b) b = q_b; + if (b * b + q_c > rem || b * b + q_c <= 0) { + ok = false; + break; + } + s.push_back(elem(b, 0)); + rem -= b * b + q_c; + } + if (ok && rem > 0) { // one z = 29 element with (b^2 + q_c) >> 29 == rem + int64_t b = ISqrtFloor((rem << 29) > q_c ? (rem << 29) - q_c : 0); + while (b > 0 && ((b * b + q_c) >> 29) >= rem) --b; + while (((b * b + q_c) >> 29) < rem) ++b; + if (((b * b + q_c) >> 29) != rem || b > q_b || q_b - b >= q_ln2) ok = false; + else s.push_back(elem(b, 29)); + } + if (!ok || s.size() > kSmMaxWidth) continue; + out.push_back(SmCase{label, q_ln2, q_b, q_c, std::move(s), false}); + ++kept; + } + } + } +} + +// Calls fn(const SmCase&) for every case of the S4 set, in a fixed order. +template +void ForEachSoftmaxCase(Fn&& fn) { + Rng rng(0x5334534D53455431ULL); // "S4SMSET1" + const std::vector> triples = SmRealisticConstants(rng, 2); + // 4.S4: the grid. Plan widths {1, 2, 3, 4, 5, 2^14}, plus the kernels' block edges (4 and 8 lanes) and + // the forward's widths, at every realistic triple (inside and outside the guard), four row kinds. + for (size_t w : {size_t{1}, size_t{2}, size_t{3}, size_t{4}, size_t{5}, size_t{7}, size_t{8}, size_t{9}, + size_t{15}, size_t{16}, size_t{17}, size_t{63}, size_t{64}, size_t{65}, size_t{301}}) + for (const auto& t : triples) + for (int kind = 0; kind < 4; ++kind) { + const bool aliased = (rng.Next() & 7) == 0; + fn(SmCase{aliased ? "4.S4 grid, aliased" : "4.S4 grid", t[0], t[1], t[2], + SmScoreRow(w, kind, t[0], rng), aliased}); + } + for (int kind = 0; kind < 4; ++kind) { + const auto& t = triples[static_cast(rng.Next() % triples.size())]; + fn(SmCase{"4.S4 grid, width 1,024", t[0], t[1], t[2], SmScoreRow(1024, kind, t[0], rng), kind == 3}); + fn(SmCase{"4.S4 grid, width 2^14", t[0], t[1], t[2], SmScoreRow(kSmMaxWidth, kind, t[0], rng), kind == 1}); + } + // 4.S4: total = 1 (M = 1 and every other element clipped to e = 0) and one element equal to total + // (p = 2^15); z at 0 and 30 on every clip-heavy row above. + fn(SmCase{"4.S4 total = 1, p = 2^15", 3, 1, 0, {0, -1000, -2000}, false}); + fn(SmCase{"4.S4 total = 1, p = 2^15, aliased", 3, 1, 0, {0, -1000, -2000}, true}); + // 4.S4 inside corners, each fast. + fn(SmCase{"4.S4 inside: M = 2^47, width 2^14, equal scores (denom = 2^61)", 1, 0, kSmMaxM, + std::vector(kSmMaxWidth, 5), false}); + { + const int64_t qb = 636211, qc = 212166321733; // the -49 triple's q_b and q_c, q_ln2 at its ceiling + fn(SmCase{"4.S4 inside: q_ln2 = 2 q_b + 1, realistic spread", 2 * qb + 1, qb, qc, + SmScoreRow(301, 0, 2 * qb + 1, rng), false}); + fn(SmCase{"4.S4 inside: q_ln2 = 2 q_b + 1, exact multiples", 2 * qb + 1, qb, qc, + SmScoreRow(301, 3, 2 * qb + 1, rng), false}); + } + fn(SmCase{"4.S4 inside: scores of exactly +-2^61", 9, 4, 0, {kSmScoreLimit, -kSmScoreLimit, 0}, false}); + fn(SmCase{"4.S4 inside: M = 1", 3, 1, 0, {0, -1, -2, -100}, false}); + fn(SmCase{"4.S4 inside: q_c = 0", 9, 4, 0, {0, -3, -9, -1000}, false}); + fn(SmCase{"4.S4 inside: q_ln2 = 1", 1, 0, kSmMaxM, {0, -1, -5, -29, -30, -31}, false}); + // 4.S4 correction rows, chosen by the replica (at least 16 each; 24 are kept). z up: scores at exact + // multiples of q_ln2 below the max, over many q_ln2. p up: ordinary rows. A row is kept only where the + // replica says the correction fires AND leaving it out changes the row's output (the i-exp construction + // halves its value per ln 2 step, so a skipped z correction often lands on the same probability; those + // rows would not decide the §9 mutant). + { + int z_up = 0, p_up = 0; + std::vector good, skipped; + for (int it = 0; it < 20000 && (z_up < 24 || p_up < 24); ++it) { + const auto& t = triples[static_cast(rng.Next() % triples.size())]; + const size_t w = static_cast(rng.InRange(2, 40)); + std::vector s = SmScoreRow(w, it % 2 == 0 ? 3 : 0, t[0], rng); + good.assign(w, 0); + SoftmaxCorrections c; + if (!SoftmaxEstimateReplica(s.data(), w, t[0], t[1], t[2], good.data(), &c)) continue; + const auto decides = [&](int skip) { + SoftmaxCorrections unused; + skipped.assign(w, 0); + SoftmaxEstimateReplica(s.data(), w, t[0], t[1], t[2], skipped.data(), &unused, skip); + return skipped != good; + }; + if (c.z_up > 0 && z_up < 24 && decides(1)) { + ++z_up; + fn(SmCase{"4.S4 correction row: z up", t[0], t[1], t[2], s, false}); + } else if (c.p_up > 0 && p_up < 24 && decides(2)) { + ++p_up; + fn(SmCase{"4.S4 correction row: p up", t[0], t[1], t[2], s, false}); + } + } + } + // p down: the steered generator. The plan's constants put M within 2^-24 of 2^47, where M R / 2^47 sits + // just below the integer R and the integer estimate for e = M can never overshoot (progress file, + // deviation 1), so the constants here put M near 3/4 and 7/10 of 2^47, in the plan's two shapes: q_ln2 = + // 2 q_b + 1 with q_c = 0, and q_ln2 = q_b with q_c = 12,345. The widths run from 6 to 8,193 (the second + // set reaches N = 4, width 2^15 / 4, because 4 / 0.7 rounds up). + { + std::vector rows; + const std::vector ks = {3, 24, 71, 200, 553, 1648, 2600, 3500, 5052, 7000, 9000, 12000}; + SmSteeredPDownRows(2 * 10273742 + 1, 10273742, 0, ks, 2, "4.S4 correction row: p down (steered)", rows); + SmSteeredPDownRows(9925730, 9925730, 12345, ks, 2, "4.S4 correction row: p down (steered, q_c = 12,345)", rows); + for (const SmCase& c : rows) fn(c); + } + // 2.S4: each row fails exactly one conjunct of §5.4's guard. + // The width and score rows use the first realistic triple inside the guard, so each fails one conjunct. + std::array rt{}; + for (const auto& t : triples) { + const int64_t zero = 0; + if (TestSoftmaxGuard(&zero, 1, t[0], t[1], t[2]).Fast()) { + rt = t; + break; + } + } + fn(SmCase{"2.S4 q_ln2 = 0", 0, 0, 1, {0, -3, -7}, false}); + fn(SmCase{"2.S4 q_ln2 = -1", -1, 0, 1, {0, -3, -7}, false}); + fn(SmCase{"2.S4 q_c = -50 (q_b 10, q_ln2 21)", 21, 10, -50, {0, -10}, false}); + fn(SmCase{"2.S4 M = 0 (q_b 0, q_c 0, q_ln2 1), width 3", 1, 0, 0, {0, 0, 0}, false}); + fn(SmCase{"2.S4 M = 2^47 + 1", 1, 0, kSmMaxM + 1, {0, -1}, false}); + fn(SmCase{"2.S4 q_ln2 = 2 q_b + 2 (q_b 4, q_ln2 10)", 10, 4, 0, {0, -9}, false}); + fn(SmCase{"2.S4 width 2^14 + 1", rt[0], rt[1], rt[2], SmScoreRow(kSmMaxWidth + 1, 0, rt[0], rng), false}); + { + std::vector s(64, 0); + s[17] = kSmScoreLimit + 1; + fn(SmCase{"2.S4 one score 2^61 + 1", rt[0], rt[1], rt[2], s, false}); + s[17] = -kSmScoreLimit - 1; + fn(SmCase{"2.S4 one score -(2^61 + 1)", rt[0], rt[1], rt[2], s, false}); + } + // The off-ratio witness (tests/sslm_c32_softmax_row_width_gate_fixtures.h, kSoftmaxRowOffRatioWitness), + // restated here so the generator needs no other header: a realistic hostile row that fails more than one + // conjunct (q_c >= 0, M >= 1 and q_ln2 <= 2 q_b + 1), so it kills no single-conjunct mutant. + fn(SmCase{"2.S4 off-ratio witness", INT64_C(3000000001), 10, -100, {0, INT64_C(-3000000000), INT64_C(-2999999999)}, + false}); +} + +// The whole S4 set through SoftmaxRowQ15: per call, width, the bool and every output value (the output +// row is poisoned before the call; an aliased call writes over its own scores). +template +void RunSoftmaxCases(Emit& emit) { + ForEachSoftmaxCase([&emit](const SmCase& c) { + const size_t w = c.scores.size(); + std::vector out(w, kSmPoison); + bool ok; + if (c.aliased) { + out = c.scores; + ok = superslm::SoftmaxRowQ15(out.data(), w, c.q_ln2, c.q_b, c.q_c, out.data()); + } else { + ok = superslm::SoftmaxRowQ15(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, out.data()); + } + emit(static_cast(w)); + emit(ok ? 1 : 0); + for (int64_t x : out) emit(x); + }); +} + +// ==== Slice S5: the Q31 score in three 16-bit pieces, per head (plan §4.5, §5.5, §8 4.S5 and 7.S5) ===== +// +// The S5 set scores one query head against `width` keys through a caller-supplied row function: the +// digest and the suite pass the build's QkQ31ScoreRow; the golden-pin generator, built against the v1.9.0 +// tag (which has no row entry), passes a loop over the v1.9.0 per-key QkQ31Score. The set is §8 4.S5's grid, +// head_dim {4, 8, 60, 64, 128, 132, 256, 512, 513, 516} x width {1, 7, 8, 9, 1,024} plus widths 15, 16 +// and 17 (the AVX-512 body's 16-key block, full and partial), in three operand kinds; 7.S5b's margin +// corners at head_dim 512 (and the same rows at 516, past the guard); and 7.S5c's rounding ties. +// +// Every ratio here is in [1, 2^31], the loader's range (G6), inside S5's fast-path range [0, 2^32): §3.3 +// keeps pin and digest inputs in contract, because outside [0, 2^32) the v1.9.0 SIMD tiers of QkQ31Score +// already differ from its scalar reference, so no single pin could hold there. The out-of-contract rows +// (2.S5) are suite-only cells compared with the same binary's per-key QkQ31Score. + +inline constexpr size_t kQ31HeadDims[] = {4, 8, 60, 64, 128, 132, 256, 512, 513, 516}; +inline constexpr size_t kQ31Widths[] = {1, 7, 8, 9, 15, 16, 17, 1024}; +inline constexpr int64_t kQ31Poison = INT64_C(0x2B2B2B2B2B2B2B2B); +inline constexpr int64_t kQ31RatioMax = INT64_C(1) << 31; // the loader's maximum + +// One score-row call: one query head's q, `width` key rows of head_dim (the K store's layout for one KV +// head, positions 0 .. width - 1), and the KV head's per-channel ratios. +struct Q31Case { + const char* label; + size_t head_dim; + size_t width; + std::vector q; // head_dim + std::vector keys; // width * head_dim + std::vector ratio; // head_dim +}; + +// kind 0: uniform q and k in [-128, 127], ratio in [1, 2^31]; kind 1: q and k at the int8 extremes +// {-128, 127}, ratio in {1, 2^31 - 1, 2^31}; kind 2: uniform q and k, ratio 2^31 on every channel (the +// real value, as the QK-norm fixture carries). +inline Q31Case MakeQ31GridCase(size_t head_dim, size_t width, int kind, Rng& rng) { + static const char* const kLabels[] = {"4.S5 grid, uniform", "4.S5 grid, int8 extremes", "4.S5 grid, ratio 2^31"}; + Q31Case c{kLabels[kind], head_dim, width, std::vector(head_dim), std::vector(width * head_dim), + std::vector(head_dim)}; + auto code = [&]() -> int8_t { + if (kind == 1) return rng.Next() & 1 ? int8_t{127} : int8_t{-128}; + return static_cast(rng.InRange(-128, 127)); + }; + for (auto& x : c.q) x = code(); + for (auto& x : c.keys) x = code(); + for (auto& r : c.ratio) { + if (kind == 0) r = rng.InRange(1, kQ31RatioMax); + else if (kind == 1) r = (rng.Next() % 3 == 0) ? 1 : (rng.Next() & 1 ? kQ31RatioMax : kQ31RatioMax - 1); + else r = kQ31RatioMax; + } + return c; +} + +// 7.S5b: every limb sum at §5.5's int32 margin. Channel products w = q * ratio with a0 = a1 = 32,767 (w +// is -1, or 2^30 - 1) against keys of -128 (or 127) on every channel: at head_dim 512 each limb's per-key +// sum is -128 * 32,767 * 512 = -2,147,418,112 (margin 65,535 to int32's minimum) or 127 * 32,767 * 512 = +// 2,130,690,048. At 516 the same rows are past the guard (and would wrap an int32 lane). +inline Q31Case MakeQ31MarginCase(size_t head_dim, size_t width, bool negative_w, int8_t key) { + Q31Case c{negative_w ? "7.S5b margin corner, w = -1, key fill" : "7.S5b margin corner, w = 2^30 - 1, key fill", + head_dim, width, std::vector(head_dim, negative_w ? int8_t{-1} : int8_t{1}), + std::vector(width * head_dim, key), + std::vector(head_dim, negative_w ? INT64_C(1) : (INT64_C(1) << 30) - 1)}; + return c; +} + +// 7.S5c: totals on and beside the rounding ties x = +-2^30 (mod 2^31). Two channels carry the products: +// ratio 2^30 - 5 and 5 (every limb nonzero), q = +-1 on both, key t on both, so the key's total is exactly +// q * t * 2^30; odd t is a tie, which RoundingDivideByPOT rounds away from zero. The remaining channels +// add +-1 (ratio 1) on some keys to land one beside the tie. +inline Q31Case MakeQ31TieCase(int8_t q_sign) { + const size_t hd = 64, width = 24; + Q31Case c{q_sign > 0 ? "7.S5c ties, q = +1" : "7.S5c ties, q = -1", hd, width, std::vector(hd, 0), + std::vector(width * hd, 0), std::vector(hd, 1)}; + c.q[0] = q_sign; + c.q[1] = q_sign; + c.q[2] = 1; + c.ratio[0] = (INT64_C(1) << 30) - 5; + c.ratio[1] = 5; + static constexpr int8_t kT[] = {1, -1, 3, -3, 5, -5, 127, -127, -128, 2, -2, 0}; + for (size_t j = 0; j < width; ++j) { + const int8_t t = kT[j % 12]; + c.keys[j * hd + 0] = t; + c.keys[j * hd + 1] = t; + c.keys[j * hd + 2] = j < 12 ? int8_t{0} : (j % 2 ? int8_t{1} : int8_t{-1}); // beside the tie + } + return c; +} + +template +void ForEachQ31Case(Fn&& fn) { + Rng rng(0x5E5E'0031'7153'0005ULL); + for (size_t hd : kQ31HeadDims) + for (size_t w : kQ31Widths) + for (int kind = 0; kind < 3; ++kind) fn(MakeQ31GridCase(hd, w, kind, rng)); + for (size_t hd : {size_t{512}, size_t{516}}) + for (size_t w : {size_t{1}, size_t{17}}) + for (bool neg : {true, false}) + for (int8_t key : {int8_t{-128}, int8_t{127}}) fn(MakeQ31MarginCase(hd, w, neg, key)); + fn(MakeQ31TieCase(1)); + fn(MakeQ31TieCase(-1)); +} + +// The whole S5 set through `row(q, keys, ratio, head_dim, width, out)`: per call, width, head_dim and every +// score (the output row is poisoned before the call). +template +void RunQ31Cases(Emit& emit, Row&& row) { + ForEachQ31Case([&](const Q31Case& c) { + std::vector out(c.width, kQ31Poison); + row(c.q.data(), c.keys.data(), c.ratio.data(), c.head_dim, c.width, out.data()); + emit(static_cast(c.width)); + emit(static_cast(c.head_dim)); + for (int64_t x : out) emit(x); + }); +} + +} // namespace superslm_attention_cases + +#endif // SUPERSLM_TESTS_SUPPORT_ATTENTION_CASES_H diff --git a/tests/support/matmul_dispatch_instrument.h b/tests/support/matmul_dispatch_instrument.h index 6b2a8fe4..023087fb 100644 --- a/tests/support/matmul_dispatch_instrument.h +++ b/tests/support/matmul_dispatch_instrument.h @@ -37,12 +37,31 @@ #include "superslm/matmul.h" -#if SUPERSLM_MATMUL_HAVE_SIMD_X64 - #include namespace superslm_test { +// Attention and per-row sites plan, slice S1 (§3.6): the row-table path counters. Each of the three +// per-row sites (src/forward/forward_sites.cpp) increments exactly one of its pair once per call, +// after its table decision: `taken` when it builds the 256-entry table (n >= kRowTableMinWidth, 512), +// `skipped` when it runs the per-element loop. RmsNormSite counts every call; MlpActSite counts after +// its gate-scale domain check accepts; ResidualReconcileSite counts once per call (not per candidate) +// after its scale checks accept. They are not tier-split and sit OUTSIDE the x64 block below, because +// the tables are on in every build except forced scalar, arm64 included; forced scalar builds no +// test binary and no instrument. +inline std::atomic g_rowtable_norm_taken{0}; +inline std::atomic g_rowtable_norm_skipped{0}; +inline std::atomic g_rowtable_silu_taken{0}; +inline std::atomic g_rowtable_silu_skipped{0}; +inline std::atomic g_rowtable_landing_taken{0}; +inline std::atomic g_rowtable_landing_skipped{0}; + +} // namespace superslm_test + +#if SUPERSLM_MATMUL_HAVE_SIMD_X64 + +namespace superslm_test { + // Atomic, not a plain counter: design §10 dimension 3's entire cell is N threads racing the // very first call to the magic-static initializer that is expected to perform this // increment. A non-atomic counter observed under that same contention would itself be a data @@ -66,6 +85,44 @@ inline long long TiledEntryInvocationsTotal() { return g_tiled_entry_invocations_avx2.load() + g_tiled_entry_invocations_avx512.load(); } +// Attention and per-row sites plan, slice S2 (§3.6): the prob·V path counters, per tier. src/matmul.cpp's +// GemmProbQ15Accumulate increments exactly one of them per call on the AVX2 and AVX-512 tiers, after its +// guard has decided: `fast` when head_dim % 16 == 0 and the row passes the int16 condition (every p in +// [0, 32767], Sum p <= 2^15), `fallback` otherwise. On the scalar and SSE2 tiers, and on an MSVC build's +// AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off (v1.9.0 code), none moves. +inline std::atomic g_pv_fast_avx2{0}; +inline std::atomic g_pv_fallback_avx2{0}; +inline std::atomic g_pv_fast_avx512{0}; +inline std::atomic g_pv_fallback_avx512{0}; + +// Attention and per-row sites plan, slice S3 (§3.6): the requant row counters, per tier. src/intmath.cpp's +// RequantRowWide increments exactly one of them per call on the AVX2 and AVX-512 tiers, inside the tier's +// own body (S3 has no runtime guard, §5.3, so there is no fallback counter). On the scalar and SSE2 tiers, +// and on an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off (the element loop), none moves. +inline std::atomic g_requant_row_avx2{0}; +inline std::atomic g_requant_row_avx512{0}; + +// Attention and per-row sites plan, slice S4 (§3.6): the softmax path counters, per tier. src/intmath.cpp's +// SoftmaxRowQ15 increments exactly one of them per call with width >= 1 on the AVX2 and AVX-512 tiers, +// after the row guard (§5.4) has decided: `fast` inside the tier's own body, `fallback` in the dispatcher +// when the guard fails (the shipped body then runs). On the scalar and SSE2 tiers, and on an MSVC build's +// AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off (v1.9.0 code), none moves. +inline std::atomic g_softmax_fast_avx2{0}; +inline std::atomic g_softmax_fallback_avx2{0}; +inline std::atomic g_softmax_fast_avx512{0}; +inline std::atomic g_softmax_fallback_avx512{0}; + +// Attention and per-row sites plan, slice S5 (§3.6): the Q31 score-row path counters, per tier. +// src/forward/forward_sites.cpp's QkQ31ScoreRow increments exactly one of them per call on the AVX2 and +// AVX-512 tiers, after its guard has decided: `fast` inside the tier's own body (head_dim <= 512 and every +// ratio in [0, 2^32)), `fallback` in the entry when the guard fails (the per-key QkQ31Score loop then runs). +// On the scalar and SSE2 tiers, and on an MSVC build's AVX-512 tier with SUPERSLM_SITES_AVX512_MSVC off +// (v1.9.0 code), none moves. +inline std::atomic g_q31_row_fast_avx2{0}; +inline std::atomic g_q31_row_fallback_avx2{0}; +inline std::atomic g_q31_row_fast_avx512{0}; +inline std::atomic g_q31_row_fallback_avx512{0}; + } // namespace superslm_test #endif // SUPERSLM_MATMUL_HAVE_SIMD_X64 diff --git a/tests/support/qk_attention_fixture.h b/tests/support/qk_attention_fixture.h new file mode 100644 index 00000000..111db75b --- /dev/null +++ b/tests/support/qk_attention_fixture.h @@ -0,0 +1,327 @@ +// Attention and per-row sites plan (rev 3.1), cell 11.1(c): the widened QK-norm attention fixture. +// +// Built the way test_main.cpp's QkNormWiringFixture is (a minimal loaded artifact for the RoPE table, +// then LayerWeights wired to this object's own arrays, driving the real RunLayerLoop and +// RunLayerLoopChunkBatched), and widened so the Q31 score path runs at a real geometry: hidden 256, +// 4 query heads over 2 KV heads (G = 2), head_dim 64, intermediate 256, one layer, context_cap 32, +// q_norm and k_norm gains set, k_channel_ratio = 2^31 on every channel (the loader's maximum, G6), and +// int8 weights from a fixed-seed generator so the scores spread. +// +// Header-only and integer-only: the suite (tests/test_attn_rowsites.cpp) and the golden-pin generator +// (tools/gen_attn_rowsite_golden.cpp, built against the v1.9.0 tag's library) both include it, so it +// needs no build entry and takes nothing from the code it grades. It calls only entry points whose +// signatures are unchanged since v1.9.0 (SslmModel::Load, RunLayerLoop, RunLayerLoopChunkBatched, +// DynamicScaleReciprocal). The RoPE table is built from Pythagorean triples, never libm, so every +// platform builds the same bytes. +// +// Every run feeds the same 24 positions (fixed-seed hidden codes and scales) and emits, as int64 values: +// per position, the layer's 256 output codes and its output scale (m, e); then every byte of the K/V +// workspace. The decode loop (one RunLayerLoop per position), the chunk loop (one +// RunLayerLoopChunkBatched over all 24) and the decode loop with a capture sink installed must all emit +// the same stream, and its hash is the golden pin's fixture hash (plan §3.3 evidence 3, cell 6.3). +#ifndef SUPERSLM_TESTS_SUPPORT_QK_ATTENTION_FIXTURE_H +#define SUPERSLM_TESTS_SUPPORT_QK_ATTENTION_FIXTURE_H + +#include +#include +#include +#include + +#include "superslm/forward_sites.h" +#include "superslm/intmath.h" +#include "superslm/model.h" +#include "../sslm_cfg1_hostile_fixtures.h" // Cfg1Spec, BuildCfg1 +#include "../sslm_fixtures.h" // BuildArtifact, MakeSection +#include "../sslm_model_hostile_fixtures.h" // ManifestTensorSpec, BuildManifest, PutU64 +#include "../sslm_sil1_hostile_fixtures.h" // MakeSigmoidLutSection + +namespace superslm_qk_fixture { + +inline constexpr size_t kHidden = 256; +inline constexpr size_t kHeads = 4; +inline constexpr size_t kKvHeads = 2; +inline constexpr size_t kHeadDim = 64; +inline constexpr size_t kInter = 256; +inline constexpr int64_t kCap = 32; +inline constexpr size_t kPositions = 24; +inline constexpr size_t kKvWidth = kKvHeads * kHeadDim; +inline constexpr size_t kWorkspaceBytes = size_t{1} * static_cast(kCap) * kKvHeads * kHeadDim * 2; +inline constexpr int64_t kRatioQ31 = INT64_C(2147483648); // 2^31, every channel + +// splitmix64, integer-only (the generator rowsite_cases.h and attention_cases.h use). +struct FixtureRng { + uint64_t s; + explicit FixtureRng(uint64_t seed) : s(seed) {} + uint64_t Next() { + s += 0x9e3779b97f4a7c15ULL; + uint64_t z = s; + z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL; + z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL; + return z ^ (z >> 31); + } + int64_t InRange(int64_t lo, int64_t hi) { + return lo + static_cast(Next() % (static_cast(hi - lo) + 1ULL)); + } +}; + +// A Q2.30 rotation per (position, pair) from Pythagorean triples: cos = a/c, sin = +-b/c, exact integer +// division, so |cos|, |sin| <= 2^30 (the loader's RoPE entry bound) and no platform's libm is involved. +// Position 0 is the identity, as a real table's is. +inline superslm_test::FixtureSection MakeRopeSection() { + static constexpr int64_t kTriples[][3] = {{3, 4, 5}, {5, 12, 13}, {8, 15, 17}, {7, 24, 25}, {20, 21, 29}, + {12, 35, 37}, {9, 40, 41}, {28, 45, 53}, {11, 60, 61}, {33, 56, 65}}; + const size_t pairs = kHeadDim / 2; + const size_t n = static_cast(kCap) * pairs; + std::vector cos_flat(n), sin_flat(n); + for (size_t p = 0; p < static_cast(kCap); ++p) { + for (size_t i = 0; i < pairs; ++i) { + const size_t at = p * pairs + i; + if (p == 0) { + cos_flat[at] = INT64_C(1) << 30; + sin_flat[at] = 0; + continue; + } + const int64_t* t = kTriples[(p * 7 + i * 3) % 10]; + const bool swap = ((p + i) & 1) != 0; + const int64_t a = swap ? t[1] : t[0], b = swap ? t[0] : t[1]; + cos_flat[at] = ((p + 2 * i) % 3 == 0 ? -1 : 1) * ((a << 30) / t[2]); + sin_flat[at] = ((p + i) % 4 < 2 ? -1 : 1) * ((b << 30) / t[2]); + } + } + std::vector tensors = { + {"cos", {static_cast(kCap), static_cast(pairs)}}, + {"sin", {static_cast(kCap), static_cast(pairs)}}, + }; + auto manifest = superslm_test::BuildManifest(superslm::kRopeMagic, /*element_size=*/8, tensors); + for (size_t i = 0; i < n; ++i) { + superslm_test::PutU64(manifest.bytes, static_cast(manifest.tensor_data_off[0]) + i * 8, + static_cast(cos_flat[i])); + superslm_test::PutU64(manifest.bytes, static_cast(manifest.tensor_data_off[1]) + i * 8, + static_cast(sin_flat[i])); + } + return superslm_test::MakeSection(superslm::SslmSectionType::RopeTables, superslm::SslmDtype::Int64, + manifest.bytes, /*alignment=*/64); +} + +struct QkAttentionFixture { + superslm::SslmModelView view; + superslm::LayerWeights layer{}; + bool loaded = false; + std::string load_error; + + std::vector q_w, k_w, v_w, o_w, gate_w, up_w, down_w; + std::vector fold_identity, fold_zero; + std::vector hidden_gain, q_norm_gain, k_norm_gain; + int64_t kv_landing_r_t[kKvHeads], kv_landing_e_t[kKvHeads]; + int64_t k_channel_r_t[kKvWidth], k_channel_e_t[kKvWidth], k_channel_ratio[kKvWidth]; + int64_t softmax_khead_m[kKvHeads], softmax_khead_e[kKvHeads]; + int32_t ctx_fold_identity[kHeads], ctx_fold_mult[kHeads], ctx_fold_shift[kHeads]; + + // The 24 positions' layer inputs. + std::vector hidden_in; // kPositions * kHidden + std::vector scale_in; // kPositions + + QkAttentionFixture() { + using superslm::CarriedScale; + superslm_test::Cfg1Spec spec{}; + spec.hidden_size = static_cast(kHidden); + spec.num_hidden_layers = 1; + spec.num_attention_heads = static_cast(kHeads); + spec.num_key_value_heads = static_cast(kKvHeads); + spec.head_dim = static_cast(kHeadDim); + spec.intermediate_size = static_cast(kInter); + spec.context_cap = static_cast(kCap); + spec.kv_precision = 0; + spec.kv_block_size = 1; + auto built = superslm_test::BuildArtifact( + {superslm_test::MakeSection(superslm::SslmSectionType::Config, superslm::SslmDtype::Raw, + superslm_test::BuildCfg1(spec)), + superslm_test::MakeSigmoidLutSection(), MakeRopeSection()}); + artifact_bytes = std::move(built.bytes); + loaded = superslm::SslmModel::Load(artifact_bytes.data(), artifact_bytes.size(), view, &load_error) == + superslm::SslmModelStatus::Ok; + + FixtureRng rng(0x51A7'7E57'0005'0011ULL); + auto weights = [&](size_t n) { + std::vector w(n); + for (auto& x : w) x = static_cast(rng.InRange(-127, 127)); + return w; + }; + q_w = weights(kHeads * kHeadDim * kHidden); + k_w = weights(kKvWidth * kHidden); + v_w = weights(kKvWidth * kHidden); + o_w = weights(kHidden * kHeads * kHeadDim); + gate_w = weights(kInter * kHidden); + up_w = weights(kInter * kHidden); + down_w = weights(kHidden * kInter); + fold_identity.assign(kInter > kHidden ? kInter : kHidden, 1); + fold_zero.assign(fold_identity.size(), 0); + hidden_gain.assign(kHidden, 8192); + q_norm_gain.resize(kHeadDim); + k_norm_gain.resize(kHeadDim); + for (size_t d = 0; d < kHeadDim; ++d) { + q_norm_gain[d] = static_cast(rng.InRange(4096, 8192)); + k_norm_gain[d] = static_cast(rng.InRange(4096, 8192)); + } + + const CarriedScale canonical{INT64_C(1073741824), INT64_C(-30)}; + const int64_t canonical_r_t = superslm::DynamicScaleReciprocal(canonical.m); + for (size_t h = 0; h < kKvHeads; ++h) { + kv_landing_r_t[h] = canonical_r_t; + kv_landing_e_t[h] = kKvLandingE; + softmax_khead_m[h] = INT64_C(1073741824); + softmax_khead_e[h] = kSoftmaxKheadE; + } + for (size_t c = 0; c < kKvWidth; ++c) { + k_channel_r_t[c] = canonical_r_t; + k_channel_e_t[c] = kKChannelE; + k_channel_ratio[c] = kRatioQ31; + } + for (size_t h = 0; h < kHeads; ++h) { + ctx_fold_identity[h] = 1; + ctx_fold_mult[h] = 0; + ctx_fold_shift[h] = 0; + } + + superslm::LayerWeights& lw = layer; + const int32_t* id = fold_identity.data(); + const int32_t* zero = fold_zero.data(); + lw.attn_norm_gain = hidden_gain.data(); + lw.attn_norm_site_constant = canonical; + lw.q_weight = q_w.data(); + lw.k_weight = k_w.data(); + lw.v_weight = v_w.data(); + lw.o_weight = o_w.data(); + lw.q_fold_identity = id; lw.q_fold_mult = zero; lw.q_fold_shift = zero; + lw.k_fold_identity = id; lw.k_fold_mult = zero; lw.k_fold_shift = zero; + lw.v_fold_identity = id; lw.v_fold_mult = zero; lw.v_fold_shift = zero; + lw.o_fold_identity = id; lw.o_fold_mult = zero; lw.o_fold_shift = zero; + lw.gate_fold_identity = id; lw.gate_fold_mult = zero; lw.gate_fold_shift = zero; + lw.up_fold_identity = id; lw.up_fold_mult = zero; lw.up_fold_shift = zero; + lw.down_fold_identity = id; lw.down_fold_mult = zero; lw.down_fold_shift = zero; + lw.q_site_constant = canonical; + lw.o_site_constant = canonical; + lw.kv_landing_r_t_k = kv_landing_r_t; + lw.kv_landing_e_t_k = kv_landing_e_t; + lw.kv_landing_r_t_v = kv_landing_r_t; + lw.kv_landing_e_t_v = kv_landing_e_t; + lw.ctx_fold_identity = ctx_fold_identity; + lw.ctx_fold_mult = ctx_fold_mult; + lw.ctx_fold_shift = ctx_fold_shift; + lw.ctx_fold_site_constant = canonical; + lw.attn_residual_site_constant = canonical; + lw.iexp_softmax_khead_m = softmax_khead_m; + lw.iexp_softmax_khead_e = softmax_khead_e; + lw.mlp_norm_gain = hidden_gain.data(); + lw.mlp_norm_site_constant = canonical; + lw.gate_weight = gate_w.data(); + lw.up_weight = up_w.data(); + lw.down_weight = down_w.data(); + lw.gate_site_constant = CarriedScale{INT64_C(1073741824), kGateSiteE}; + lw.up_site_constant = canonical; + lw.mlp_act_site_constant = CarriedScale{INT64_C(1073741824), INT64_C(-96)}; + lw.down_site_constant = canonical; + lw.mlp_residual_site_constant = canonical; + lw.q_norm_gain = q_norm_gain.data(); + lw.q_norm_site_constant = canonical; + lw.k_norm_gain = k_norm_gain.data(); + lw.k_norm_site_constant = canonical; + lw.k_wide_source_scale = CarriedScale{INT64_C(1073741824), kKWideSourceE}; + lw.k_channel_r_t = k_channel_r_t; + lw.k_channel_e_t = k_channel_e_t; + lw.k_channel_ratio = k_channel_ratio; + + hidden_in.resize(kPositions * kHidden); + for (auto& x : hidden_in) x = static_cast(rng.InRange(-127, 127)); + scale_in.resize(kPositions); + for (size_t p = 0; p < kPositions; ++p) + scale_in[p] = CarriedScale{INT64_C(1073741824) + rng.InRange(0, (INT64_C(1) << 30) - 1), kHiddenE}; + } + + QkAttentionFixture(const QkAttentionFixture&) = delete; + QkAttentionFixture& operator=(const QkAttentionFixture&) = delete; + + // Tuned on the base (the v1.9.0 code) so that every step returns Ok, K lands mostly unclamped, and the + // softmax rows are spread and inside §5.4's guard (docs/attention-rowsites/s5/fixture-premise.txt). + static constexpr int64_t kHiddenE = -30; + static constexpr int64_t kKvLandingE = 12; + static constexpr int64_t kKWideSourceE = -60; + static constexpr int64_t kKChannelE = -36; + static constexpr int64_t kSoftmaxKheadE = -72; + static constexpr int64_t kGateSiteE = -52; + + private: + std::vector artifact_bytes; +}; + +template +void EmitPosition(Emit& emit, const int8_t* codes, const superslm::CarriedScale& scale) { + for (size_t i = 0; i < kHidden; ++i) emit(static_cast(codes[i])); + emit(scale.m); + emit(scale.e); +} + +template +void EmitWorkspace(Emit& emit, const std::vector& workspace) { + for (uint8_t b : workspace) emit(static_cast(b)); +} + +struct NoPositionHook { + void operator()(size_t) const {} +}; + +// Run (i), or run (iii) when `sink` is non-null: the decode loop, one RunLayerLoop call per position. +// `after(p)` runs after position p's call returns Ok (the suite reads its path counters there). Returns +// the first non-Ok status (the stream then stops there). +template +superslm::SslmForwardStatus RunFixtureDecode(QkAttentionFixture& f, Emit& emit, + superslm::AttentionCaptureSink* sink = nullptr, + After after = After{}) { + using superslm::SslmForwardStatus; + f.layer.attention_capture_sink = sink; + std::vector workspace(kWorkspaceBytes, 0); + std::vector hidden(kHidden); + superslm::SequenceLayerState seq; + seq.hidden_codes = hidden.data(); + SslmForwardStatus st = SslmForwardStatus::Ok; + for (size_t p = 0; p < kPositions && st == SslmForwardStatus::Ok; ++p) { + for (size_t i = 0; i < kHidden; ++i) hidden[i] = f.hidden_in[p * kHidden + i]; + seq.hidden_scale = f.scale_in[p]; + seq.layer_index = 0; + st = superslm::RunLayerLoop(seq, &f.layer, 1, 1, kHidden, kHeadDim, kKvHeads, kInter, kCap, + f.view.rope_tables, workspace.data(), workspace.size(), {}, p, nullptr, + kHeads * kHeadDim); + if (st != SslmForwardStatus::Ok) break; + EmitPosition(emit, hidden.data(), seq.hidden_scale); + after(p); + } + f.layer.attention_capture_sink = nullptr; + if (st == SslmForwardStatus::Ok) EmitWorkspace(emit, workspace); + return st; +} + +// Run (ii): the chunk loop, one RunLayerLoopChunkBatched call over all 24 positions from position 0. +// `sink` is installed on the layer (t2701's chunk-mode configuration) and must never be called. +template +superslm::SslmForwardStatus RunFixtureChunk(QkAttentionFixture& f, Emit& emit, + superslm::AttentionCaptureSink* sink = nullptr) { + using superslm::SslmForwardStatus; + f.layer.attention_capture_sink = sink; + std::vector workspace(kWorkspaceBytes, 0); + std::vector chunk(f.hidden_in); + std::vector scales(f.scale_in); + uint64_t saturation = 0; + const SslmForwardStatus st = superslm::RunLayerLoopChunkBatched( + chunk.data(), scales.data(), kPositions, &f.layer, 1, kHidden, kHeadDim, kKvHeads, kInter, kCap, 0, + f.view.rope_tables, workspace.data(), workspace.size(), /*option_g_fused_k_landing=*/false, &saturation, + {}, nullptr, kHeads * kHeadDim); + f.layer.attention_capture_sink = nullptr; + if (st != SslmForwardStatus::Ok) return st; + for (size_t p = 0; p < kPositions; ++p) EmitPosition(emit, chunk.data() + p * kHidden, scales[p]); + EmitWorkspace(emit, workspace); + return st; +} + +} // namespace superslm_qk_fixture + +#endif // SUPERSLM_TESTS_SUPPORT_QK_ATTENTION_FIXTURE_H diff --git a/tests/support/rowsite_cases.h b/tests/support/rowsite_cases.h new file mode 100644 index 00000000..18414b50 --- /dev/null +++ b/tests/support/rowsite_cases.h @@ -0,0 +1,279 @@ +// Attention and per-row sites plan, slices S1 (the per-row tables) and S3 (the requant element loop): +// the fixed input sets that the digest section `c_rowsites`, the golden-pin generator +// (tools/gen_attn_rowsite_golden.cpp) and the suite's golden cells run through the three per-row sites +// (S1, RunRowTableCases) and through the checked chain funnel (S3, RunRequantRowCases). Header-only, so it needs no build +// entry: the digest and the generator include it by relative path, the suite through `tests/`. +// +// Everything here calls only entry points whose signatures are unchanged since v1.9.0 +// (RmsNormSite, MlpActSite, ResidualReconcileSite, RequantChainChecked), so the generator can be built against the +// v1.9.0 tag's library and the pin takes no input from the code it grades (plan §3.3 evidence 3). +// +// The set covers both sides of the table threshold (widths 1 to 4,864 around 512), the -128 code +// at the first, middle and last element, uniform rows, a gate scale small enough that +// sig(-128) differs from sig(-127) and from 0 (plan §8 4.S1), residual rows whose first candidate +// is refused mid-row by the landing flag, and rows the funnel refuses. Each call contributes its +// status, its output scale and its whole output row (poisoned before the call, so a refused call's +// untouched row is digested too). +#ifndef SUPERSLM_TESTS_SUPPORT_ROWSITE_CASES_H +#define SUPERSLM_TESTS_SUPPORT_ROWSITE_CASES_H + +#include +#include +#include +#include + +#include "superslm/checked_chain_funnel.h" +#include "superslm/forward_sites.h" +#include "superslm/silu_lut_canonical.h" + +namespace superslm_rowsite_cases { + +// splitmix64, integer-only: the same values on every conforming C++20 implementation. +struct Rng { + uint64_t s; + explicit Rng(uint64_t seed) : s(seed) {} + uint64_t Next() { + s += 0x9e3779b97f4a7c15ULL; + uint64_t z = s; + z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL; + z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL; + return z ^ (z >> 31); + } + int64_t InRange(int64_t lo, int64_t hi) { + const uint64_t span = static_cast(hi - lo) + 1ULL; + return lo + static_cast(Next() % span); + } +}; + +// The widths of plan §8 4.S1: both sides of kRowTableMinWidth (512) and the 0.5B's two widths. +inline constexpr size_t kWidths[] = {1, 64, 255, 511, 512, 513, 896, 4864}; + +// Row shapes: random codes over the full int8 range with -128 placed at the first, middle and last +// element, random codes in [-127, 127] with no -128, all -127, all 127, all -128. +enum class RowShape : int { kRandomWithMinus128 = 0, kRandomNo128, kAllMinus127, kAll127, kAllMinus128 }; +inline constexpr RowShape kShapes[] = {RowShape::kRandomWithMinus128, RowShape::kRandomNo128, + RowShape::kAllMinus127, RowShape::kAll127, RowShape::kAllMinus128}; + +inline void FillRow(std::vector& row, RowShape shape, Rng& rng) { + const size_t n = row.size(); + for (size_t i = 0; i < n; ++i) { + switch (shape) { + case RowShape::kRandomWithMinus128: row[i] = static_cast(rng.InRange(-128, 127)); break; + case RowShape::kRandomNo128: row[i] = static_cast(rng.InRange(-127, 127)); break; + case RowShape::kAllMinus127: row[i] = -127; break; + case RowShape::kAll127: row[i] = 127; break; + case RowShape::kAllMinus128: row[i] = -128; break; + } + } + if (shape == RowShape::kRandomWithMinus128 && n > 0) { + row[0] = -128; + row[n / 2] = -128; + row[n - 1] = -128; + } +} + +// One site call's observable result, emitted as int64 values through `emit`: status, scale (m, e), +// then every output code. `out` was poisoned with 0x5A before the call. +template +void EmitResult(Emit& emit, superslm::SslmForwardStatus status, const superslm::CarriedScale& scale, + const std::vector& out) { + emit(static_cast(status)); + emit(scale.m); + emit(scale.e); + for (int8_t c : out) emit(static_cast(c)); +} + +inline constexpr int8_t kPoison = 0x5A; + +// RMSNorm: every width x shape, gains drawn per case (one small range the funnel accepts at every +// width, one wide range it refuses at some), and two site constants. +template +void RunNormCases(Emit& emit) { + using superslm::CarriedScale; + Rng rng(0x5331524F574E4F52ULL); // "S1ROWNOR" + const CarriedScale site_constants[] = {{INT64_C(1073741824), 0}, {INT64_C(1518500250), -3}}; + const int64_t gain_limits[] = {300, 40000}; + for (size_t n : kWidths) + for (RowShape shape : kShapes) + for (int64_t gain_limit : gain_limits) + for (const CarriedScale& site_constant : site_constants) { + std::vector h(n); + FillRow(h, shape, rng); + std::vector g(n); + for (auto& v : g) v = static_cast(rng.InRange(-gain_limit, gain_limit)); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const auto st = superslm::RmsNormSite(h.data(), g.data(), n, CarriedScale{}, site_constant, + out.data(), &scale); + EmitResult(emit, st, scale, out); + } +} + +// SwiGLU: every width x shape of the gate row, random up rows, and gate scales from the forward's +// range plus the small scale (m = 2^30, e = -35) at which sig(-128) = 589, sig(-127) = 608. +template +void RunSiluCases(Emit& emit) { + using superslm::CarriedScale; + Rng rng(0x533153494C555F43ULL); // "S1SILU_C" + const CarriedScale gate_scales[] = {{INT64_C(1073741824), -35}, {INT64_C(1073741824), -34}, + {INT64_C(1631069115), -32}, {INT64_C(1216311211), -28}, + {INT64_C(2147483647), -36}}; + const CarriedScale up_scale{INT64_C(1340958474), -18}; + const CarriedScale site_constant{INT64_C(1073741824), 0}; + for (size_t n : kWidths) + for (RowShape shape : kShapes) + for (const CarriedScale& gate_scale : gate_scales) { + std::vector gate(n), up(n); + FillRow(gate, shape, rng); + FillRow(up, RowShape::kRandomNo128, rng); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const auto st = superslm::MlpActSite(gate.data(), gate_scale, up.data(), up_scale, n, + superslm::kSiluLutCanonicalTable, site_constant, + out.data(), &scale); + EmitResult(emit, st, scale, out); + } +} + +// A residual scale pair and what it is for. +struct ResidualScalePair { + superslm::CarriedScale branch; + superslm::CarriedScale stream; +}; + +// Residual: every width x shape of both rows over scale pairs from the forward's range (exponent +// gaps 0 to 20, both signs of the mantissa, both candidate orders), gaps beyond 31 both ways, and +// the landing-flag rows: the first candidate's "other" row is all 0 except one 127 at the middle +// element (refused mid-row, the second candidate commits), all 0 (never refused although code 127 +// would be), and a refused first candidate with a second candidate the funnel refuses too. +template +void RunResidualCases(Emit& emit) { + using superslm::CarriedScale; + Rng rng(0x5331524553494455ULL); // "S1RESIDU" + const CarriedScale site_constant{INT64_C(1073741824), 0}; + const ResidualScalePair pairs[] = { + {{INT64_C(1518500250), -40}, {INT64_C(1073741824), -40}}, + {{INT64_C(1234567890), -37}, {INT64_C(1987654321), -44}}, + {{INT64_C(1987654321), -44}, {INT64_C(1234567890), -37}}, + {{INT64_C(-1400000000), -39}, {INT64_C(1100000000), -41}}, + {{INT64_C(1100000000), -52}, {INT64_C(-1900000000), -32}}, + {{INT64_C(1300000000), -70}, {INT64_C(1300000000), -36}}, // gap 34: stream selected first + {{INT64_C(1300000000), -36}, {INT64_C(1300000000), -70}}, // gap 34: branch selected first + }; + for (size_t n : kWidths) + for (RowShape shape : kShapes) + for (const ResidualScalePair& p : pairs) { + std::vector branch(n), stream(n); + FillRow(branch, shape, rng); + FillRow(stream, RowShape::kRandomWithMinus128, rng); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const auto st = superslm::ResidualReconcileSite(branch.data(), p.branch, stream.data(), p.stream, + n, site_constant, out.data(), &scale); + EmitResult(emit, st, scale, out); + } + + // The landing-flag rows. Branch e = 30, stream e = -30: the gap exceeds 31, so the stream + // (finer) candidate is built first and lands the branch codes; code 127 overflows int64 there + // and code 0 does not. The branch row is the "other" row of that first candidate. + const CarriedScale coarse{INT64_C(1300000000), 30}; + const CarriedScale fine{INT64_C(1300000000), -30}; + const CarriedScale bad_site_constant{INT64_C(4294967296), 0}; // the funnel refuses it (m > int32) + for (size_t n : kWidths) { + std::vector stream(n); + FillRow(stream, RowShape::kRandomWithMinus128, rng); + for (int variant = 0; variant < 3; ++variant) { + std::vector branch(n, 0); + if (variant != 1) branch[n / 2] = 127; // variants 0 and 2: refused mid-row + const CarriedScale sc = variant == 2 ? bad_site_constant : site_constant; + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const auto st = superslm::ResidualReconcileSite(branch.data(), coarse, stream.data(), fine, n, sc, + out.data(), &scale); + EmitResult(emit, st, scale, out); + } + } +} + +// ---- Slice S3: the requant element loop (plan §4.3, §5.3, §8 4.S3) ------------------------------------ +// +// The funnel's element loop runs through RequantChainChecked, whose signature is v1.9.0's, so the +// generator can drive the same rows through the v1.9.0 tag's library. Every row carries +d' or -d' +// at a chosen element (so the row's own max-abs is d', and NormalizeScale/DynamicScaleReciprocal of +// that d' are the r and s the element loop receives), the other elements drawn from [-d', d']. + +// The widths of 4.S3: lanes of 4 (AVX2) and 8 (AVX-512) with every tail length around them, plus the +// 0.5B's hidden and intermediate widths. +inline constexpr size_t kRequantWidths[] = {1, 3, 4, 5, 7, 8, 9, 896, 4864}; + +// d' values: 1, 2, 2^30 and 2^31 (the P = 2^63 corner at r = 2^32, s = -1) and their neighbours, then +// for every s in [0, 30] the octave [2^(30-s), 2^(31-s)) that NormalizeScale maps to that s: its low +// end, its high end and one draw inside. +inline std::vector RequantDPrimes(Rng& rng) { + std::vector d = {1, 2, 3, (INT64_C(1) << 30) - 1, INT64_C(1) << 30, (INT64_C(1) << 30) + 1, + (INT64_C(1) << 31) - 1, INT64_C(1) << 31}; + for (int s = 0; s <= 30; ++s) { + const int64_t lo = INT64_C(1) << (30 - s), hi = (INT64_C(1) << (31 - s)) - 1; + d.push_back(lo); + d.push_back(hi); + d.push_back(rng.InRange(lo, hi)); + } + return d; +} + +// One funnel call on `x` with the unit site constant; status, output scale and codes are emitted. +template +void EmitRequantRow(Emit& emit, const std::vector& x) { + using superslm::CarriedScale; + const CarriedScale site_constant{INT64_C(1073741824), 0}; + std::vector out(x.size(), kPoison); + CarriedScale scale{-1, -1}; + const auto st = superslm::RequantChainChecked(x.data(), x.size(), std::span{}, + site_constant, out.data(), &scale) + .status; + EmitResult(emit, st, scale, out); +} + +// Every width x every d': in rows of up to 9 elements, +d' and -d' at every position in turn (every lane +// position of both SIMD widths and every tail slot); in the wide rows, at the first element, the last, +// and one drawn position. Then the empty row and two rows the preflight refuses (d' = 2^31 + 1), whose +// untouched output is digested too. +template +void RunRequantRowCases(Emit& emit) { + Rng rng(0x5333524551524F57ULL); // "S3REQROW" + const std::vector dprimes = RequantDPrimes(rng); + for (size_t n : kRequantWidths) + for (int64_t dp : dprimes) { + std::vector positions; + if (n <= 9) { + for (size_t p = 0; p < n; ++p) positions.push_back(p); + } else { + positions = {0, n - 1, static_cast(rng.InRange(1, static_cast(n) - 2))}; + } + for (size_t p : positions) + for (int sign : {1, -1}) { + std::vector x(n); + for (auto& v : x) v = rng.InRange(-dp, dp); + x[p] = sign * dp; + EmitRequantRow(emit, x); + } + } + EmitRequantRow(emit, std::vector{}); + for (size_t n : {size_t{5}, size_t{896}}) { + std::vector x(n, 7); + x[n / 2] = (INT64_C(1) << 31) + 1; + EmitRequantRow(emit, x); + } +} + +// The whole S1 set, in a fixed order. +template +void RunRowTableCases(Emit& emit) { + RunNormCases(emit); + RunSiluCases(emit); + RunResidualCases(emit); +} + +} // namespace superslm_rowsite_cases + +#endif // SUPERSLM_TESTS_SUPPORT_ROWSITE_CASES_H diff --git a/tests/t2296-fp-free-open-red-suite/build_link_red.bat b/tests/t2296-fp-free-open-red-suite/build_link_red.bat index ad93416a..0c9b2660 100644 --- a/tests/t2296-fp-free-open-red-suite/build_link_red.bat +++ b/tests/t2296-fp-free-open-red-suite/build_link_red.bat @@ -26,6 +26,9 @@ rem (T-2297, 2026-08-26: the original source list omitted it; the conductor repr rem passing once linked, and added it below.) Kept deliberately minimal rather than matching rem tests/t2138-abi-red-suite's own full CPU-core source list: the two headers plus the one rem .cpp their templates call out to is everything these cells need. +rem Since the attention and per-row sites plan's slice S3, intmath.cpp dispatches its requant row +rem leaf on detail::ActiveGemmTier(), which src/matmul.cpp defines, so matmul.cpp is on the compile +rem line beside it: without it all five cells fail to LINK (LNK2019), the same false-red class. rem rem dim7_contract_red.cpp is compiled WITH /DNDEBUG (this file's own header comment: the rem probe-exhaustion std::abort() is documented release-safe/NDEBUG-independent, and building @@ -65,7 +68,7 @@ for %%f in (dim4_shape_red.cpp dim6_determinism_red.cpp dim7_contract_red.cpp di set EXTRASOURCES= if "%%f"=="dim7_capacity_red.cpp" set EXTRASOURCES="%ENG%\src\damped_greedy_antilm.cpp" cl /nologo /std:c++20 /O2 /W4 /EHsc !EXTRAFLAGS! /I%ENG%\src /I%ENG%\include -I. ^ - "%%f" "%ENG%\src\intmath.cpp" !EXTRASOURCES! /Fo:"obj\\" /Fe:"obj\%%~nf.exe" ^ + "%%f" "%ENG%\src\intmath.cpp" "%ENG%\src\matmul.cpp" !EXTRASOURCES! /Fo:"obj\\" /Fe:"obj\%%~nf.exe" ^ /link > "obj\%%~nf.log" 2>&1 findstr /C:"error C" "obj\%%~nf.log" >nul if not errorlevel 1 ( diff --git a/tests/test_attn_rowsites.cpp b/tests/test_attn_rowsites.cpp new file mode 100644 index 00000000..cb4f19f4 --- /dev/null +++ b/tests/test_attn_rowsites.cpp @@ -0,0 +1,2173 @@ +// Attention and per-row sites plan (rev 3.1): the cells of the plan's coverage model (§8) that its +// slices own, in their own translation unit. test_main.cpp calls RunAttnRowsiteCells and adds this +// unit's check and failure counts to its own totals, the pattern test_tiled_gemm.cpp set. +// +// Slice S1 (§4.1): 256-entry per-row tables in RmsNormSite (the divide), MlpActSite (the SiLU +// sigmoid) and ResidualReconcileSite (the landing rescale). Every table holds the same pure +// function's value at each code, computed with exactly the arguments the per-element loop passes +// (§5.1), so every S1 cell is an equality cell against the v1.9.0 per-element construction plus a +// path assertion: the row-table counter delta of the one call (§8 path rule, §3.6). +// +// "Reference" below is a test-side copy of the v1.9.0 per-element loop of each site, built from +// the public pure functions (FloorDivI64, SiluSigmoidQ15, LandingRescale) and the unchanged funnel +// (RequantChainChecked, PreflightRequantChain), never the build under test's site body; the golden +// cell (6.3) additionally pins the digest input set to a hash generated from the v1.9.0 tag's own +// library (tools/gen_attn_rowsite_golden.cpp). The expected path is a test-side copy of the guard, +// written from §4.1: tables are taken exactly when n >= 512, in every build that has a test binary. +// +// Slice S2 (§4.2): prob·V on int16 multiply-add; slice S3 (§4.3): the requant element loop in 64-bit +// lanes (RequantRowWide), checked against the unchanged per-element RequantTokenCodeWide with sentinel +// fences and exact-size buffers, and counted per tier (requant_row); slice S4 (§4.4): the guarded softmax, +// checked against a test-side restatement of the v1.9.0 SoftmaxRowQ15 body (the unchanged public +// IExpConstruct / IExpEvaluate leaves), bool and output, with the path chosen by a test-side copy of the +// §5.4 guard and the correction rows chosen by a test-side replica of the estimates; slice S5 (§4.5): the Q31 +// score row (QkQ31ScoreRow) in three 16-bit pieces, checked against the in-tree scalar reference (the same +// binary's per-key QkQ31Score outside the loader's ratio range), with the path chosen by a test-side copy of +// its guard, and driven through both layer loops on the widened QK-norm fixture (cell 11.1(c)). +// +// A build without the instrument seam (build.bat's MSVC recipe) runs the value checks alone and +// says so. Cell numbers are the plan's §8 numbering. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "superslm/checked_chain_funnel.h" +#include "superslm/forward_sites.h" +#include "superslm/intmath.h" +#include "superslm/layer_marshal.h" +#include "superslm/matmul.h" +#include "superslm/model.h" +#include "superslm/sha256.h" +#include "superslm/silu_lut.h" +#include "superslm/silu_lut_canonical.h" +#include "superslm/trace_hook.h" +#include "attn_rowsite_golden_pin.h" +#include "sslm_c32_softmax_row_width_gate_fixtures.h" +#include "support/attention_cases.h" +#include "support/matmul_dispatch_instrument.h" +#include "support/qk_attention_fixture.h" +#include "support/rowsite_cases.h" +#include "sslm_model_hostile_fixtures.h" + +#include +#include +#ifdef _WIN32 +#include // _dup, _dup2, _close, _fileno -- the stderr capture in the preflight cell +#include // _getpid +#else +#include // dup, dup2, close, fileno, getpid -- the stderr capture in the preflight cell +#endif + +static int GChecks = 0; +static int GFailures = 0; + +#define CHECK_MSG(cond, ...) \ + do { \ + ++GChecks; \ + if (!(cond)) { \ + ++GFailures; \ + std::printf("FAIL %s:%d: %s -- ", __FILE__, __LINE__, #cond); \ + std::printf(__VA_ARGS__); \ + std::printf("\n"); \ + } \ + } while (0) + +namespace { + +using superslm::CarriedScale; +using superslm::SslmForwardStatus; +using superslm::SslmForwardStatusName; +using superslm_rowsite_cases::FillRow; +using superslm_rowsite_cases::kPoison; +using superslm_rowsite_cases::Rng; +using superslm_rowsite_cases::RowShape; + +// The test-side copy of the S1 guard (§4.1): the threshold, and "tables off" only under forced +// scalar (which builds no test binary, §3.3). +constexpr size_t kTestRowTableMinWidth = 512; +bool ExpectTableTaken(size_t n) { +#if defined(SUPERSLM_FORCE_SCALAR_MATMUL) + (void)n; + return false; +#else + return n >= kTestRowTableMinWidth; +#endif +} + +// ---- the row-table counters (§3.6) -------------------------------------------------------------- + +enum RowSite : int { kNorm = 0, kSilu = 1, kLanding = 2 }; +const char* const kRowSiteNames[] = {"norm", "silu", "landing"}; + +struct RowCounters { + long long taken[3] = {0, 0, 0}; + long long skipped[3] = {0, 0, 0}; +}; + +#if defined(SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT) +constexpr bool kHaveRowCounters = true; +RowCounters ReadRowCounters() { + RowCounters c; + c.taken[kNorm] = superslm_test::g_rowtable_norm_taken.load(); + c.skipped[kNorm] = superslm_test::g_rowtable_norm_skipped.load(); + c.taken[kSilu] = superslm_test::g_rowtable_silu_taken.load(); + c.skipped[kSilu] = superslm_test::g_rowtable_silu_skipped.load(); + c.taken[kLanding] = superslm_test::g_rowtable_landing_taken.load(); + c.skipped[kLanding] = superslm_test::g_rowtable_landing_skipped.load(); + return c; +} +#else +constexpr bool kHaveRowCounters = false; +RowCounters ReadRowCounters() { return RowCounters{}; } +#endif + +RowCounters Delta(const RowCounters& a, const RowCounters& b) { + RowCounters d; + for (int s = 0; s < 3; ++s) { + d.taken[s] = b.taken[s] - a.taken[s]; + d.skipped[s] = b.skipped[s] - a.skipped[s]; + } + return d; +} + +// The path rule for one site call: the site's own pair moves by exactly one on the side the guard +// copy names (or not at all when the call is refused before its table decision), and every other +// row-table counter stays put. +void CheckPath(const char* label, const RowCounters& d, RowSite site, bool counted, size_t n) { + if (!kHaveRowCounters) return; + const bool taken = ExpectTableTaken(n); + for (int s = 0; s < 3; ++s) { + const long long want_taken = (s == site && counted && taken) ? 1 : 0; + const long long want_skipped = (s == site && counted && !taken) ? 1 : 0; + CHECK_MSG(d.taken[s] == want_taken && d.skipped[s] == want_skipped, + "%s n=%zu: rowtable_%s taken +%lld skipped +%lld, want +%lld/+%lld", label, n, + kRowSiteNames[s], d.taken[s], d.skipped[s], want_taken, want_skipped); + } +} + +// ---- the v1.9.0 per-element references ---------------------------------------------------------- + +struct SiteResult { + SslmForwardStatus status = SslmForwardStatus::Ok; + CarriedScale scale{-1, -1}; + std::vector out; +}; + +// RmsNormSite as v1.9.0 writes it: sumsq, root, then FloorDivI64(h << 32, root) * g per element. +SiteResult RefNorm(const int8_t* h, const int32_t* g, size_t n, CarriedScale site_constant) { + SiteResult r; + r.out.assign(n, kPoison); + int64_t sumsq = 0; + for (size_t i = 0; i < n; ++i) sumsq += static_cast(h[i]) * static_cast(h[i]); + int64_t root = superslm::ISqrt(superslm::FloorDivI64(sumsq << 32, static_cast(n))); + root = root > 1 ? root : 1; + std::vector wide(n); + for (size_t i = 0; i < n; ++i) + wide[i] = superslm::FloorDivI64(static_cast(h[i]) << 32, root) * static_cast(g[i]); + r.status = superslm::RequantChainChecked(wide.data(), n, std::span{}, site_constant, + r.out.data(), &r.scale) + .status; + return r; +} + +// MlpActSite as v1.9.0 writes it: domain check, then g * SiluSigmoidQ15(g) * u per element. +SiteResult RefSilu(const int8_t* gate, CarriedScale gate_scale, const int8_t* up, CarriedScale up_scale, size_t n, + CarriedScale site_constant) { + SiteResult r; + r.out.assign(n, kPoison); + r.status = superslm::CheckSiluCompositionScaleDomain(gate_scale.m, gate_scale.e); + if (r.status != SslmForwardStatus::Ok) return r; + std::vector wide(n); + for (size_t i = 0; i < n; ++i) { + const int32_t sig = superslm::SiluSigmoidQ15(superslm::kSiluLutCanonicalTable, gate[i], gate_scale.m, + static_cast(gate_scale.e)); + wide[i] = static_cast(gate[i]) * static_cast(sig) * static_cast(up[i]); + } + const CarriedScale incoming[2] = {gate_scale, up_scale}; + r.status = superslm::RequantChainChecked(wide.data(), n, std::span{incoming, 2}, + site_constant, r.out.data(), &r.scale) + .status; + return r; +} + +// ResidualReconcileSite as v1.9.0 writes it (its selection rule, its per-element landing loop with +// the first-flag return and the overflow checks, its second candidate, its funnel). +SiteResult RefResidual(const int8_t* branch_code, CarriedScale branch_scale, const int8_t* stream_code, + CarriedScale stream_scale, size_t n, CarriedScale site_constant) { + SiteResult r; + r.out.assign(n, kPoison); + if (branch_scale.m == 0 || stream_scale.m == 0) { + r.status = SslmForwardStatus::ResidualReconciliationScaleOutOfDomain; + return r; + } + const int64_t i32min = INT64_C(-2147483648), i32max = INT64_C(2147483647); + if (branch_scale.m < i32min || branch_scale.m > i32max || stream_scale.m < i32min || stream_scale.m > i32max) { + r.status = SslmForwardStatus::CarriedScaleMantissaOutOfDomain; + return r; + } + const auto magnitude = [](int64_t m) -> uint64_t { + return m < 0 ? ~static_cast(m) + 1u : static_cast(m); + }; + const auto exceeds_by_31 = [](int64_t a, int64_t b) { + return a > b && static_cast(a) - static_cast(b) > 31u; + }; + bool branch_selected; + if (exceeds_by_31(branch_scale.e, stream_scale.e)) { + branch_selected = false; + } else if (exceeds_by_31(stream_scale.e, branch_scale.e)) { + branch_selected = true; + } else { + const int d = static_cast(branch_scale.e - stream_scale.e); + branch_selected = d >= 0 ? (magnitude(branch_scale.m) << d) < magnitude(stream_scale.m) + : magnitude(branch_scale.m) < (magnitude(stream_scale.m) << -d); + } + struct Candidate { + SslmForwardStatus status = SslmForwardStatus::Ok; + CarriedScale scale{}; + std::vector wide; + }; + const auto build = [&](bool select_branch) { + Candidate c; + c.scale = select_branch ? branch_scale : stream_scale; + const CarriedScale other_scale = select_branch ? stream_scale : branch_scale; + const int8_t* direct_code = select_branch ? branch_code : stream_code; + const int8_t* other_code = select_branch ? stream_code : branch_code; + const auto reciprocal = superslm::CarriedScaleNormalizedReciprocal(magnitude(c.scale.m)); + c.wide.resize(n); + for (size_t i = 0; i < n; ++i) { + bool exceeded = false; + int64_t landed = superslm::LandingRescale(static_cast(other_code[i]), other_scale.m, reciprocal.r, + other_scale.e, c.scale.e, nullptr, &exceeded, nullptr, + reciprocal.s); + if (exceeded || (c.scale.m < 0 && landed == INT64_MIN)) { + c.status = SslmForwardStatus::ResidualReconciliationMagnitudeOutOfDomain; + return c; + } + if (c.scale.m < 0) landed = -landed; + const int64_t direct = static_cast(direct_code[i]); + if ((landed > 0 && direct > INT64_MAX - landed) || (landed < 0 && direct < INT64_MIN - landed)) { + c.status = SslmForwardStatus::ResidualReconciliationMagnitudeOutOfDomain; + return c; + } + c.wide[i] = direct + landed; + } + const CarriedScale incoming[1] = {c.scale}; + c.status = superslm::PreflightRequantChain(c.wide.data(), n, std::span{incoming, 1}, + site_constant) + .status; + return c; + }; + Candidate c = build(branch_selected); + if (c.status != SslmForwardStatus::Ok) c = build(!branch_selected); + if (c.status != SslmForwardStatus::Ok) { + r.status = c.status; + return r; + } + const CarriedScale incoming[1] = {c.scale}; + r.status = superslm::RequantChainChecked(c.wide.data(), n, std::span{incoming, 1}, + site_constant, r.out.data(), &r.scale) + .status; + return r; +} + +// Whether a residual call reaches its table decision (the scale checks accept). +bool ResidualCounted(CarriedScale b, CarriedScale s) { + const int64_t i32min = INT64_C(-2147483648), i32max = INT64_C(2147483647); + return b.m != 0 && s.m != 0 && b.m >= i32min && b.m <= i32max && s.m >= i32min && s.m <= i32max; +} + +void CheckEqual(const char* label, size_t n, const SiteResult& want, SslmForwardStatus st, const CarriedScale& scale, + const std::vector& out) { + CHECK_MSG(st == want.status, "%s n=%zu: status %s, reference %s", label, n, SslmForwardStatusName(st), + SslmForwardStatusName(want.status)); + if (st == SslmForwardStatus::Ok && want.status == SslmForwardStatus::Ok) + CHECK_MSG(scale.m == want.scale.m && scale.e == want.scale.e, + "%s n=%zu: scale (%lld, %lld), reference (%lld, %lld)", label, n, static_cast(scale.m), + static_cast(scale.e), static_cast(want.scale.m), + static_cast(want.scale.e)); + size_t first = n, count = 0; + for (size_t i = 0; i < n; ++i) + if (out[i] != want.out[i]) { + if (first == n) first = i; + ++count; + } + CHECK_MSG(count == 0, "%s n=%zu: %zu output bytes differ from the reference, first at %zu (%d vs %d)", label, n, + count, first, first < n ? out[first] : 0, first < n ? want.out[first] : 0); +} + +// One call through each site against its reference, with the path assertion. +void RunNorm(const char* label, const std::vector& h, const std::vector& g, + CarriedScale site_constant) { + const size_t n = h.size(); + const SiteResult want = RefNorm(h.data(), g.data(), n, site_constant); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const RowCounters c0 = ReadRowCounters(); + const SslmForwardStatus st = + superslm::RmsNormSite(h.data(), g.data(), n, CarriedScale{}, site_constant, out.data(), &scale); + const RowCounters c1 = ReadRowCounters(); + CheckEqual(label, n, want, st, scale, out); + CheckPath(label, Delta(c0, c1), kNorm, /*counted=*/true, n); +} + +void RunSilu(const char* label, const std::vector& gate, CarriedScale gate_scale, + const std::vector& up, CarriedScale up_scale, CarriedScale site_constant) { + const size_t n = gate.size(); + const SiteResult want = RefSilu(gate.data(), gate_scale, up.data(), up_scale, n, site_constant); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const RowCounters c0 = ReadRowCounters(); + const SslmForwardStatus st = superslm::MlpActSite(gate.data(), gate_scale, up.data(), up_scale, n, + superslm::kSiluLutCanonicalTable, site_constant, + out.data(), &scale); + const RowCounters c1 = ReadRowCounters(); + CheckEqual(label, n, want, st, scale, out); + const bool counted = + superslm::CheckSiluCompositionScaleDomain(gate_scale.m, gate_scale.e) == SslmForwardStatus::Ok; + CheckPath(label, Delta(c0, c1), kSilu, counted, n); +} + +SslmForwardStatus RunResidual(const char* label, const std::vector& branch, CarriedScale branch_scale, + const std::vector& stream, CarriedScale stream_scale, + CarriedScale site_constant) { + const size_t n = branch.size(); + const SiteResult want = RefResidual(branch.data(), branch_scale, stream.data(), stream_scale, n, site_constant); + std::vector out(n, kPoison); + CarriedScale scale{-1, -1}; + const RowCounters c0 = ReadRowCounters(); + const SslmForwardStatus st = superslm::ResidualReconcileSite(branch.data(), branch_scale, stream.data(), + stream_scale, n, site_constant, out.data(), &scale); + const RowCounters c1 = ReadRowCounters(); + CheckEqual(label, n, want, st, scale, out); + CheckPath(label, Delta(c0, c1), kLanding, ResidualCounted(branch_scale, stream_scale), n); + return st; +} + +// The widths of 4.S1 (both sides of 512 and the 0.5B's widths). +constexpr size_t kGridWidths[] = {1, 64, 255, 511, 512, 513, 896, 4864}; + +// ---- 4.S1: the shape grid, every site ---------------------------------------------------------- + +// The −128 rows at a small gate scale (4.S1, §9 "−128 read from the table"): at m = 2^30, e = −35 +// (128·scale ≤ 4) sig(−128) = 589 is neither 0 nor sig(−127) = 608, so a table entry left unset or +// filled from −127 changes the output. The premises are asserted, not assumed. +void TestS1SmallGateScalePremise() { + const CarriedScale small{INT64_C(1073741824), -35}; + CHECK_MSG(superslm::CheckSiluCompositionScaleDomain(small.m, small.e) == SslmForwardStatus::Ok, + "4.S1 premise: MlpActSite's domain check accepts the small gate scale (2^30, -35)"); + const int32_t s128 = superslm::SiluSigmoidQ15(superslm::kSiluLutCanonicalTable, -128, small.m, -35); + const int32_t s127 = superslm::SiluSigmoidQ15(superslm::kSiluLutCanonicalTable, -127, small.m, -35); + CHECK_MSG(s128 == 589 && s127 == 608, "4.S1 premise: sig(-128) = %d (want 589), sig(-127) = %d (want 608)", + s128, s127); +} + +void TestS1Grid() { + Rng rng(0x3453314752494431ULL); // "4S1GRID1" + const RowShape shapes[] = {RowShape::kRandomWithMinus128, RowShape::kRandomNo128, RowShape::kAllMinus127, + RowShape::kAll127}; + const CarriedScale unit{INT64_C(1073741824), 0}; + const CarriedScale gate_scales[] = {{INT64_C(1073741824), -35}, {INT64_C(1073741824), -34}, + {INT64_C(1631069115), -30}}; + const CarriedScale up_scale{INT64_C(1340958474), -18}; + for (size_t n : kGridWidths) + for (RowShape shape : shapes) { + char label[96]; + std::snprintf(label, sizeof label, "4.S1 norm shape %d", static_cast(shape)); + std::vector h(n); + FillRow(h, shape, rng); + std::vector g(n); + for (auto& v : g) v = static_cast(rng.InRange(-300, 300)); + RunNorm(label, h, g, unit); + + for (const CarriedScale& gs : gate_scales) { + std::snprintf(label, sizeof label, "4.S1 silu shape %d gate e=%lld", static_cast(shape), + static_cast(gs.e)); + std::vector gate(n), up(n); + FillRow(gate, shape, rng); + FillRow(up, RowShape::kRandomNo128, rng); + RunSilu(label, gate, gs, up, up_scale, unit); + } + + std::snprintf(label, sizeof label, "4.S1 residual shape %d", static_cast(shape)); + std::vector branch(n), stream(n); + FillRow(branch, shape, rng); + FillRow(stream, RowShape::kRandomWithMinus128, rng); + RunResidual(label, branch, {INT64_C(1234567890), -37}, stream, {INT64_C(1987654321), -44}, unit); + RunResidual(label, stream, {INT64_C(-1400000000), -39}, branch, {INT64_C(1100000000), -41}, unit); + } + + // −128 at exactly one position (first, middle, last), everything else a code whose value differs + // from −128's, at the small gate scale, both sides of the threshold. + for (size_t n : {size_t{64}, size_t{511}, size_t{512}, size_t{896}}) + for (size_t pos : {size_t{0}, n / 2, n - 1}) { + char label[96]; + std::snprintf(label, sizeof label, "4.S1 silu -128 at %zu, small gate scale", pos); + std::vector gate(n, -127), up(n, 127); + gate[pos] = -128; + RunSilu(label, gate, {INT64_C(1073741824), -35}, up, up_scale, unit); + std::snprintf(label, sizeof label, "4.S1 norm -128 at %zu", pos); + std::vector h(n, 1); + h[pos] = -128; + std::vector g(n, 7); + RunNorm(label, h, g, unit); + std::snprintf(label, sizeof label, "4.S1 residual -128 at %zu", pos); + std::vector other(n, 3); + other[pos] = -128; + std::vector direct(n, -5); + RunResidual(label, direct, {INT64_C(1234567890), -37}, other, {INT64_C(1987654321), -44}, unit); + } + + // A refused call counts nothing: a gate scale outside the SiLU domain, a zero residual mantissa. + { + std::vector gate(896, 5), up(896, 5); + RunSilu("4.S1 silu refused by the domain check", gate, {INT64_C(1073741824), 9}, up, up_scale, unit); + RunResidual("4.S1 residual refused by a zero mantissa", gate, {0, -37}, up, {INT64_C(1987654321), -44}, + unit); + } +} + +// ---- 1.S1: no table kept across calls ---------------------------------------------------------- + +void TestS1ConsecutiveRowsDifferentConstants() { + Rng rng(0x3153315245555345ULL); + const CarriedScale unit{INT64_C(1073741824), 0}; + const size_t n = 896; + for (int round = 0; round < 2; ++round) { + // Norm: the row constant is the root, so two rows with different magnitudes. + std::vector h(n); + for (auto& v : h) v = static_cast(rng.InRange(round == 0 ? -20 : -127, round == 0 ? 20 : 127)); + std::vector g(n); + for (auto& v : g) v = static_cast(rng.InRange(-300, 300)); + RunNorm(round == 0 ? "1.S1 norm row 1" : "1.S1 norm row 2", h, g, unit); + // SiLU: the row constants are the gate scale. + std::vector gate(n), up(n); + FillRow(gate, RowShape::kRandomNo128, rng); + FillRow(up, RowShape::kRandomNo128, rng); + RunSilu(round == 0 ? "1.S1 silu row 1" : "1.S1 silu row 2", gate, + round == 0 ? CarriedScale{INT64_C(1073741824), -34} : CarriedScale{INT64_C(1631069115), -30}, up, + {INT64_C(1340958474), -18}, unit); + // Residual: the row constants are the scale pair. + std::vector branch(n), stream(n); + FillRow(branch, RowShape::kRandomNo128, rng); + FillRow(stream, RowShape::kRandomNo128, rng); + RunResidual(round == 0 ? "1.S1 residual row 1" : "1.S1 residual row 2", branch, + round == 0 ? CarriedScale{INT64_C(1234567890), -37} : CarriedScale{INT64_C(1100000000), -52}, + stream, + round == 0 ? CarriedScale{INT64_C(1987654321), -44} : CarriedScale{INT64_C(-1900000000), -32}, + unit); + } +} + +// ---- 3.S1: concurrent calls share no table ------------------------------------------------------ + +// The plan's 3.S1 names "the existing concurrent-read stress cell ... on the 0.5B-width artifact"; no +// such artifact-driven cell exists in the suite, so this is its own: 8 threads call all three sites at +// table-taking widths on their own rows, each result compared with the single-threaded reference. +// The tables are stack arrays, so nothing is shared; the hosted TSan leg runs this cell. +void TestS1ConcurrentCalls() { + constexpr int kThreads = 8, kCalls = 24; + const CarriedScale unit{INT64_C(1073741824), 0}; + struct Job { + std::vector h, gate, up, branch, stream; + std::vector g; + SiteResult want_norm, want_silu, want_res; + int mismatches = 0; + }; + std::vector jobs(kThreads); + for (int t = 0; t < kThreads; ++t) { + Rng rng(0x3353314354485200ULL + static_cast(t)); + Job& j = jobs[t]; + j.h.resize(896); j.g.resize(896); j.branch.resize(896); j.stream.resize(896); j.gate.resize(4864); j.up.resize(4864); + FillRow(j.h, RowShape::kRandomWithMinus128, rng); + for (auto& v : j.g) v = static_cast(rng.InRange(-300, 300)); + FillRow(j.gate, RowShape::kRandomWithMinus128, rng); + FillRow(j.up, RowShape::kRandomNo128, rng); + FillRow(j.branch, RowShape::kRandomWithMinus128, rng); + FillRow(j.stream, RowShape::kRandomWithMinus128, rng); + j.want_norm = RefNorm(j.h.data(), j.g.data(), 896, unit); + j.want_silu = RefSilu(j.gate.data(), {INT64_C(1073741824), -34 + t % 3}, j.up.data(), {INT64_C(1340958474), -18}, + 4864, unit); + j.want_res = RefResidual(j.branch.data(), {INT64_C(1234567890), -37 - t % 4}, j.stream.data(), + {INT64_C(1987654321), -44}, 896, unit); + } + std::vector threads; + for (int t = 0; t < kThreads; ++t) + threads.emplace_back([&jobs, t, unit] { + Job& j = jobs[t]; + for (int c = 0; c < kCalls; ++c) { + std::vector out(4864, kPoison); + CarriedScale sc{-1, -1}; + SslmForwardStatus st = superslm::RmsNormSite(j.h.data(), j.g.data(), 896, CarriedScale{}, unit, out.data(), &sc); + if (st != j.want_norm.status || sc.m != j.want_norm.scale.m || sc.e != j.want_norm.scale.e || + !std::equal(j.want_norm.out.begin(), j.want_norm.out.end(), out.begin())) + ++j.mismatches; + st = superslm::MlpActSite(j.gate.data(), {INT64_C(1073741824), -34 + t % 3}, j.up.data(), + {INT64_C(1340958474), -18}, 4864, superslm::kSiluLutCanonicalTable, unit, + out.data(), &sc); + if (st != j.want_silu.status || sc.m != j.want_silu.scale.m || sc.e != j.want_silu.scale.e || + !std::equal(j.want_silu.out.begin(), j.want_silu.out.end(), out.begin())) + ++j.mismatches; + std::fill(out.begin(), out.end(), kPoison); + st = superslm::ResidualReconcileSite(j.branch.data(), {INT64_C(1234567890), -37 - t % 4}, j.stream.data(), + {INT64_C(1987654321), -44}, 896, unit, out.data(), &sc); + if (st != j.want_res.status || sc.m != j.want_res.scale.m || sc.e != j.want_res.scale.e || + !std::equal(j.want_res.out.begin(), j.want_res.out.end(), out.begin())) + ++j.mismatches; + } + }); + for (auto& th : threads) th.join(); + int total = 0; + for (const Job& j : jobs) total += j.mismatches; + CHECK_MSG(total == 0, "3.S1: %d of %d concurrent site calls (8 threads) differ from the single-threaded reference", + total, kThreads * kCalls * 3); +} + +// ---- 5.S1 and 7.S1c: the landing flag, read per element present --------------------------------- + +// Branch e = 30, stream e = −30: the gap exceeds 31, so the stream candidate is built first and lands +// the branch codes. Code 127 overflows int64 there; code 0 does not. +void TestS1LandingFlag() { + const CarriedScale coarse{INT64_C(1300000000), 30}; + const CarriedScale fine{INT64_C(1300000000), -30}; + const CarriedScale unit{INT64_C(1073741824), 0}; + const CarriedScale bad_site_constant{INT64_C(4294967296), 0}; + + // The premises: the table's entry for 127 carries the flag and the entry for 0 does not. + const auto rec = superslm::CarriedScaleNormalizedReciprocal(static_cast(fine.m)); + bool flag127 = false, flag0 = true; + (void)superslm::LandingRescale(127, coarse.m, rec.r, coarse.e, fine.e, nullptr, &flag127, nullptr, rec.s); + (void)superslm::LandingRescale(0, coarse.m, rec.r, coarse.e, fine.e, nullptr, &flag0, nullptr, rec.s); + CHECK_MSG(flag127 && !flag0, "5.S1 premise: landing flag at code 127 = %d (want 1), at code 0 = %d (want 0)", + flag127 ? 1 : 0, flag0 ? 1 : 0); + + for (size_t n : {size_t{64}, size_t{511}, size_t{512}, size_t{896}, size_t{4864}}) { + Rng rng(0x3553314C414E4400ULL + n); + std::vector stream(n); + FillRow(stream, RowShape::kRandomWithMinus128, rng); + + // 5.S1 (a): the first candidate is refused mid-row, the second commits. + std::vector branch(n, 0); + branch[n / 2] = 127; + const SslmForwardStatus a = RunResidual("5.S1 first candidate refused mid-row", branch, coarse, stream, fine, unit); + CHECK_MSG(a == SslmForwardStatus::Ok, "5.S1 n=%zu: the second candidate commits (status %s)", n, + SslmForwardStatusName(a)); + + // 5.S1 (b): both candidates fail: the first by the landing flag mid-row, the second by the funnel. + const SslmForwardStatus b = + RunResidual("5.S1 both candidates refused", branch, coarse, stream, fine, bad_site_constant); + CHECK_MSG(b == SslmForwardStatus::CarriedScaleMantissaOutOfDomain, + "5.S1 n=%zu: both refused returns the second candidate's status (got %s)", n, + SslmForwardStatusName(b)); + + // 7.S1c: the row's codes never overflow (every branch code is 0) while code 127 would. The + // first candidate commits: a flag OR-ed over the whole table would refuse it and commit the + // second candidate's different scale. + std::vector zeros(n, 0); + const SslmForwardStatus c = RunResidual("7.S1c flag read per element present", zeros, coarse, stream, fine, unit); + CHECK_MSG(c == SslmForwardStatus::Ok, "7.S1c n=%zu: the first candidate commits (status %s)", n, + SslmForwardStatusName(c)); + const SiteResult first = RefResidual(zeros.data(), coarse, stream.data(), fine, n, unit); + CHECK_MSG(first.status == SslmForwardStatus::Ok && first.scale.e != coarse.e, + "7.S1c premise n=%zu: the committed scale is the fine candidate's, not the coarse one's", n); + } +} + +// ---- 6.3: the golden pin (the v1.9.0 tag's hash over the digest input set) ---------------------- + +void TestS1GoldenPin() { + superslm::Sha256 h; + uint64_t values = 0; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++values; + }; + const RowCounters c0 = ReadRowCounters(); + superslm_rowsite_cases::RunRowTableCases(emit); + const RowCounters d = Delta(c0, ReadRowCounters()); + uint8_t digest[32]; + h.Final(digest); + const std::string hex = superslm::ToHex(digest); + std::printf("attn-rowsites S1 golden hash: %s (%llu values)\n", hex.c_str(), + static_cast(values)); + CHECK_MSG(hex == std::string(superslm_test::kAttnRowsiteS1GoldenHash) && + values == superslm_test::kAttnRowsiteS1GoldenValues, + "6.3 S1 golden: %s over %llu values, pin %s over %llu (v1.9.0 tag)", hex.c_str(), + static_cast(values), superslm_test::kAttnRowsiteS1GoldenHash, + static_cast(superslm_test::kAttnRowsiteS1GoldenValues)); + // The input set reaches both sides of the threshold at every site. + if (kHaveRowCounters) + for (int s = 0; s < 3; ++s) + CHECK_MSG(d.taken[s] > 0 && d.skipped[s] > 0, "6.3: the golden set takes and skips the %s table (+%lld/+%lld)", + kRowSiteNames[s], d.taken[s], d.skipped[s]); +} + +// ==== Slice S2: prob·V on int16 multiply-add, per head (§4.2, §5.2) ============================== +// +// Every S2 cell calls GemmProbQ15Accumulate directly and asserts two things per call (§8 path rule): +// the output equals a test-side copy of the v1.9.0 loop, and the prob·V path counters moved by +// exactly the delta a test-side copy of the guard names, on the active kernel's own tier only. Which +// kernel runs at all comes first, through a test-side copy of 11.2's selector. + +using superslm::detail::GemmTier; +using superslm::detail::SitesKernel; + +// The tier the build under test dispatches on: known from the force macro in a forced binary, read +// from the build in the auto binary (the CPU decides there). +GemmTier ExpectedGemmTier() { +#if defined(SUPERSLM_FORCE_SCALAR_MATMUL) + return GemmTier::kScalar; +#elif defined(SUPERSLM_FORCE_SSE2_MATMUL) + return GemmTier::kSse2; +#elif defined(SUPERSLM_FORCE_AVX2_MATMUL) + return GemmTier::kAvx2; +#elif defined(SUPERSLM_FORCE_AVX512_MATMUL) + return GemmTier::kAvx512; +#else + return superslm::detail::ActiveGemmTier(); +#endif +} + +// The test-side switch and compiler identity (§3.2), read from the macros the test itself sees. +#if defined(SUPERSLM_SITES_AVX512_MSVC) +constexpr int kTestSitesAvx512MsvcSwitch = SUPERSLM_SITES_AVX512_MSVC; +#else +constexpr int kTestSitesAvx512MsvcSwitch = 0; +#endif +#if defined(_MSC_VER) +constexpr bool kTestIsMsvcBuild = true; +#else +constexpr bool kTestIsMsvcBuild = false; +#endif + +// The test-side copy of 11.2's selector, written from §3.2. +SitesKernel TestSelectSitesKernel(GemmTier tier, int sw, bool msvc) { + if (tier == GemmTier::kAvx2) return SitesKernel::kAvx2; + if (tier == GemmTier::kAvx512) return (msvc && sw == 0) ? SitesKernel::kShipped : SitesKernel::kAvx512; + return SitesKernel::kShipped; +} + +SitesKernel ExpectedSitesKernel() { + return TestSelectSitesKernel(ExpectedGemmTier(), kTestSitesAvx512MsvcSwitch, kTestIsMsvcBuild); +} + +const char* SitesKernelName(SitesKernel k) { + switch (k) { + case SitesKernel::kShipped: return "v1.9.0 code"; + case SitesKernel::kAvx2: return "AVX2"; + case SitesKernel::kAvx512: return "AVX-512"; + } + return "?"; +} + +// The test-side copy of the S2 guard (§4.2): head_dim % 16 == 0 and the int16 condition (every p in +// [0, 32767] and Sum p <= 2^15). The three conjuncts are reported separately so a 2.S2 row can be +// shown to fail exactly one. +struct PvGuard { + bool hd16 = true, p_nonneg = true, p_le_max = true, sum_le = true; + bool Fast() const { return hd16 && p_nonneg && p_le_max && sum_le; } + int Failing() const { return !hd16 + !p_nonneg + !p_le_max + !sum_le; } +}; + +PvGuard TestPvGuard(const int64_t* probs, size_t width, size_t head_dim) { + PvGuard g; + g.hd16 = head_dim % 16 == 0; + int64_t sum = 0; // every row of the set is bounded (|p| <= 32,769, width <= 4,097), so int64 is exact + for (size_t k = 0; k < width; ++k) { + if (probs[k] < 0) g.p_nonneg = false; + if (probs[k] > 32767) g.p_le_max = false; + sum += probs[k]; + } + g.sum_le = sum <= 32768; + return g; +} + +// The prob·V counters (§3.6), inside the x64 block of the seam. +struct PvCounters { + long long fast2 = 0, fb2 = 0, fast5 = 0, fb5 = 0; +}; +#if defined(SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT) && SUPERSLM_MATMUL_HAVE_SIMD_X64 +constexpr bool kHavePvCounters = true; +PvCounters ReadPvCounters() { + PvCounters c; + c.fast2 = superslm_test::g_pv_fast_avx2.load(); + c.fb2 = superslm_test::g_pv_fallback_avx2.load(); + c.fast5 = superslm_test::g_pv_fast_avx512.load(); + c.fb5 = superslm_test::g_pv_fallback_avx512.load(); + return c; +} +#else +constexpr bool kHavePvCounters = false; +PvCounters ReadPvCounters() { return PvCounters{}; } +#endif + +PvCounters PvDelta(const PvCounters& a, const PvCounters& b) { + return PvCounters{b.fast2 - a.fast2, b.fb2 - a.fb2, b.fast5 - a.fast5, b.fb5 - a.fb5}; +} + +// The expected delta of `calls` calls of which `fast` take the fast path, on this build (11.1(b)). +PvCounters ExpectedPvDelta(long long fast, long long fallback) { + PvCounters w; + switch (ExpectedSitesKernel()) { + case SitesKernel::kAvx2: w.fast2 = fast; w.fb2 = fallback; break; + case SitesKernel::kAvx512: w.fast5 = fast; w.fb5 = fallback; break; + case SitesKernel::kShipped: break; + } + return w; +} + +void CheckPvDelta(const char* label, const PvCounters& d, const PvCounters& want) { + if (!kHavePvCounters) return; + CHECK_MSG(d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5, + "%s: pv_fast_avx2 +%lld pv_fallback_avx2 +%lld pv_fast_avx512 +%lld pv_fallback_avx512 +%lld; want " + "+%lld/+%lld/+%lld/+%lld (kernel: %s)", + label, d.fast2, d.fb2, d.fast5, d.fb5, want.fast2, want.fb2, want.fast5, want.fb5, + SitesKernelName(ExpectedSitesKernel())); +} + +// The v1.9.0 loop, restated: zero, then key outer, dimension inner, exact int64. +std::vector RefProbV(const int64_t* probs, const int8_t* values, size_t width, size_t head_dim) { + std::vector out(head_dim, 0); + for (size_t k = 0; k < width; ++k) + for (size_t d = 0; d < head_dim; ++d) + out[d] += probs[k] * static_cast(values[k * head_dim + d]); + return out; +} + +// One call against the reference, with its path assertion. Every buffer is an exact-size heap +// vector, so the hosted ASan leg sees any read or write past one (4.S2's head_dim 60 and 100 rows). +// Returns whether the guard copy expected the fast path. +bool RunPv(const char* label, const std::vector& probs, const std::vector& values, size_t width, + size_t head_dim) { + const std::vector want = RefProbV(probs.data(), values.data(), width, head_dim); + std::vector out(head_dim, superslm_attention_cases::kPvPoison); + const PvGuard g = TestPvGuard(probs.data(), width, head_dim); + const PvCounters c0 = ReadPvCounters(); + superslm::GemmProbQ15Accumulate(probs.data(), values.data(), width, head_dim, out.data()); + const PvCounters c1 = ReadPvCounters(); + size_t bad = 0, first = head_dim; + for (size_t d = 0; d < head_dim; ++d) + if (out[d] != want[d]) { + if (first == head_dim) first = d; + ++bad; + } + CHECK_MSG(bad == 0, "%s (head_dim %zu, width %zu): %zu of %zu outputs differ from the v1.9.0 loop, first at %zu " + "(%lld vs %lld)", label, head_dim, width, bad, head_dim, first, + first < head_dim ? static_cast(out[first]) : 0LL, + first < head_dim ? static_cast(want[first]) : 0LL); + char full[192]; + std::snprintf(full, sizeof full, "%s (head_dim %zu, width %zu, guard copy: %s)", label, head_dim, width, + g.Fast() ? "fast" : "fallback"); + CheckPvDelta(full, PvDelta(c0, c1), ExpectedPvDelta(g.Fast() ? 1 : 0, g.Fast() ? 0 : 1)); + return g.Fast(); +} + +// ---- 11.2: the MSVC switch, a pure selector with its full truth table, and its wiring --------------- + +void TestS2SitesSelector() { + using superslm::detail::SelectSitesKernel; + const GemmTier tiers[] = {GemmTier::kScalar, GemmTier::kSse2, GemmTier::kAvx2, GemmTier::kAvx512}; + size_t bad = 0; + for (GemmTier t : tiers) + for (bool msvc : {false, true}) + for (int sw : {0, 1}) { + const SitesKernel got = SelectSitesKernel(t, sw, msvc); + const SitesKernel want = TestSelectSitesKernel(t, sw, msvc); + if (got != want) { + ++bad; + std::printf(" cell 11.2: SelectSitesKernel(tier %d, switch %d, msvc %d) = %s, want %s\n", + static_cast(t), sw, msvc ? 1 : 0, SitesKernelName(got), SitesKernelName(want)); + } + } + CHECK_MSG(bad == 0, "cell 11.2: %zu of 16 selector rows wrong", bad); + // The wiring: this build's own switch and compiler identity, on every tier (a hosted runner of any + // CPU can observe it; the MSVC legs are where the switch's value decides the AVX-512 row). + for (GemmTier t : tiers) + CHECK_MSG(superslm::detail::DispatchSitesKernel(t) == + TestSelectSitesKernel(t, kTestSitesAvx512MsvcSwitch, kTestIsMsvcBuild), + "cell 11.2 wiring: DispatchSitesKernel(tier %d) = %s, want %s (switch %d, msvc %d)", + static_cast(t), SitesKernelName(superslm::detail::DispatchSitesKernel(t)), + SitesKernelName(TestSelectSitesKernel(t, kTestSitesAvx512MsvcSwitch, kTestIsMsvcBuild)), + kTestSitesAvx512MsvcSwitch, kTestIsMsvcBuild ? 1 : 0); + std::printf("attn-rowsites S2: tier %d, kernel %s (switch %d, msvc %d), prob-V counters %s\n", + static_cast(ExpectedGemmTier()), SitesKernelName(ExpectedSitesKernel()), + kTestSitesAvx512MsvcSwitch, kTestIsMsvcBuild ? 1 : 0, kHavePvCounters ? "read" : "not compiled"); +} + +// ---- 4.S2, 7.S2, 6.1: the shape grid and the int16 condition's corners ----------------------------- + +void TestS2Grid() { + size_t fast = 0, fallback = 0; + superslm_attention_cases::ForEachProbVCase([&](const superslm_attention_cases::PvCase& c) { + (RunPv(c.label, c.probs, c.values, c.width, c.head_dim) ? fast : fallback) += 1; + }); + // The kernels' own blocking (not in the golden set, so the pin stays the plan's grid): AVX2 runs + // 16-dimension units four at a time, AVX-512 runs 32-dimension units four at a time plus one + // 16-dimension tail when head_dim % 32 == 16. These head_dims reach every block count and the tail + // behind a full block, each at even and odd widths. + { + superslm_attention_cases::Rng rng(0x5332424C4F434B53ULL); // "S2BLOCKS" + for (size_t hd : {size_t{32}, size_t{48}, size_t{80}, size_t{96}, size_t{144}, size_t{160}, size_t{208}, + size_t{224}, size_t{240}}) + for (size_t w : {size_t{1}, size_t{2}, size_t{3}, size_t{64}, size_t{65}}) { + const std::vector p = superslm_attention_cases::RealisticRow(w, 10, rng); + const std::vector v = superslm_attention_cases::RandomValues(w * hd, rng); + (RunPv("4.S2 kernel blocking", p, v, w, hd) ? fast : fallback) += 1; + } + } + // The grid reaches both sides of every conjunct: fast and fallback both occur. + CHECK_MSG(fast > 0 && fallback > 0, "4.S2: the set takes the fast path %zu times and falls back %zu times", fast, + fallback); + // The named corners, asserted against the guard copy's own verdict, so a guard copy that drifted + // from §4.2 cannot quietly move a corner to the other side. + const struct { + std::vector p; + size_t hd; + bool fast; + const char* what; + } corners[] = { + {{32767, 1}, 64, true, "p = 32,767 with Sum p = 2^15 is inside"}, + {{32767, 2}, 64, false, "Sum p = 2^15 + 1 is outside"}, + {{32768}, 64, false, "the width-1 one-hot row is outside"}, + {{32767, 1}, 16, true, "head_dim 16 is inside"}, + {{100, 200}, 60, false, "head_dim 60 is outside"}, + {{100, 200}, 100, false, "head_dim 100 is outside"}, + }; + for (const auto& k : corners) + CHECK_MSG(TestPvGuard(k.p.data(), k.p.size(), k.hd).Fast() == k.fast, "7.S2 guard copy: %s", k.what); +} + +// ---- 2.S2: hostile rows, each failing exactly one conjunct ------------------------------------------ + +void TestS2HostileRows() { + size_t rows = 0; + superslm_attention_cases::ForEachProbVCase([&](const superslm_attention_cases::PvCase& c) { + if (std::strncmp(c.label, "2.S2", 4) != 0) return; + ++rows; + const PvGuard g = TestPvGuard(c.probs.data(), c.width, c.head_dim); + CHECK_MSG(g.Failing() == 1, "%s: fails %d conjuncts of the guard copy, want exactly 1", c.label, g.Failing()); + // RunPv (in TestS2Grid) already asserted output and fallback +1; the int32-lane premise of the + // two rows that would wrap is asserted here, so the rows keep deciding what they were built for. + }); + CHECK_MSG(rows == 4, "2.S2: %zu hostile rows in the set, want 4", rows); + // The premise of the alternating and the p = 32,767 rows: each lane's true sum, 1,024 x 32,767 x 127, + // passes INT32_MAX, so a kernel that dropped the conjunct they fail would wrap. + const int64_t lane = 1024LL * 32767 * 127; + CHECK_MSG(lane > INT32_MAX, "2.S2 premise: the lane sum %lld exceeds INT32_MAX", static_cast(lane)); +} + +// ---- 6.3: the S2 golden pin (the v1.9.0 tag's hash over the prob·V input set) ----------------------- + +void TestS2GoldenPin() { + superslm::Sha256 h; + uint64_t values = 0; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++values; + }; + superslm_attention_cases::RunProbVCases(emit); + uint8_t digest[32]; + h.Final(digest); + const std::string hex = superslm::ToHex(digest); + std::printf("attn-rowsites S2 golden hash: %s (%llu values)\n", hex.c_str(), + static_cast(values)); + CHECK_MSG(hex == std::string(superslm_test::kAttnRowsiteS2GoldenHash) && + values == superslm_test::kAttnRowsiteS2GoldenValues, + "6.3 S2 golden: %s over %llu values, pin %s over %llu (v1.9.0 tag)", hex.c_str(), + static_cast(values), superslm_test::kAttnRowsiteS2GoldenHash, + static_cast(superslm_test::kAttnRowsiteS2GoldenValues)); +} + +// ==== Slice S3: the requant element loop in 64-bit lanes (§4.3, §5.3) ============================== +// +// Every S3 cell calls the row leaf RequantRowWide directly (or the funnel that calls it) and asserts +// per call (§8 path rule): the codes equal RequantTokenCodeWide element by element (the v1.9.0 leaf, +// unchanged and never the build's row code), nothing outside the row was written, and the requant_row +// counter moved +1 on the selected kernel's own tier and nowhere else. S3 has no runtime guard (§5.3), +// so the expected path is the selector's alone: AVX2 and AVX-512 run the lanes, every other kernel the +// element loop. + +struct RqCounters { + long long avx2 = 0, avx512 = 0; +}; +#if defined(SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT) && SUPERSLM_MATMUL_HAVE_SIMD_X64 +constexpr bool kHaveRqCounters = true; +RqCounters ReadRqCounters() { + return RqCounters{superslm_test::g_requant_row_avx2.load(), superslm_test::g_requant_row_avx512.load()}; +} +#else +constexpr bool kHaveRqCounters = false; +RqCounters ReadRqCounters() { return RqCounters{}; } +#endif + +RqCounters RqDelta(const RqCounters& a, const RqCounters& b) { return RqCounters{b.avx2 - a.avx2, b.avx512 - a.avx512}; } + +// `calls` row-leaf calls on this build (11.1(b)): the selected kernel's own tier moves, the other does not. +RqCounters ExpectedRqDelta(long long calls) { + RqCounters w; + switch (ExpectedSitesKernel()) { + case SitesKernel::kAvx2: w.avx2 = calls; break; + case SitesKernel::kAvx512: w.avx512 = calls; break; + case SitesKernel::kShipped: break; + } + return w; +} + +bool CheckRqDelta(const char* label, const RqCounters& d, const RqCounters& want) { + if (!kHaveRqCounters) return true; + const bool ok = d.avx2 == want.avx2 && d.avx512 == want.avx512; + CHECK_MSG(ok, "%s: requant_row_avx2 +%lld requant_row_avx512 +%lld; want +%lld/+%lld (kernel: %s)", label, d.avx2, + d.avx512, want.avx2, want.avx512, SitesKernelName(ExpectedSitesKernel())); + return ok; +} + +// r and s exactly as the funnel's preflight derives them from d' (checked_chain_funnel.cpp step 4). +struct RqConstants { + int64_t r; + int s; +}; +RqConstants RqConstantsFor(int64_t d_prime) { + const superslm::NormalizedScale ns = superslm::NormalizeScale(d_prime); + return RqConstants{superslm::DynamicScaleReciprocal(ns.dn), ns.s}; +} + +constexpr size_t kRqFence = 16; +constexpr uint8_t kRqSentinel = 0xA5; + +// One row-leaf call against the element loop, with its path assertion. `fenced`: the output row sits +// between 16 sentinel bytes before out[0] and after out[n - 1], checked unchanged after the call (every +// binary). Otherwise the input and the output are exact-size heap buffers, so the hosted ASan leg sees +// any read or write past either (4.S3's second pass). The value assertions are one check each; the +// path assertion is tallied in `path_mismatches` (with the first few printed) and asserted once per +// pass by the caller, so a build whose counter never moves reports one failure per pass, not one per +// row. Returns the number of failed value assertions. +int RunRequantRow(const char* label, const std::vector& row, int64_t d_prime, bool fenced, + size_t* path_mismatches) { + const int f0 = GFailures; + const size_t n = row.size(); + const RqConstants k = RqConstantsFor(d_prime); + std::vector want(n); + for (size_t i = 0; i < n; ++i) want[i] = superslm::RequantTokenCodeWide(row[i], k.r, k.s); + std::vector x(row); // exact size + std::vector fencedbuf; + std::vector exact; + int8_t* out; + if (fenced) { + fencedbuf.assign(n + 2 * kRqFence, kRqSentinel); + out = reinterpret_cast(fencedbuf.data() + kRqFence); + std::memset(out, static_cast(kPoison), n); + } else { + exact.assign(n, kPoison); + out = exact.data(); + } + const RqCounters c0 = ReadRqCounters(); + superslm::RequantRowWide(x.data(), n, k.r, k.s, out); + const RqCounters c1 = ReadRqCounters(); + size_t bad = 0, first = n; + for (size_t i = 0; i < n; ++i) + if (out[i] != want[i]) { + if (first == n) first = i; + ++bad; + } + CHECK_MSG(bad == 0, "%s (n %zu, d' %lld, r %lld, s %d): %zu of %zu codes differ from RequantTokenCodeWide, first at " + "%zu (x %lld: %d vs %d)", label, n, static_cast(d_prime), static_cast(k.r), k.s, bad, + n, first, first < n ? static_cast(row[first]) : 0LL, first < n ? out[first] : 0, + first < n ? want[first] : 0); + if (fenced) { + size_t clobbered = 0; + for (size_t i = 0; i < kRqFence; ++i) { + clobbered += fencedbuf[i] != kRqSentinel; + clobbered += fencedbuf[kRqFence + n + i] != kRqSentinel; + } + CHECK_MSG(clobbered == 0, "%s (n %zu, d' %lld): %zu of the 32 sentinel bytes around the row changed", label, n, + static_cast(d_prime), clobbered); + } + CHECK_MSG(x == row, "%s (n %zu): the input row changed", label, n); + const int value_failures = GFailures - f0; + const RqCounters d = RqDelta(c0, c1), want_d = ExpectedRqDelta(1); + if (kHaveRqCounters && (d.avx2 != want_d.avx2 || d.avx512 != want_d.avx512)) { + if (++*path_mismatches <= 3) + std::printf(" %s (n %zu, d' %lld): requant_row_avx2 +%lld requant_row_avx512 +%lld, want +%lld/+%lld\n", label, + n, static_cast(d_prime), d.avx2, d.avx512, want_d.avx2, want_d.avx512); + } + return value_failures; +} + +// ---- 7.S3: the P = 2^63 corner's premises (§5.3) -------------------------------------------------- + +void TestS3CornerPremise() { + const superslm::NormalizedScale ns = superslm::NormalizeScale(INT64_C(1) << 31); + CHECK_MSG(ns.dn == (INT64_C(1) << 30) && ns.s == -1, "7.S3 premise: NormalizeScale(2^31) = (%lld, %d), want (2^30, -1)", + static_cast(ns.dn), ns.s); + const int64_t r = superslm::DynamicScaleReciprocal(ns.dn); + CHECK_MSG(r == (INT64_C(1) << 32), "7.S3 premise: DynamicScaleReciprocal(2^30) = %lld, want 2^32", static_cast(r)); + // |x| = 2^31 times r = 2^32 is P = 2^63 exactly: one past INT64_MAX, so a signed lane cannot hold it. + const uint64_t p = (uint64_t{1} << 31) * static_cast(r); + CHECK_MSG(p == (uint64_t{1} << 63), "7.S3 premise: P at the corner is 2^63"); + // Both corner codes are the true +-127 (§5.3: the clamp never fires inside the contract, and the corner + // itself lands exactly on 127). + CHECK_MSG(superslm::RequantTokenCodeWide(INT64_C(1) << 31, r, -1) == 127 && + superslm::RequantTokenCodeWide(-(INT64_C(1) << 31), r, -1) == -127, + "7.S3 premise: the corner's codes are +-127"); + // Every s the preflight can produce, in [-1, 30], is reached by RequantDPrimes' d' list. + superslm_rowsite_cases::Rng rng(1); + bool seen[32] = {}; + for (int64_t dp : superslm_rowsite_cases::RequantDPrimes(rng)) { + const int s = superslm::NormalizeScale(dp).s; + if (s >= -1 && s <= 30) seen[s + 1] = true; + } + int missing = 0; + for (bool b : seen) missing += !b; + CHECK_MSG(missing == 0, "4.S3 premise: %d of the 32 shifts s in [-1, 30] are not reached by the d' list", missing); +} + +// ---- 4.S3, 7.S3, 6.1: the shape grid, both fences -------------------------------------------------- + +// The grid: n {1, 3, 4, 5, 7, 8, 9, 896, 4,864} x d' (1, 2, 2^30, 2^31, their neighbours, and three in +// every octave, so every s in [-1, 30]) x +-d' at every lane position (every position of rows up to 9 +// elements, positions 0-8, 15, 16 and the last of the wide rows), the rest drawn from [-d', d']. Then +// random rows: n in [1, 40], d' log-uniform over [1, 2^31], with values biased to +-d' and to rounding +// ties. `fenced` selects the sentinel pass or the exact-size heap pass. +void RunS3Grid(bool fenced) { + const char* const pass = fenced ? "4.S3 sentinel pass" : "4.S3 exact-size pass"; + superslm_rowsite_cases::Rng rng(fenced ? 0x3453334752494431ULL : 0x3453334752494432ULL); // "4S3GRID1/2" + const std::vector dprimes = superslm_rowsite_cases::RequantDPrimes(rng); + size_t calls = 0, path_bad = 0; + int failed_calls = 0; + const RqCounters c0 = ReadRqCounters(); + // A broken body would otherwise print thousands of rows: each loop stops after 20 failing calls. + for (size_t n : superslm_rowsite_cases::kRequantWidths) + for (int64_t dp : dprimes) { + std::vector positions; + if (n <= 9) { + for (size_t p = 0; p < n; ++p) positions.push_back(p); + } else { + positions = {0, 1, 2, 3, 4, 5, 6, 7, 8, 15, 16, n - 1}; + } + for (size_t p : positions) + for (int sign : {1, -1}) { + if (failed_calls > 20) continue; + std::vector x(n); + for (auto& v : x) v = rng.InRange(-dp, dp); + x[p] = sign * dp; + ++calls; + failed_calls += RunRequantRow(pass, x, dp, fenced, &path_bad) != 0; + } + } + // Every element at +-d' (the corner in every lane at once, at d' = 2^31). + for (size_t n : superslm_rowsite_cases::kRequantWidths) + for (int64_t dp : {INT64_C(1), INT64_C(2), INT64_C(1) << 30, INT64_C(1) << 31}) { + std::vector x(n); + for (size_t i = 0; i < n; ++i) x[i] = (i % 3 == 1) ? -dp : dp; + ++calls; + failed_calls += RunRequantRow(pass, x, dp, fenced, &path_bad) != 0; + } + // Random rows, and the empty row (reads and writes nothing, still one call). + for (int t = 0; t < 4000 && failed_calls <= 20; ++t) { + const size_t n = static_cast(rng.InRange(1, 40)); + const int octave = static_cast(rng.InRange(0, 31)); + const int64_t dp = octave == 31 ? (INT64_C(1) << 31) : rng.InRange(INT64_C(1) << octave, (INT64_C(2) << octave) - 1); + const RqConstants k = RqConstantsFor(dp); + std::vector x(n); + for (auto& v : x) { + switch (rng.InRange(0, 3)) { + case 0: v = rng.Next() & 1 ? dp : -dp; break; + case 1: { + // A value whose |x|·127·r sits next to a rounding tie (e = 62 - s): x = floor(j · 2^(e-1) / + // (127·r)) + {-1, 0, 1} for a random odd j < 256, clamped into [-d', d']. The quotient is + // formed exactly in 64 bits: 2^(e-1) = Q·127r + R with R < 127r < 2^39. + const int e = 62 - k.s; + const uint64_t j = static_cast(2 * rng.InRange(0, 127) + 1); + const uint64_t div = 127u * static_cast(k.r); + const uint64_t half = uint64_t{1} << (e - 1); + int64_t m = static_cast(j * (half / div) + (j * (half % div)) / div); + m += rng.InRange(-1, 1); + m = std::clamp(m, 0, dp); + v = rng.Next() & 1 ? m : -m; + break; + } + default: v = rng.InRange(-dp, dp); break; + } + } + x[static_cast(rng.InRange(0, static_cast(n) - 1))] = rng.Next() & 1 ? dp : -dp; + ++calls; + failed_calls += RunRequantRow(pass, x, dp, fenced, &path_bad) != 0; + } + ++calls; + failed_calls += RunRequantRow(pass, std::vector{}, 1, fenced, &path_bad) != 0; + // The path rule per call (tallied above), then the whole pass as one delta, so a counter that moved on a + // call the loop above did not see is caught too. + CHECK_MSG(path_bad == 0, "%s: %zu of %zu row-leaf calls moved the requant_row counters wrongly (kernel: %s)", pass, + path_bad, calls, SitesKernelName(ExpectedSitesKernel())); + CheckRqDelta(pass, RqDelta(c0, ReadRqCounters()), ExpectedRqDelta(static_cast(calls))); + std::printf("attn-rowsites S3 %s: %zu row-leaf calls, %d with a wrong code or fence, %zu with a wrong path\n", pass, + calls, failed_calls, path_bad); +} + +void TestS3Grid() { + RunS3Grid(/*fenced=*/true); + RunS3Grid(/*fenced=*/false); +} + +// ---- the funnel's call site: one row-leaf call per funnel call that passes its preflight ------------ + +void TestS3FunnelCallSite() { + const CarriedScale unit{INT64_C(1073741824), 0}; + superslm_rowsite_cases::Rng rng(0x5333464E4E4C0000ULL); + for (size_t n : {size_t{0}, size_t{5}, size_t{896}, size_t{4864}}) { + std::vector x(n); + for (auto& v : x) v = rng.InRange(-(INT64_C(1) << 31), INT64_C(1) << 31); + std::vector out(n, kPoison), want(n, kPoison); + CarriedScale scale{-1, -1}; + const int64_t dp = superslm::MaxAbsReduceWide(x.data(), n); + const RqConstants k = RqConstantsFor(dp); + for (size_t i = 0; i < n; ++i) want[i] = superslm::RequantTokenCodeWide(x[i], k.r, k.s); + const RqCounters c0 = ReadRqCounters(); + const SslmForwardStatus st = + superslm::RequantChainChecked(x.data(), n, std::span{}, unit, out.data(), &scale).status; + const RqCounters c1 = ReadRqCounters(); + CHECK_MSG(st == SslmForwardStatus::Ok && out == want, "S3 funnel n=%zu: status %s, codes %s the element loop's", n, + SslmForwardStatusName(st), out == want ? "equal" : "differ from"); + char label[64]; + std::snprintf(label, sizeof label, "S3 funnel call n=%zu", n); + CheckRqDelta(label, RqDelta(c0, c1), ExpectedRqDelta(1)); + } + // A refused preflight (d' = 2^31 + 1) writes nothing and never reaches the row leaf. + std::vector x(896, 3); + x[100] = (INT64_C(1) << 31) + 1; + std::vector out(896, kPoison); + CarriedScale scale{-1, -1}; + const RqCounters c0 = ReadRqCounters(); + const SslmForwardStatus st = + superslm::RequantChainChecked(x.data(), x.size(), std::span{}, unit, out.data(), &scale).status; + CHECK_MSG(st == SslmForwardStatus::ChainInputOutOfDomain && + std::all_of(out.begin(), out.end(), [](int8_t c) { return c == kPoison; }), + "S3 funnel refused: status %s, codes untouched", SslmForwardStatusName(st)); + CheckRqDelta("S3 funnel refused by the preflight", RqDelta(c0, ReadRqCounters()), ExpectedRqDelta(0)); +} + +// ---- 6.3: the S3 golden pin (the v1.9.0 tag's hash over the requant row set) ------------------------- + +void TestS3GoldenPin() { + superslm::Sha256 h; + uint64_t values = 0; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++values; + }; + superslm_rowsite_cases::RunRequantRowCases(emit); + uint8_t digest[32]; + h.Final(digest); + const std::string hex = superslm::ToHex(digest); + std::printf("attn-rowsites S3 golden hash: %s (%llu values)\n", hex.c_str(), static_cast(values)); + CHECK_MSG(hex == std::string(superslm_test::kAttnRowsiteS3GoldenHash) && + values == superslm_test::kAttnRowsiteS3GoldenValues, + "6.3 S3 golden: %s over %llu values, pin %s over %llu (v1.9.0 tag)", hex.c_str(), + static_cast(values), superslm_test::kAttnRowsiteS3GoldenHash, + static_cast(superslm_test::kAttnRowsiteS3GoldenValues)); +} + +// ==== Slice S4: the guarded softmax (§4.4, §5.4) ===================================================== +// +// Every S4 cell calls SoftmaxRowQ15 directly and asserts per call (§8 path rule): the bool and every +// probability equal a test-side restatement of the v1.9.0 body, and the softmax path counters moved by +// exactly the delta that the test-side guard copy (tests/support/attention_cases.h, TestSoftmaxGuard) names, +// on the selected kernel's own tier only. Width 0 counts nowhere (§3.6: "call with width >= 1"). + +using superslm_attention_cases::SmCase; +using superslm_attention_cases::SoftmaxCorrections; +using superslm_attention_cases::SoftmaxGuard; +using superslm_attention_cases::SoftmaxEstimateReplica; +using superslm_attention_cases::TestSoftmaxGuard; + +struct SmCounters { + long long fast2 = 0, fb2 = 0, fast5 = 0, fb5 = 0; +}; +#if defined(SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT) && SUPERSLM_MATMUL_HAVE_SIMD_X64 +constexpr bool kHaveSmCounters = true; +SmCounters ReadSmCounters() { + return SmCounters{superslm_test::g_softmax_fast_avx2.load(), superslm_test::g_softmax_fallback_avx2.load(), + superslm_test::g_softmax_fast_avx512.load(), superslm_test::g_softmax_fallback_avx512.load()}; +} +#else +constexpr bool kHaveSmCounters = false; +SmCounters ReadSmCounters() { return SmCounters{}; } +#endif + +SmCounters SmDelta(const SmCounters& a, const SmCounters& b) { + return SmCounters{b.fast2 - a.fast2, b.fb2 - a.fb2, b.fast5 - a.fast5, b.fb5 - a.fb5}; +} + +// `fast` fast and `fallback` fallback calls on this build (11.1(b)). +SmCounters ExpectedSmDelta(long long fast, long long fallback) { + SmCounters w; + switch (ExpectedSitesKernel()) { + case SitesKernel::kAvx2: w.fast2 = fast; w.fb2 = fallback; break; + case SitesKernel::kAvx512: w.fast5 = fast; w.fb5 = fallback; break; + case SitesKernel::kShipped: break; + } + return w; +} + +bool CheckSmDelta(const char* label, const SmCounters& d, const SmCounters& want) { + if (!kHaveSmCounters) return true; + const bool ok = d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5; + CHECK_MSG(ok, "%s: softmax_fast_avx2 +%lld softmax_fallback_avx2 +%lld softmax_fast_avx512 +%lld " + "softmax_fallback_avx512 +%lld; want +%lld/+%lld/+%lld/+%lld (kernel: %s)", + label, d.fast2, d.fb2, d.fast5, d.fb5, want.fast2, want.fb2, want.fast5, want.fb5, + SitesKernelName(ExpectedSitesKernel())); + return ok; +} + +// The v1.9.0 SoftmaxRowQ15 body, restated from the unchanged public leaves: M usable exactly when +// M = q_b^2 + q_c is in [1, 2^47] (read from the guard copy's two M flags, which judge the same 128-bit +// value); max shift; per element IExpConstruct, a kBad* outcome or a value outside [0, M] refuses the element +// (0) and the row (false); denom = max(total, 1); p = (e << 15) / denom. +bool RefSoftmax(const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, int64_t q_c, int64_t* out) { + if (width == 0) return true; + const SoftmaxGuard g = TestSoftmaxGuard(scores, 1, q_ln2, q_b, q_c); + const bool m_usable = g.m_ge1 && g.m_le; + // M fits int64 when usable; formed with unsigned wrap so no intermediate overflows. + const int64_t m = m_usable ? static_cast(static_cast(q_b) * static_cast(q_b) + + static_cast(q_c)) + : 0; + int64_t peak = scores[0]; + for (size_t k = 1; k < width; ++k) + if (scores[k] > peak) peak = scores[k]; + std::vector e(width, 0); + int64_t total = 0; + bool ok = true; + for (size_t k = 0; k < width; ++k) { + superslm::IExpConstruction c; + const superslm::IExpDomain d = superslm::IExpConstruct(scores[k] - peak, q_ln2, q_b, q_c, &c); + if (d == superslm::IExpDomain::kBadQ || d == superslm::IExpDomain::kBadQLn2 || d == superslm::IExpDomain::kBadQB) { + ok = false; + continue; + } + const int64_t v = superslm::IExpEvaluate(c); + if (!m_usable || v < 0 || v > m) { + ok = false; + continue; + } + e[k] = v; + total += v; + } + const int64_t denom = total > 1 ? total : 1; + for (size_t k = 0; k < width; ++k) out[k] = (e[k] << superslm::kProbFracBits) / denom; + return ok; +} + +// One call against the reference, with its path assertion. Exact-size heap buffers (the hosted ASan leg). +// Returns whether the guard copy expected the fast path. +bool RunSm(const SmCase& c) { + const size_t w = c.scores.size(); + std::vector want(w, 0); + const bool want_ok = RefSoftmax(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, want.data()); + const SoftmaxGuard g = TestSoftmaxGuard(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c); + std::vector out = c.aliased ? c.scores : std::vector(w, superslm_attention_cases::kSmPoison); + const int64_t* in = c.aliased ? out.data() : c.scores.data(); + const SmCounters c0 = ReadSmCounters(); + const bool ok = superslm::SoftmaxRowQ15(in, w, c.q_ln2, c.q_b, c.q_c, out.data()); + const SmCounters c1 = ReadSmCounters(); + size_t bad = 0, first = w; + for (size_t k = 0; k < w; ++k) + if (out[k] != want[k]) { + if (first == w) first = k; + ++bad; + } + CHECK_MSG(ok == want_ok && bad == 0, + "%s (width %zu, q_ln2 %lld, q_b %lld, q_c %lld): bool %d (v1.9.0 %d); %zu of %zu probabilities differ, " + "first at %zu (%lld vs %lld)", c.label, w, static_cast(c.q_ln2), static_cast(c.q_b), + static_cast(c.q_c), ok ? 1 : 0, want_ok ? 1 : 0, bad, w, first, + first < w ? static_cast(out[first]) : 0LL, first < w ? static_cast(want[first]) : 0LL); + char full[224]; + std::snprintf(full, sizeof full, "%s (width %zu, guard copy: %s)", c.label, w, g.Fast() ? "fast" : "fallback"); + CheckSmDelta(full, SmDelta(c0, c1), ExpectedSmDelta(g.Fast() ? 1 : 0, g.Fast() ? 0 : 1)); + return g.Fast(); +} + +// ---- 4.S4, 7.S4, 6.1: the grid, the inside corners, the correction rows, the aliased rows ---------- + +void TestS4Grid() { + size_t fast = 0, fallback = 0, aliased = 0; + std::map corrections; // label -> rows on which the named correction decides the output + superslm_attention_cases::ForEachSoftmaxCase([&](const SmCase& c) { + (RunSm(c) ? fast : fallback) += 1; + aliased += c.aliased ? 1 : 0; + const std::string label(c.label); + if (label.find("correction row") == std::string::npos) return; + // The row's premise (7.S4b): the replica says the named correction fires on it, and the same + // arithmetic without that correction gives a different row, so the §9 "skipped" mutant dies here. + const size_t w = c.scores.size(); + const int skip = label.find("z up") != std::string::npos ? 1 : label.find("p up") != std::string::npos ? 2 : 3; + std::vector good(w), without(w), want(w); + SoftmaxCorrections k, unused; + const bool in = SoftmaxEstimateReplica(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, good.data(), &k); + SoftmaxEstimateReplica(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, without.data(), &unused, skip); + const bool ref_ok = RefSoftmax(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, want.data()); + const long long fired = skip == 1 ? k.z_up : skip == 2 ? k.p_up : k.p_down; + CHECK_MSG(in && ref_ok && fired > 0 && good == want && without != want && k.z_down == 0, + "7.S4b premise, %s (width %zu): inside %d, fires %lld times, replica == v1.9.0 %d, skipped differs %d, " + "z down %lld", c.label, w, in ? 1 : 0, fired, good == want ? 1 : 0, without != want ? 1 : 0, k.z_down); + ++corrections[label.substr(0, label.find(" (steered"))]; + }); + CHECK_MSG(fast > 0 && fallback > 0 && aliased > 0, "4.S4: %zu fast, %zu fallback, %zu aliased calls", fast, fallback, + aliased); + for (const char* want : {"4.S4 correction row: z up", "4.S4 correction row: p up", "4.S4 correction row: p down"}) + CHECK_MSG(corrections[want] >= 16, "4.S4: %d rows for \"%s\", want at least 16", corrections[want], want); + // Width 0: true, nothing written, no counter (§3.6 counts calls with width >= 1). + { + int64_t sentinel = superslm_attention_cases::kSmPoison; + const SmCounters c0 = ReadSmCounters(); + const bool ok = superslm::SoftmaxRowQ15(&sentinel, 0, 636211, 1272422, 848665286933, &sentinel); + const SmCounters c1 = ReadSmCounters(); + CHECK_MSG(ok && sentinel == superslm_attention_cases::kSmPoison, "4.S4 width 0: bool %d, sentinel %s", ok ? 1 : 0, + sentinel == superslm_attention_cases::kSmPoison ? "kept" : "written"); + CheckSmDelta("4.S4 width 0", SmDelta(c0, c1), SmCounters{}); + } + // The named inside corners, asserted against the guard copy's own verdict (a guard copy that drifted + // from §5.4 cannot quietly move a corner to the other side). + const int64_t zero = 0; + const struct { + int64_t q_ln2, q_b, q_c; + size_t width; + int64_t score; + bool fast; + const char* what; + } corners[] = { + {1, 0, INT64_C(1) << 47, size_t{1} << 14, 5, true, "q_ln2 = 1 with M = 2^47 at width 2^14 is inside"}, + {1, 0, (INT64_C(1) << 47) + 1, 1, 0, false, "M = 2^47 + 1 is outside"}, + {3, 1, 0, 1, 0, true, "M = 1 is inside"}, + {1, 0, 0, 1, 0, false, "M = 0 is outside"}, + {9, 4, 0, 1, 0, true, "q_c = 0 is inside"}, + {9, 4, -1, 1, 0, false, "q_c = -1 is outside"}, + {9, 4, 0, 1, INT64_C(1) << 61, true, "a score of 2^61 is inside"}, + {9, 4, 0, 1, -(INT64_C(1) << 61), true, "a score of -2^61 is inside"}, + {9, 4, 0, 1, (INT64_C(1) << 61) + 1, false, "a score of 2^61 + 1 is outside"}, + {10, 4, 0, 1, 0, false, "q_ln2 = 2 q_b + 2 is outside"}, + {9, 4, 0, (size_t{1} << 14) + 1, 0, false, "width 2^14 + 1 is outside"}, + }; + for (const auto& k : corners) { + std::vector row(k.width, 0); + row[0] = k.score; + CHECK_MSG(TestSoftmaxGuard(row.data(), row.size(), k.q_ln2, k.q_b, k.q_c).Fast() == k.fast, "7.S4a guard copy: %s", + k.what); + } + (void)zero; +} + +// ---- 2.S4 and 7.S4c: hostile rows, each failing exactly one conjunct ---------------------------------- + +void TestS4HostileRows() { + size_t rows = 0; + superslm_attention_cases::ForEachSoftmaxCase([&](const SmCase& c) { + if (std::strncmp(c.label, "2.S4", 4) != 0) return; + ++rows; + const size_t w = c.scores.size(); + const SoftmaxGuard g = TestSoftmaxGuard(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c); + const bool witness = std::strstr(c.label, "witness") != nullptr; + // RunSm (in TestS4Grid) already asserted bool, output and fallback +1. + if (witness) + CHECK_MSG(g.Failing() >= 2, "%s: fails %d conjuncts, want more than one", c.label, g.Failing()); + else + CHECK_MSG(g.Failing() == 1, "%s: fails %d conjuncts of the guard copy, want exactly 1", c.label, g.Failing()); + // Which rows the v1.9.0 body refuses (the bool is the signal there) and which it accepts (the + // counter is then the only signal): the constant rows refuse; the width and score rows are + // output-equivalent (§8 2.S4). + std::vector want(w); + const bool ref_ok = RefSoftmax(c.scores.data(), w, c.q_ln2, c.q_b, c.q_c, want.data()); + const bool output_equivalent = !g.width_ok || !g.scores_ok; + CHECK_MSG(ref_ok == output_equivalent, "%s: v1.9.0 bool %d, want %d", c.label, ref_ok ? 1 : 0, + output_equivalent ? 1 : 0); + }); + CHECK_MSG(rows == 10, "2.S4: %zu hostile rows in the set, want 10", rows); + // The witness restated in the set is the fixture's own. + const auto& w = superslm_test::kSoftmaxRowOffRatioWitness; + bool same = false; + superslm_attention_cases::ForEachSoftmaxCase([&](const SmCase& c) { + if (std::strstr(c.label, "witness") == nullptr) return; + same = c.q_ln2 == w.q_ln2 && c.q_b == w.q_b && c.q_c == w.q_c && c.scores.size() == w.width && + c.scores[0] == w.scores[0] && c.scores[1] == w.scores[1] && c.scores[2] == w.scores[2]; + }); + CHECK_MSG(same, "2.S4: the set's off-ratio witness equals kSoftmaxRowOffRatioWitness"); +} + +// ---- the replica's own premise: equal to the v1.9.0 body on realistic rows inside the guard ------------ + +void TestS4ReplicaPremise() { + superslm_attention_cases::Rng rng(0x5334524550524D53ULL); // "S4REPRMS" + const auto triples = superslm_attention_cases::SmRealisticConstants(rng, 4); + size_t rows = 0, bad = 0, z_down = 0; + for (int it = 0; it < 4000; ++it) { + const auto& t = triples[static_cast(rng.Next() % triples.size())]; + const size_t w = static_cast(rng.InRange(1, 200)); + const std::vector s = superslm_attention_cases::SmScoreRow(w, it % 4, t[0], rng); + std::vector a(w), b(w); + SoftmaxCorrections k; + if (!SoftmaxEstimateReplica(s.data(), w, t[0], t[1], t[2], a.data(), &k)) continue; + ++rows; + z_down += static_cast(k.z_down); + if (!RefSoftmax(s.data(), w, t[0], t[1], t[2], b.data()) || a != b) ++bad; + } + CHECK_MSG(rows > 1000 && bad == 0 && z_down == 0, + "4.S4 replica premise: %zu rows inside the guard, %zu differ from v1.9.0, z down fired %zu times", rows, + bad, z_down); +} + +// ---- 6.3: the S4 golden pin (the v1.9.0 tag's hash over the softmax input set) ------------------------ + +void TestS4GoldenPin() { + superslm::Sha256 h; + uint64_t values = 0; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++values; + }; + superslm_attention_cases::RunSoftmaxCases(emit); + uint8_t digest[32]; + h.Final(digest); + const std::string hex = superslm::ToHex(digest); + std::printf("attn-rowsites S4 golden hash: %s (%llu values)\n", hex.c_str(), + static_cast(values)); + CHECK_MSG(hex == std::string(superslm_test::kAttnRowsiteS4GoldenHash) && + values == superslm_test::kAttnRowsiteS4GoldenValues, + "6.3 S4 golden: %s over %llu values, pin %s over %llu (v1.9.0 tag)", hex.c_str(), + static_cast(values), superslm_test::kAttnRowsiteS4GoldenHash, + static_cast(superslm_test::kAttnRowsiteS4GoldenValues)); +} + +// ==== Slice S5: the Q31 score in three 16-bit pieces, per head (§4.5, §5.5) =========================== +// +// Every S5 kernel cell calls QkQ31ScoreRow directly and asserts per call (§8 path rule): every score equals +// the reference, and the q31_row path counters moved by exactly the delta a test-side copy of the guard names +// (head_dim <= 512 and every ratio in [0, 2^32), written from §4.5), on the selected kernel's own tier only. +// The reference is the in-tree scalar reference QkQ31ScoreScalarRef for in-contract ratios, and the SAME +// binary's per-key QkQ31Score for the out-of-contract rows of 2.S5 (§3.3: outside [0, 2^32) the v1.9.0 SIMD +// tiers already differ from the scalar reference, and the fallback is that per-key loop). Cell 11.1(c) drives +// the Q31 call sites of both layer loops on the widened QK-norm fixture. + +using superslm_attention_cases::Q31Case; + +// The test-side copy of the S5 guard (§4.5), one flag per conjunct so a 2.S5 row can be shown to fail +// exactly one. +struct Q31Guard { + bool hd_ok = true, ratio_ge0 = true, ratio_lt = true; + bool Fast() const { return hd_ok && ratio_ge0 && ratio_lt; } + int Failing() const { return !hd_ok + !ratio_ge0 + !ratio_lt; } +}; + +Q31Guard TestQ31Guard(const int64_t* ratio, size_t head_dim) { + Q31Guard g; + g.hd_ok = head_dim <= 512; + for (size_t d = 0; d < head_dim; ++d) { + if (ratio[d] < 0) g.ratio_ge0 = false; + if (ratio[d] > INT64_C(0xFFFFFFFF)) g.ratio_lt = false; + } + return g; +} + +struct Q3Counters { + long long fast2 = 0, fb2 = 0, fast5 = 0, fb5 = 0; +}; +#if defined(SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT) && SUPERSLM_MATMUL_HAVE_SIMD_X64 +constexpr bool kHaveQ3Counters = true; +Q3Counters ReadQ3Counters() { + return Q3Counters{superslm_test::g_q31_row_fast_avx2.load(), superslm_test::g_q31_row_fallback_avx2.load(), + superslm_test::g_q31_row_fast_avx512.load(), superslm_test::g_q31_row_fallback_avx512.load()}; +} +#else +constexpr bool kHaveQ3Counters = false; +Q3Counters ReadQ3Counters() { return Q3Counters{}; } +#endif + +Q3Counters Q3Delta(const Q3Counters& a, const Q3Counters& b) { + return Q3Counters{b.fast2 - a.fast2, b.fb2 - a.fb2, b.fast5 - a.fast5, b.fb5 - a.fb5}; +} + +// `fast` fast and `fallback` fallback calls on this build (11.1(b)). +Q3Counters ExpectedQ3Delta(long long fast, long long fallback) { + Q3Counters w; + switch (ExpectedSitesKernel()) { + case SitesKernel::kAvx2: w.fast2 = fast; w.fb2 = fallback; break; + case SitesKernel::kAvx512: w.fast5 = fast; w.fb5 = fallback; break; + case SitesKernel::kShipped: break; + } + return w; +} + +bool CheckQ3Delta(const char* label, const Q3Counters& d, const Q3Counters& want) { + if (!kHaveQ3Counters) return true; + const bool ok = d.fast2 == want.fast2 && d.fb2 == want.fb2 && d.fast5 == want.fast5 && d.fb5 == want.fb5; + CHECK_MSG(ok, "%s: q31_row_fast_avx2 +%lld q31_row_fallback_avx2 +%lld q31_row_fast_avx512 +%lld " + "q31_row_fallback_avx512 +%lld; want +%lld/+%lld/+%lld/+%lld (kernel: %s)", + label, d.fast2, d.fb2, d.fast5, d.fb5, want.fast2, want.fb2, want.fast5, want.fb5, + SitesKernelName(ExpectedSitesKernel())); + return ok; +} + +enum class Q31Ref { kScalarRef, kSameBinaryPerKey }; + +// One call against its reference, with its path assertion. The output is an exact-size heap vector, +// poisoned, so the hosted ASan leg sees any write past the row. Returns whether the guard copy expected +// the fast path. +bool RunQ31(const Q31Case& c, Q31Ref ref) { + std::vector want(c.width); + for (size_t j = 0; j < c.width; ++j) { + const int8_t* k = c.keys.data() + j * c.head_dim; + want[j] = ref == Q31Ref::kScalarRef + ? superslm::QkQ31ScoreScalarRef(c.q.data(), k, c.ratio.data(), c.head_dim) + : superslm::QkQ31Score(c.q.data(), k, c.ratio.data(), c.head_dim); + } + std::vector out(c.width, superslm_attention_cases::kQ31Poison); + const Q31Guard g = TestQ31Guard(c.ratio.data(), c.head_dim); + const Q3Counters c0 = ReadQ3Counters(); + superslm::QkQ31ScoreRow(c.q.data(), c.keys.data(), c.ratio.data(), c.head_dim, c.width, out.data()); + const Q3Counters c1 = ReadQ3Counters(); + size_t bad = 0, first = c.width; + for (size_t j = 0; j < c.width; ++j) + if (out[j] != want[j]) { + if (first == c.width) first = j; + ++bad; + } + CHECK_MSG(bad == 0, "%s (head_dim %zu, width %zu): %zu of %zu scores differ from %s, first at key %zu (%lld vs %lld)", + c.label, c.head_dim, c.width, bad, c.width, + ref == Q31Ref::kScalarRef ? "QkQ31ScoreScalarRef" : "this binary's per-key QkQ31Score", first, + first < c.width ? static_cast(out[first]) : 0LL, + first < c.width ? static_cast(want[first]) : 0LL); + char full[200]; + std::snprintf(full, sizeof full, "%s (head_dim %zu, width %zu, guard copy: %s)", c.label, c.head_dim, c.width, + g.Fast() ? "fast" : "fallback"); + CheckQ3Delta(full, Q3Delta(c0, c1), ExpectedQ3Delta(g.Fast() ? 1 : 0, g.Fast() ? 0 : 1)); + return g.Fast(); +} + +// §5.5's limbs, restated: w = a2 * 2^30 + a1 * 2^15 + a0, a0 = w & 0x7FFF, a1 = (w >> 15) & 0x7FFF, +// a2 = w >> 30 (arithmetic). Returns a key's three limb sums in int64 (no wrap), for the premise cells. +void LimbSums(const Q31Case& c, size_t key, int64_t sums[3]) { + sums[0] = sums[1] = sums[2] = 0; + for (size_t d = 0; d < c.head_dim; ++d) { + const int64_t w = static_cast(c.q[d]) * c.ratio[d]; + const int64_t k = c.keys[key * c.head_dim + d]; + sums[0] += k * (w & 0x7FFF); + sums[1] += k * ((w >> 15) & 0x7FFF); + sums[2] += k * (w >> 30); + } +} + +// ---- 4.S5, 6.1, 7.S5a, 7.S5b, 7.S5c: the shape grid, the margin corners and the ties -------------------- + +void TestS5Grid() { + size_t rows = 0, fast = 0, margin = 0, ties = 0; + superslm_attention_cases::ForEachQ31Case([&](const Q31Case& c) { + ++rows; + const bool f = RunQ31(c, Q31Ref::kScalarRef); + fast += f ? 1 : 0; + CHECK_MSG(f == (c.head_dim <= 512), "%s (head_dim %zu): guard copy says %s, want fast exactly when head_dim " + "<= 512 (every ratio of the set is in [1, 2^31])", c.label, c.head_dim, f ? "fast" : "fallback"); + if (std::strncmp(c.label, "7.S5b", 5) == 0) { + // The premise: every limb sum of every key is the margin row §5.5 names. + ++margin; + int64_t s[3]; + LimbSums(c, 0, s); + const int64_t per = static_cast(c.keys[0]) * 32767 * static_cast(c.head_dim); + CHECK_MSG(s[0] == per && s[1] == per, "%s head_dim %zu: limb sums %lld, %lld, want %lld", c.label, + c.head_dim, static_cast(s[0]), static_cast(s[1]), + static_cast(per)); + if (c.head_dim == 512 && c.keys[0] == -128) + CHECK_MSG(per == INT64_C(-2147418112) && per - INT32_MIN == 65536, + "7.S5b: the corner's limb sum %lld is not 65,536 above int32's minimum", static_cast(per)); + if (c.head_dim == 516 && c.keys[0] == -128) + CHECK_MSG(per < INT32_MIN, "7.S5b: at head_dim 516 the limb sum %lld should pass int32's minimum", + static_cast(per)); + } + if (std::strncmp(c.label, "7.S5c", 5) == 0) { + for (size_t j = 0; j < c.width; ++j) { + int64_t total = 0; + for (size_t d = 0; d < c.head_dim; ++d) + total += static_cast(c.q[d]) * c.keys[j * c.head_dim + d] * c.ratio[d]; + const int64_t r = total & ((INT64_C(1) << 31) - 1); + if (r == (INT64_C(1) << 30)) ++ties; + } + } + }); + CHECK_MSG(rows == 10 * 8 * 3 + 16 + 2, "4.S5: %zu rows in the set, want %d", rows, 10 * 8 * 3 + 16 + 2); + CHECK_MSG(fast == 8 * 8 * 3 + 8 + 2, "4.S5: %zu rows on the fast side of the guard copy, want %d", fast, + 8 * 8 * 3 + 8 + 2); + CHECK_MSG(margin == 16, "7.S5b: %zu margin rows, want 16", margin); + CHECK_MSG(ties == 16, "7.S5c: %zu keys whose total is a rounding tie (+-2^30 mod 2^31), want 16", ties); +} + +// ---- 7.S5d and the guard's inside corners: ratios 0, 2^31 and 2^32 - 1 run fast -------------------------- + +void TestS5GuardInsideRows() { + superslm_attention_cases::Rng rng(0x5335494E53494445ULL); // "S5INSIDE" + for (size_t hd : {size_t{64}, size_t{128}, size_t{512}}) + for (size_t w : {size_t{9}, size_t{17}}) + for (int64_t r : {INT64_C(0), INT64_C(1) << 31, INT64_C(0xFFFFFFFF)}) { + Q31Case c = superslm_attention_cases::MakeQ31GridCase(hd, w, 1, rng); + c.label = "7.S5d inside corner"; + for (size_t d = 0; d < hd; d += 2) c.ratio[d] = r; // every other channel at the corner + const bool f = RunQ31(c, Q31Ref::kScalarRef); + CHECK_MSG(f, "7.S5d: ratio %lld at head_dim %zu should be inside the guard copy", static_cast(r), + hd); + } + // Channel tails inside the guard: head_dim not a multiple of 4 (the last quad is padded in both the limbs and + // the key pack) and not a multiple of 16 (the pack's scalar tail). 4.S5's in-guard head_dims are all multiples + // of 4, so the pad sides ran on no row (found by the coverage replica, plan §3.4 step 1). Not in the golden set. + size_t tails = 0; + for (size_t hd : {size_t{1}, size_t{2}, size_t{3}, size_t{5}, size_t{63}, size_t{66}, size_t{127}, size_t{130}, + size_t{509}, size_t{511}}) + for (size_t w : {size_t{1}, size_t{9}, size_t{17}}) + for (int kind = 0; kind < 3; ++kind) { + Q31Case c = superslm_attention_cases::MakeQ31GridCase(hd, w, kind, rng); + c.label = "4.S5 channel tail"; + tails += RunQ31(c, Q31Ref::kScalarRef) ? 1 : 0; + } + CHECK_MSG(tails == 90, "4.S5 channel tail: %zu of 90 rows inside the guard copy, want all", tails); + // Width 0: nothing is written; the call still counts once on its tier (§3.6: once per call). + Q31Case z = superslm_attention_cases::MakeQ31GridCase(64, 1, 0, rng); + z.label = "4.S5 width 0"; + z.width = 0; + z.keys.clear(); + RunQ31(z, Q31Ref::kScalarRef); +} + +// ---- 2.S5: hostile rows, each failing exactly one conjunct ----------------------------------------------- + +void TestS5HostileRows() { + superslm_attention_cases::Rng rng(0x3253354853544C45ULL); // "2S5HSTLE" + size_t rows = 0; + // Ratio -1 and 2^32 break a kernel that forms w from 32-bit pieces; 2^33 and 2^38 are outside the guard but + // inside a 64-bit formation's int16 range (§5.5); 2^48 breaks every formation while 128 * 128 * 2^48 = 2^62 + // keeps the scalar reference's int64 sum defined. INT64_MAX is not a row (§5.5). Each at the first, a middle + // and the last channel (the last is the SIMD tiers' scalar tail at head_dim 63). + const int64_t hostile[] = {INT64_C(-1), INT64_C(1) << 32, INT64_C(1) << 33, INT64_C(1) << 38, INT64_C(1) << 48}; + for (size_t hd : {size_t{63}, size_t{64}}) + for (int64_t r : hostile) + for (size_t at : {size_t{0}, hd / 2, hd - 1}) { + Q31Case c = superslm_attention_cases::MakeQ31GridCase(hd, 17, 0, rng); + c.label = "2.S5 ratio outside [0, 2^32)"; + c.ratio[at] = r; + const Q31Guard g = TestQ31Guard(c.ratio.data(), c.head_dim); + CHECK_MSG(g.Failing() == 1 && (r < 0 ? !g.ratio_ge0 : !g.ratio_lt), + "2.S5 ratio %lld: fails %d conjuncts of the guard copy, want exactly the ratio one", + static_cast(r), g.Failing()); + RunQ31(c, Q31Ref::kSameBinaryPerKey); + ++rows; + } + for (size_t w : {size_t{1}, size_t{17}}) { + Q31Case c = superslm_attention_cases::MakeQ31GridCase(513, w, 0, rng); + c.label = "2.S5 head_dim 513"; + const Q31Guard g = TestQ31Guard(c.ratio.data(), c.head_dim); + CHECK_MSG(g.Failing() == 1 && !g.hd_ok, "2.S5 head_dim 513: fails %d conjuncts, want exactly head_dim", + g.Failing()); + RunQ31(c, Q31Ref::kSameBinaryPerKey); + ++rows; + } + CHECK_MSG(rows == 2 * 5 * 3 + 2, "2.S5: %zu hostile rows, want %d", rows, 2 * 5 * 3 + 2); +} + +// ---- 6.3: the S5 golden pin (the v1.9.0 tag's hashes over the Q31 set and the QK-norm fixture) ---------- + +template +std::string HashRun(Run run, uint64_t* values) { + superslm::Sha256 h; + *values = 0; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++*values; + }; + run(emit); + uint8_t digest[32]; + h.Final(digest); + return superslm::ToHex(digest); +} + +void TestS5GoldenPin() { + uint64_t values = 0; + const std::string hex = HashRun( + [](auto& emit) { superslm_attention_cases::RunQ31Cases(emit, superslm::QkQ31ScoreRow); }, &values); + std::printf("attn-rowsites S5 golden hash: %s (%llu values)\n", hex.c_str(), static_cast(values)); + CHECK_MSG(hex == std::string(superslm_test::kAttnRowsiteS5GoldenHash) && + values == superslm_test::kAttnRowsiteS5GoldenValues, + "6.3 S5 golden: %s over %llu values, pin %s over %llu (v1.9.0 tag)", hex.c_str(), + static_cast(values), superslm_test::kAttnRowsiteS5GoldenHash, + static_cast(superslm_test::kAttnRowsiteS5GoldenValues)); +} + +// ---- 11.1(c): the QK-norm attention fixture, the Q31 call sites of both loops ---------------------------- + +// Run (iii)'s sink: every row the decode loop hands it, classified by the test-side guard copies (§5.4's +// softmax guard, §5.2's int16 condition), so the softmax and prob·V counters are exact too. +struct FixtureRows { + long long observes = 0, sm_fast = 0, sm_fallback = 0, pv_fast = 0, pv_fallback = 0; +}; + +void ClassifyRow(void* ctx, uint32_t, int64_t, size_t, const int64_t* scores, size_t width, int64_t q_ln2, int64_t q_b, + int64_t q_c, const int64_t* probs, const int64_t*, const int64_t*, size_t head_dim) { + auto* r = static_cast(ctx); + ++r->observes; + (TestSoftmaxGuard(scores, width, q_ln2, q_b, q_c).Fast() ? r->sm_fast : r->sm_fallback) += 1; + (TestPvGuard(probs, width, head_dim).Fast() ? r->pv_fast : r->pv_fallback) += 1; +} + +struct AllCounters { + RowCounters row; + Q3Counters q3; + SmCounters sm; + PvCounters pv; + RqCounters rq; +}; + +AllCounters ReadAll() { + return AllCounters{ReadRowCounters(), ReadQ3Counters(), ReadSmCounters(), ReadPvCounters(), ReadRqCounters()}; +} + +// The per-position structural deltas of §8 11.1(c)'s table, "S5 landed, S6 not" column, over `n` +// positions: q31_row fast L·H = 4 per position and fallback 0; rowtable_norm_skipped 2L + L·H = 6 (two +// hidden norms at 256, four q_norm at 64); silu 1 and landing 2 skipped; every taken 0. +void CheckFixtureStructural(const char* label, const AllCounters& a, const AllCounters& b, long long n) { + CheckQ3Delta(label, Q3Delta(a.q3, b.q3), ExpectedQ3Delta(4 * n, 0)); + if (!kHaveRowCounters) return; + const RowCounters d = Delta(a.row, b.row); + const long long want_skipped[3] = {6 * n, n, 2 * n}; + for (int s = 0; s < 3; ++s) + CHECK_MSG(d.taken[s] == 0 && d.skipped[s] == want_skipped[s], + "%s: rowtable_%s taken +%lld skipped +%lld, want +0/+%lld", label, kRowSiteNames[s], d.taken[s], + d.skipped[s], want_skipped[s]); +} + +void TestS5QkNormFixture() { + using superslm::SslmForwardStatus; + namespace fx = superslm_qk_fixture; + fx::QkAttentionFixture f; + CHECK_MSG(f.loaded, "11.1(c): the fixture's own minimal artifact failed to load: %s", f.load_error.c_str()); + if (!f.loaded) return; + const long long P = static_cast(fx::kPositions); + + // Run (iii) first: the decode loop with the classifying sink. Its rows give the data terms. + FixtureRows rows; + superslm::AttentionCaptureSink sink{&rows, &ClassifyRow}; + AllCounters at = ReadAll(); + const AllCounters start3 = at; + uint64_t n3 = 0; + SslmForwardStatus st3 = SslmForwardStatus::Ok; + const std::string h3 = HashRun( + [&](auto& emit) { + st3 = fx::RunFixtureDecode(f, emit, &sink, [&](size_t p) { + const AllCounters now = ReadAll(); + char label[64]; + std::snprintf(label, sizeof label, "11.1(c) run (iii) position %zu", p); + CheckFixtureStructural(label, at, now, 1); + at = now; + }); + }, + &n3); + const AllCounters end3 = ReadAll(); + CHECK_MSG(st3 == SslmForwardStatus::Ok, "11.1(c) run (iii): status %s", SslmForwardStatusName(st3)); + // The premise (checked on the base before any kernel landed): one observe per (position, head), every + // softmax row inside §5.4's guard; the only rows failing the int16 condition are position 0's width-1 rows. + CHECK_MSG(rows.observes == 4 * P && rows.sm_fallback == 0 && rows.pv_fallback == 4, + "11.1(c) premise: %lld rows observed (want %lld), %lld outside the softmax guard (want 0), %lld failing " + "the int16 condition (want 4)", rows.observes, 4 * P, rows.sm_fallback, rows.pv_fallback); + + // Run (i): the decode loop, no sink. + at = ReadAll(); + const AllCounters start1 = at; + uint64_t n1 = 0; + SslmForwardStatus st1 = SslmForwardStatus::Ok; + const std::string h1 = HashRun( + [&](auto& emit) { + st1 = fx::RunFixtureDecode(f, emit, nullptr, [&](size_t p) { + const AllCounters now = ReadAll(); + char label[64]; + std::snprintf(label, sizeof label, "11.1(c) run (i) position %zu", p); + CheckFixtureStructural(label, at, now, 1); + at = now; + }); + }, + &n1); + const AllCounters end1 = ReadAll(); + CHECK_MSG(st1 == SslmForwardStatus::Ok, "11.1(c) run (i): status %s", SslmForwardStatusName(st1)); + + // Run (ii): the chunk loop over all 24 positions, with a counting sink installed on the layer (t2701's + // chunk-mode configuration): the chunk loop never observes. + FixtureRows chunk_rows; + superslm::AttentionCaptureSink chunk_sink{&chunk_rows, &ClassifyRow}; + const AllCounters start2 = ReadAll(); + uint64_t n2 = 0; + SslmForwardStatus st2 = SslmForwardStatus::Ok; + const std::string h2 = HashRun([&](auto& emit) { st2 = fx::RunFixtureChunk(f, emit, &chunk_sink); }, &n2); + const AllCounters end2 = ReadAll(); + CHECK_MSG(st2 == SslmForwardStatus::Ok, "11.1(c) run (ii): status %s", SslmForwardStatusName(st2)); + CHECK_MSG(chunk_rows.observes == 0, "11.1(c) run (ii): the chunk loop made %lld observe calls, want 0", + chunk_rows.observes); + CheckFixtureStructural("11.1(c) run (ii), 24 positions", start2, end2, P); + + // Every run hashes to the golden pin's fixture hash (6.3, from the v1.9.0 tag). + const struct { + const char* name; + const std::string& hex; + uint64_t n; + const AllCounters &a, &b; + } runs[] = {{"(i)", h1, n1, start1, end1}, {"(ii)", h2, n2, start2, end2}, {"(iii)", h3, n3, start3, end3}}; + for (const auto& r : runs) { + CHECK_MSG(r.hex == std::string(superslm_test::kAttnRowsiteS5FixtureGoldenHash) && + r.n == superslm_test::kAttnRowsiteS5FixtureGoldenValues, + "11.1(c)/6.3 run %s: fixture hash %s over %llu values, pin %s over %llu (v1.9.0 tag)", r.name, + r.hex.c_str(), static_cast(r.n), superslm_test::kAttnRowsiteS5FixtureGoldenHash, + static_cast(superslm_test::kAttnRowsiteS5FixtureGoldenValues)); + // Softmax and prob·V: exact, from run (iii)'s classified rows, on every run. + char label[64]; + std::snprintf(label, sizeof label, "11.1(c) run %s softmax", r.name); + CheckSmDelta(label, SmDelta(r.a.sm, r.b.sm), ExpectedSmDelta(rows.sm_fast, rows.sm_fallback)); + std::snprintf(label, sizeof label, "11.1(c) run %s prob-V", r.name); + CheckPvDelta(label, PvDelta(r.a.pv, r.b.pv), ExpectedPvDelta(rows.pv_fast, rows.pv_fallback)); + } + std::printf("attn-rowsites 11.1(c) fixture hash: %s (%llu values); %lld rows, softmax %lld/%lld, prob-V %lld/%lld " + "(fast/fallback by the guard copies)\n", h1.c_str(), static_cast(n1), rows.observes, + rows.sm_fast, rows.sm_fallback, rows.pv_fast, rows.pv_fallback); +} + +// ---- PreflightScanWscFolds with no artifact ----------------------------------------------------- +// +// 11.1(d) is the only other caller of PreflightScanWscFolds in this binary, and it runs only when +// SUPERSLM_ATTN_ROWSITES_ARTIFACT names an uncommitted artifact. Without this cell the function is +// linked into superslm_tests and never run, which leaves every one of its branches uncovered. The +// view here is built from a committed WSC1 manifest, so the scan runs on every build. + +// Redirects stderr to a temp file for the life of the object and restores it on scope exit. +struct StderrCapture { + int saved_fd = -1; + std::string path; + bool active = false; + StderrCapture() { + std::error_code ec; + const std::filesystem::path dir = std::filesystem::temp_directory_path(ec); + if (ec) return; +#ifdef _WIN32 + const long pid = static_cast(_getpid()); +#else + const long pid = static_cast(getpid()); +#endif + path = (dir / ("sslm_preflight_capture_" + std::to_string(pid) + ".txt")).string(); + std::fflush(stderr); +#ifdef _WIN32 + saved_fd = _dup(_fileno(stderr)); +#else + saved_fd = dup(fileno(stderr)); +#endif + if (saved_fd == -1) return; + if (!std::freopen(path.c_str(), "w", stderr)) { + Restore(); + return; + } + active = true; + } + void Restore() { + if (saved_fd == -1) return; + std::fflush(stderr); +#ifdef _WIN32 + _dup2(saved_fd, _fileno(stderr)); + _close(saved_fd); +#else + dup2(saved_fd, fileno(stderr)); + close(saved_fd); +#endif + saved_fd = -1; + } + std::string Read() { + std::fflush(stderr); + std::ifstream f(path, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); + } + ~StderrCapture() { + Restore(); + if (!path.empty()) std::remove(path.c_str()); + } +}; + +std::string RunPreflight(const superslm::SslmModelView& view) { + StderrCapture cap; + superslm_marshal::PreflightScanWscFolds(view); + return cap.active ? cap.Read() : std::string(); +} + +void TestPreflightScanWscFolds() { + using namespace superslm; + using superslm_test::BuildManifest; + using superslm_test::MakeManifestSectionView; + // Three layers. layer0 carries one degenerate (one-row) fold tensor and no other projection; + // layer1 carries two per-channel tensors, the larger first, so the running maximum is both + // raised and not raised; layer2 carries none. Expected: 1 of 3 layers affected, worst case 4. + const auto manifest = BuildManifest(kWeightScalesMagic, /*element_size=*/4, + {{"layer0.q_proj", {1, 3}}, {"layer1.k_proj", {4, 3}}, {"layer1.v_proj", {2, 3}}}); + const SslmSectionView section = MakeManifestSectionView(SslmSectionType::WeightScales, SslmDtype::Int32, manifest.bytes); + SslmModelView view; + std::string err; + const SslmModelStatus st = SslmTensorManifest::Parse(section, view.weight_scales, &err); + CHECK_MSG(st == SslmModelStatus::Ok, "preflight: WSC1 fixture did not parse: %s", err.c_str()); + if (st != SslmModelStatus::Ok) return; + view.has_weight_scales = true; + + view.config.num_hidden_layers = 3; + std::string out = RunPreflight(view); + CHECK_MSG(out.find("preflight: 1/3 layers carry a non-degenerate") != std::string::npos && + out.find("worst case 4 rows") != std::string::npos, + "preflight over three layers: got \"%s\"", out.c_str()); + + // Only layer0 in range: its one-row tensor raises the maximum to 1 but does not mark the layer. + view.config.num_hidden_layers = 1; + out = RunPreflight(view); + CHECK_MSG(out.find("preflight: 0/1 layers carry a non-degenerate") != std::string::npos && + out.find("worst case 1 rows") != std::string::npos, + "preflight over layer0 only: got \"%s\"", out.c_str()); + + // No layers: the loop never runs. + view.config.num_hidden_layers = 0; + out = RunPreflight(view); + CHECK_MSG(out.find("preflight: 0/0 layers carry a non-degenerate") != std::string::npos && + out.find("worst case 0 rows") != std::string::npos, + "preflight over no layers: got \"%s\"", out.c_str()); +} + +// ---- 11.1(d): the 0.5B-width 1-layer artifact through the two layer loops ----------------------- + +struct TraceCount { + uint64_t chain = 0; + std::map by_site; +}; + +void CountTrace(const superslm::SslmChainTraceRecord* chain, const superslm::SslmKvLandingTraceRecord*, void* user) { + if (!chain) return; + auto* t = static_cast(user); + ++t->chain; + std::string s(chain->site); + if (s.rfind("layer", 0) == 0) s = s.substr(s.find('.') + 1); + ++t->by_site[s]; +} + +bool ReadWholeFile(const char* path, std::vector& out) { + std::ifstream f(path, std::ios::binary); + if (!f) return false; + out.assign(std::istreambuf_iterator(f), std::istreambuf_iterator()); + return !out.empty(); +} + +// The drive is the plan's (§8 11.1(d)): no ABI, the model's trace hook installed, pinned tokens +// t_i = (37·i + 11) mod 256, a 128-token chunk through RunLayerLoopChunkBatched, then 32 single +// tokens through RunLayerLoop, no final norm and no logits. The artifact is not committed (about +// 16 MB); the cell runs when SUPERSLM_ATTN_ROWSITES_ARTIFACT names it and says SKIPPED otherwise. +void TestS1LayerLoopWindows() { + const char* path = std::getenv("SUPERSLM_ATTN_ROWSITES_ARTIFACT"); + if (path == nullptr || *path == '\0') { + std::printf("attn-rowsites 11.1(d): SKIPPED (set SUPERSLM_ATTN_ROWSITES_ARTIFACT to the 0.5B-width 1-layer " + "synthetic artifact)\n"); + return; + } + using namespace superslm; + std::vector bytes; + CHECK_MSG(ReadWholeFile(path, bytes), "11.1(d): cannot read %s", path); + if (bytes.empty()) return; + SslmModelView model; + std::string error; + const SslmModelStatus ls = SslmModel::Load(bytes.data(), bytes.size(), model, &error); + CHECK_MSG(ls == SslmModelStatus::Ok, "11.1(d): load failed: %s", error.c_str()); + if (ls != SslmModelStatus::Ok) return; + const uint32_t L = model.config.num_hidden_layers; + const size_t hidden = model.config.hidden_size, head_dim = model.config.head_dim, + kv = model.config.num_key_value_heads, H = model.config.num_attention_heads; + const int64_t cap = model.config.context_cap; + CHECK_MSG(L == 1 && H == 14 && kv == 2 && head_dim == 64 && hidden == 896 && model.config.intermediate_size == 4864, + "11.1(d): the artifact is not the 0.5B-width 1-layer geometry"); + superslm_marshal::PreflightScanWscFolds(model); + std::vector backing(L); + std::vector layers(L); + for (uint32_t l = 0; l < L; ++l) + CHECK_MSG(superslm_marshal::MarshalLayer(model, l, static_cast(H), static_cast(kv), + backing[l], layers[l], &error), + "11.1(d): marshal layer %u: %s", l, error.c_str()); + const int8_t* embed = reinterpret_cast(model.weights.Tensor("embed")->data); + bool ok = true; + const CarriedScale embed_scale = superslm_marshal::ReadCarriedScale(model.composition_constants, "embed", &ok); + std::vector workspace(size_t(L) * size_t(cap) * kv * head_dim * 2); + TraceCount trace; + SslmSetTraceHook(model.trace_hook, &CountTrace, &trace); + const OptionGKLandingMode k_mode = + model.option_g_fused_k_landing ? OptionGKLandingMode::kFused : OptionGKLandingMode::kLegacy; + const auto tok = [](size_t i) { return static_cast((i * 37 + 11) % 256); }; + const size_t T = 128, D = 32; + + // The prefill window. + const RowCounters p0 = ReadRowCounters(); + const PvCounters v0 = ReadPvCounters(); + const RqCounters q0 = ReadRqCounters(); + const SmCounters m0 = ReadSmCounters(); + const Q3Counters x0 = ReadQ3Counters(); + const TraceCount tp0 = trace; + std::vector chunk(T * hidden); + std::vector scales(T); + for (size_t i = 0; i < T; ++i) + CHECK_MSG(EmbedEntry(tok(i), static_cast(model.config.vocab_size), embed, hidden, embed_scale, + chunk.data() + i * hidden, &scales[i], "embed", i, &model.trace_hook) == + SslmForwardStatus::Ok, + "11.1(d): embed %zu", i); + SequenceLayerState seq; + std::vector hc(hidden); + seq.hidden_codes = hc.data(); + SslmForwardStatus st = RunLayerLoopChunkBatched( + chunk.data(), scales.data(), T, layers.data(), L, hidden, head_dim, kv, model.config.intermediate_size, cap, 0, + model.rope_tables, workspace.data(), workspace.size(), model.option_g_fused_k_landing, &seq.kv_saturation_count, + {}, &model.trace_hook, H * head_dim); + CHECK_MSG(st == SslmForwardStatus::Ok, "11.1(d): chunk loop status %s", SslmForwardStatusName(st)); + seq.context_length = static_cast(T); + const RowCounters p1 = ReadRowCounters(); + const PvCounters v1 = ReadPvCounters(); + const RqCounters q1 = ReadRqCounters(); + const SmCounters m1 = ReadSmCounters(); + const Q3Counters x1 = ReadQ3Counters(); + const TraceCount tp1 = trace; + + // The decode window. + for (size_t i = 0; i < D && st == SslmForwardStatus::Ok; ++i) { + CarriedScale sc{}; + CHECK_MSG(EmbedEntry(tok(T + i), static_cast(model.config.vocab_size), embed, hidden, embed_scale, + hc.data(), &sc, "embed", T + i, &model.trace_hook) == SslmForwardStatus::Ok, + "11.1(d): embed %zu", T + i); + seq.hidden_scale = sc; + seq.layer_index = 0; + st = RunLayerLoop(seq, layers.data(), L, L, hidden, head_dim, kv, model.config.intermediate_size, cap, + model.rope_tables, workspace.data(), workspace.size(), k_mode, {}, T + i, &model.trace_hook, + H * head_dim); + CHECK_MSG(st == SslmForwardStatus::Ok, "11.1(d): decode step %zu status %s", i, SslmForwardStatusName(st)); + } + CHECK_MSG(seq.context_length == static_cast(T + D), "11.1(d): context_length %lld, want %zu", + static_cast(seq.context_length), T + D); + const RowCounters p2 = ReadRowCounters(); + const PvCounters v2 = ReadPvCounters(); + const RqCounters q2 = ReadRqCounters(); + const SmCounters m2 = ReadSmCounters(); + const Q3Counters x2 = ReadQ3Counters(); + const TraceCount tp2 = trace; + SslmSetTraceHook(model.trace_hook, nullptr, nullptr); + + static const char* const kSites[] = {"embed", "q_proj.requant", "attn_ctx", "o_proj.requant", + "attn_norm", "attn_residual", "mlp_norm", "gate_proj.requant", + "up_proj.requant", "mlp_act", "down_proj.requant", "mlp_residual"}; + // The prob·V data terms (§8 11.1(d)): rows failing the int16 condition on this artifact and these + // pinned tokens, measured on the base (docs/attention-rowsites/s2/pv-data-terms.txt): 15 in the prefill + // window (position 0's 14 width-1 rows and position 2, head 13's {0, 32,768, 0}), 0 in decode. + // The softmax data term: rows outside §5.4's guard, measured on the base (the S3 head) the same way + // (docs/attention-rowsites/s4/softmax-data-terms.txt): 0 in both windows. + const struct { + const char* name; + RowCounters a, b; + PvCounters va, vb; + RqCounters qa, qb; + SmCounters ma, mb; + Q3Counters xa, xb; + TraceCount ta, tb; + size_t N; + long long pv_fallback, softmax_fallback; + } windows[] = {{"prefill", p0, p1, v0, v1, q0, q1, m0, m1, x0, x1, tp0, tp1, T, 15, 0}, + {"decode", p1, p2, v1, v2, q1, q2, m1, m2, x1, x2, tp1, tp2, D, 0, 0}}; + for (const auto& w : windows) { + // Structural terms (G26): the closed forms, every width here being >= 512. + const RowCounters d = Delta(w.a, w.b); + const long long N = static_cast(w.N), Ll = static_cast(L); + const uint64_t records = w.tb.chain - w.ta.chain; + CHECK_MSG(records == static_cast((11 * Ll + 1) * N), "11.1(d) %s: %llu chain trace records, want %lld", + w.name, static_cast(records), (11 * Ll + 1) * N); + for (const char* site : kSites) { + const auto after = w.tb.by_site.find(site); + const auto before = w.ta.by_site.find(site); + const uint64_t got = (after == w.tb.by_site.end() ? 0 : after->second) - + (before == w.ta.by_site.end() ? 0 : before->second); + CHECK_MSG(got == static_cast(N), "11.1(d) %s: %llu records named %s, want %lld", w.name, + static_cast(got), site, N); + } + // S2: one GemmProbQ15Accumulate call per (layer, head, token pass), L·H·N in all (structural), + // of which the data term falls back; on the active kernel's tier only (11.1(b)). + const long long calls = Ll * static_cast(H) * N; + char pv_label[64]; + std::snprintf(pv_label, sizeof pv_label, "11.1(d) %s prob-V", w.name); + CheckPvDelta(pv_label, PvDelta(w.va, w.vb), ExpectedPvDelta(calls - w.pv_fallback, w.pv_fallback)); + // S3: one row-leaf call per funnel call that passes its preflight, (11·L + 1)·N (structural: the + // same count as the chain trace records above), on the active kernel's tier only (11.1(b)); 0 on + // forced SSE2 and on an MSVC build's AVX-512 tier with its switch off, where the records still count. + std::snprintf(pv_label, sizeof pv_label, "11.1(d) %s requant_row", w.name); + CheckRqDelta(pv_label, RqDelta(w.qa, w.qb), ExpectedRqDelta((11 * Ll + 1) * N)); + // S4: one SoftmaxRowQ15 call per (layer, head, token pass), L·H·N (structural), of which the data term + // falls back; on the active kernel's tier only (11.1(b)). + std::snprintf(pv_label, sizeof pv_label, "11.1(d) %s softmax", w.name); + CheckSmDelta(pv_label, SmDelta(w.ma, w.mb), ExpectedSmDelta(calls - w.softmax_fallback, w.softmax_fallback)); + // S5: this artifact carries no QK-norm, so the plain score path runs and QkQ31ScoreRow is never called. + std::snprintf(pv_label, sizeof pv_label, "11.1(d) %s q31_row", w.name); + CheckQ3Delta(pv_label, Q3Delta(w.xa, w.xb), Q3Counters{}); + if (!kHaveRowCounters) continue; + const long long want_taken[3] = {2 * Ll * N, Ll * N, 2 * Ll * N}; + for (int s = 0; s < 3; ++s) { + CHECK_MSG(d.taken[s] == want_taken[s] && d.skipped[s] == 0, + "11.1(d) %s: rowtable_%s taken +%lld skipped +%lld, want +%lld/+0", w.name, kRowSiteNames[s], + d.taken[s], d.skipped[s], want_taken[s]); + } + } + std::printf("attn-rowsites 11.1(d): prefill and decode windows driven on %s\n", path); +} + +} // namespace + +void RunAttnRowsiteCells(int& checks, int& failures) { + GChecks = 0; + GFailures = 0; + if (!kHaveRowCounters) + std::printf("attn-rowsites: this build has no instrument seam; value checks only, path counters not read\n"); + TestS1SmallGateScalePremise(); + TestS1Grid(); + TestS1ConsecutiveRowsDifferentConstants(); + TestS1ConcurrentCalls(); + TestS1LandingFlag(); + TestS1GoldenPin(); + TestS2SitesSelector(); + TestS2Grid(); + TestS2HostileRows(); + TestS2GoldenPin(); + TestS3CornerPremise(); + TestS3Grid(); + TestS3FunnelCallSite(); + TestS3GoldenPin(); + TestS4ReplicaPremise(); + TestS4Grid(); + TestS4HostileRows(); + TestS4GoldenPin(); + TestS5Grid(); + TestS5GuardInsideRows(); + TestS5HostileRows(); + TestS5GoldenPin(); + TestS5QkNormFixture(); // 11.1(c): the Q31 call sites of both loops + TestPreflightScanWscFolds(); + TestS1LayerLoopWindows(); // 11.1(d): S1's, S2's, S3's, S4's and S5's rows over one drive + std::printf("attn-rowsites cells (plan slices S1, S2, S3, S4, S5): %d checks, %d failures\n", GChecks, GFailures); + checks += GChecks; + failures += GFailures; +} diff --git a/tests/test_main.cpp b/tests/test_main.cpp index 5ce3df02..4631537e 100644 --- a/tests/test_main.cpp +++ b/tests/test_main.cpp @@ -28267,6 +28267,8 @@ void RunSlm18xSaturationCensusCells(int& checks, int& failures); void RunSlm19xSchemaDampedGreedyCells(int& checks, int& failures); // Tiled-matmul plan slice 1: the tiled GEMM cells (tests/test_tiled_gemm.cpp). void RunTiledGemmCells(int& checks, int& failures); +// Attention and per-row sites plan: its slices' cells (tests/test_attn_rowsites.cpp). +void RunAttnRowsiteCells(int& checks, int& failures); #ifdef _WIN32 // resumes the Windows/D3D12 block closed above T-2572 @@ -30139,6 +30141,7 @@ int main(int argc, char** argv) { RunSlm18xSaturationCensusCells(GChecks, GFailures); RunSlm19xSchemaDampedGreedyCells(GChecks, GFailures); RunTiledGemmCells(GChecks, GFailures); + RunAttnRowsiteCells(GChecks, GFailures); TestKvStoreEarlyWriteSurvivesLateReadAcrossEightFurtherPositions(); TestRunLayerLoopContextAxisAndCapacityExhaustedFailFast(); TestRunLayerLoopColdPrefillAndIncrementalDecodeAgreeAtSamePosition(); diff --git a/tools/build_inspect.bat b/tools/build_inspect.bat index 99dba435..11116b90 100644 --- a/tools/build_inspect.bat +++ b/tools/build_inspect.bat @@ -7,11 +7,11 @@ call %VSDEVCMD% -arch=x64 -no_logo pushd %~dp0\.. if not exist out mkdir out cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude ^ - src\artifact.cpp src\sha256.cpp src\model.cpp src\tokenizer.cpp src\intmath.cpp src\proof_manifest.cpp tools\sslm_inspect.cpp ^ + src\artifact.cpp src\sha256.cpp src\model.cpp src\tokenizer.cpp src\intmath.cpp src\matmul.cpp src\proof_manifest.cpp tools\sslm_inspect.cpp ^ /Fo:out\ /Fe:out\sslm_inspect.exe if errorlevel 1 (popd & exit /b 1) cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude ^ - src\artifact.cpp src\sha256.cpp src\tokenizer.cpp src\intmath.cpp tools\tok_verify.cpp ^ + src\artifact.cpp src\sha256.cpp src\tokenizer.cpp src\intmath.cpp src\matmul.cpp tools\tok_verify.cpp ^ /Fo:out\ /Fe:out\tok_verify.exe set ec=%errorlevel% popd diff --git a/tools/ci/branch_coverage_allowlist.txt b/tools/ci/branch_coverage_allowlist.txt index 004a04a2..7b8a4145 100644 --- a/tools/ci/branch_coverage_allowlist.txt +++ b/tools/ci/branch_coverage_allowlist.txt @@ -31,3 +31,30 @@ # on one CPU (the CPUID/XGETBV bits it reads are a fixed property of the # running hardware) -- a genuine, permanent gap no union of same-machine # binaries closes (design §10 dimension 7 item (h)). +# Attention and per-row sites plan, slice S2 (prob-V on int16 multiply-add; plan §3.4, cell 11.6): +# src/matmul.cpp:821-901 -- ProbVBlockAvx512, ProbVTail16Avx512 and +# ProbVAccumulateIntoAvx512 (all branches) cannot execute on a runner +# without AVX-512, for the same reason as DotRowAvx512 above. +# src/matmul.cpp:923-924 -- ProbQ15AccumulateInto's AVX-512 arm: the same +# runner restriction (DispatchSitesKernel returns kAvx512 only on the +# AVX-512 tier). +# src/matmul.cpp:913,1059 -- the implicit default of the switches in +# ProbQ15AccumulateInto (over SitesKernel) and SelectSitesKernel (over +# GemmTier): every enumerator has a case, so no input reaches it. +# Attention and per-row sites plan, slice S3 (the requant row leaf; plan §3.4, cell 11.6): +# src/intmath.cpp:625-650 -- RequantRowAvx512 (its one branch, the lane +# loop at 637) cannot execute on a runner without AVX-512, for the same +# reason as DotRowAvx512 above. +# src/intmath.cpp:660,664-665 -- RequantRowWide's AVX-512 arm: the same +# runner restriction (DispatchSitesKernel returns kAvx512 only on the +# AVX-512 tier). The switch has no default and every SitesKernel +# enumerator has a case, as in ProbQ15AccumulateInto. +# Attention and per-row sites plan, slice S4 (the guarded softmax; plan §3.4, cell 11.6): +# src/intmath.cpp:1189-1238 -- SoftmaxRowAvx512 (its branches: the lane +# loops at 1204 and 1231, the lane sum at 1212, the tail tests at 1214 +# and 1232 and the tail copies at 1218 and 1236) cannot execute on a +# runner without AVX-512, for the same reason as RequantRowAvx512 above. +# src/intmath.cpp:1268,1275 -- SoftmaxRowQ15's kAvx2-or-kAvx512 choices +# (the body call and the fallback counter): the kAvx512 side needs the +# AVX-512 tier, as RequantRowWide's arm above. The z estimate's downward +# correction in both bodies is branch-free and adds no branch here. diff --git a/tools/ci/branch_coverage_floors.json b/tools/ci/branch_coverage_floors.json index 6ad97c36..9092cee1 100644 --- a/tools/ci/branch_coverage_floors.json +++ b/tools/ci/branch_coverage_floors.json @@ -1,5 +1,5 @@ { - "_measured_cell": "Every numeric floor in this file is pinned from the branch-coverage CI job's own run (.github/workflows/tests.yml): GitHub-hosted ubuntu-latest runner, clang-18/clang++-18, LLVM 18 llvm-cov/llvm-profdata, RelWithDebInfo, the full committed suite, all five binaries this job builds (the non-forced superslm_tests plus the sse2/avx2/avx512-forced full suites and the scalar-forced digest) merged via llvm-profdata merge before the branch percentage is computed. A number measured on any other machine, toolchain, or binary subset is not a floor for this file (T-2149 design \u00a710 dimension 7 item (h), fold round 5, D-SLM3584). T-2529 (Brunel): src/model.cpp and src/proof_manifest.cpp re-pinned to 67.31343283582089 and 53.84615384615385 respectively (from 68.26347305389223 and 63.013698630136986). T-2531 fix round (Poirot 5e128ee-t2530-superslm-ci-green-review.md S-2): the T-2529 entry here claimed 'no GitHub-hosted runner reachable from this session' as the reason for measuring on WSL instead -- false, and refuted by the T-2530 reviewer's own execution: run 33545319929's measured-branch-coverage-floors artifact was present, unexpired, and retrieved in one gh run download. Re-verified here the same way (T-2531, this session): `gh run download 33545319929 -n measured-branch-coverage-floors` succeeds and its own recorded values are model.cpp 67.31343283582089, proof_manifest.cpp 53.84615384615385, matmul.cpp 74.39024390243902 -- digit for digit what this file already carries for all three, so no value here changes. Both files DID gain real branches since their floors were set: five and four commits respectively (T-2296/T-2367/T-2371/T-2432/T-2441/T-2450), including new SslmSectionType enum members (CalibrationBand/DeltaFoldScales/UFoldScales/DampedGreedyConstants) whose own SectionTypeName mapping branches (proof_manifest.cpp) no fixture artifact yet constructs a section of, and T-2432's own new geometry-gate validation branches (model.cpp) -- confirmed by git blame on the specific missed lines, not inferred from the commit count alone. This is genuine, not-this-ticket coverage debt from feature work the branch-coverage floors were never re-measured against, not a regression T-2529 introduced; flagged as a follow-up (board) rather than backfilled here, since authoring the missing suite is coverage-audit/test-design work (Mendeleev/Curie's craft), not a CI-greening fix. Re-pinning follows this file's own established, standing process (a human copies the leg's own recorded measurement into this file) -- the same process src/matmul.cpp's very first entry used (see git history: e8b809f). src/matmul.cpp's own floor (74.39024390243902) is UNCHANGED by T-2529: CI run 33545319929 -- the file's only sanctioned measurement cell -- reported exactly two files below floor (model.cpp, proof_manifest.cpp) and did not report matmul.cpp, so on that cell matmul.cpp measured at or above 74.39024390243902 (confirmed above, digit for digit, against the downloaded artifact). T-2537 structural fix (Poirot 67bfcbf-t2536-superslm-ci-green-confirmation3.md S-4, following the T-2526 precedent on this same repository): this comment carried a running tally of same-session, off-cell WSL readings of src/matmul.cpp -- 'over a sample of two', then 'over a sample of three', each stale the moment the next round took another reading, three rewordings failing in three consecutive rounds. The count is deleted from this prose rather than reworded a fourth time: this comment's own FIRST rule already states it -- only a number measured on the sanctioned cell (GitHub-hosted ubuntu-latest, clang-18/clang++-18, the full suite, all five binaries) is a floor for this file, so no off-cell reading, however many have been taken, ever becomes one. The individual off-cell readings themselves, and whichever causal account (a static per-box AVX-512-exposure mechanism, or genuine build-to-build measurement noise) they do or do not support, live in the decision log, by ID, not here: D-SLM5973 (T-2533), D-SLM5996 (T-2535 M-3), D-SLM6003 (T-2535 umbrella), D-SLM6010 (T-2536, this finding's own record) -- each dated, each carrying its own reading and its own cell, appended as new entries rather than folded into a prose tally this file's own comment has to keep re-deriving. Settling the underlying question -- does one branch of src/matmul.cpp genuinely cover nondeterministically build-to-build on a WSL clang-18 toolchain, and if so why -- needs two independent five-binary clang-18 coverage builds under WSL (T-2532's own tool ceiling, hours-scale) and remains open; a fifth reading, if one is ever taken, is filed as its own decision entry, not appended to this sentence. What is not in question either way: matmul.cpp's floor stays at the CI leg's own last reported value (74.39024390243902) unchanged, regardless of what any off-cell reading shows. T-2783 (1.5.0, principal's ruling): the branch-coverage leg had not run since 1.3.1 -- 1.4.0 and 1.5.0 broke the Linux build before it -- so coverage debt accumulated unmeasured. Once the build was fixed, CI run 35385978298 recorded src/artifact.cpp 82.1969696969697 (was 84.12698412698413), src/matmul.cpp 72.22222222222221 (was 74.39024390243902), src/model.cpp 49.05857740585774 (was 67.31343283582089) and src/proof_manifest.cpp 53.16455696202531 (was 53.84615384615385); the principal chose to re-pin to those values rather than hold 1.5.0 for new tests. Most of model.cpp's drop is expected to be 1.5.0's QKC1 loader refusal branches, which the Python loader suite exercises through sslm_verify but this C++-only leg does not instrument; counting them is the follow-up. 1.8.1: the saturation census cells (tests/test_slm18x_saturation_census.cpp) are the first superslm_tests cells to call the C ABI, so src/sslm_abi.cpp and the objects it pulls in are now linked into the measured binaries for the first time. CI run 36163807227 (tag v1.8.1) recorded them with no floor; pinned here from that run's artifact: src/sslm_abi.cpp 13.135593220338984; include/superslm/schema_masks.h, src/damped_greedy_antilm.cpp, src/damped_greedy_phaseD.cpp and src/damped_greedy_topk.cpp 0.0 (linked, not yet exercised by the C++ suite). 1.9.0: the C-ABI cells in tests/test_slm19x_schema_damped_greedy.cpp now exercise the four files pinned at 0.0 above. CI run 36188435744 (tag v1.9.0) recorded include/superslm/schema_masks.h 90.54054054054053, src/damped_greedy_antilm.cpp 94.23076923076923, src/damped_greedy_phaseD.cpp 62.5 and src/damped_greedy_topk.cpp 43.292682926829265; pinned here from that run's artifact. src/sslm_abi.cpp raised to 47.58403361344538 from the same run (36188435744), at the principal's word.", + "_measured_cell": "Every numeric floor in this file is pinned from the branch-coverage CI job's own run (.github/workflows/tests.yml): GitHub-hosted ubuntu-latest runner, clang-18/clang++-18, LLVM 18 llvm-cov/llvm-profdata, RelWithDebInfo, the full committed suite, all five binaries this job builds (the non-forced superslm_tests plus the sse2/avx2/avx512-forced full suites and the scalar-forced digest) merged via llvm-profdata merge before the branch percentage is computed. A number measured on any other machine, toolchain, or binary subset is not a floor for this file (T-2149 design \u00a710 dimension 7 item (h), fold round 5, D-SLM3584). T-2529 (Brunel): src/model.cpp and src/proof_manifest.cpp re-pinned to 67.31343283582089 and 53.84615384615385 respectively (from 68.26347305389223 and 63.013698630136986). T-2531 fix round (Poirot 5e128ee-t2530-superslm-ci-green-review.md S-2): the T-2529 entry here claimed 'no GitHub-hosted runner reachable from this session' as the reason for measuring on WSL instead -- false, and refuted by the T-2530 reviewer's own execution: run 33545319929's measured-branch-coverage-floors artifact was present, unexpired, and retrieved in one gh run download. Re-verified here the same way (T-2531, this session): `gh run download 33545319929 -n measured-branch-coverage-floors` succeeds and its own recorded values are model.cpp 67.31343283582089, proof_manifest.cpp 53.84615384615385, matmul.cpp 74.39024390243902 -- digit for digit what this file already carries for all three, so no value here changes. Both files DID gain real branches since their floors were set: five and four commits respectively (T-2296/T-2367/T-2371/T-2432/T-2441/T-2450), including new SslmSectionType enum members (CalibrationBand/DeltaFoldScales/UFoldScales/DampedGreedyConstants) whose own SectionTypeName mapping branches (proof_manifest.cpp) no fixture artifact yet constructs a section of, and T-2432's own new geometry-gate validation branches (model.cpp) -- confirmed by git blame on the specific missed lines, not inferred from the commit count alone. This is genuine, not-this-ticket coverage debt from feature work the branch-coverage floors were never re-measured against, not a regression T-2529 introduced; flagged as a follow-up (board) rather than backfilled here, since authoring the missing suite is coverage-audit/test-design work (Mendeleev/Curie's craft), not a CI-greening fix. Re-pinning follows this file's own established, standing process (a human copies the leg's own recorded measurement into this file) -- the same process src/matmul.cpp's very first entry used (see git history: e8b809f). src/matmul.cpp's own floor (74.39024390243902) is UNCHANGED by T-2529: CI run 33545319929 -- the file's only sanctioned measurement cell -- reported exactly two files below floor (model.cpp, proof_manifest.cpp) and did not report matmul.cpp, so on that cell matmul.cpp measured at or above 74.39024390243902 (confirmed above, digit for digit, against the downloaded artifact). T-2537 structural fix (Poirot 67bfcbf-t2536-superslm-ci-green-confirmation3.md S-4, following the T-2526 precedent on this same repository): this comment carried a running tally of same-session, off-cell WSL readings of src/matmul.cpp -- 'over a sample of two', then 'over a sample of three', each stale the moment the next round took another reading, three rewordings failing in three consecutive rounds. The count is deleted from this prose rather than reworded a fourth time: this comment's own FIRST rule already states it -- only a number measured on the sanctioned cell (GitHub-hosted ubuntu-latest, clang-18/clang++-18, the full suite, all five binaries) is a floor for this file, so no off-cell reading, however many have been taken, ever becomes one. The individual off-cell readings themselves, and whichever causal account (a static per-box AVX-512-exposure mechanism, or genuine build-to-build measurement noise) they do or do not support, live in the decision log, by ID, not here: D-SLM5973 (T-2533), D-SLM5996 (T-2535 M-3), D-SLM6003 (T-2535 umbrella), D-SLM6010 (T-2536, this finding's own record) -- each dated, each carrying its own reading and its own cell, appended as new entries rather than folded into a prose tally this file's own comment has to keep re-deriving. Settling the underlying question -- does one branch of src/matmul.cpp genuinely cover nondeterministically build-to-build on a WSL clang-18 toolchain, and if so why -- needs two independent five-binary clang-18 coverage builds under WSL (T-2532's own tool ceiling, hours-scale) and remains open; a fifth reading, if one is ever taken, is filed as its own decision entry, not appended to this sentence. What is not in question either way: matmul.cpp's floor stays at the CI leg's own last reported value (74.39024390243902) unchanged, regardless of what any off-cell reading shows. T-2783 (1.5.0, principal's ruling): the branch-coverage leg had not run since 1.3.1 -- 1.4.0 and 1.5.0 broke the Linux build before it -- so coverage debt accumulated unmeasured. Once the build was fixed, CI run 35385978298 recorded src/artifact.cpp 82.1969696969697 (was 84.12698412698413), src/matmul.cpp 72.22222222222221 (was 74.39024390243902), src/model.cpp 49.05857740585774 (was 67.31343283582089) and src/proof_manifest.cpp 53.16455696202531 (was 53.84615384615385); the principal chose to re-pin to those values rather than hold 1.5.0 for new tests. Most of model.cpp's drop is expected to be 1.5.0's QKC1 loader refusal branches, which the Python loader suite exercises through sslm_verify but this C++-only leg does not instrument; counting them is the follow-up. 1.8.1: the saturation census cells (tests/test_slm18x_saturation_census.cpp) are the first superslm_tests cells to call the C ABI, so src/sslm_abi.cpp and the objects it pulls in are now linked into the measured binaries for the first time. CI run 36163807227 (tag v1.8.1) recorded them with no floor; pinned here from that run's artifact: src/sslm_abi.cpp 13.135593220338984; include/superslm/schema_masks.h, src/damped_greedy_antilm.cpp, src/damped_greedy_phaseD.cpp and src/damped_greedy_topk.cpp 0.0 (linked, not yet exercised by the C++ suite). 1.9.0: the C-ABI cells in tests/test_slm19x_schema_damped_greedy.cpp now exercise the four files pinned at 0.0 above. CI run 36188435744 (tag v1.9.0) recorded include/superslm/schema_masks.h 90.54054054054053, src/damped_greedy_antilm.cpp 94.23076923076923, src/damped_greedy_phaseD.cpp 62.5 and src/damped_greedy_topk.cpp 43.292682926829265; pinned here from that run's artifact. src/sslm_abi.cpp raised to 47.58403361344538 from the same run (36188435744), at the principal's word. Attention and per-row sites (S1-S5): the new AVX-512 kernels in src/matmul.cpp and src/intmath.cpp cannot run on GitHub-hosted runners, which lack AVX-512, so their branches are uncovered on this cell; at the principal's word they are re-pinned from this PR's own CI run 36584892310 (pull_request, head 4d7acc2), whose runner had no AVX-512 (superslm_tests_avx512_forced exited 132, SIGILL): src/matmul.cpp 69.23076923076923 and src/intmath.cpp 83.47107438016529. A floor pinned from a no-AVX-512 run holds on runners with AVX-512 too, which only add covered branches; pinning from an AVX-512 runner would fail every later run that lands on one without it.", "include/superslm/adapter_marshal.h": 44.91525423728814, "include/superslm/artifact.h": 100.0, "include/superslm/layer_marshal.h": 40.0, @@ -10,8 +10,8 @@ "src/damped_greedy_phaseD.cpp": 62.5, "src/damped_greedy_topk.cpp": 43.292682926829265, "src/decode_digest.cpp": 100.0, - "src/intmath.cpp": 87.93103448275862, - "src/matmul.cpp": 72.22222222222221, + "src/intmath.cpp": 83.47107438016529, + "src/matmul.cpp": 69.23076923076923, "src/model.cpp": 49.05857740585774, "src/proof_manifest.cpp": 53.16455696202531, "src/sha256.cpp": 100.0, diff --git a/tools/ci/check_matmul_avx_isolation.py b/tools/ci/check_matmul_avx_isolation.py index 63a83f8e..df56adfe 100644 --- a/tools/ci/check_matmul_avx_isolation.py +++ b/tools/ci/check_matmul_avx_isolation.py @@ -7,7 +7,26 @@ are the ONLY functions in the whole project that may compile with AVX2/AVX-512 instructions enabled. Since the tiled-matmul plan's slice 1 they are `DotRowAvx2` and `DotRowAvx512` (the shipped per-row tiers) plus the tiled kernel's `TiledMicroAvx2`, `TiledMicroAvx512`, -`TiledGemmAvx2` and `TiledGemmAvx512` (all in the anonymous namespace). The packer +`TiledGemmAvx2` and `TiledGemmAvx512`, plus (the attention and per-row sites plan's slice S2) the +prob·V bodies `ProbVBlockAvx2`, `ProbVBlockAvx512`, `ProbVTail16Avx512`, `ProbVAccumulateIntoAvx2` and +`ProbVAccumulateIntoAvx512` (all in the anonymous namespace; their guard `ProbVFastPathAdmits` and +the dispatching core `ProbQ15AccumulateInto` carry no target attribute), plus (that plan's slice S3) +src/intmath.cpp's requant row bodies `RequantRowAvx2` and `RequantRowAvx512` (anonymous namespace; the +exported leaf `RequantRowWide` that dispatches to them carries no target attribute), plus (slice S4) the +softmax fast path's bodies `SoftmaxRowAvx2` and `SoftmaxRowAvx512` and the step helpers they inline, +`SoftmaxExpAvx2`, `SoftmaxExpAvx512`, `SoftmaxProbAvx2` and `SoftmaxProbAvx512` (anonymous namespace; the +guard `SoftmaxFastGuard`, the row set-up `MakeSoftmaxFastRow` and `SoftmaxProbReciprocal`, and the exported +`SoftmaxRowQ15` that dispatches carry no target attribute), plus (slice S5) src/forward/forward_sites.cpp's +Q31 score row bodies `QkQ31RowAvx2` and `QkQ31RowAvx512` and the rounding steps they inline, `Q31RoundAvx2` +and `Q31RoundAvx512` (anonymous namespace; the key packers `Q31PackKeyBlock` and `Q31PackQuads4x16` are +baseline SSE2, and the guard `Q31RowFastPathAdmits`, the limb set-up `MakeQ31RowLimbs` and the exported +`QkQ31ScoreRow` that dispatches carry no target attribute; that file's per-key `QkQ31ScoreAvx2` and +`QkQ31ScoreAvx512` predate these plans). Since slice S5 forward_sites.cpp calls `detail::ActiveGemmTier()` +too; every direct recipe that compiles it already names matmul.cpp. intmath.cpp is +compiled by the same CMake targets as matmul.cpp, and since slice S3 it calls `detail::ActiveGemmTier()`, +so every direct compiler recipe that compiles it must name matmul.cpp too (the four recipes that did not +gained it in that slice) and is therefore inside this checker's `matmul.cpp`-gated direct-invocation +channel. The packer (`TiledPackPanel16`, `TiledTranspose8x8Epi16`) and the activation prep (`TiledWidenActivations`) are baseline SSE2 and carry no target attribute. All of them use GCC/Clang's per-function `__attribute__((target("avx2")))` / diff --git a/tools/ci/check_tiled_matmul_linkage.py b/tools/ci/check_tiled_matmul_linkage.py index 6d8b271e..eb6bcde6 100644 --- a/tools/ci/check_tiled_matmul_linkage.py +++ b/tools/ci/check_tiled_matmul_linkage.py @@ -7,15 +7,32 @@ class the forced-tier design exists to prevent. So every tiled and packer symbol must be LOCAL. The population is every symbol of the given matmul.cpp objects whose demangled name contains `Tiled` -(functions and data alike: the kernels, the packer, the activation prep, the constants), plus the -build-configuration record `superslm_build_config_record`. The engine's public API in the same object +(functions and data alike: the kernels, the packer, the activation prep, the constants) or `ProbV` / +`ProbQ15AccumulateInto` (the attention and per-row sites plan's slice S2: the AVX2 and AVX-512BW prob·V +bodies, their guard and the accumulate-into core, cell 11.3 of that plan), plus the +build-configuration record `superslm_build_config_record`. Slice S3 of the same plan gives src/intmath.cpp +its first target-attributed functions, the requant row bodies `RequantRowAvx2` and `RequantRowAvx512` +(`RequantRowAvx` in the population; the exported leaf `RequantRowWide` that dispatches to them carries no +target attribute and is outside it), so the intmath.cpp objects are passed too. Slice S4 adds the softmax +fast path's bodies there: `SoftmaxRowAvx2` / `SoftmaxRowAvx512` and the per-step helpers they inline, +`SoftmaxExpAvx2` / `SoftmaxExpAvx512` and `SoftmaxProbAvx2` / `SoftmaxProbAvx512` (`SoftmaxRowAvx`, +`SoftmaxExpAvx` and `SoftmaxProbAvx` in the population), and their shared, unattributed guard and row +set-up (`SoftmaxFastGuard`, `MakeSoftmaxFastRow`, `SoftmaxProbReciprocal`; `SoftmaxFast` and +`SoftmaxProbReciprocal`); the exported `SoftmaxRowQ15` that dispatches to them is outside it. Slice S5 adds +src/forward/forward_sites.cpp's Q31 score row: the bodies `QkQ31RowAvx2` / `QkQ31RowAvx512` (`QkQ31RowAvx` in +the population), the rounding steps they inline (`Q31RoundAvx2` / `Q31RoundAvx512`; `Q31RoundAvx`), the key +packers (`Q31PackKeyBlock`, `Q31PackQuads4x16`; `Q31Pack`) and the unattributed guard and limb set-up +(`Q31RowFastPathAdmits`, `MakeQ31RowLimbs`, the `Q31RowLimbs` struct), so the forward_sites.cpp objects are +passed too; the exported `QkQ31ScoreRow` that dispatches to them is outside it, and so is that file's older +per-key `QkQ31Score` family. The engine's public API in the same object is outside it (including the `superslm::detail::` entries the header declares, which carry no target attribute), and so are the test seam's own `superslm_test::` variables, which exist only in seam builds and are shared with the test translation unit on purpose. Rules: 1. every population symbol except the record is local: never global, weak, unique or COMDAT; - 2. the record is present exactly once per object and is external (the single allowed exception); + 2. the record is present exactly once per matmul.cpp object and is external (the single allowed + exception); any other object (intmath.cpp's) holds no record; 3. the name list below is not stale: each listed function resolves to a symbol in at least one of the given objects, or is proven inlined by its signature instruction in its caller's disassembly (at -O3 the packer, the micro-kernels and the activation prep are all inlined). @@ -24,6 +41,7 @@ class the forced-tier design exists to prevent. So every tiled and packer symbol packer table, must each turn rule 1 red. Usage: check_tiled_matmul_linkage.py OBJECT [OBJECT...] (Linux/ELF: nm, readelf, objdump) + (the matmul.cpp objects are recognized by "matmul" in the object's file name) Windows (dumpbin /symbols) is not implemented here; see the plan's §11.3 Windows leg. """ @@ -34,7 +52,10 @@ class the forced-tier design exists to prevent. So every tiled and packer symbol import sys RECORD = "superslm_build_config_record" -POPULATION = re.compile(r"Tiled|" + RECORD) +POPULATION = re.compile( + r"Tiled|ProbV|ProbQ15AccumulateInto|RequantRowAvx|SoftmaxRowAvx|SoftmaxExpAvx|SoftmaxProbAvx|SoftmaxFast|" + r"SoftmaxProbReciprocal|QkQ31RowAvx|Q31RoundAvx|Q31Pack|Q31RowFastPathAdmits|MakeQ31RowLimbs|Q31RowLimbs|" + + RECORD) SEAM = "superslm_test::" DETAIL_API = "superslm::detail::" # declared in include/superslm/matmul.h; never target-attributed @@ -48,6 +69,35 @@ class the forced-tier design exists to prevent. So every tiled and packer symbol "TiledTranspose8x8Epi16": ("TiledGemmAvx", r"punpck[lh]wd"), "TiledWidenActivations": ("GemmInt8AccumulateCols", r"(movsbw|pmovsxbw)"), "RunTiledGemm": ("GemmInt8AccumulateCols", r"call.*TiledGemmAvx"), + # Attention and per-row sites plan, slice S2 (cell 11.3). + "ProbVAccumulateIntoAvx2": ("GemmProbQ15Accumulate", r"(call|jmp).*ProbVAccumulateIntoAvx2"), + "ProbVAccumulateIntoAvx512": ("GemmProbQ15Accumulate", r"(call|jmp).*ProbVAccumulateIntoAvx512"), + "ProbVBlockAvx2": ("ProbVAccumulateIntoAvx2", r"vpmaddwd\s.*%ymm"), + "ProbVBlockAvx512": ("ProbVAccumulateIntoAvx512", r"vpmaddwd\s.*%zmm"), + "ProbVTail16Avx512": ("ProbVAccumulateIntoAvx512", r"vinserti128"), + "ProbQ15AccumulateInto": ("GemmProbQ15Accumulate", r"(call|jmp).*ProbVAccumulateIntoAvx"), + # Attention and per-row sites plan, slice S3 (cell 11.3): src/intmath.cpp's requant row bodies. + "RequantRowAvx2": ("RequantRowWide", r"vpmuludq\s.*%ymm"), + "RequantRowAvx512": ("RequantRowWide", r"vpmuludq\s.*%zmm"), + # Attention and per-row sites plan, slice S4 (cell 11.3): src/intmath.cpp's softmax fast path. + "SoftmaxRowAvx2": ("SoftmaxRowQ15", r"(call|jmp).*SoftmaxRowAvx2"), + "SoftmaxRowAvx512": ("SoftmaxRowQ15", r"(call|jmp).*SoftmaxRowAvx512"), + "SoftmaxExpAvx2": ("SoftmaxRowAvx2", r"vpsrlvq\s.*%ymm"), + "SoftmaxExpAvx512": ("SoftmaxRowAvx512", r"vpsrlvq\s.*%zmm"), + "SoftmaxProbAvx2": ("SoftmaxRowAvx2", r"vpsllq\s+\$0xf,.*%ymm"), + "SoftmaxProbAvx512": ("SoftmaxRowAvx512", r"vpsllq\s+\$0xf,.*%zmm"), + "SoftmaxFastGuard": ("SoftmaxRowQ15", r"(0x2000000000000000|0xe000000000000000|0x4000000000000000)"), # +-2^61 + "MakeSoftmaxFastRow": ("SoftmaxRowQ15", r"(lzcnt|bsr)"), + "SoftmaxProbReciprocal": ("SoftmaxRowAvx", r"\bdiv"), + # Attention and per-row sites plan, slice S5 (cell 11.3): src/forward/forward_sites.cpp's Q31 score row. + "QkQ31RowAvx2": ("QkQ31ScoreRow", r"(call|jmp).*QkQ31RowAvx2"), + "QkQ31RowAvx512": ("QkQ31ScoreRow", r"(call|jmp).*QkQ31RowAvx512"), + "Q31RoundAvx2": ("QkQ31RowAvx2", r"vpsrlq\s+\$0x3f,.*%ymm"), + "Q31RoundAvx512": ("QkQ31RowAvx512", r"vpsrlq\s+\$0x3f,.*%zmm"), + "Q31PackKeyBlock": ("QkQ31RowAvx", r"vpsraw\s+\$0x8,"), + "Q31PackQuads4x16": ("QkQ31RowAvx", r"vpsraw\s+\$0x8,"), + "Q31RowFastPathAdmits": ("QkQ31ScoreRow", r"\$0x200,"), # head_dim <= 512 + "MakeQ31RowLimbs": ("QkQ31ScoreRow", r"\$0x1e,"), # w >> 30 } @@ -115,8 +165,9 @@ def main(argv: list[str]) -> int: for exp in EXPECTED: if re.search(r"(^|::)" + exp + r"\b", name) and resolved[exp] is None: resolved[exp] = f"symbol in {obj}" - if records != 1: - failures.append(f"{obj}: the record appears {records} times, want exactly 1") + want_records = 1 if "matmul" in obj.rsplit("/", 1)[-1] else 0 + if records != want_records: + failures.append(f"{obj}: the record appears {records} times, want exactly {want_records}") dis = disassembly_by_function(obj) for exp, (caller, sig) in EXPECTED.items(): if resolved[exp] is not None: diff --git a/tools/ci/sslm_axis_digest.cpp b/tools/ci/sslm_axis_digest.cpp index 145fb917..4531a4b2 100644 --- a/tools/ci/sslm_axis_digest.cpp +++ b/tools/ci/sslm_axis_digest.cpp @@ -37,10 +37,15 @@ // Build (see run_axes.sh / run_axes.ps1): // -std=c++20 -I/include /src/*.cpp sslm_axis_digest.cpp -o sslm_axis_digest +#include "superslm/forward_sites.h" // QkQ31ScoreRow (section 11, slice S5) #include "superslm/intmath.h" #include "superslm/matmul.h" #include "superslm/sha256.h" #include "superslm/silu_lut.h" +// Attention and per-row sites plan, section 10's fixed input set (header-only, shared with the golden +// generator and the suite; included by path so this tool keeps its one-line build recipe). +#include "../../tests/support/attention_cases.h" +#include "../../tests/support/rowsite_cases.h" #include #include @@ -675,6 +680,49 @@ void SectionMatmulTiled() { } } +// --- 10. The per-row sites: tables and element loops ------------------------------ +// +// Attention and per-row sites plan (rev 3.1), §3.3 evidence 2, cell 6.2. RmsNormSite, MlpActSite and +// ResidualReconcileSite over tests/support/rowsite_cases.h's fixed set, which sits on both sides of the +// per-row table threshold (widths 1 to 4,864 around 512), carries -128 codes, uniform rows, a small +// gate scale and landing-flag rows (slice S1's entries), then slice S3's requant rows through +// RequantChainChecked (every width around the 4- and 8-lane SIMD bodies, +-d' in every lane position, +// every s the funnel's preflight can produce, the P = 2^63 corner). Every call's status, output scale +// and whole output row are digested. The forced-scalar leg runs the v1.9.0 +// per-element loops (tables off), so it is the reference axis this section is compared against; the +// suite pins the same stream to a hash from the v1.9.0 tag. A new section, so sections 1-9 keep their +// previous values exactly. +void SectionRowsites() { + Section& sec = NewSection("c_rowsites"); + auto emit = [&](int64_t v) { sec.sink.I64(v); }; + superslm_rowsite_cases::RunRowTableCases(emit); + superslm_rowsite_cases::RunRequantRowCases(emit); +} + +// --- 11. The attention kernels ------------------------------------------------------ +// +// Attention and per-row sites plan (rev 3.1), §3.3 evidence 2, cell 6.2. GemmProbQ15Accumulate over +// tests/support/attention_cases.h's fixed set (slice S2's entries: plan §8 4.S2's head_dim x width grid, +// the int16 condition's corners, width 0 and 2.S2's hostile rows; slices S4-S6 append theirs). Every +// call's width, head_dim and whole output row are digested. The forced-scalar and forced-SSE2 legs run +// the v1.9.0 loop, so they are the reference axes; the suite pins the same stream to a hash from the +// v1.9.0 tag. A new section, so sections 1-10 keep their previous values exactly. +// +// Slice S4 appends SoftmaxRowQ15 over the same header's S4 set (§8 4.S4's grid, inside corners and +// correction rows, 2.S4's hostile rows; per call its width, bool and output row), as §3.3 evidence 2 has +// c32_attention carry the S2, S4, S5 and S6 entries. The suite pins the S4 stream to its own hash. +// +// Slice S5 appends QkQ31ScoreRow over the same header's S5 set (§8 4.S5's head_dim x width grid in three +// operand kinds, 7.S5b's margin corners, 7.S5c's rounding ties; per call its width, head_dim and scores). Every +// ratio is in [1, 2^31] (§3.3). The suite pins the S5 stream to its own hash, from the v1.9.0 per-key loop. +void SectionAttention() { + Section& sec = NewSection("c32_attention"); + auto emit = [&](int64_t v) { sec.sink.I64(v); }; + superslm_attention_cases::RunProbVCases(emit); + superslm_attention_cases::RunSoftmaxCases(emit); + superslm_attention_cases::RunQ31Cases(emit, superslm::QkQ31ScoreRow); +} + // --- driver --------------------------------------------------------------------- void PrintBuildIdentity() { @@ -736,6 +784,8 @@ int main() { SectionSiluLut(); SectionMatmul(); SectionMatmulTiled(); + SectionRowsites(); + SectionAttention(); Sha256 global; int failures = 0; diff --git a/tools/gen_attn_rowsite_golden.cpp b/tools/gen_attn_rowsite_golden.cpp new file mode 100644 index 00000000..f67bbd79 --- /dev/null +++ b/tools/gen_attn_rowsite_golden.cpp @@ -0,0 +1,152 @@ +// Attention and per-row sites plan (rev 3.1), §3.3 evidence 3: the golden-pin generator. +// +// Built against the v1.9.0 TAG's library -- the normative per-element code, independent of every +// kernel and table under test -- it runs each slice's fixed input set and writes +// tests/attn_rowsite_golden_pin.h, one SHA-256 per slice, each serialized as little-endian int64 +// exactly as the digest serializes the same stream: +// - S1: tests/support/rowsite_cases.h through the three per-row sites (every call's status, output +// scale and output row; the digest's `c_rowsites` section); +// - S2: tests/support/attention_cases.h through GemmProbQ15Accumulate (every call's width, head_dim +// and output row; the digest's `c32_attention` section); +// - S3: tests/support/rowsite_cases.h's requant rows through RequantChainChecked (every call's +// status, output scale and codes; appended to the digest's `c_rowsites` section); +// - S4: tests/support/attention_cases.h's softmax rows through SoftmaxRowQ15 (every call's width, +// bool and output row; appended to the digest's `c32_attention` section); +// - S5: tests/support/attention_cases.h's Q31 rows through the v1.9.0 per-key QkQ31Score (every call's +// width, head_dim and scores; the build under test runs them through QkQ31ScoreRow, and the digest +// appends them to `c32_attention`), plus a second hash: tests/support/qk_attention_fixture.h's widened +// QK-norm fixture through the decode loop (24 positions' output codes and scales, then the K/V bytes), +// which the suite's cell 11.1(c) requires of the decode loop, the chunk loop and the decode loop with a +// capture sink alike. +// Every tier of every build must reproduce every hash (tests/test_attn_rowsites.cpp, cell 6.3), so +// the reference takes no input from the code it grades. One hash per slice, so no later slice +// regenerates an earlier one's. +// +// Recipe (the one used for the committed pin; see docs/attention-rowsites/s1/golden.txt, s2/golden.txt, +// s3/golden.txt, s4/golden.txt and s5/golden.txt): +// git worktree add /tmp/v190 v1.9.0 +// cmake -S /tmp/v190 -B /tmp/v190/build -DCMAKE_BUILD_TYPE=Release && cmake --build /tmp/v190/build --target superslm +// c++ -std=c++20 -O2 -I/tmp/v190/include tools/gen_attn_rowsite_golden.cpp /tmp/v190/build/libsuperslm.a +// -o gen_attn_rowsite_golden +// ./gen_attn_rowsite_golden tests/attn_rowsite_golden_pin.h +// It also builds against the current tree (the CMake target of the same name), where it must +// print the same hash; that is a consistency check, not the pin's provenance. + +#include +#include +#include + +#include "superslm/forward_sites.h" +#include "superslm/sha256.h" +#include "../tests/support/attention_cases.h" +#include "../tests/support/qk_attention_fixture.h" +#include "../tests/support/rowsite_cases.h" + +namespace { + +struct Hashed { + std::string hex; + unsigned long long values = 0; +}; + +template +Hashed HashStream(Run run) { + superslm::Sha256 h; + Hashed r; + auto emit = [&](int64_t v) { + uint8_t b[8]; + for (int i = 0; i < 8; ++i) b[i] = static_cast((static_cast(v) >> (8 * i)) & 0xffU); + h.Update(b, 8); + ++r.values; + }; + run(emit); + uint8_t digest[32]; + h.Final(digest); + r.hex = superslm::ToHex(digest); + return r; +} + +} // namespace + +int main(int argc, char** argv) { + const Hashed s1 = HashStream([](auto& emit) { superslm_rowsite_cases::RunRowTableCases(emit); }); + std::printf("S1 row-table golden: %s over %llu values\n", s1.hex.c_str(), s1.values); + const Hashed s2 = HashStream([](auto& emit) { superslm_attention_cases::RunProbVCases(emit); }); + std::printf("S2 prob-V golden: %s over %llu values\n", s2.hex.c_str(), s2.values); + const Hashed s3 = HashStream([](auto& emit) { superslm_rowsite_cases::RunRequantRowCases(emit); }); + std::printf("S3 requant-row golden: %s over %llu values\n", s3.hex.c_str(), s3.values); + const Hashed s4 = HashStream([](auto& emit) { superslm_attention_cases::RunSoftmaxCases(emit); }); + std::printf("S4 softmax golden: %s over %llu values\n", s4.hex.c_str(), s4.values); + // v1.9.0 has no row entry: the per-key loop the layer loops ran. + auto per_key_row = [](const int8_t* q, const int8_t* keys, const int64_t* ratio, size_t hd, size_t width, + int64_t* out) { + for (size_t j = 0; j < width; ++j) out[j] = superslm::QkQ31Score(q, keys + j * hd, ratio, hd); + }; + const Hashed s5 = HashStream([&](auto& emit) { superslm_attention_cases::RunQ31Cases(emit, per_key_row); }); + std::printf("S5 Q31-row golden: %s over %llu values\n", s5.hex.c_str(), s5.values); + int fixture_status = -1; + const Hashed s5f = HashStream([&](auto& emit) { + superslm_qk_fixture::QkAttentionFixture f; + if (!f.loaded) return; + fixture_status = static_cast(superslm_qk_fixture::RunFixtureDecode(f, emit)); + }); + std::printf("S5 QK-norm fixture golden: %s over %llu values (status %d)\n", s5f.hex.c_str(), s5f.values, + fixture_status); + if (fixture_status != static_cast(superslm::SslmForwardStatus::Ok)) + return std::fprintf(stderr, "the QK-norm fixture did not run Ok\n"), 1; + if (argc < 2) return 0; + FILE* f = std::fopen(argv[1], "wb"); + if (!f) return std::fprintf(stderr, "cannot write %s\n", argv[1]), 1; + std::fprintf(f, + "// GENERATED FILE. Do not hand-edit.\n" + "//\n" + "// Produced by tools/gen_attn_rowsite_golden.cpp built against the v1.9.0 tag's library (the\n" + "// normative per-element code), over tests/support/rowsite_cases.h's and\n" + "// tests/support/attention_cases.h's fixed input sets, one hash per slice. Attention and per-row\n" + "// sites plan, §3.3 evidence 3, coverage cell 6.3. Re-running the generator against v1.9.0 must\n" + "// reproduce this file byte-for-byte.\n" + "#ifndef SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H\n" + "#define SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H\n" + "\n" + "#include \n" + "\n" + "namespace superslm_test {\n" + "\n" + "// Slice S1: RmsNormSite, MlpActSite and ResidualReconcileSite over RunRowTableCases.\n" + "inline constexpr const char* kAttnRowsiteS1GoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS1GoldenValues = %lluULL;\n" + "\n" + "// Slice S2: GemmProbQ15Accumulate over RunProbVCases.\n" + "inline constexpr const char* kAttnRowsiteS2GoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS2GoldenValues = %lluULL;\n" + "\n" + "// Slice S3: RequantChainChecked's element loop over RunRequantRowCases.\n" + "inline constexpr const char* kAttnRowsiteS3GoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS3GoldenValues = %lluULL;\n" + "\n" + "// Slice S4: SoftmaxRowQ15 over RunSoftmaxCases.\n" + "inline constexpr const char* kAttnRowsiteS4GoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS4GoldenValues = %lluULL;\n" + "\n" + "// Slice S5: QkQ31Score per key over RunQ31Cases.\n" + "inline constexpr const char* kAttnRowsiteS5GoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS5GoldenValues = %lluULL;\n" + "\n" + "// Slice S5, cell 11.1(c): the QK-norm fixture's forward through the decode loop.\n" + "inline constexpr const char* kAttnRowsiteS5FixtureGoldenHash =\n" + " \"%s\";\n" + "inline constexpr uint64_t kAttnRowsiteS5FixtureGoldenValues = %lluULL;\n" + "\n" + "} // namespace superslm_test\n" + "\n" + "#endif // SUPERSLM_TESTS_ATTN_ROWSITE_GOLDEN_PIN_H\n", + s1.hex.c_str(), s1.values, s2.hex.c_str(), s2.values, s3.hex.c_str(), s3.values, s4.hex.c_str(), s4.values, + s5.hex.c_str(), s5.values, s5f.hex.c_str(), s5f.values); + std::fclose(f); + return 0; +} diff --git a/tools/sslm_sites_bench.cpp b/tools/sslm_sites_bench.cpp new file mode 100644 index 00000000..af192aba --- /dev/null +++ b/tools/sslm_sites_bench.cpp @@ -0,0 +1,428 @@ +// sslm_sites_bench -- the attention and per-row sites plan's engine bench (rev 3.1, cell 10.1). +// +// One source, built against either library (the base's or the candidate's), so a ratio between two +// builds' readings is the slice's own effect. Reports only; nothing here gates a slice. +// +// sslm_sites_bench kernel [--repeat=R] +// Each per-row site at the Qwen2.5-0.5B geometry, through its public entry point with realistic +// constants: RmsNormSite at 896, MlpActSite at 4,864, ResidualReconcileSite at 896. Prints the +// best-of-R microseconds per call (each rep times a batch of calls over 16 distinct rows). +// sslm_sites_bench pv [--repeat=R] +// Slice S2: GemmProbQ15Accumulate at head_dim 64 (the 0.5B's) on realistic probability rows (every p +// formed as the softmax forms it, Sum p <= 2^15) over one KV head's value rows. Prints best-of-R +// microseconds per call at single widths, then the per-token cost at the 0.5B's full depth (24 +// layers x 14 query heads, one call per head per token): prefill of T tokens sums the calls at +// widths 1..T and divides by T; decode at context C is one call at width C + 1 per head. +// sslm_sites_bench requant [--repeat=R] +// Slice S3: RequantChainChecked (the whole funnel call: max-abs, preflight, the element loop that S3 +// replaces, the scale fold) at the 0.5B's two funnel widths, 896 and 4,864, over 16 rows whose +// max-abs spans 2^16 to 2^31. Prints best-of-R microseconds per call, then the per-token cost at +// full depth: 24 layers x (8 calls at 896 + 3 at 4,864), plus the embed's one call at 896 (plan §4.3, +// G26). Prefill and decode make the same calls per token. +// sslm_sites_bench softmax [--repeat=R] +// Slice S4: SoftmaxRowQ15 on rows with the forward's own constants (IExpScaleConstants with the format-30 +// coefficients the forward passes, kept when q_ln2 falls in [347, 944], the range cell 11.1(d)'s data term +// measured on the 0.5B-width synthetic; docs/attention-rowsites/s4/softmax-data-terms.txt) and scores spread +// over about 16 q_ln2, as int8 dot products at head_dim 64 give there. Same per-token accounting as `pv`. +// sslm_sites_bench q31 [--repeat=R] +// Slice S5: one query head's Q31 scores (the Qwen3 QK-norm path) at head_dim 128 (Qwen3-0.6B's), over one KV +// head's key rows, ratios in [2^29, 2^31] (inside the loader's [1, 2^31]). Times the layer loops' v1.9.0 +// per-key QkQ31Score loop and QkQ31ScoreRow in the same process, alternating, so their difference is S5's +// own effect on whichever tier the library runs. Same per-token accounting as `pv`, at Qwen3-0.6B's depth +// (28 layers x 16 query heads). +// sslm_sites_bench prefill [--layers=L] [--repeat=R] +// One sslm_prefill of T token ids at chunk_budget = T; best-of-R ms per prompt token. +// sslm_sites_bench decode [--layers=L] [--repeat=R] +// Prefill `context` token ids, then D greedy sslm_decode_step calls; best-of-R ms per decode token. +// +// Token ids are 1 + (i mod 200), inside every synthetic artifact's vocabulary (tools/t2147's "ids:N"). +// Slice S1's method (docs/attention-rowsites/s1/bench.md): kernel mode's per-call savings times the +// sites' per-token counts at full depth, checked against prefill and decode on reduced-layer artifacts. + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "superslm/checked_chain_funnel.h" +#include "superslm/forward_sites.h" +#include "superslm/intmath.h" +#include "superslm/matmul.h" +#include "superslm/silu_lut_canonical.h" +#include "superslm/sslm_abi.h" + +namespace { + +using Clock = std::chrono::steady_clock; + +struct Rng { + uint64_t s; + explicit Rng(uint64_t seed) : s(seed) {} + uint64_t Next() { + s += 0x9e3779b97f4a7c15ULL; + uint64_t z = s; + z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL; + z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL; + return z ^ (z >> 31); + } + int64_t InRange(int64_t lo, int64_t hi) { + return lo + static_cast(Next() % (static_cast(hi - lo) + 1ULL)); + } +}; + +volatile int64_t g_sink = 0; + +template +double BestMicrosPerCall(int repeat, int calls, F&& f) { + double best = 1e300; + for (int r = 0; r < repeat; ++r) { + const auto t0 = Clock::now(); + for (int c = 0; c < calls; ++c) f(c); + const auto t1 = Clock::now(); + best = std::min(best, std::chrono::duration(t1 - t0).count() / calls); + } + return best; +} + +int KernelMode(int repeat) { + using superslm::CarriedScale; + constexpr int kRows = 16; + Rng rng(0x5155454E424E4348ULL); + const size_t hidden = 896, inter = 4864; + std::vector> h(kRows, std::vector(hidden)), gate(kRows, std::vector(inter)), + up(kRows, std::vector(inter)), branch(kRows, std::vector(hidden)), + stream(kRows, std::vector(hidden)); + for (int r = 0; r < kRows; ++r) { + for (auto& v : h[r]) v = static_cast(rng.InRange(-127, 127)); + for (auto& v : gate[r]) v = static_cast(rng.InRange(-127, 127)); + for (auto& v : up[r]) v = static_cast(rng.InRange(-127, 127)); + for (auto& v : branch[r]) v = static_cast(rng.InRange(-127, 127)); + for (auto& v : stream[r]) v = static_cast(rng.InRange(-127, 127)); + } + std::vector g(hidden); + for (auto& v : g) v = static_cast(rng.InRange(-300, 300)); + const CarriedScale unit{INT64_C(1073741824), 0}; + std::vector out(inter); + CarriedScale scale{}; + int bad = 0; + + const double norm = BestMicrosPerCall(repeat, 256, [&](int c) { + bad += superslm::RmsNormSite(h[c % kRows].data(), g.data(), hidden, CarriedScale{}, unit, out.data(), &scale) != + superslm::SslmForwardStatus::Ok; + }); + const double silu = BestMicrosPerCall(repeat, 64, [&](int c) { + bad += superslm::MlpActSite(gate[c % kRows].data(), {INT64_C(1073741824), -34}, up[c % kRows].data(), + {INT64_C(1340958474), -18}, inter, superslm::kSiluLutCanonicalTable, unit, + out.data(), &scale) != superslm::SslmForwardStatus::Ok; + }); + const double residual = BestMicrosPerCall(repeat, 256, [&](int c) { + bad += superslm::ResidualReconcileSite(branch[c % kRows].data(), {INT64_C(1234567890), -37}, + stream[c % kRows].data(), {INT64_C(1987654321), -44}, hidden, unit, + out.data(), &scale) != superslm::SslmForwardStatus::Ok; + }); + g_sink = g_sink + out[0]; + std::printf("kernel best-of-%d us/call: rmsnorm_896 %.3f mlp_act_4864 %.3f residual_896 %.3f (refused %d)\n", + repeat, norm, silu, residual, bad); + return bad == 0 ? 0 : 1; +} + +// Slice S2's bench (see the header): the prob·V call at the 0.5B's head_dim over realistic rows. +int PvMode(int repeat) { + constexpr size_t kHeadDim = 64, kMaxWidth = 1025, kLayers = 24, kHeads = 14; + Rng rng(0x5332505642454E43ULL); + std::vector values(kMaxWidth * kHeadDim); + for (auto& v : values) v = static_cast(rng.InRange(-128, 127)); + // probs[w] is a realistic row of width w: weights 2^(0..12) with jitter, p = floor(e * 2^15 / Sum e). + std::vector> probs(kMaxWidth + 1); + for (size_t w = 1; w <= kMaxWidth; ++w) { + std::vector e(w); + int64_t total = 0; + for (auto& x : e) { + const int sh = static_cast(rng.InRange(0, 12)); + x = (INT64_C(1) << sh) + rng.InRange(0, (INT64_C(1) << sh) - 1); + total += x; + } + probs[w].resize(w); + for (size_t k = 0; k < w; ++k) probs[w][k] = (e[k] << 15) / total; + } + std::vector out(kHeadDim); + const auto call = [&](size_t w) { + superslm::GemmProbQ15Accumulate(probs[w].data(), values.data(), w, kHeadDim, out.data()); + g_sink = g_sink + out[0]; + }; + // Single widths. + std::printf("pv best-of-%d us/call at head_dim 64:", repeat); + for (size_t w : {size_t{1}, size_t{128}, size_t{301}, size_t{512}, size_t{601}, size_t{1024}}) { + const int calls = w < 64 ? 4096 : 256; + std::printf(" w=%zu %.4f", w, BestMicrosPerCall(repeat, calls, [&](int) { call(w); })); + } + std::printf("\n"); + // Prefill: every width 1..T once (one head of one layer), best of R, scaled to 24 layers x 14 heads / T. + for (size_t T : {size_t{128}, size_t{512}, size_t{1024}}) { + const double us = BestMicrosPerCall(repeat, 1, [&](int) { + for (size_t w = 1; w <= T; ++w) call(w); + }); + std::printf("pv prefill T=%zu: %.4f ms/token at 24 layers x 14 heads\n", T, + us * kLayers * kHeads / static_cast(T) / 1000.0); + } + // Decode at context C: one call at width C + 1 per head per layer. + for (size_t C : {size_t{300}, size_t{600}}) { + const double us = BestMicrosPerCall(repeat, 256, [&](int) { call(C + 1); }); + std::printf("pv decode ctx=%zu: %.4f ms/token at 24 layers x 14 heads\n", C, us * kLayers * kHeads / 1000.0); + } + return 0; +} + +// Slice S3's bench (see the header): the funnel call at the 0.5B's two widths. +int RequantMode(int repeat) { + using superslm::CarriedScale; + constexpr int kRows = 16; + constexpr size_t kLayers = 24; + Rng rng(0x5333524551424E43ULL); + const CarriedScale unit{INT64_C(1073741824), 0}; + int bad = 0; + double us[2] = {0, 0}; + const size_t widths[2] = {896, 4864}; + for (int wi = 0; wi < 2; ++wi) { + const size_t n = widths[wi]; + std::vector> rows(kRows, std::vector(n)); + for (int r = 0; r < kRows; ++r) { + const int64_t amp = INT64_C(1) << (16 + r % 16); // max-abs 2^16 .. 2^31 + for (auto& v : rows[r]) v = rng.InRange(-amp, amp); + rows[r][static_cast(r) % n] = amp; + } + std::vector out(n); + CarriedScale scale{}; + us[wi] = BestMicrosPerCall(repeat, 256, [&](int c) { + bad += superslm::RequantChainChecked(rows[static_cast(c) % kRows].data(), n, + std::span{}, unit, out.data(), &scale) + .status != superslm::SslmForwardStatus::Ok; + g_sink = g_sink + out[0]; + }); + } + std::printf("requant best-of-%d us/call: funnel_896 %.4f funnel_4864 %.4f (refused %d)\n", repeat, us[0], us[1], + bad); + std::printf("requant per token: %.4f ms at 24 layers x (8 x 896 + 3 x 4864) + embed\n", + (static_cast(kLayers) * (8 * us[0] + 3 * us[1]) + us[0]) / 1000.0); + return bad == 0 ? 0 : 1; +} + +// Slice S4's bench (see the header): the softmax row at the forward's widths and constants. +int SoftmaxMode(int repeat) { + constexpr size_t kMaxWidth = 1025, kLayers = 24, kHeads = 14; + Rng rng(0x5334534F46544D58ULL); + std::vector> triples; + while (triples.size() < 16) { + const int64_t m = (INT64_C(1) << 30) + rng.InRange(0, (INT64_C(1) << 30) - 1); + const int64_t e = rng.InRange(-44, -34); + int64_t a = 0, b = 0, c = 0; + if (superslm::IExpScaleConstants(m, e, superslm::kIExpLn2Q, 30, superslm::kIExpBQ, 30, superslm::kIExpCaQ, 30, &a, + &b, &c) == superslm::IExpScaleDomain::kOk && + a >= 347 && a <= 944) + triples.push_back({a, b, c}); + } + struct Row { + std::vector scores; + std::array t; + }; + std::vector rows(kMaxWidth + 1); + for (size_t w = 1; w <= kMaxWidth; ++w) { + rows[w].t = triples[w % triples.size()]; + rows[w].scores.resize(w); + const int64_t half = 8 * rows[w].t[0]; + for (auto& x : rows[w].scores) x = rng.InRange(-half, half); + } + std::vector out(kMaxWidth); + int bad = 0; + const auto call = [&](size_t w) { + const Row& r = rows[w]; + bad += !superslm::SoftmaxRowQ15(r.scores.data(), w, r.t[0], r.t[1], r.t[2], out.data()); + g_sink = g_sink + out[0]; + }; + int64_t lo = triples[0][0], hi = lo; + for (const auto& t : triples) lo = std::min(lo, t[0]), hi = std::max(hi, t[0]); + std::printf("softmax constants: %zu triples, q_ln2 %lld..%lld\n", triples.size(), static_cast(lo), + static_cast(hi)); + std::printf("softmax best-of-%d us/call:", repeat); + for (size_t w : {size_t{1}, size_t{128}, size_t{301}, size_t{512}, size_t{601}, size_t{1024}}) { + const int calls = w < 64 ? 4096 : 256; + std::printf(" w=%zu %.4f", w, BestMicrosPerCall(repeat, calls, [&](int) { call(w); })); + } + std::printf("\n"); + for (size_t T : {size_t{128}, size_t{512}, size_t{1024}}) { + const double us = BestMicrosPerCall(repeat, 1, [&](int) { + for (size_t w = 1; w <= T; ++w) call(w); + }); + std::printf("softmax prefill T=%zu: %.4f ms/token at 24 layers x 14 heads\n", T, + us * kLayers * kHeads / static_cast(T) / 1000.0); + } + for (size_t C : {size_t{300}, size_t{600}}) { + const double us = BestMicrosPerCall(repeat, 256, [&](int) { call(C + 1); }); + std::printf("softmax decode ctx=%zu: %.4f ms/token at 24 layers x 14 heads\n", C, + us * kLayers * kHeads / 1000.0); + } + std::printf("softmax refused rows: %d\n", bad); + return bad == 0 ? 0 : 1; +} + +// Slice S5's bench (see the header): the Q31 score row against the per-key loop it replaces. +int Q31Mode(int repeat) { + constexpr size_t kMaxWidth = 1025, kHeadDim = 128, kLayers = 28, kHeads = 16; + Rng rng(0x5335513331524F57ULL); + std::vector q(kHeadDim), keys(kMaxWidth * kHeadDim); + std::vector ratio(kHeadDim), out(kMaxWidth); + for (auto& x : q) x = static_cast(rng.InRange(-127, 127)); + for (auto& x : keys) x = static_cast(rng.InRange(-127, 127)); + for (auto& r : ratio) r = rng.InRange(INT64_C(1) << 29, INT64_C(1) << 31); + const auto per_key = [&](size_t w) { + for (size_t j = 0; j < w; ++j) + out[j] = superslm::QkQ31Score(q.data(), keys.data() + j * kHeadDim, ratio.data(), kHeadDim); + g_sink = g_sink + out[w - 1]; + }; + const auto row = [&](size_t w) { + superslm::QkQ31ScoreRow(q.data(), keys.data(), ratio.data(), kHeadDim, w, out.data()); + g_sink = g_sink + out[w - 1]; + }; + // Equality first: the row must match the per-key loop on every key. + std::vector ref(kMaxWidth); + for (size_t j = 0; j < kMaxWidth; ++j) + ref[j] = superslm::QkQ31Score(q.data(), keys.data() + j * kHeadDim, ratio.data(), kHeadDim); + row(kMaxWidth); + int bad = 0; + for (size_t j = 0; j < kMaxWidth; ++j) bad += out[j] != ref[j]; + std::printf("q31 head_dim %zu, ratios in [2^29, 2^31]; row vs per-key mismatches: %d\n", kHeadDim, bad); + std::printf("q31 best-of-%d ns per (head x key), per-key / row:", repeat); + for (size_t w : {size_t{1}, size_t{8}, size_t{64}, size_t{128}, size_t{512}, size_t{1024}}) { + const int calls = w < 64 ? 2048 : w < 512 ? 128 : 32; + const double a = BestMicrosPerCall(repeat, calls, [&](int) { per_key(w); }); + const double b = BestMicrosPerCall(repeat, calls, [&](int) { row(w); }); + std::printf(" w=%zu %.2f / %.2f", w, 1000.0 * a / static_cast(w), + 1000.0 * b / static_cast(w)); + } + std::printf("\n"); + for (size_t T : {size_t{128}, size_t{512}, size_t{1024}}) { + const double a = BestMicrosPerCall(repeat, 1, [&](int) { for (size_t w = 1; w <= T; ++w) per_key(w); }); + const double b = BestMicrosPerCall(repeat, 1, [&](int) { for (size_t w = 1; w <= T; ++w) row(w); }); + const double f = static_cast(kLayers * kHeads) / static_cast(T) / 1000.0; + std::printf("q31 prefill T=%zu: per-key %.4f, row %.4f ms/token at 28 layers x 16 heads\n", T, a * f, b * f); + } + for (size_t C : {size_t{300}, size_t{600}}) { + const double a = BestMicrosPerCall(repeat, 64, [&](int) { per_key(C + 1); }); + const double b = BestMicrosPerCall(repeat, 64, [&](int) { row(C + 1); }); + const double f = static_cast(kLayers * kHeads) / 1000.0; + std::printf("q31 decode ctx=%zu: per-key %.4f, row %.4f ms/token at 28 layers x 16 heads\n", C, a * f, b * f); + } + return bad == 0 ? 0 : 1; +} + +bool ReadFile(const char* path, std::vector& out) { + std::ifstream f(path, std::ios::binary); + if (!f) return false; + out.assign(std::istreambuf_iterator(f), std::istreambuf_iterator()); + return !out.empty(); +} + +[[noreturn]] void Fail(const char* what, int st) { + std::fprintf(stderr, "sslm_sites_bench: %s failed (status %d)\n", what, st); + std::exit(2); +} + +// One fresh model, pool, workspace and sequence; prefill T ids in one call, then D decode steps. +// Returns the prefill and decode wall times in ms. +void RunOnce(const std::vector& bytes, int32_t layers, int32_t T, int32_t D, double* prefill_ms, + double* decode_ms) { + sslm_model model = nullptr; + sslm_status st = sslm_model_map(bytes.data(), bytes.size(), &model); + if (st != SSLM_OK) Fail("sslm_model_map", st); + const size_t kv_required = sslm_kv_block_size(model) + sslm_kv_pool_overhead_size(model, 1); + std::vector pool_raw(kv_required + SSLM_ABI_ALIGNMENT_BYTES); + void* pa = pool_raw.data(); + size_t ps = pool_raw.size(); + std::align(SSLM_ABI_ALIGNMENT_BYTES, kv_required, pa, ps); + sslm_kv_pool pool = nullptr; + if ((st = sslm_kv_pool_create(model, pa, kv_required, 1, &pool)) != SSLM_OK) Fail("sslm_kv_pool_create", st); + sslm_config cfg{}; + cfg.max_batch = 1; + cfg.max_chunk_budget = T; + cfg.max_layer_budget = layers; + const size_t wsb = sslm_workspace_size(model, &cfg); + std::vector ws_raw(wsb + SSLM_ABI_ALIGNMENT_BYTES); + void* wa = ws_raw.data(); + size_t wsz = ws_raw.size(); + std::align(SSLM_ABI_ALIGNMENT_BYTES, wsb, wa, wsz); + sslm_workspace ws = nullptr; + if ((st = sslm_workspace_create(model, &cfg, wa, wsb, &ws)) != SSLM_OK) Fail("sslm_workspace_create", st); + sslm_seq seq = nullptr; + if ((st = sslm_seq_create(model, &pool, &seq)) != SSLM_OK) Fail("sslm_seq_create", st); + std::vector ids(static_cast(T)); + for (int32_t i = 0; i < T; ++i) ids[static_cast(i)] = 1 + (i % 200); + const auto t0 = Clock::now(); + int32_t consumed = 0; + if ((st = sslm_prefill(model, seq, ids.data(), T, T, SSLM_SPAN_PROMPT, ws, &consumed)) != SSLM_OK || consumed != T) + Fail("sslm_prefill", st); + const auto t1 = Clock::now(); + if (D > 0) { + sslm_decode_params params{}; + if ((st = sslm_decode_params_init(model, SSLM_DECODE_MODE_GREEDY, layers, ¶ms)) != SSLM_OK) + Fail("sslm_decode_params_init", st); + for (int32_t d = 0; d < D; ++d) { + int32_t next = -1; + if ((st = sslm_decode_step(model, &seq, 1, ¶ms, ws, &next)) != SSLM_OK) Fail("sslm_decode_step", st); + } + } + const auto t2 = Clock::now(); + *prefill_ms = std::chrono::duration(t1 - t0).count(); + *decode_ms = std::chrono::duration(t2 - t1).count(); + sslm_seq_release(seq); + sslm_workspace_destroy(ws); + sslm_kv_pool_destroy(pool); + sslm_model_unmap(model); +} + +} // namespace + +int main(int argc, char** argv) { + if (argc < 2) { + std::fprintf(stderr, + "usage: %s kernel [--repeat=R] | pv [--repeat=R] | requant [--repeat=R] | softmax [--repeat=R] | q31 [--repeat=R] | prefill [--layers=L] [--repeat=R] | " + "decode [--layers=L] [--repeat=R]\n", + argv[0]); + return 2; + } + int repeat = 7; + int32_t layers = 1; + for (int i = 2; i < argc; ++i) { + if (std::strncmp(argv[i], "--repeat=", 9) == 0) repeat = std::max(1, std::atoi(argv[i] + 9)); + if (std::strncmp(argv[i], "--layers=", 9) == 0) layers = std::max(1, std::atoi(argv[i] + 9)); + } + const std::string mode = argv[1]; + if (mode == "kernel") return KernelMode(repeat); + if (mode == "pv") return PvMode(repeat); + if (mode == "requant") return RequantMode(repeat); + if (mode == "softmax") return SoftmaxMode(repeat); + if (mode == "q31") return Q31Mode(repeat); + if ((mode == "prefill" && argc >= 4) || (mode == "decode" && argc >= 5)) { + std::vector bytes; + if (!ReadFile(argv[2], bytes)) Fail("reading the artifact", 0); + const int32_t T = std::atoi(argv[3]); + const int32_t D = mode == "decode" ? std::atoi(argv[4]) : 0; + double best = 1e300; + for (int r = 0; r < repeat; ++r) { + double p = 0, d = 0; + RunOnce(bytes, layers, T, D, &p, &d); + best = std::min(best, mode == "prefill" ? p / T : d / D); + } + std::printf("%s best-of-%d ms/token: %.4f (T=%d D=%d layers=%d)\n", mode.c_str(), repeat, best, T, D, layers); + return 0; + } + std::fprintf(stderr, "bad arguments\n"); + return 2; +} diff --git a/tools/t2147_chunk_batched_pins.cpp b/tools/t2147_chunk_batched_pins.cpp index ef654e79..f7d65288 100644 --- a/tools/t2147_chunk_batched_pins.cpp +++ b/tools/t2147_chunk_batched_pins.cpp @@ -36,6 +36,11 @@ // sets max_layer_budget to the artifact's own layer count (a reduced-layer artifact has fewer than 28). // --repeat=R times each arm of --speedup R times and reports the best (the host is shared; best of R is // the noise-robust figure). +// +// Blob-protocol options (attention and per-row sites plan, §3.3's end-to-end blobs): with --dump-blob, +// --chunk-budget=B prefills in calls of at most B tokens (default: the whole prompt in one call), and +// --decode=D then runs D greedy sslm_decode_step calls before the blob is saved; the decoded token ids +// are printed. Both arms of a base-vs-candidate comparison run this same source. #include #include @@ -98,7 +103,8 @@ int32_t g_num_hidden_layers = 28; std::vector PrefillAndSave(const std::vector& bytes, const int32_t* tokens, int32_t count, const std::vector& splits, int32_t chunk_budget_per_call, sslm_span_kind kind, - const char* schema_name, double* out_elapsed_ms) { + const char* schema_name, double* out_elapsed_ms, + int32_t decode_steps = 0, std::vector* out_decoded = nullptr) { sslm_model model = nullptr; sslm_status st = sslm_model_map(bytes.data(), bytes.size(), &model); if (st != SSLM_OK || !model) Fail("sslm_model_map", static_cast(st)); @@ -163,6 +169,17 @@ std::vector PrefillAndSave(const std::vector& bytes, const int } offset += span_len; } + if (decode_steps > 0) { + sslm_decode_params params{}; + st = sslm_decode_params_init(model, SSLM_DECODE_MODE_GREEDY, g_num_hidden_layers, ¶ms); + if (st != SSLM_OK) Fail("sslm_decode_params_init", static_cast(st)); + for (int32_t d = 0; d < decode_steps; ++d) { + int32_t next = -1; + st = sslm_decode_step(model, &seq, 1, ¶ms, ws, &next); + if (st != SSLM_OK || next < 0) Fail("sslm_decode_step", static_cast(st)); + if (out_decoded) out_decoded->push_back(next); + } + } const auto t1 = std::chrono::steady_clock::now(); if (out_elapsed_ms) { *out_elapsed_ms = std::chrono::duration(t1 - t0).count(); @@ -200,6 +217,8 @@ int main(int argc, char** argv) { int32_t boundary_sweep = -1; // -1 = every split point std::string dump_blob_path; int32_t repeat = 1; + int32_t dump_chunk_budget = 0; // 0: the whole prompt in one call + int32_t dump_decode_steps = 0; for (int i = 3; i < argc; ++i) { const std::string a = argv[i]; if (a == "--speedup") do_speedup = true; @@ -207,6 +226,8 @@ int main(int argc, char** argv) { else if (a.rfind("--dump-blob=", 0) == 0) dump_blob_path = a.substr(12); else if (a.rfind("--layers=", 0) == 0) g_num_hidden_layers = std::atoi(a.c_str() + 9); else if (a.rfind("--repeat=", 0) == 0) repeat = std::max(1, std::atoi(a.c_str() + 9)); + else if (a.rfind("--chunk-budget=", 0) == 0) dump_chunk_budget = std::max(1, std::atoi(a.c_str() + 15)); + else if (a.rfind("--decode=", 0) == 0) dump_decode_steps = std::max(0, std::atoi(a.c_str() + 9)); } std::vector bytes; @@ -250,8 +271,15 @@ int main(int argc, char** argv) { // "reproduces the existing per-token path bit-for-bit" oracle (T-2133 §9 C4's own shape). if (!dump_blob_path.empty()) { const std::vector one_call_dump = {N}; - const auto blob = PrefillAndSave(bytes, tokens.data(), N, one_call_dump, /*chunk_budget=*/N, - SSLM_SPAN_PROMPT, nullptr, nullptr); + std::vector decoded; + const auto blob = PrefillAndSave(bytes, tokens.data(), N, one_call_dump, + dump_chunk_budget > 0 ? dump_chunk_budget : N, SSLM_SPAN_PROMPT, nullptr, + nullptr, dump_decode_steps, &decoded); + if (dump_decode_steps > 0) { + std::printf("decoded %d tokens:", dump_decode_steps); + for (int32_t t : decoded) std::printf(" %d", t); + std::printf("\n"); + } std::ofstream out(dump_blob_path, std::ios::binary); if (!out) Fail("open dump-blob output", 0); out.write(reinterpret_cast(blob.data()), static_cast(blob.size()));