Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
ac30f47
[skip ci] Attention and per-row sites, S1: the red cells for the per-…
claude Sep 29, 2026
2b3dae9
[skip ci] Attention and per-row sites, S1: the per-row tables, green …
claude Sep 29, 2026
3aeb698
[skip ci] Attention and per-row sites, S1: evidence, the 3.S1 cell, t…
claude Sep 29, 2026
ccfe19b
Attention and per-row sites, S2: the red cells for prob-V on int16 mu…
claude Sep 29, 2026
b2765f2
Attention and per-row sites, S2: prob-V on int16 multiply-add, green …
claude Sep 29, 2026
4f83ade
Attention and per-row sites, S2: evidence, the blocking rows, the bench
claude Sep 29, 2026
a591cb7
Attention and per-row sites, S3: the red cells for the requant row leaf
claude Sep 29, 2026
c29f258
Attention and per-row sites, S3: the requant row leaf in 64-bit lanes…
claude Sep 29, 2026
1f09e13
Attention and per-row sites, S3: evidence, the bench
claude Sep 29, 2026
19e7ac0
Attention and per-row sites, S4: the red cells for the guarded softmax
dansupergameprogrammer Sep 29, 2026
597f8b1
Attention and per-row sites, S4: the guarded softmax, green on every …
dansupergameprogrammer Sep 29, 2026
65c2b59
Attention and per-row sites, S4: evidence, the bench
dansupergameprogrammer Sep 29, 2026
288e2dc
Attention and per-row sites: re-cite the CPU guard ladder, clear a CI…
claude Sep 29, 2026
4a2674f
Attention and per-row sites, S5: the red cells for the Q31 score row
dansupergameprogrammer Sep 29, 2026
8facc48
Attention and per-row sites, S5: the Q31 score row, green on every tier
dansupergameprogrammer Sep 29, 2026
766a22f
Attention and per-row sites, S5: evidence, the bench
dansupergameprogrammer Sep 29, 2026
c28f171
Attention and per-row sites, S5: re-cite the CPU guard ladder [skip ci]
claude Sep 29, 2026
9763947
Attention and per-row sites: provisional branch-coverage floors for m…
claude Sep 29, 2026
4d7acc2
Attention and per-row sites: run PreflightScanWscFolds on every build…
claude Sep 29, 2026
0242af8
Attention and per-row sites: pin matmul.cpp and intmath.cpp floors fr…
claude Sep 29, 2026
99058f3
Attention and per-row sites: pin generated golden-pin headers to LF […
claude Sep 30, 2026
1cfe368
Attention and per-row sites: mark the new sites declarations internal
claude Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .gitattributes
Original file line number Diff line number Diff line change
Expand Up @@ -31,3 +31,8 @@ tests/golden/** -text
# T-2814: the recorded dumpbin read that tests/t2807-api-slot/consumed_symbols.txt is generated from, and
# whose SHA-256 its header states. Byte-exact on every platform.
tests/t2807-api-slot/recorded/*.tsv -text

# Generated golden-pin headers claim byte-for-byte regeneration by their
# generators, which emit LF; pin them to LF so a Windows checkout matches.
tests/attn_rowsite_golden_pin.h text eol=lf
tests/matmul_tiled_golden_pin.h text eol=lf
29 changes: 23 additions & 6 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -361,7 +361,8 @@ jobs:
# built by MSVC and by clang-cl, the full suite and the digest each. Before these legs no MSVC build ever executed
# an AVX2 or AVX-512 kernel in CI (the auto legs dispatch on whatever the runner has). The AVX-512
# leg builds with SUPERSLM_TILED_AVX512_MSVC=1, so it runs the tiled AVX-512 kernel that the
# default MSVC build still holds off; it probes the CPU first (IsProcessorFeaturePresent 41 =
# default MSVC build still holds off, and (attention and per-row sites plan §3.2) with
# SUPERSLM_SITES_AVX512_MSVC=1, so it runs the attention kernels' AVX-512 bodies too; it probes the CPU first (IsProcessorFeaturePresent 41 =
# AVX-512F, which the hosted image's CPUs pair with BW) and reports SKIPPED without it.
windows-msvc-avx2-forced:
runs-on: windows-latest
Expand Down Expand Up @@ -389,7 +390,7 @@ jobs:
runs-on: windows-latest
steps:
- uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
- run: cmake -B build -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1"
- run: cmake -B build -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1 /DSUPERSLM_SITES_AVX512_MSVC=1"
- run: cmake --build build --config Release --target superslm_tests_avx512_forced sslm_axis_digest_avx512_forced
- name: Probe for AVX-512, run the forced suite and extract the GLOBAL digest if present
shell: pwsh
Expand Down Expand Up @@ -439,7 +440,7 @@ jobs:
runs-on: windows-latest
steps:
- uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
- run: cmake -B build -T ClangCL -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1"
- run: cmake -B build -T ClangCL -DCMAKE_CXX_FLAGS="/DWIN32 /D_WINDOWS /EHsc /DSUPERSLM_TILED_AVX512_MSVC=1 /DSUPERSLM_SITES_AVX512_MSVC=1"
- run: cmake --build build --config Release --target superslm_tests_avx512_forced sslm_axis_digest_avx512_forced
- name: Probe for AVX-512, run the forced suite and extract the GLOBAL digest if present
shell: pwsh
Expand All @@ -466,7 +467,10 @@ jobs:
# Tiled-matmul plan slice 1, cell 11.3: every tiled and packer symbol in the matmul object is local
# (never global, weak, unique or COMDAT), in the auto and both forced builds; the build-configuration
# record is the single external exception. The check's own vitality plants are recorded in
# docs/tiled-matmul-slice1-progress.md.
# docs/tiled-matmul-slice1-progress.md. The attention and per-row sites plan's slice S3 adds the
# intmath.cpp objects, whose requant row bodies are that file's first target-attributed functions;
# slice S4 adds the softmax fast path's bodies to the same objects; slice S5 adds the forward_sites.cpp
# objects, for the Q31 score row's bodies.
tiled-matmul-linkage:
runs-on: ubuntu-latest
steps:
Expand All @@ -477,7 +481,13 @@ jobs:
python3 tools/ci/check_tiled_matmul_linkage.py \
build/CMakeFiles/superslm.dir/src/matmul.cpp.o \
build/CMakeFiles/superslm_avx2_forced.dir/src/matmul.cpp.o \
build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o
build/CMakeFiles/superslm_avx512_forced.dir/src/matmul.cpp.o \
build/CMakeFiles/superslm.dir/src/intmath.cpp.o \
build/CMakeFiles/superslm_avx2_forced.dir/src/intmath.cpp.o \
build/CMakeFiles/superslm_avx512_forced.dir/src/intmath.cpp.o \
build/CMakeFiles/superslm.dir/src/forward/forward_sites.cpp.o \
build/CMakeFiles/superslm_avx2_forced.dir/src/forward/forward_sites.cpp.o \
build/CMakeFiles/superslm_avx512_forced.dir/src/forward/forward_sites.cpp.o

# Ratified into this slot's scope (Sec7.1, D-SLM170): the first executed
# x64-vs-ARM comparison this plan's own text (SuperSLM_Plan.md:1436, T-182)
Expand Down Expand Up @@ -843,6 +853,12 @@ jobs:
python3 tools/ci/check_branch_coverage_floors.py \
--record-floors-to build/measured_branch_coverage_floors.json \
build/coverage.json
# Re-pinning a floor copies this leg's recorded value into tools/ci/branch_coverage_floors.json,
# and the artifact store that holds the record is not reachable from every session that does
# the re-pin. Printing the record puts the exact values in the job log as well.
- name: Print the recorded branch-coverage floors
if: always()
run: cat build/measured_branch_coverage_floors.json
- name: Check per-file branch-coverage floors
run: python3 tools/ci/check_branch_coverage_floors.py build/coverage.json
# Tiled-matmul plan slice 1, cell 11.4: nonzero counts on the tiled kernel's named spans (flush
Expand Down Expand Up @@ -975,7 +991,8 @@ jobs:
# C1): the checked chain funnel's structural half -- no translation unit in
# the forward composition may name a funnel leaf (MaxAbsReduce,
# MaxAbsReduceWide, RowBoundsWide, NormalizeScale, DynamicScaleReciprocal,
# RequantTokenCode, RequantTokenCodeWide, NarrowAccumulatorToI32) directly,
# RequantTokenCode, RequantTokenCodeWide, NarrowAccumulatorToI32, and since the
# attention and per-row sites plan's slice S3 the row leaf RequantRowWide) directly,
# outside the funnel's own file and the leaf certification TUs. Mirrors the
# two precedents already CI-wired (tests/check_no_pow_operator.py above,
# tools/ci/check_bad_alloc_contract.py's own job): a test step, then the
Expand Down
46 changes: 46 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,52 @@ set in `check_fp_free_scan.py`. Each is 32-bit integer arithmetic on xmm lanes (
modular adds, rotates, shifts and boolean ops; no rounding, no MXCSR, no floating-point operand).
The SHA-1 instructions stay rejected.

On the AVX2 and AVX-512 tiers, attention's probability-times-value step (`GemmProbQ15Accumulate`)
now runs on 16-bit multiply-add when the head dimension is a multiple of 16 and the probability
row fits 16 bits (every p in [0, 32767], sum at most 2^15). Other rows take the 1.9.0 loop. This
covers every softmax row except a one-hot row. Outputs are bit-identical to 1.9.0: the same
tokens, save blobs and digest. Engine level, on one cloud host, at head dimension 64 the step is
about 16-21x faster. At Qwen2.5-0.5B depth (24 layers, 14 heads) that saves about 0.8 / 3.4 /
6.7 ms per prompt token at 128 / 512 / 1,024 tokens, and about 4 ms per decode token at context
300. MSVC and clang-cl builds keep the AVX-512 tier on the 1.9.0 loop until it has executed
there; `SUPERSLM_SITES_AVX512_MSVC=1` turns it on. There is no ABI, format or status change.

On the AVX2 and AVX-512 tiers, the requantization step every checked projection, norm, activation
and residual ends in (the funnel's per-element conversion to int8 codes) now runs 4 or 8 elements
at a time in 64-bit integer lanes, through the new row function `RequantRowWide`. Every code is
bit-identical to 1.9.0's per-element `RequantTokenCodeWide`: the same tokens, save blobs and
digest. Engine level, on one cloud host, a funnel call at Qwen2.5-0.5B's widths (896 and 4,864)
is about 4.5x faster on AVX2 and 5.5x on AVX-512, which saves about 2.5 ms per token at 24
layers, prefill and decode alike. The scalar and SSE2 tiers keep the 1.9.0 loop. MSVC and
clang-cl builds keep the AVX-512 tier on it too, under the same `SUPERSLM_SITES_AVX512_MSVC`
switch. There is no ABI, format or status change.

On the AVX2 and AVX-512 tiers, attention's softmax row (`SoftmaxRowQ15`) now runs 4 or 8 elements
at a time when the row is inside a guard: width at most 2^14, q_ln2 >= 1, q_c >= 0,
q_b^2 + q_c in [1, 2^47], q_ln2 <= 2 q_b + 1 and every score within 2^61. The two divides per
element are integer estimates that one exact integer correction each way makes exact. Rows outside
the guard run the 1.9.0 body unchanged. Every probability and the returned bool are bit-identical
to 1.9.0: the same tokens, save blobs and digest. Engine level, on one cloud host, a row of 512
keys is about 3.6x faster on AVX2. At Qwen2.5-0.5B depth (24 layers, 14 heads) that saves about
0.1 / 0.46 / 0.91 ms per prompt token at 128 / 512 / 1,024 tokens, and about 0.53 ms per decode
token at context 300. A one-key row is slightly slower. The scalar and SSE2 tiers keep
the 1.9.0 body. MSVC and clang-cl builds keep the AVX-512 tier on it too, under the same
`SUPERSLM_SITES_AVX512_MSVC` switch. There is no ABI, format or status change.

On the AVX2 and AVX-512 tiers, the Q31 attention score used by QK-norm models (the Qwen3 path) is
now computed for all of a query head's keys in one call (`QkQ31ScoreRow`), instead of one
`QkQ31Score` call per key. Inside a guard (head_dim at most 512, every K-channel ratio in
[0, 2^32)) each channel's q x ratio product is split exactly into three 16-bit pieces, and each
piece's sum over the channels is a 16-bit multiply-add. The rounding is `RoundingDivideByPOT`'s, ties away from zero. Rows outside the
guard run the 1.9.0 per-key loop. Every score is bit-identical to 1.9.0: the same tokens, save
blobs and digest. Engine level, on one cloud host, a score costs about 14 ns per head and key
instead of about 410 on AVX2 (about 330 on AVX-512). At Qwen3-0.6B depth (28 layers, 16 heads)
that saves about 11 / 46 / 94 ms per prompt token at 128 / 512 / 1,024 tokens on AVX2, and about
55 ms per decode token at context 300. Qwen2.5 models do not take this path. A one-key row on
AVX-512 is slightly slower. The scalar and SSE2 tiers keep the 1.9.0 loop. MSVC and clang-cl
builds keep the AVX-512 tier on it too, under the same `SUPERSLM_SITES_AVX512_MSVC` switch. There
is no ABI, format or status change.

## [1.9.0] - 2026-09-25

`sslm_seq_save` writes a new save format, `SSB5`: the `SSB4` layout with the four per-site
Expand Down
32 changes: 30 additions & 2 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -91,7 +91,7 @@ else()
endif()

add_executable(superslm_tests tests/test_main.cpp tests/test_slm18x_saturation_census.cpp
tests/test_slm19x_schema_damped_greedy.cpp tests/test_tiled_gemm.cpp)
tests/test_slm19x_schema_damped_greedy.cpp tests/test_tiled_gemm.cpp tests/test_attn_rowsites.cpp)
target_link_libraries(superslm_tests PRIVATE superslm_test_injection)
if(MSVC)
target_compile_options(superslm_tests PRIVATE /W4 /fp:precise)
Expand Down Expand Up @@ -358,7 +358,7 @@ foreach(_sslm_tier SSE2 AVX2 AVX512)
# target_compile_definitions call.
add_executable(superslm_tests_${_sslm_tier_lower}_forced EXCLUDE_FROM_ALL tests/test_main.cpp
tests/test_slm18x_saturation_census.cpp tests/test_slm19x_schema_damped_greedy.cpp
tests/test_tiled_gemm.cpp)
tests/test_tiled_gemm.cpp tests/test_attn_rowsites.cpp)
target_link_libraries(superslm_tests_${_sslm_tier_lower}_forced PRIVATE superslm_${_sslm_tier_lower}_forced)
target_include_directories(superslm_tests_${_sslm_tier_lower}_forced PRIVATE tests)
if(MSVC)
Expand Down Expand Up @@ -417,6 +417,11 @@ foreach(_sslm_bench_tier AVX2 AVX512)
string(TOLOWER "${_sslm_bench_tier}" _sslm_bench_tier_lower)
add_executable(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced EXCLUDE_FROM_ALL tools/t2147_chunk_batched_pins.cpp)
target_link_libraries(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE superslm_${_sslm_bench_tier_lower}_forced)
# The forced library carries SUPERSLM_ENABLE_MATMUL_DISPATCH_INSTRUMENT PUBLIC, so this tool's own
# compile takes its instrument branch (#include "support/matmul_dispatch_instrument.h") and needs
# tests/ on its include path; without it the target does not compile (found by the attention and
# per-row sites plan's S1 bench).
target_include_directories(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE tests)
if(MSVC)
target_compile_options(sslm_bench_prefill_${_sslm_bench_tier_lower}_forced PRIVATE /W4 /fp:precise)
else()
Expand All @@ -432,6 +437,29 @@ endforeach()
# its D-infinity twin is the tiled kernel's own speedup on that tier. The bench calls only matmul.cpp's
# entry points, so the D-infinity library needs no other engine source. All EXCLUDE_FROM_ALL except the
# production bench, for the same non-x64 reason as the forced targets above.
# Attention and per-row sites plan, cell 10.1: the engine bench (tools/sslm_sites_bench.cpp; kernel, prefill
# and decode modes). Reports, not gates. Linked here against the production library; the base arm of a
# comparison is the same source linked against the base's library.
add_executable(sslm_sites_bench tools/sslm_sites_bench.cpp)
target_link_libraries(sslm_sites_bench PRIVATE superslm)
if(MSVC)
target_compile_options(sslm_sites_bench PRIVATE /W4 /fp:precise)
else()
target_compile_options(sslm_sites_bench PRIVATE -Wall -Wextra -ffp-contract=off)
endif()

# Attention and per-row sites plan, §3.3 evidence 3: the golden-pin generator. The committed pin
# (tests/attn_rowsite_golden_pin.h) is generated by this source built against the v1.9.0 TAG's library
# (recipe in the source's header); this target builds it against the current tree, where it must print
# the same hash -- a consistency check on the generator, never the pin's provenance.
add_executable(gen_attn_rowsite_golden tools/gen_attn_rowsite_golden.cpp)
target_link_libraries(gen_attn_rowsite_golden PRIVATE superslm)
if(MSVC)
target_compile_options(gen_attn_rowsite_golden PRIVATE /W4 /fp:precise)
else()
target_compile_options(gen_attn_rowsite_golden PRIVATE -Wall -Wextra -ffp-contract=off)
endif()

add_executable(sslm_gemm_bench tools/sslm_gemm_bench.cpp)
target_link_libraries(sslm_gemm_bench PRIVATE superslm)
if(MSVC)
Expand Down
2 changes: 1 addition & 1 deletion build.bat
Original file line number Diff line number Diff line change
Expand Up @@ -56,7 +56,7 @@ cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude /Itests /DSUPERSLM_ENA
src\forward\checked_chain_funnel.cpp src\forward\forward_sites.cpp src\decode_digest.cpp ^
src\sslm_abi.cpp src\damped_greedy_antilm.cpp src\damped_greedy_topk.cpp src\damped_greedy_phaseD.cpp src\damped_greedy_phaseD_loop.cpp ^
src\gpu\superslm_gpu.cpp src\gpu\gpu_1p0.cpp ^
tests\test_main.cpp tests\test_slm18x_saturation_census.cpp tests\test_slm19x_schema_damped_greedy.cpp tests\test_tiled_gemm.cpp /Fo:out\ /Fe:out\superslm_tests.exe ^
tests\test_main.cpp tests\test_slm18x_saturation_census.cpp tests\test_slm19x_schema_damped_greedy.cpp tests\test_tiled_gemm.cpp tests\test_attn_rowsites.cpp /Fo:out\ /Fe:out\superslm_tests.exe ^
/link d3d12.lib dxgi.lib dxguid.lib
if errorlevel 1 (
goto :hard_fail
Expand Down
2 changes: 1 addition & 1 deletion build_cert.bat
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ call %VSDEVCMD% -arch=x64 -no_logo
pushd %~dp0
if not exist out mkdir out
cl /nologo /std:c++20 /O2 /W4 /fp:precise /EHsc /Iinclude ^
src\intmath.cpp tests\cert_intmath.cpp /Fo:out\ /Fe:out\cert_intmath.exe
src\intmath.cpp src\matmul.cpp tests\cert_intmath.cpp /Fo:out\ /Fe:out\cert_intmath.exe
if errorlevel 1 (popd & exit /b 1)
popd
exit /b 0
Loading
Loading