Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
35 commits
Select commit Hold shift + click to select a range
6c2bd06
[None][feat] Add GLM-5.3-Flash support
ruocheng-nv Sep 14, 2026
439f801
[None][fix] Validate GLM FP8 KV cache and sparse forward contracts
ruocheng-nv Sep 14, 2026
9033427
[None][feat] Enable GLM-5.3-Flash attention DP with MTP and FP8 KV cache
ruocheng-nv Sep 15, 2026
bab6f06
[None][docs] Clarify GLM KV cache auto dtype behavior
ruocheng-nv Sep 16, 2026
55d9550
[None][fix] Honor all stop tokens in greedy sampling
ruocheng-nv Sep 16, 2026
86fcd93
[None][test] Include greedy stop-token regression in GPU CI
ruocheng-nv Sep 16, 2026
bc4f9d7
[None][refactor] Simplify GLM-5.3-Flash runtime and sparse backend
ruocheng-nv Sep 16, 2026
6e44815
[None][refactor] Trim GLM-5.3-Flash model helpers and comments
ruocheng-nv Sep 16, 2026
80a2ef9
[None][docs] Pin GLM-5.3-Flash deployment to Transformers 5.17.0
ruocheng-nv Sep 16, 2026
6795727
[None][refactor] Consolidate GLM sparse attention and KDA runtime setup
ruocheng-nv Sep 16, 2026
34a7706
[None][perf] Enable compact KDA decode for 64 heads at small batches
ruocheng-nv Sep 17, 2026
971d447
[None][docs] Simplify GLM-5.3-Flash supported-model note
ruocheng-nv Sep 17, 2026
e18a345
[None][fix] Honor GLM indexer head-weight strides
ruocheng-nv Sep 18, 2026
d765474
[None][docs] Trim GLM-5.3-Flash benchmark commentary
ruocheng-nv Sep 18, 2026
a02fbb6
[TRTLLM-16480][perf] Extend mHC pre-mapping tuning coverage
ruocheng-nv Sep 20, 2026
f048260
[TRTLLM-16480][perf] Optimize GLM-5.3-Flash small-batch decoding
ruocheng-nv Sep 21, 2026
5dea8f3
[TRTLLM-16018][fix] Fix GLM k-pool offsets and TopK fallback
ruocheng-nv Sep 21, 2026
901fc65
[None][refactor] Reuse FP32 linear for GLM router
ruocheng-nv Sep 22, 2026
2a02428
[None][refactor] Align GLM sparse attention with shared interfaces
ruocheng-nv Sep 22, 2026
76d5471
[None][refactor] Query GLM raw cache slots through the public layer API
ruocheng-nv Sep 22, 2026
5892029
[None][fix] Preserve GLM 64-head dispatch in the updated KDA backend
ruocheng-nv Sep 22, 2026
eaf9e32
[None][fix] Declare GLM transceiver preference and simplify attribute…
ruocheng-nv Sep 22, 2026
41b5db8
[None][refactor] Validate GLM sparse metadata and forward argument types
ruocheng-nv Sep 23, 2026
71a6860
[None][test] Add GLM-5.3-Flash coverage to GB300 CI
ruocheng-nv Sep 23, 2026
24fe42a
[None][fix] Separate GLM image and video preprocessing settings
ruocheng-nv Sep 23, 2026
2ce8304
[None][test] Reduce GLM-5.3-Flash end-to-end CI coverage
ruocheng-nv Sep 23, 2026
6bbd17c
[None][doc] Document GLM-5.3-Flash long-input benchmark results
ruocheng-nv Sep 23, 2026
55ce509
[None][doc] Restore GLM-5.3-Flash performance guide
ruocheng-nv Sep 23, 2026
c1140b4
[None][fix] Account for GLM indexer cache memory
ruocheng-nv Sep 24, 2026
de9f05c
[None][refactor] Address GLM-5.3-Flash review follow-ups
ruocheng-nv Sep 25, 2026
f6a796e
[None][chore] Trim GLM KV cache manager routing comments
ruocheng-nv Sep 25, 2026
70017b8
[None][test] Skip MoE A2A CFT regression under CUDA forward compatibi…
ruocheng-nv Sep 26, 2026
4fa804d
[None][test] Gate MoE A2A CFT skip on a newer user-mode driver
ruocheng-nv Sep 26, 2026
2d17aae
[None][fix] Gate MoE A2A CFT on the NVIDIA kernel driver version
zhangcl Sep 26, 2026
1ec4cfe
[None][doc] Update GLM-5.3-Flash 1K/1K performance chart
ruocheng-nv Sep 27, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion cpp/tensorrt_llm/kernels/kdaDecode/kdaDecode.h
Comment thread
BowenFu marked this conversation as resolved.
Original file line number Diff line number Diff line change
Expand Up @@ -31,7 +31,8 @@ constexpr int kCompactHeadsWorkThreshold = 144;
constexpr bool isSupportedHeadCount(int numHeads)
{
return numHeads == 1 || numHeads == 2 || numHeads == 3 || numHeads == 4 || numHeads == 6 || numHeads == 8
|| numHeads == 12 || numHeads == 16 || numHeads == 24 || numHeads == 32 || numHeads == 48 || numHeads == 96;
|| numHeads == 12 || numHeads == 16 || numHeads == 24 || numHeads == 32 || numHeads == 48 || numHeads == 64
|| numHeads == 96;
}

//! Select the compact-head kernel within the measured KDA decode work threshold.
Expand Down
1 change: 1 addition & 0 deletions cpp/tensorrt_llm/kernels/kdaDecode/kdaDecodeLegacy.cu
Original file line number Diff line number Diff line change
Expand Up @@ -1560,6 +1560,7 @@ void dispatch_kda_decode_heads(KdaDecodeLaunchParams const& p)
case 24: dispatch_kda_decode_layout<kCompact, 24>(p); break;
case 32: dispatch_kda_decode_layout<kCompact, 32>(p); break;
case 48: dispatch_kda_decode_layout<kCompact, 48>(p); break;
case 64: dispatch_kda_decode_layout<kCompact, 64>(p); break;
case 96: dispatch_kda_decode_layout<kCompact, 96>(p); break;
default:
if constexpr (kCompact)
Expand Down
1 change: 1 addition & 0 deletions cpp/tensorrt_llm/kernels/kdaDecode/kdaDecodeOptimized.cu
Original file line number Diff line number Diff line change
Expand Up @@ -740,6 +740,7 @@ void dispatchKdaDecodeOptimizedHeads(KdaDecodeParams const& params, cudaStream_t
case 24: TLLM_CUDA_CHECK((launchKernelSchedule<kSchedule, 24, kUpdateConvCache>(params, stream))); break;
case 32: TLLM_CUDA_CHECK((launchKernelSchedule<kSchedule, 32, kUpdateConvCache>(params, stream))); break;
case 48: TLLM_CUDA_CHECK((launchKernelSchedule<kSchedule, 48, kUpdateConvCache>(params, stream))); break;
case 64: TLLM_CUDA_CHECK((launchKernelSchedule<kSchedule, 64, kUpdateConvCache>(params, stream))); break;
case 96: TLLM_CUDA_CHECK((launchKernelSchedule<kSchedule, 96, kUpdateConvCache>(params, stream))); break;
default: TLLM_CHECK_WITH_INFO(false, "Optimized KDA decode does not support numHeads=%d", params.numHeads);
}
Expand Down
4 changes: 2 additions & 2 deletions cpp/tensorrt_llm/thop/kdaDecodeOp.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -97,9 +97,9 @@ void validate_kda_decode_fusion_inputs(at::Tensor x_q, at::Tensor x_k, at::Tenso
int const HV = static_cast<int>(x_v.size(2));
TORCH_CHECK(B > 0, "KDA decode requires a non-empty batch");
bool const supportedHeads = H == 1 || H == 2 || H == 3 || H == 4 || H == 6 || H == 8 || H == 12 || H == 16
|| H == 24 || H == 32 || H == 48 || H == 96;
|| H == 24 || H == 32 || H == 48 || H == 64 || H == 96;
TORCH_CHECK(
H == HV && supportedHeads, "KDA decode fusion CUDA supports H == HV in {1,2,3,4,6,8,12,16,24,32,48,96}");
H == HV && supportedHeads, "KDA decode fusion CUDA supports H == HV in {1,2,3,4,6,8,12,16,24,32,48,64,96}");
TORCH_CHECK(x_k.size(1) == B && x_k.size(2) == H && x_v.size(1) == B,
"x_q, x_k, and x_v batch/head dimensions are inconsistent");
TORCH_CHECK(HV % H == 0, "HV must be divisible by H");
Expand Down
Loading
Loading