Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
ffb177e
[None][fix] make CUDA graph kernel profiling CFT-compatible
bobboli Sep 3, 2026
643f616
[None][fix] add CFT driver-version detection helpers
zhangcl Sep 23, 2026
e026c5c
[None][fix] enable automatic CFT selection and CUPTI event profiling
bobboli Sep 16, 2026
2917c1b
[None][refactor] simplify MoE communication benchmark and add model p…
bobboli Sep 16, 2026
b34e46d
[None][refactor] namespace one-sided A2A controls and fix block sizes
bobboli Sep 16, 2026
c80d983
[None][refactor] isolate one-sided A2A workspace regions and add roun…
bobboli Sep 17, 2026
ca50319
[None][refactor] consolidate NVLink one-sided communication tests
bobboli Sep 17, 2026
8759219
[None][refactor] separate MNNVL memory from two-sided MoE communication
bobboli Sep 17, 2026
bafb66d
[None][test] consolidate one-sided CFT policy coverage
bobboli Sep 17, 2026
fc13ffa
[None][perf] use CUPTI kernel spans for MoE communication benchmarks
bobboli Sep 18, 2026
4fd8715
[None][refactor] select NVLink one-sided A2A timeouts in Python
bobboli Sep 21, 2026
b663bb0
[None][refactor] tidy CFT kernel naming and synchronization
bobboli Sep 21, 2026
7b48014
[None][fix] guard CFT dispatch workspace reuse across rounds
bobboli Sep 21, 2026
f602a13
[None][perf] compact NVLink one-sided dispatch and combine fanout
bobboli Sep 21, 2026
7fbd2cb
[None][refactor] share NVLink one-sided round flag helpers
bobboli Sep 21, 2026
8ad4d24
[None][fix] skip invalid expert routes in NVLink one-sided dispatch
bobboli Sep 23, 2026
23b2799
[None][fix] reconcile the one-sided overhaul with main-only callers
zhangcl Sep 23, 2026
2f1d9cd
[None][fix] drop the Python combine-offset reservation superseded by …
zhangcl Sep 25, 2026
750294d
[None][refactor] simplify NVLink one-sided dispatch and combine
bobboli Sep 23, 2026
130dc98
[None][refactor] organize NVLink one-sided workspace by phase
bobboli Sep 23, 2026
3640eff
[None][fix] re-initialize the planned one-sided layout on MNNVL restore
zhangcl Sep 25, 2026
3737131
[None][perf] pair FP8 conversions in one-sided combine
bobboli Sep 23, 2026
7eeb699
[None][docs] clarify one-sided workspace metadata fields
bobboli Sep 24, 2026
a7695a0
[None][fix] use main's MoE A2A paths in ported sources
zhangcl Sep 25, 2026
0d9a563
[None][fix] honor an explicit CFT choice when sizing the one-sided wo…
zhangcl Sep 25, 2026
6136b7a
[None][fix] import NVLinkOneSided from main's MoE package in model_en…
zhangcl Sep 25, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .claude/skills/trtllm-moe-develop/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -673,7 +673,7 @@ Prefer the unified MoE tests:
- Communication changes: `pytest tests/unittest/_torch/moe/test_moe_comm.py -k '<strategy>'`.
- Routing changes: `pytest tests/unittest/_torch/moe/test_moe_routing.py -k '<routing>'`.
- Load balancer changes: `pytest tests/unittest/_torch/moe/test_moe_load_balancer.py -k '<case>'`.
- Multi-GPU EP/all-to-all behavior: `pytest tests/unittest/_torch/moe/multi_gpu/test_moe_a2a.py -k '<case>'`.
- Multi-GPU EP/all-to-all behavior: `pytest tests/unittest/_torch/moe/multi_gpu/test_nvlink_one_sided.py -k '<case>'`.

When GPU resources are required, use the TRT-LLM GPU allocation/test-runner
skills first and record skipped tests with reasons.
Original file line number Diff line number Diff line change
Expand Up @@ -371,7 +371,7 @@ Use these examples when wrapper forward policy grows complicated:
should move into the scheduler.
- `tests/unittest/_torch/moe/test_moe_module.py`
- Module-level multi-GPU, chunking, routing, and EPLB cases.
- `tests/unittest/_torch/moe/multi_gpu/test_moe_a2a.py`
- `tests/unittest/_torch/moe/multi_gpu/test_nvlink_one_sided.py`
- Multi-GPU all-to-all behavior when relevant.

Good uses:
Expand Down
4 changes: 0 additions & 4 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -273,7 +273,6 @@ common-files: &common_files |
tensorrt_llm/_torch/modules/triton_linear.py |
tensorrt_llm/_torch/moe/expert_statistic.py |
tensorrt_llm/_torch/moe/fused_moe/__init__.py |
tensorrt_llm/_torch/moe/fused_moe/communication/moe_alltoall.py |
tensorrt_llm/_torch/moe/fused_moe/create_moe.py |
tensorrt_llm/_torch/moe/fused_moe/deep_ep_utils.py |
tensorrt_llm/_torch/moe/fused_moe/fused_moe_cute_dsl.py |
Expand Down Expand Up @@ -582,7 +581,6 @@ common-files: &common_files |
tests/unittest/_torch/modeling/test_modeling_vila.py |
tests/unittest/_torch/modules/test_group_rmn_norm.py |
tests/unittest/_torch/modules/test_triton_linear.py |
tests/unittest/_torch/moe/multi_gpu/test_moe_a2a.py |
tests/unittest/_torch/moe/test_fused_moe.py |
tests/unittest/_torch/moe/test_moe_host_sharer.py |
tests/unittest/_torch/moe/test_moe_load_balancer.py |
Expand Down Expand Up @@ -1029,7 +1027,6 @@ legacy-files: &legacy_files |
tensorrt_llm/_torch/modules/triton_linear.py |
tensorrt_llm/_torch/moe/expert_statistic.py |
tensorrt_llm/_torch/moe/fused_moe/__init__.py |
tensorrt_llm/_torch/moe/fused_moe/communication/moe_alltoall.py |
tensorrt_llm/_torch/moe/fused_moe/create_moe.py |
tensorrt_llm/_torch/moe/fused_moe/deep_ep_utils.py |
tensorrt_llm/_torch/moe/fused_moe/fused_moe_cute_dsl.py |
Expand Down Expand Up @@ -1338,7 +1335,6 @@ legacy-files: &legacy_files |
tests/unittest/_torch/modeling/test_modeling_vila.py |
tests/unittest/_torch/modules/test_group_rmn_norm.py |
tests/unittest/_torch/modules/test_triton_linear.py |
tests/unittest/_torch/moe/multi_gpu/test_moe_a2a.py |
tests/unittest/_torch/moe/test_fused_moe.py |
tests/unittest/_torch/moe/test_moe_host_sharer.py |
tests/unittest/_torch/moe/test_moe_load_balancer.py |
Expand Down
44 changes: 0 additions & 44 deletions cpp/tensorrt_llm/common/envUtils.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -542,50 +542,6 @@ bool getEnvDisableChunkedAttentionInGenPhase()
return getBoolEnv("TRTLLM_DISABLE_CHUNKED_ATTENTION_IN_GEN_PHASE");
}

static int sanitizeBlockSize(std::optional<int32_t> const& val)
{
// Default 256 when not set or invalid
int block = val.value_or(256);
// Clamp to sane CUDA bounds and warp multiples
if (block <= 0)
block = 256;
if (block > 1024)
block = 1024;
// Round to nearest multiple of 32 (warp size)
block = (block + 31) / 32 * 32;
if (block == 0)
block = 256;
return block;
}

// Read an integer env var and sanitize it as a CUDA block size. Treats malformed
// values (e.g. non-numeric strings that would throw inside std::stoi) as unset and
// falls back to the default, so this debug knob never becomes a hard failure.
static int getSanitizedBlockSizeFromEnv(char const* name)
{
try
{
return sanitizeBlockSize(getIntEnv(name));
}
catch (std::exception const&)
{
TLLM_LOG_WARNING("Invalid value for %s. Falling back to default block size.", name);
return sanitizeBlockSize(std::nullopt);
}
}

int getEnvMoeA2ADispatchBlockSize()
{
static int const kBlock = getSanitizedBlockSizeFromEnv("TLLM_MOE_A2A_DISPATCH_BLOCK_SIZE");
return kBlock;
}

int getEnvMoeA2ACombineBlockSize()
{
static int const kBlock = getSanitizedBlockSizeFromEnv("TLLM_MOE_A2A_COMBINE_BLOCK_SIZE");
return kBlock;
}

bool getEnvEplbForceGdrcopy()
{
return getBoolEnv("TRTLLM_EPLB_FORCE_GDRCOPY");
Expand Down
6 changes: 0 additions & 6 deletions cpp/tensorrt_llm/common/envUtils.h
Original file line number Diff line number Diff line change
Expand Up @@ -170,12 +170,6 @@ bool getEnvDisaggBenchmarkGenOnly();
// Whether to disable the chunked-attention in the generation phase.
bool getEnvDisableChunkedAttentionInGenPhase();

// TODO: For DEV purpose temporarily.
// Block size (threads per block) for MoE A2A Dispatch kernels (default 256 if unset or invalid)
int getEnvMoeA2ADispatchBlockSize();
// Block size (threads per block) for MoE A2A Combine kernels (default 256 if unset or invalid)
int getEnvMoeA2ACombineBlockSize();

bool getEnvKVCacheTransferAllBlocksForWindow();

bool getEnvEplbForceGdrcopy();
Expand Down
Loading
Loading