Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion .github/workflows/cuda.yml
Original file line number Diff line number Diff line change
Expand Up @@ -480,8 +480,11 @@ jobs:
pip install gguf
python -m pytest examples/models/gemma4_31b/tests/ --ignore=examples/models/gemma4_31b/tests/test_mlx_pipeline.py -v -o "addopts="

# Muse Glimmer batched CUDA export on a tiny model
# Muse Glimmer batched CUDA export on a tiny model, and the batched
# runner end to end on it
python -m pytest examples/models/muse-glimmer/tests/test_cuda_batching_pipeline.py -v -o "addopts="
make muse-glimmer-cuda
python -m pytest examples/models/muse-glimmer/tests/test_run_solo_batching.py -v -o "addopts="

unittest-cuda-runtime:
name: unittest-cuda-runtime
Expand Down
1 change: 1 addition & 0 deletions Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -492,6 +492,7 @@ muse-glimmer-cuda:
@echo ""
@echo "✓ Build complete!"
@echo " Solo runner: cmake-out/examples/models/muse-glimmer/solo_runner"
@echo " Batched runner: cmake-out/examples/models/muse-glimmer/run_solo_batching"
@echo " DFlash runner: cmake-out/examples/models/muse-glimmer/dflash_runner"
@echo " Worker: cmake-out/examples/models/muse-glimmer/muse_glimmer_worker"

Expand Down
16 changes: 16 additions & 0 deletions examples/models/muse-glimmer/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -189,3 +189,19 @@ if(TARGET mlxdelegate)
executorch_target_copy_mlx_metallib(dflash_runner)
executorch_target_copy_mlx_metallib(muse_glimmer_worker)
endif()

# Batched text generation on the batching extension (CUDA only): runs what
# export/export_solo_batching.py writes.
if(EXECUTORCH_BUILD_CUDA)
add_executable(run_solo_batching runtime/runners/run_solo_batching.cpp)
target_include_directories(
run_solo_batching PUBLIC ${_common_include_directories} ${_json_include}
)
target_link_libraries(
run_solo_batching PUBLIC ${link_libraries} cuda_batching Threads::Threads
)
# CUDA AOTI blobs resolve symbols from the host executable.
if(NOT APPLE AND NOT MSVC)
target_link_options(run_solo_batching PRIVATE "LINKER:--export-dynamic")
endif()
endif()
2 changes: 1 addition & 1 deletion examples/models/muse-glimmer/CMakePresets.json
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@
"name": "muse-glimmer-cuda",
"displayName": "Build Muse Glimmer runner (CUDA)",
"configurePreset": "muse-glimmer-cuda",
"targets": ["solo_runner", "dflash_runner", "muse_glimmer_worker"]
"targets": ["solo_runner", "run_solo_batching", "dflash_runner", "muse_glimmer_worker"]
},
{
"name": "muse-glimmer-mlx",
Expand Down
Loading
Loading