Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .github/workflows/cuda.yml
Original file line number Diff line number Diff line change
Expand Up @@ -515,11 +515,12 @@ jobs:
-v -o "addopts="

cmake --preset llm-release-cuda -DEXECUTORCH_BUILD_TESTS=ON
cmake --build cmake-out --target test_cuda_allocator test_cuda_mutable_state test_cuda_weight_cache test_cuda_kv_cache test_cuda_guard test_cuda_stream_guard -j$(nproc)
cmake --build cmake-out --target test_cuda_allocator test_cuda_mutable_state test_cuda_weight_cache test_cuda_kv_cache test_cuda_guard test_cuda_stream_guard test_step_plan -j$(nproc)
ctest --test-dir cmake-out -R test_cuda_allocator --output-on-failure -V
ctest --test-dir cmake-out -R test_cuda_mutable_state --output-on-failure -V
ctest --test-dir cmake-out -R test_cuda_weight_cache --output-on-failure -V
ctest --test-dir cmake-out -R test_cuda_kv_cache --output-on-failure -V
ctest --test-dir cmake-out -R test_step_plan --output-on-failure -V
ctest --test-dir cmake-out -R test_cuda_guard --output-on-failure -V
ctest --test-dir cmake-out -R test_cuda_stream_guard --output-on-failure -V

Expand Down
5 changes: 5 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
# Copyright 2024-2026 Arm Limited and/or its affiliates.
Expand Down Expand Up @@ -1017,6 +1017,11 @@
)
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/extension/llm/serving)
list(APPEND _executorch_extensions extension_llm_serving)
if(EXECUTORCH_BUILD_CUDA)
# The CUDA batching executor implements extension/llm/batching's seam, so
# it builds here rather than with backends/cuda.
add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/backends/cuda/batching)
endif()
endif()

if(EXECUTORCH_BUILD_EXTENSION_RUNNER_UTIL)
Expand Down
9 changes: 9 additions & 0 deletions backends/cuda/batching/BUCK
Original file line number Diff line number Diff line change
@@ -0,0 +1,9 @@
# Any targets that should be shared between fbcode and xplat must be defined in
# targets.bzl.

load("@fbsource//tools/build_defs:fbsource_utils.bzl", "is_fbcode")
load(":targets.bzl", "define_common_targets")

oncall("executorch")

define_common_targets(is_fbcode = is_fbcode())
23 changes: 23 additions & 0 deletions backends/cuda/batching/CMakeLists.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.

# Step slicing for the CUDA batching executor: cuts a batch of packed tokens
# into forwards the exported decode/prefill methods can run. Added from the
# root after extension/llm/batching.

if(NOT EXECUTORCH_ROOT)
set(EXECUTORCH_ROOT ${CMAKE_CURRENT_SOURCE_DIR}/../../..)
endif()

install(FILES step_plan.h
DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/executorch/backends/cuda/batching
)

if(BUILD_TESTING)
include(${EXECUTORCH_ROOT}/tools/cmake/Test.cmake)

et_cxx_test(test_step_plan SOURCES test/test_step_plan.cpp)
endif()
54 changes: 54 additions & 0 deletions backends/cuda/batching/step_plan.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#pragma once

#include <algorithm>
#include <vector>

namespace executorch::backends::cuda::batching {

// Which exported method runs a slice of a batch.
enum class StepMethod {
// Static, one token: the method a CUDA graph is captured for.
Decode,
// Dynamic over [min_prefill_tokens, max_step_tokens].
Prefill,
};

struct StepSlice {
int offset;
int length;
StepMethod method;
};

// Cuts a batch of `total` packed tokens into forwards the exported methods can
// run, in order, so a slice attends every cell its predecessors wrote.
//
// Slices take up to `max_step_tokens` each. A one-token slice runs Decode. A
// slice shorter than `min_prefill_tokens` -- the lower bound the prefill
// method was exported with -- runs as that many Decode forwards; anything
// else runs Prefill.
inline std::vector<StepSlice>
plan_slices(int total, int max_step_tokens, int min_prefill_tokens) {
std::vector<StepSlice> slices;
for (int offset = 0; offset < total;) {
const int length = std::min(max_step_tokens, total - offset);
if (length >= min_prefill_tokens && length > 1) {
slices.push_back({offset, length, StepMethod::Prefill});
} else {
for (int i = 0; i < length; ++i) {
slices.push_back({offset + i, 1, StepMethod::Decode});
}
}
offset += length;
}
return slices;
}

} // namespace executorch::backends::cuda::batching
24 changes: 24 additions & 0 deletions backends/cuda/batching/targets.bzl
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime")
load("@fbcode_macros//build_defs:cpp_unittest.bzl", "cpp_unittest")

def define_common_targets(is_fbcode = False):
if not is_fbcode:
return

runtime.cxx_library(
name = "step_plan",
exported_headers = [
"step_plan.h",
],
visibility = ["PUBLIC"],
)

cpp_unittest(
name = "test_step_plan",
srcs = [
"test/test_step_plan.cpp",
],
deps = [
":step_plan",
],
)
77 changes: 77 additions & 0 deletions backends/cuda/batching/test/test_step_plan.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
/*
* Copyright (c) Meta Platforms, Inc. and affiliates.
* All rights reserved.
*
* This source code is licensed under the BSD-style license found in the
* LICENSE file in the root directory of this source tree.
*/

#include <executorch/backends/cuda/batching/step_plan.h>

#include <gtest/gtest.h>

#include <vector>

namespace cb = ::executorch::backends::cuda::batching;
using cb::StepMethod;

namespace {

struct Expected {
int offset;
int length;
StepMethod method;
};

void expect_plan(
int total,
int max_step,
int min_prefill,
const std::vector<Expected>& expected) {
const auto slices = cb::plan_slices(total, max_step, min_prefill);
ASSERT_EQ(slices.size(), expected.size());
int covered = 0;
for (size_t i = 0; i < slices.size(); ++i) {
EXPECT_EQ(slices[i].offset, expected[i].offset) << i;
EXPECT_EQ(slices[i].length, expected[i].length) << i;
EXPECT_EQ(slices[i].method, expected[i].method) << i;
// In order and without gaps: a slice attends what its predecessors wrote.
EXPECT_EQ(slices[i].offset, covered) << i;
covered += slices[i].length;
}
EXPECT_EQ(covered, total);
}

constexpr auto D = StepMethod::Decode;
constexpr auto P = StepMethod::Prefill;

} // namespace

TEST(StepPlanTest, OneTokenRunsDecode) {
expect_plan(1, 8, 2, {{0, 1, D}});
}

TEST(StepPlanTest, TwoTokensRunPrefill) {
expect_plan(2, 8, 2, {{0, 2, P}});
}

TEST(StepPlanTest, UpToTheWidestStepIsOneForward) {
expect_plan(8, 8, 2, {{0, 8, P}});
}

TEST(StepPlanTest, WiderBatchesSliceAndALoneTailTokenRunsDecode) {
expect_plan(9, 8, 2, {{0, 8, P}, {8, 1, D}});
expect_plan(10, 8, 2, {{0, 8, P}, {8, 2, P}});
expect_plan(16, 8, 2, {{0, 8, P}, {8, 8, P}});
}

TEST(StepPlanTest, ShortSlicesBelowThePrefillBoundRunAsDecodes) {
// A prefill exported from five tokens up: two to four run token by token.
expect_plan(4, 8, 5, {{0, 1, D}, {1, 1, D}, {2, 1, D}, {3, 1, D}});
expect_plan(5, 8, 5, {{0, 5, P}});
expect_plan(11, 8, 5, {{0, 8, P}, {8, 1, D}, {9, 1, D}, {10, 1, D}});
}

TEST(StepPlanTest, EmptyBatchPlansNothing) {
expect_plan(0, 8, 2, {});
}
Loading