diff --git a/.github/workflows/cuda.yml b/.github/workflows/cuda.yml index b30da2a99de..65d755e5280 100644 --- a/.github/workflows/cuda.yml +++ b/.github/workflows/cuda.yml @@ -515,11 +515,12 @@ jobs: -v -o "addopts=" cmake --preset llm-release-cuda -DEXECUTORCH_BUILD_TESTS=ON - cmake --build cmake-out --target test_cuda_allocator test_cuda_mutable_state test_cuda_weight_cache test_cuda_kv_cache test_cuda_guard test_cuda_stream_guard -j$(nproc) + cmake --build cmake-out --target test_cuda_allocator test_cuda_mutable_state test_cuda_weight_cache test_cuda_kv_cache test_cuda_guard test_cuda_stream_guard test_step_plan -j$(nproc) ctest --test-dir cmake-out -R test_cuda_allocator --output-on-failure -V ctest --test-dir cmake-out -R test_cuda_mutable_state --output-on-failure -V ctest --test-dir cmake-out -R test_cuda_weight_cache --output-on-failure -V ctest --test-dir cmake-out -R test_cuda_kv_cache --output-on-failure -V + ctest --test-dir cmake-out -R test_step_plan --output-on-failure -V ctest --test-dir cmake-out -R test_cuda_guard --output-on-failure -V ctest --test-dir cmake-out -R test_cuda_stream_guard --output-on-failure -V diff --git a/CMakeLists.txt b/CMakeLists.txt index 737133db121..d8fd79ba98f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1017,6 +1017,11 @@ if(EXECUTORCH_BUILD_EXTENSION_LLM) ) add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/extension/llm/serving) list(APPEND _executorch_extensions extension_llm_serving) + if(EXECUTORCH_BUILD_CUDA) + # The CUDA batching executor implements extension/llm/batching's seam, so + # it builds here rather than with backends/cuda. + add_subdirectory(${CMAKE_CURRENT_SOURCE_DIR}/backends/cuda/batching) + endif() endif() if(EXECUTORCH_BUILD_EXTENSION_RUNNER_UTIL) diff --git a/backends/cuda/batching/BUCK b/backends/cuda/batching/BUCK new file mode 100644 index 00000000000..f559a6f1cfe --- /dev/null +++ b/backends/cuda/batching/BUCK @@ -0,0 +1,9 @@ +# Any targets that should be shared between fbcode and xplat must be defined in +# targets.bzl. + +load("@fbsource//tools/build_defs:fbsource_utils.bzl", "is_fbcode") +load(":targets.bzl", "define_common_targets") + +oncall("executorch") + +define_common_targets(is_fbcode = is_fbcode()) diff --git a/backends/cuda/batching/CMakeLists.txt b/backends/cuda/batching/CMakeLists.txt new file mode 100644 index 00000000000..a46dd2b8e2d --- /dev/null +++ b/backends/cuda/batching/CMakeLists.txt @@ -0,0 +1,23 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the BSD-style license found in the +# LICENSE file in the root directory of this source tree. + +# Step slicing for the CUDA batching executor: cuts a batch of packed tokens +# into forwards the exported decode/prefill methods can run. Added from the +# root after extension/llm/batching. + +if(NOT EXECUTORCH_ROOT) + set(EXECUTORCH_ROOT ${CMAKE_CURRENT_SOURCE_DIR}/../../..) +endif() + +install(FILES step_plan.h + DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/executorch/backends/cuda/batching +) + +if(BUILD_TESTING) + include(${EXECUTORCH_ROOT}/tools/cmake/Test.cmake) + + et_cxx_test(test_step_plan SOURCES test/test_step_plan.cpp) +endif() diff --git a/backends/cuda/batching/step_plan.h b/backends/cuda/batching/step_plan.h new file mode 100644 index 00000000000..a8184ecf5c4 --- /dev/null +++ b/backends/cuda/batching/step_plan.h @@ -0,0 +1,54 @@ +/* + * Copyright (c) Meta Platforms, Inc. and affiliates. + * All rights reserved. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ + +#pragma once + +#include +#include + +namespace executorch::backends::cuda::batching { + +// Which exported method runs a slice of a batch. +enum class StepMethod { + // Static, one token: the method a CUDA graph is captured for. + Decode, + // Dynamic over [min_prefill_tokens, max_step_tokens]. + Prefill, +}; + +struct StepSlice { + int offset; + int length; + StepMethod method; +}; + +// Cuts a batch of `total` packed tokens into forwards the exported methods can +// run, in order, so a slice attends every cell its predecessors wrote. +// +// Slices take up to `max_step_tokens` each. A one-token slice runs Decode. A +// slice shorter than `min_prefill_tokens` -- the lower bound the prefill +// method was exported with -- runs as that many Decode forwards; anything +// else runs Prefill. +inline std::vector +plan_slices(int total, int max_step_tokens, int min_prefill_tokens) { + std::vector slices; + for (int offset = 0; offset < total;) { + const int length = std::min(max_step_tokens, total - offset); + if (length >= min_prefill_tokens && length > 1) { + slices.push_back({offset, length, StepMethod::Prefill}); + } else { + for (int i = 0; i < length; ++i) { + slices.push_back({offset + i, 1, StepMethod::Decode}); + } + } + offset += length; + } + return slices; +} + +} // namespace executorch::backends::cuda::batching diff --git a/backends/cuda/batching/targets.bzl b/backends/cuda/batching/targets.bzl new file mode 100644 index 00000000000..338cb39b77b --- /dev/null +++ b/backends/cuda/batching/targets.bzl @@ -0,0 +1,24 @@ +load("@fbsource//xplat/executorch/build:runtime_wrapper.bzl", "runtime") +load("@fbcode_macros//build_defs:cpp_unittest.bzl", "cpp_unittest") + +def define_common_targets(is_fbcode = False): + if not is_fbcode: + return + + runtime.cxx_library( + name = "step_plan", + exported_headers = [ + "step_plan.h", + ], + visibility = ["PUBLIC"], + ) + + cpp_unittest( + name = "test_step_plan", + srcs = [ + "test/test_step_plan.cpp", + ], + deps = [ + ":step_plan", + ], + ) diff --git a/backends/cuda/batching/test/test_step_plan.cpp b/backends/cuda/batching/test/test_step_plan.cpp new file mode 100644 index 00000000000..ca91464d6bb --- /dev/null +++ b/backends/cuda/batching/test/test_step_plan.cpp @@ -0,0 +1,77 @@ +/* + * Copyright (c) Meta Platforms, Inc. and affiliates. + * All rights reserved. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ + +#include + +#include + +#include + +namespace cb = ::executorch::backends::cuda::batching; +using cb::StepMethod; + +namespace { + +struct Expected { + int offset; + int length; + StepMethod method; +}; + +void expect_plan( + int total, + int max_step, + int min_prefill, + const std::vector& expected) { + const auto slices = cb::plan_slices(total, max_step, min_prefill); + ASSERT_EQ(slices.size(), expected.size()); + int covered = 0; + for (size_t i = 0; i < slices.size(); ++i) { + EXPECT_EQ(slices[i].offset, expected[i].offset) << i; + EXPECT_EQ(slices[i].length, expected[i].length) << i; + EXPECT_EQ(slices[i].method, expected[i].method) << i; + // In order and without gaps: a slice attends what its predecessors wrote. + EXPECT_EQ(slices[i].offset, covered) << i; + covered += slices[i].length; + } + EXPECT_EQ(covered, total); +} + +constexpr auto D = StepMethod::Decode; +constexpr auto P = StepMethod::Prefill; + +} // namespace + +TEST(StepPlanTest, OneTokenRunsDecode) { + expect_plan(1, 8, 2, {{0, 1, D}}); +} + +TEST(StepPlanTest, TwoTokensRunPrefill) { + expect_plan(2, 8, 2, {{0, 2, P}}); +} + +TEST(StepPlanTest, UpToTheWidestStepIsOneForward) { + expect_plan(8, 8, 2, {{0, 8, P}}); +} + +TEST(StepPlanTest, WiderBatchesSliceAndALoneTailTokenRunsDecode) { + expect_plan(9, 8, 2, {{0, 8, P}, {8, 1, D}}); + expect_plan(10, 8, 2, {{0, 8, P}, {8, 2, P}}); + expect_plan(16, 8, 2, {{0, 8, P}, {8, 8, P}}); +} + +TEST(StepPlanTest, ShortSlicesBelowThePrefillBoundRunAsDecodes) { + // A prefill exported from five tokens up: two to four run token by token. + expect_plan(4, 8, 5, {{0, 1, D}, {1, 1, D}, {2, 1, D}, {3, 1, D}}); + expect_plan(5, 8, 5, {{0, 5, P}}); + expect_plan(11, 8, 5, {{0, 8, P}, {8, 1, D}, {9, 1, D}, {10, 1, D}}); +} + +TEST(StepPlanTest, EmptyBatchPlansNothing) { + expect_plan(0, 8, 2, {}); +}