From a0a7329a60afd9e83d5790e26c7cdad7ea1b6490 Mon Sep 17 00:00:00 2001 From: gasoonjia Date: Fri, 2 Oct 2026 12:07:02 -0700 Subject: [PATCH] Update [ghstack-poisoned] --- backends/cuda/batching/cuda_executor.cpp | 39 +++++++++++++++---- backends/cuda/batching/cuda_executor.h | 7 +++- .../cuda/batching/test/export_toy_decoder.py | 25 +++++++----- .../cuda/batching/test/test_cuda_executor.cpp | 27 ++++++++----- 4 files changed, 70 insertions(+), 28 deletions(-) diff --git a/backends/cuda/batching/cuda_executor.cpp b/backends/cuda/batching/cuda_executor.cpp index 904a30ad4b0..edde2359a06 100644 --- a/backends/cuda/batching/cuda_executor.cpp +++ b/backends/cuda/batching/cuda_executor.cpp @@ -43,7 +43,8 @@ namespace metadata = ::executorch::extension::llm; namespace { -// A prefill forward carries at least this many tokens; one token is decode's. +// The least a prefill forward may carry when the program names no bound: one +// token is decode's. constexpr int kMinPrefillTokens = 2; bool is_supported_logits_type(::executorch::aten::ScalarType type) { @@ -146,7 +147,8 @@ CudaExecutor::CudaExecutor( int max_session_tokens, std::string backend_id, std::int32_t vocab_size, - int max_step_tokens) + int max_step_tokens, + int min_prefill_tokens) : install_guard_(cache), module_(std::move(module)), ctl_(cache->as()), @@ -154,6 +156,7 @@ CudaExecutor::CudaExecutor( backend_id_(std::move(backend_id)), vocab_size_(vocab_size), max_step_tokens_(max_step_tokens), + min_prefill_tokens_(min_prefill_tokens), sessions_(*ctl_, max_sessions, max_session_tokens, vocab_size) {} CudaExecutor::~CudaExecutor() = default; @@ -206,14 +209,23 @@ Result> CudaExecutor::create( decode_width, step_width(decode_meta, kDecodeMethod)); ET_ASSIGN_OR_RETURN( max_step_tokens, step_width(prefill_meta, kPrefillMethod)); + // A program may export prefill from more than two tokens -- one whose + // kernels switch at a small width cannot trace narrower -- and say so here. + ET_ASSIGN_OR_RETURN( + declared_min_prefill, + metadata::detail::read_int_method(*module, kMinPrefillTokensMethod)); + const int min_prefill_tokens = static_cast( + declared_min_prefill.value_or(kMinPrefillTokens)); ET_CHECK_OR_RETURN_ERROR( - decode_width == 1 && max_step_tokens >= kMinPrefillTokens && - max_step_tokens <= max_cells, + decode_width == 1 && min_prefill_tokens >= kMinPrefillTokens && + min_prefill_tokens <= max_step_tokens && max_step_tokens <= max_cells, InvalidProgram, - "CudaExecutor: decode must take one token and prefill [%d, max_cells]; " - "got %d and %d", + "CudaExecutor: decode must take one token and prefill [%d, max_cells] " + "tokens from at least %d; got %d, and [%d, %d]", + kMinPrefillTokens, kMinPrefillTokens, decode_width, + min_prefill_tokens, max_step_tokens); ET_ASSIGN_OR_RETURN( decode_vocab, logits_width(decode_meta, kDecodeMethod)); @@ -283,7 +295,8 @@ Result> CudaExecutor::create( max_session_tokens, std::move(backend_id), vocab_size, - max_step_tokens)); + max_step_tokens, + min_prefill_tokens)); } bool CudaExecutor::initialize() { @@ -341,7 +354,7 @@ bool CudaExecutor::execute(const BatchInput& batch, BatchOutput& out) { // input's logits row falls in exactly one slice. const int total = static_cast(step->tokens.size()); for (const StepSlice& slice : - plan_slices(total, max_step_tokens_, kMinPrefillTokens)) { + plan_slices(total, max_step_tokens_, min_prefill_tokens_)) { const int off = slice.offset; const int n = slice.length; const char* method = @@ -362,6 +375,16 @@ bool CudaExecutor::execute(const BatchInput& batch, BatchOutput& out) { std::vector( step->positions.begin() + off, step->positions.begin() + off + n)); auto selected = llm_batching::util::select_rows(*step, off, n); + if (slice.method == StepMethod::Prefill) { + // The selected-rows dimension shares the tokens' lower bound: the LM + // head runs over those rows. Extra rows repeat the last; nothing reads + // them. + selected.selector.resize( + std::max( + selected.selector.size(), + static_cast(min_prefill_tokens_)), + selected.selector.back()); + } const int rows = static_cast(selected.selector.size()); auto selector = make_tensor_ptr({rows}, std::move(selected.selector)); diff --git a/backends/cuda/batching/cuda_executor.h b/backends/cuda/batching/cuda_executor.h index ce609c6fd38..6b7d758479a 100644 --- a/backends/cuda/batching/cuda_executor.h +++ b/backends/cuda/batching/cuda_executor.h @@ -41,6 +41,9 @@ inline constexpr char kDecodeMethod[] = "decode"; inline constexpr char kPrefillMethod[] = "prefill"; // The cell layout's pool size, which fixes the shape of its step buffers. inline constexpr char kMaxCellsMethod[] = "get_offgraph_kv_max_cells"; +// Optional: the fewest tokens, and selected rows, prefill was exported for. +// Narrower slices run as decodes. Defaults to 2. +inline constexpr char kMinPrefillTokensMethod[] = "get_min_prefill_chunk"; // Process-wide CUDA backend options create() sets before the methods load. struct CudaExecutorOptions { @@ -106,7 +109,8 @@ class ET_EXPERIMENTAL CudaExecutor : public llm_batching::Executor { int max_session_tokens, std::string backend_id, std::int32_t vocab_size, - int max_step_tokens); + int max_step_tokens, + int min_prefill_tokens); // Ordered so the module dies first, releasing the delegates that resolved @@ -118,6 +122,7 @@ class ET_EXPERIMENTAL CudaExecutor : public llm_batching::Executor { std::string backend_id_; std::int32_t vocab_size_; int max_step_tokens_; + int min_prefill_tokens_; llm_batching::util::SessionTable sessions_; }; diff --git a/backends/cuda/batching/test/export_toy_decoder.py b/backends/cuda/batching/test/export_toy_decoder.py index f925d877b56..7e7e3b527d6 100644 --- a/backends/cuda/batching/test/export_toy_decoder.py +++ b/backends/cuda/batching/test/export_toy_decoder.py @@ -6,11 +6,14 @@ """Export the toy decoder the CudaExecutor GPU test runs. -Writes ``model.pte`` and ``aoti_cuda_blob.ptd`` -- a two-layer decoder (one -full-history layer, one sliding-window layer) exported as the ``decode`` and -``prefill`` methods CudaExecutor drives, lowered in the cell layout -- plus -``expected.txt``: one ``prompt;continuation`` line per prompt, the greedy -continuation computed eagerly with the neutral reference cache. +Writes two artifacts, ``min2/`` and ``min5/``, each ``model.pte`` and +``aoti_cuda_blob.ptd`` -- a two-layer decoder (one full-history layer, one +sliding-window layer) exported as the ``decode`` and ``prefill`` methods +CudaExecutor drives, lowered in the cell layout. ``min2`` exports prefill from +two tokens; ``min5`` from five, as a model whose kernels switch at a small +width must, and publishes that bound. Each has ``expected.txt``: one +``prompt;continuation`` line per prompt, the greedy continuation computed +eagerly with the neutral reference cache. The residual stream carries each token's embedding at a large scale and the LM head maps it to a fixed successor, while attention adds a smaller term. The @@ -140,7 +143,7 @@ def _greedy(model: ToyDecoder, prompt) -> list: REGISTRY.uninstall(key) -def export(output_dir: str) -> None: +def export(output_dir: str, min_prefill: int) -> None: import torch._inductor.config as inductor_config from executorch.backends.cuda.cuda_backend import CudaBackend from executorch.backends.cuda.cuda_partitioner import CudaPartitioner @@ -170,8 +173,8 @@ def export(output_dir: str) -> None: expected = [_greedy(model, prompt) for prompt in PROMPTS] model = model.to(dtype=torch.bfloat16) - width = Dim("width", min=2, max=MAX_STEP) - rows = Dim("rows", min=1, max=MAX_STEP) + width = Dim("width", min=min_prefill, max=MAX_STEP) + rows = Dim("rows", min=1 if min_prefill == 2 else min_prefill, max=MAX_STEP) long = {"dtype": torch.long} with torch.no_grad(): programs = { @@ -216,6 +219,8 @@ def partitioner(name: str) -> CudaPartitioner: ), "get_offgraph_kv_max_cells": MAX_CELLS, } + if min_prefill != 2: + constant_methods["get_min_prefill_chunk"] = min_prefill program = to_edge_transform_and_lower( programs, partitioner={name: [partitioner(name)] for name in programs}, @@ -246,7 +251,9 @@ def partitioner(name: str) -> CudaPartitioner: def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--output-dir", required=True) - export(parser.parse_args().output_dir) + output_dir = parser.parse_args().output_dir + for min_prefill in (2, 5): + export(os.path.join(output_dir, f"min{min_prefill}"), min_prefill) if __name__ == "__main__": diff --git a/backends/cuda/batching/test/test_cuda_executor.cpp b/backends/cuda/batching/test/test_cuda_executor.cpp index a3650c59841..bc8809f85d3 100644 --- a/backends/cuda/batching/test/test_cuda_executor.cpp +++ b/backends/cuda/batching/test/test_cuda_executor.cpp @@ -6,9 +6,11 @@ * LICENSE file in the root directory of this source tree. */ -// Drives CudaExecutor through the batching Runner on the toy decoder that -// export_toy_decoder.py writes to $ET_CUDA_BATCHING_TOY_DIR, and checks every -// generation against the eager greedy continuation it recorded. +// Drives CudaExecutor through the batching Runner on the toy decoders that +// export_toy_decoder.py writes under $ET_CUDA_BATCHING_TOY_DIR, and checks +// every generation against the eager greedy continuation it recorded. Runs +// once per artifact: prefill exported from two tokens, and from five, where +// narrower slices must run as decodes and prefill's selector is padded. #include #include @@ -54,7 +56,7 @@ std::vector parse_tokens(const std::string& text) { return tokens; } -class CudaExecutorTest : public ::testing::Test { +class CudaExecutorTest : public ::testing::TestWithParam { protected: void SetUp() override { int devices = 0; @@ -66,7 +68,7 @@ class CudaExecutorTest : public ::testing::Test { GTEST_SKIP() << "ET_CUDA_BATCHING_TOY_DIR is not set; run " "export_toy_decoder.py first"; } - dir_ = dir; + dir_ = std::string(dir) + "/" + GetParam(); std::ifstream in(dir_ + "/expected.txt"); ASSERT_TRUE(in.is_open()) << dir_ << "/expected.txt"; std::string line; @@ -165,7 +167,7 @@ class CudaExecutorTest : public ::testing::Test { } // namespace -TEST_F(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) { +TEST_P(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) { auto exec = executor(); ASSERT_NE(exec, nullptr); EXPECT_EQ(exec->preferred_batch_tokens(), static_cast(kMaxStep)); @@ -195,7 +197,7 @@ TEST_F(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) { EXPECT_GT(engine.decode_sessions_total, engine.steps / 2); } -TEST_F(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) { +TEST_P(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) { cb::CudaExecutorOptions options; options.cuda_graph_for_decode = false; auto exec = executor(options); @@ -208,7 +210,7 @@ TEST_F(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) { } } -TEST_F(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) { +TEST_P(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) { auto exec = executor(); ASSERT_NE(exec, nullptr); batching::Runner runner(*exec, scheduler()); @@ -219,7 +221,7 @@ TEST_F(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) { EXPECT_EQ(generations[1].tokens, c.expected); } -TEST_F(CudaExecutorTest, SessionsReuseCellsAcrossRounds) { +TEST_P(CudaExecutorTest, SessionsReuseCellsAcrossRounds) { auto exec = executor(); ASSERT_NE(exec, nullptr); batching::Runner runner(*exec, scheduler()); @@ -235,7 +237,7 @@ TEST_F(CudaExecutorTest, SessionsReuseCellsAcrossRounds) { runner.shutdown(); } -TEST_F(CudaExecutorTest, RefusesLimitsThePoolCannotHold) { +TEST_P(CudaExecutorTest, RefusesLimitsThePoolCannotHold) { // 8 sessions of the full 64-token context need 512 cells; the program has // 256. EXPECT_EQ( @@ -248,3 +250,8 @@ TEST_F(CudaExecutorTest, RefusesLimitsThePoolCannotHold) { .error(), Error::InvalidArgument); } + +INSTANTIATE_TEST_SUITE_P( + Toy, + CudaExecutorTest, + ::testing::Values("min2", "min5"));