Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
39 changes: 31 additions & 8 deletions backends/cuda/batching/cuda_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,8 @@ namespace metadata = ::executorch::extension::llm;

namespace {

// A prefill forward carries at least this many tokens; one token is decode's.
// The least a prefill forward may carry when the program names no bound: one
// token is decode's.
constexpr int kMinPrefillTokens = 2;

bool is_supported_logits_type(::executorch::aten::ScalarType type) {
Expand Down Expand Up @@ -146,14 +147,16 @@ CudaExecutor::CudaExecutor(
int max_session_tokens,
std::string backend_id,
std::int32_t vocab_size,
int max_step_tokens)
int max_step_tokens,
int min_prefill_tokens)
: install_guard_(cache),
module_(std::move(module)),
ctl_(cache->as<llm_cache::BatchControl>()),
kv_(cache->as<CudaKVCache>()),
backend_id_(std::move(backend_id)),
vocab_size_(vocab_size),
max_step_tokens_(max_step_tokens),
min_prefill_tokens_(min_prefill_tokens),
sessions_(*ctl_, max_sessions, max_session_tokens, vocab_size) {}

CudaExecutor::~CudaExecutor() = default;
Expand Down Expand Up @@ -206,14 +209,23 @@ Result<std::unique_ptr<CudaExecutor>> CudaExecutor::create(
decode_width, step_width(decode_meta, kDecodeMethod));
ET_ASSIGN_OR_RETURN(
max_step_tokens, step_width(prefill_meta, kPrefillMethod));
// A program may export prefill from more than two tokens -- one whose
// kernels switch at a small width cannot trace narrower -- and say so here.
ET_ASSIGN_OR_RETURN(
declared_min_prefill,
metadata::detail::read_int_method(*module, kMinPrefillTokensMethod));
const int min_prefill_tokens = static_cast<int>(
declared_min_prefill.value_or(kMinPrefillTokens));
ET_CHECK_OR_RETURN_ERROR(
decode_width == 1 && max_step_tokens >= kMinPrefillTokens &&
max_step_tokens <= max_cells,
decode_width == 1 && min_prefill_tokens >= kMinPrefillTokens &&
min_prefill_tokens <= max_step_tokens && max_step_tokens <= max_cells,
InvalidProgram,
"CudaExecutor: decode must take one token and prefill [%d, max_cells]; "
"got %d and %d",
"CudaExecutor: decode must take one token and prefill [%d, max_cells] "
"tokens from at least %d; got %d, and [%d, %d]",
kMinPrefillTokens,
kMinPrefillTokens,
decode_width,
min_prefill_tokens,
max_step_tokens);
ET_ASSIGN_OR_RETURN(
decode_vocab, logits_width(decode_meta, kDecodeMethod));
Expand Down Expand Up @@ -283,7 +295,8 @@ Result<std::unique_ptr<CudaExecutor>> CudaExecutor::create(
max_session_tokens,
std::move(backend_id),
vocab_size,
max_step_tokens));
max_step_tokens,
min_prefill_tokens));
}

bool CudaExecutor::initialize() {
Expand Down Expand Up @@ -341,7 +354,7 @@ bool CudaExecutor::execute(const BatchInput& batch, BatchOutput& out) {
// input's logits row falls in exactly one slice.
const int total = static_cast<int>(step->tokens.size());
for (const StepSlice& slice :
plan_slices(total, max_step_tokens_, kMinPrefillTokens)) {
plan_slices(total, max_step_tokens_, min_prefill_tokens_)) {
const int off = slice.offset;
const int n = slice.length;
const char* method =
Expand All @@ -362,6 +375,16 @@ bool CudaExecutor::execute(const BatchInput& batch, BatchOutput& out) {
std::vector<std::int64_t>(
step->positions.begin() + off, step->positions.begin() + off + n));
auto selected = llm_batching::util::select_rows(*step, off, n);
if (slice.method == StepMethod::Prefill) {
// The selected-rows dimension shares the tokens' lower bound: the LM
// head runs over those rows. Extra rows repeat the last; nothing reads
// them.
selected.selector.resize(
std::max<std::size_t>(
selected.selector.size(),
static_cast<std::size_t>(min_prefill_tokens_)),
selected.selector.back());
}
const int rows = static_cast<int>(selected.selector.size());
auto selector = make_tensor_ptr({rows}, std::move(selected.selector));

Expand Down
7 changes: 6 additions & 1 deletion backends/cuda/batching/cuda_executor.h
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,9 @@ inline constexpr char kDecodeMethod[] = "decode";
inline constexpr char kPrefillMethod[] = "prefill";
// The cell layout's pool size, which fixes the shape of its step buffers.
inline constexpr char kMaxCellsMethod[] = "get_offgraph_kv_max_cells";
// Optional: the fewest tokens, and selected rows, prefill was exported for.
// Narrower slices run as decodes. Defaults to 2.
inline constexpr char kMinPrefillTokensMethod[] = "get_min_prefill_chunk";

// Process-wide CUDA backend options create() sets before the methods load.
struct CudaExecutorOptions {
Expand Down Expand Up @@ -106,7 +109,8 @@ class ET_EXPERIMENTAL CudaExecutor : public llm_batching::Executor {
int max_session_tokens,
std::string backend_id,
std::int32_t vocab_size,
int max_step_tokens);
int max_step_tokens,
int min_prefill_tokens);


// Ordered so the module dies first, releasing the delegates that resolved
Expand All @@ -118,6 +122,7 @@ class ET_EXPERIMENTAL CudaExecutor : public llm_batching::Executor {
std::string backend_id_;
std::int32_t vocab_size_;
int max_step_tokens_;
int min_prefill_tokens_;
llm_batching::util::SessionTable sessions_;
};

Expand Down
25 changes: 16 additions & 9 deletions backends/cuda/batching/test/export_toy_decoder.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,11 +6,14 @@

"""Export the toy decoder the CudaExecutor GPU test runs.

Writes ``model.pte`` and ``aoti_cuda_blob.ptd`` -- a two-layer decoder (one
full-history layer, one sliding-window layer) exported as the ``decode`` and
``prefill`` methods CudaExecutor drives, lowered in the cell layout -- plus
``expected.txt``: one ``prompt;continuation`` line per prompt, the greedy
continuation computed eagerly with the neutral reference cache.
Writes two artifacts, ``min2/`` and ``min5/``, each ``model.pte`` and
``aoti_cuda_blob.ptd`` -- a two-layer decoder (one full-history layer, one
sliding-window layer) exported as the ``decode`` and ``prefill`` methods
CudaExecutor drives, lowered in the cell layout. ``min2`` exports prefill from
two tokens; ``min5`` from five, as a model whose kernels switch at a small
width must, and publishes that bound. Each has ``expected.txt``: one
``prompt;continuation`` line per prompt, the greedy continuation computed
eagerly with the neutral reference cache.

The residual stream carries each token's embedding at a large scale and the
LM head maps it to a fixed successor, while attention adds a smaller term. The
Expand Down Expand Up @@ -140,7 +143,7 @@ def _greedy(model: ToyDecoder, prompt) -> list:
REGISTRY.uninstall(key)


def export(output_dir: str) -> None:
def export(output_dir: str, min_prefill: int) -> None:
import torch._inductor.config as inductor_config
from executorch.backends.cuda.cuda_backend import CudaBackend
from executorch.backends.cuda.cuda_partitioner import CudaPartitioner
Expand Down Expand Up @@ -170,8 +173,8 @@ def export(output_dir: str) -> None:
expected = [_greedy(model, prompt) for prompt in PROMPTS]

model = model.to(dtype=torch.bfloat16)
width = Dim("width", min=2, max=MAX_STEP)
rows = Dim("rows", min=1, max=MAX_STEP)
width = Dim("width", min=min_prefill, max=MAX_STEP)
rows = Dim("rows", min=1 if min_prefill == 2 else min_prefill, max=MAX_STEP)
long = {"dtype": torch.long}
with torch.no_grad():
programs = {
Expand Down Expand Up @@ -216,6 +219,8 @@ def partitioner(name: str) -> CudaPartitioner:
),
"get_offgraph_kv_max_cells": MAX_CELLS,
}
if min_prefill != 2:
constant_methods["get_min_prefill_chunk"] = min_prefill
program = to_edge_transform_and_lower(
programs,
partitioner={name: [partitioner(name)] for name in programs},
Expand Down Expand Up @@ -246,7 +251,9 @@ def partitioner(name: str) -> CudaPartitioner:
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--output-dir", required=True)
export(parser.parse_args().output_dir)
output_dir = parser.parse_args().output_dir
for min_prefill in (2, 5):
export(os.path.join(output_dir, f"min{min_prefill}"), min_prefill)


if __name__ == "__main__":
Expand Down
27 changes: 17 additions & 10 deletions backends/cuda/batching/test/test_cuda_executor.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -6,9 +6,11 @@
* LICENSE file in the root directory of this source tree.
*/

// Drives CudaExecutor through the batching Runner on the toy decoder that
// export_toy_decoder.py writes to $ET_CUDA_BATCHING_TOY_DIR, and checks every
// generation against the eager greedy continuation it recorded.
// Drives CudaExecutor through the batching Runner on the toy decoders that
// export_toy_decoder.py writes under $ET_CUDA_BATCHING_TOY_DIR, and checks
// every generation against the eager greedy continuation it recorded. Runs
// once per artifact: prefill exported from two tokens, and from five, where
// narrower slices must run as decodes and prefill's selector is padded.

#include <executorch/backends/cuda/batching/cuda_executor.h>
#include <executorch/extension/llm/batching/decode_first_scheduler.h>
Expand Down Expand Up @@ -54,7 +56,7 @@ std::vector<batching::Token> parse_tokens(const std::string& text) {
return tokens;
}

class CudaExecutorTest : public ::testing::Test {
class CudaExecutorTest : public ::testing::TestWithParam<const char*> {
protected:
void SetUp() override {
int devices = 0;
Expand All @@ -66,7 +68,7 @@ class CudaExecutorTest : public ::testing::Test {
GTEST_SKIP() << "ET_CUDA_BATCHING_TOY_DIR is not set; run "
"export_toy_decoder.py first";
}
dir_ = dir;
dir_ = std::string(dir) + "/" + GetParam();
std::ifstream in(dir_ + "/expected.txt");
ASSERT_TRUE(in.is_open()) << dir_ << "/expected.txt";
std::string line;
Expand Down Expand Up @@ -165,7 +167,7 @@ class CudaExecutorTest : public ::testing::Test {

} // namespace

TEST_F(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) {
TEST_P(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) {
auto exec = executor();
ASSERT_NE(exec, nullptr);
EXPECT_EQ(exec->preferred_batch_tokens(), static_cast<size_t>(kMaxStep));
Expand Down Expand Up @@ -195,7 +197,7 @@ TEST_F(CudaExecutorTest, ConcurrentGenerationsMatchEagerGreedy) {
EXPECT_GT(engine.decode_sessions_total, engine.steps / 2);
}

TEST_F(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) {
TEST_P(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) {
cb::CudaExecutorOptions options;
options.cuda_graph_for_decode = false;
auto exec = executor(options);
Expand All @@ -208,7 +210,7 @@ TEST_F(CudaExecutorTest, EagerDecodeMatchesTheCapturedGraph) {
}
}

TEST_F(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) {
TEST_P(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) {
auto exec = executor();
ASSERT_NE(exec, nullptr);
batching::Runner runner(*exec, scheduler());
Expand All @@ -219,7 +221,7 @@ TEST_F(CudaExecutorTest, SamePromptTwiceInOneBatchGeneratesTheSame) {
EXPECT_EQ(generations[1].tokens, c.expected);
}

TEST_F(CudaExecutorTest, SessionsReuseCellsAcrossRounds) {
TEST_P(CudaExecutorTest, SessionsReuseCellsAcrossRounds) {
auto exec = executor();
ASSERT_NE(exec, nullptr);
batching::Runner runner(*exec, scheduler());
Expand All @@ -235,7 +237,7 @@ TEST_F(CudaExecutorTest, SessionsReuseCellsAcrossRounds) {
runner.shutdown();
}

TEST_F(CudaExecutorTest, RefusesLimitsThePoolCannotHold) {
TEST_P(CudaExecutorTest, RefusesLimitsThePoolCannotHold) {
// 8 sessions of the full 64-token context need 512 cells; the program has
// 256.
EXPECT_EQ(
Expand All @@ -248,3 +250,8 @@ TEST_F(CudaExecutorTest, RefusesLimitsThePoolCannotHold) {
.error(),
Error::InvalidArgument);
}

INSTANTIATE_TEST_SUITE_P(
Toy,
CudaExecutorTest,
::testing::Values("min2", "min5"));