diff --git a/pack/.post_operation/20260924_vllm_patch_mooncake_store_pending_load/cuda/Dockerfile b/pack/.post_operation/20260924_vllm_patch_mooncake_store_pending_load/cuda/Dockerfile new file mode 100644 index 00000000..c4cd5baa --- /dev/null +++ b/pack/.post_operation/20260924_vllm_patch_mooncake_store_pending_load/cuda/Dockerfile @@ -0,0 +1,101 @@ +ARG CMAKE_MAX_JOBS +ARG CUDA_VERSION=12.9 +ARG VLLM_VERSION=0.29.0 + +FROM gpustack/runner:cuda${CUDA_VERSION}-vllm${VLLM_VERSION} AS vllm +SHELL ["/bin/bash", "-eo", "pipefail", "-c"] + +ARG TARGETPLATFORM +ARG TARGETOS +ARG TARGETARCH + +## Stop MooncakeStoreConnector from saving a load MultiConnector rejected +## +## With MultiConnector(MooncakeConnector, MooncakeStoreConnector) on the decode side, a prompt the +## prefill side has already written to the store is a hit for BOTH children. MultiConnector gives +## the request to MooncakeConnector, which loads asynchronously, and calls the store child with +## zero external tokens, so the store's LoadSpec stays can_load=False. The request is then parked +## waiting for remote KV and scheduled neither as new nor as cached, which routes it into the +## store scheduler's "pending load specs not yet scheduled" branch. That branch passes +## skip_save=None, so the rejected LoadSpec turns into a SAVE job, and _apply_current_save_block_ids +## finds no current block table for a request the core did not schedule: +## +## AssertionError: Missing current block table for store request ... +## +## EngineCore exits on the first such request. The assertion arrived with upstream #51358, first +## released in 0.29.0; 0.27.1 has the same branch without the assertion. +## +## Upstream fixed it in #54643 (34b1e9f7a6), released in 0.30.0: the branch skips a LoadSpec whose +## load this connector does not own. The hunk below is that change verbatim, rebased onto the 0.29.0 +## line numbers. It is also carried into new builds as patches/vllm/003_*.patch. + +RUN <