From 25b3f268ea95b76bd03c825a1681872c9b615428 Mon Sep 17 00:00:00 2001 From: Martin Vit Date: Sun, 2 Aug 2026 21:07:24 +0000 Subject: [PATCH 1/2] fix: retry large host registration in segments --- .../dl_binding/cuda_binding.hpp | 7 +- csrc/instant_tensor/host_registration.hpp | 69 ++++++++++++++ csrc/loader_common.cpp | 47 +++++++++- tests/cpp/test_host_registration.cpp | 89 +++++++++++++++++++ tests/test_host_registration.py | 21 +++++ 5 files changed, 228 insertions(+), 5 deletions(-) create mode 100644 csrc/instant_tensor/host_registration.hpp create mode 100644 tests/cpp/test_host_registration.cpp create mode 100644 tests/test_host_registration.py diff --git a/csrc/instant_tensor/dl_binding/cuda_binding.hpp b/csrc/instant_tensor/dl_binding/cuda_binding.hpp index ccf65b7..7bd7848 100644 --- a/csrc/instant_tensor/dl_binding/cuda_binding.hpp +++ b/csrc/instant_tensor/dl_binding/cuda_binding.hpp @@ -54,6 +54,7 @@ inline cudaError_t (*cudaEventSynchronize_fn)(cudaEvent_t event) = nullptr; inline cudaError_t (*cudaStreamWaitEvent_fn)(cudaStream_t stream, cudaEvent_t event, unsigned int flags) = nullptr; inline cudaError_t (*cudaHostRegister_fn)(void* ptr, size_t size, unsigned int flags) = nullptr; inline cudaError_t (*cudaHostUnregister_fn)(void* ptr) = nullptr; +inline cudaError_t (*cudaGetLastError_fn)() = nullptr; inline const char* (*cudaGetErrorString_fn)(cudaError_t error) = nullptr; inline bool init() { @@ -95,6 +96,7 @@ inline bool init() { cudaStreamWaitEvent_fn = resolve(lib_handle, {"cudaStreamWaitEvent", "hipStreamWaitEvent"}); cudaHostRegister_fn = resolve(lib_handle, {"cudaHostRegister", "hipHostRegister"}); cudaHostUnregister_fn = resolve(lib_handle, {"cudaHostUnregister", "hipHostUnregister"}); + cudaGetLastError_fn = resolve(lib_handle, {"cudaGetLastError", "hipGetLastError"}); cudaGetErrorString_fn = resolve(lib_handle, {"cudaGetErrorString", "hipGetErrorString"}); return true; @@ -137,9 +139,12 @@ inline cudaError_t cudaHostRegister(void* ptr, size_t size, unsigned int flags) inline cudaError_t cudaHostUnregister(void* ptr) { return cudaHostUnregister_fn(ptr); } +inline cudaError_t cudaGetLastError() { + return cudaGetLastError_fn(); +} inline const char* cudaGetErrorString(cudaError_t error) { return cudaGetErrorString_fn(error); } } // namespace cuda_binding -} \ No newline at end of file +} diff --git a/csrc/instant_tensor/host_registration.hpp b/csrc/instant_tensor/host_registration.hpp new file mode 100644 index 0000000..a506637 --- /dev/null +++ b/csrc/instant_tensor/host_registration.hpp @@ -0,0 +1,69 @@ +#pragma once + +#include +#include +#include +#include +#include +#include + +namespace instanttensor { + +struct HostRegistrationRange { + void* ptr; + size_t size; +}; + +struct HostRegistration { + std::vector ranges; + int whole_buffer_error = 0; +}; + +template +HostRegistration register_host_buffer( + void* ptr, + size_t size, + unsigned int flags, + size_t segment_size, + RegisterFn&& register_fn, + UnregisterFn&& unregister_fn, + ClearErrorFn&& clear_error_fn +) { + if (ptr == nullptr || size == 0 || segment_size == 0) { + throw std::invalid_argument("Host registration requires non-empty storage and segments"); + } + + HostRegistration registration; + int result = register_fn(ptr, size, flags); + if (result == 0) { + registration.ranges.push_back({ptr, size}); + return registration; + } + + registration.whole_buffer_error = result; + clear_error_fn(); + auto* base = static_cast(ptr); + for (size_t offset = 0; offset < size; offset += segment_size) { + const size_t current_size = std::min(segment_size, size - offset); + void* current_ptr = base + offset; + result = register_fn(current_ptr, current_size, flags); + if (result == 0) { + registration.ranges.push_back({current_ptr, current_size}); + continue; + } + + clear_error_fn(); + for (auto it = registration.ranges.rbegin(); it != registration.ranges.rend(); ++it) { + unregister_fn(it->ptr); + } + throw std::runtime_error( + "Host registration failed for the whole buffer (code " + + std::to_string(registration.whole_buffer_error) + + ") and segment at offset " + std::to_string(offset) + + " (code " + std::to_string(result) + ")" + ); + } + return registration; +} + +} // namespace instanttensor diff --git a/csrc/loader_common.cpp b/csrc/loader_common.cpp index 2b7e5d8..44f13cc 100644 --- a/csrc/loader_common.cpp +++ b/csrc/loader_common.cpp @@ -1,8 +1,13 @@ #include +#include namespace instanttensor { +namespace { +constexpr size_t HOST_REGISTER_SEGMENT_SIZE = 256ULL * 1024 * 1024; +} + int Loader::next_executor_request_id() { int request_id = this->executor_request_id; @@ -101,15 +106,49 @@ void Loader::init_buffer() { throw std::runtime_error("Failed to allocate host buffer: " + std::string(strerror(errno))); } - // NOTE: cudaHostRegisterReadOnly is not used since the host buffer is writable for the CPU - CUDA_CHECK(cudaHostRegister(this->host_buffer_entry.ptr, host_buffer_size, this->cuda_host_register_flags)); + // Some CUDA driver/kernel combinations reject a multi-GiB + // cudaHostRegister call even though the same storage can be pinned + // in smaller adjacent ranges. Keep the one-call fast path and only + // segment after that call fails. + HostRegistration registration; + try { + registration = register_host_buffer( + this->host_buffer_entry.ptr, + host_buffer_size, + this->cuda_host_register_flags, + HOST_REGISTER_SEGMENT_SIZE, + [](void* ptr, size_t size, unsigned int flags) { + return static_cast(cudaHostRegister(ptr, size, flags)); + }, + [](void* ptr) { + return static_cast(cudaHostUnregister(ptr)); + }, + []() { cudaGetLastError(); } + ); + } catch (...) { + free(this->host_buffer_entry.ptr); + this->host_buffer_entry.ptr = nullptr; + throw; + } + if (registration.whole_buffer_error != 0) { + fprintf( + stderr, + "[InstantTensor] cudaHostRegister rejected a %.2f GiB buffer " + "(code %d); registered it as %zu segments instead.\n", + host_buffer_size / static_cast(1ULL << 30), + registration.whole_buffer_error, + registration.ranges.size() + ); + } this->host_buffer_entry.size = host_buffer_size; - this->host_buffer_entry.deleter = [=](void *ptr) { + this->host_buffer_entry.deleter = [=](void* ptr) { if (this->backend == Backend::URING || this->backend == Backend::URING_BUFFERED) { this->deregister_host_buffer_uring(); } - CUDA_CHECK(cudaHostUnregister(ptr)); + for (auto it = registration.ranges.rbegin(); it != registration.ranges.rend(); ++it) { + CUDA_CHECK(cudaHostUnregister(it->ptr)); + } free(ptr); }; } diff --git a/tests/cpp/test_host_registration.cpp b/tests/cpp/test_host_registration.cpp new file mode 100644 index 0000000..85b9776 --- /dev/null +++ b/tests/cpp/test_host_registration.cpp @@ -0,0 +1,89 @@ +#include +#include +#include +#include + +#include + +using instanttensor::register_host_buffer; + +int main() { + std::vector storage(1024); + void* base = storage.data(); + + { + int register_calls = 0; + int clear_calls = 0; + auto registration = register_host_buffer( + base, + storage.size(), + 0, + 256, + [&](void*, size_t, unsigned int) { + ++register_calls; + return 0; + }, + [](void*) { return 0; }, + [&]() { ++clear_calls; } + ); + assert(register_calls == 1); + assert(clear_calls == 0); + assert(registration.whole_buffer_error == 0); + assert(registration.ranges.size() == 1); + } + + { + int register_calls = 0; + int clear_calls = 0; + auto registration = register_host_buffer( + base, + storage.size(), + 0, + 256, + [&](void*, size_t size, unsigned int) { + ++register_calls; + return size > 256 ? 1 : 0; + }, + [](void*) { return 0; }, + [&]() { ++clear_calls; } + ); + assert(register_calls == 5); + assert(clear_calls == 1); + assert(registration.whole_buffer_error == 1); + assert(registration.ranges.size() == 4); + } + + { + int register_calls = 0; + int clear_calls = 0; + std::vector unregistered; + bool threw = false; + try { + register_host_buffer( + base, + storage.size(), + 0, + 256, + [&](void*, size_t size, unsigned int) { + ++register_calls; + if (size > 256) { + return 1; + } + return register_calls == 4 ? 2 : 0; + }, + [&](void* ptr) { + unregistered.push_back(ptr); + return 0; + }, + [&]() { ++clear_calls; } + ); + } catch (const std::runtime_error&) { + threw = true; + } + assert(threw); + assert(clear_calls == 2); + assert(unregistered.size() == 2); + assert(unregistered[0] == static_cast(base) + 256); + assert(unregistered[1] == base); + } +} diff --git a/tests/test_host_registration.py b/tests/test_host_registration.py new file mode 100644 index 0000000..ceba81e --- /dev/null +++ b/tests/test_host_registration.py @@ -0,0 +1,21 @@ +import subprocess + + +def test_segmented_host_registration_fallback(tmp_path): + executable = tmp_path / "test_host_registration" + subprocess.run( + [ + "c++", + "-std=c++17", + "-Wall", + "-Wextra", + "-Werror", + "-I", + "csrc", + "tests/cpp/test_host_registration.cpp", + "-o", + str(executable), + ], + check=True, + ) + subprocess.run([str(executable)], check=True) From 49b4010afc1cae0441e71fe0b0bffc24fa05e932 Mon Sep 17 00:00:00 2001 From: Martin Vit Date: Mon, 3 Aug 2026 21:33:13 +0000 Subject: [PATCH 2/2] fix(loader): fall back to runtime pinned host buffers --- .../dl_binding/cuda_binding.hpp | 15 ++++ csrc/instant_tensor/host_registration.hpp | 74 +++++++++++++++++ csrc/loader_common.cpp | 82 +++++++++++-------- tests/cpp/test_host_registration.cpp | 66 +++++++++++++++ 4 files changed, 204 insertions(+), 33 deletions(-) diff --git a/csrc/instant_tensor/dl_binding/cuda_binding.hpp b/csrc/instant_tensor/dl_binding/cuda_binding.hpp index 7bd7848..f354f1a 100644 --- a/csrc/instant_tensor/dl_binding/cuda_binding.hpp +++ b/csrc/instant_tensor/dl_binding/cuda_binding.hpp @@ -18,6 +18,11 @@ #define cudaHostRegisterIoMemory 0x04 #define cudaHostRegisterReadOnly 0x08 +#define cudaHostAllocDefault 0x00 +#define cudaHostAllocPortable 0x01 +#define cudaHostAllocMapped 0x02 +#define cudaHostAllocWriteCombined 0x04 + namespace instanttensor { namespace cuda_binding { @@ -54,6 +59,8 @@ inline cudaError_t (*cudaEventSynchronize_fn)(cudaEvent_t event) = nullptr; inline cudaError_t (*cudaStreamWaitEvent_fn)(cudaStream_t stream, cudaEvent_t event, unsigned int flags) = nullptr; inline cudaError_t (*cudaHostRegister_fn)(void* ptr, size_t size, unsigned int flags) = nullptr; inline cudaError_t (*cudaHostUnregister_fn)(void* ptr) = nullptr; +inline cudaError_t (*cudaHostAlloc_fn)(void** ptr, size_t size, unsigned int flags) = nullptr; +inline cudaError_t (*cudaFreeHost_fn)(void* ptr) = nullptr; inline cudaError_t (*cudaGetLastError_fn)() = nullptr; inline const char* (*cudaGetErrorString_fn)(cudaError_t error) = nullptr; @@ -96,6 +103,8 @@ inline bool init() { cudaStreamWaitEvent_fn = resolve(lib_handle, {"cudaStreamWaitEvent", "hipStreamWaitEvent"}); cudaHostRegister_fn = resolve(lib_handle, {"cudaHostRegister", "hipHostRegister"}); cudaHostUnregister_fn = resolve(lib_handle, {"cudaHostUnregister", "hipHostUnregister"}); + cudaHostAlloc_fn = resolve(lib_handle, {"cudaHostAlloc", "hipHostMalloc"}); + cudaFreeHost_fn = resolve(lib_handle, {"cudaFreeHost", "hipHostFree"}); cudaGetLastError_fn = resolve(lib_handle, {"cudaGetLastError", "hipGetLastError"}); cudaGetErrorString_fn = resolve(lib_handle, {"cudaGetErrorString", "hipGetErrorString"}); @@ -139,6 +148,12 @@ inline cudaError_t cudaHostRegister(void* ptr, size_t size, unsigned int flags) inline cudaError_t cudaHostUnregister(void* ptr) { return cudaHostUnregister_fn(ptr); } +inline cudaError_t cudaHostAlloc(void** ptr, size_t size, unsigned int flags) { + return cudaHostAlloc_fn(ptr, size, flags); +} +inline cudaError_t cudaFreeHost(void* ptr) { + return cudaFreeHost_fn(ptr); +} inline cudaError_t cudaGetLastError() { return cudaGetLastError_fn(); } diff --git a/csrc/instant_tensor/host_registration.hpp b/csrc/instant_tensor/host_registration.hpp index a506637..9ac8d9f 100644 --- a/csrc/instant_tensor/host_registration.hpp +++ b/csrc/instant_tensor/host_registration.hpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -19,6 +20,13 @@ struct HostRegistration { int whole_buffer_error = 0; }; +struct HostBufferAllocation { + void* ptr = nullptr; + HostRegistration registration; + bool runtime_allocated = false; + std::string registration_failure; +}; + template HostRegistration register_host_buffer( void* ptr, @@ -66,4 +74,70 @@ HostRegistration register_host_buffer( return registration; } +template < + typename AlignedAllocFn, + typename FreeFn, + typename RegisterFn, + typename UnregisterFn, + typename ClearErrorFn, + typename RuntimeAllocFn> +HostBufferAllocation allocate_registered_host_buffer( + size_t size, + size_t alignment, + unsigned int register_flags, + unsigned int runtime_alloc_flags, + size_t segment_size, + AlignedAllocFn&& aligned_alloc_fn, + FreeFn&& free_fn, + RegisterFn&& register_fn, + UnregisterFn&& unregister_fn, + ClearErrorFn&& clear_error_fn, + RuntimeAllocFn&& runtime_alloc_fn +) { + if (size == 0 || alignment == 0) { + throw std::invalid_argument("Host allocation requires non-empty aligned storage"); + } + + HostBufferAllocation allocation; + allocation.ptr = aligned_alloc_fn(alignment, size); + if (allocation.ptr == nullptr) { + throw std::runtime_error("Failed to allocate aligned host storage"); + } + + try { + allocation.registration = register_host_buffer( + allocation.ptr, + size, + register_flags, + segment_size, + std::forward(register_fn), + std::forward(unregister_fn), + std::forward(clear_error_fn) + ); + return allocation; + } catch (const std::exception& error) { + allocation.registration_failure = error.what(); + } + + free_fn(allocation.ptr); + allocation.ptr = nullptr; + clear_error_fn(); + + const int result = runtime_alloc_fn( + &allocation.ptr, + size, + runtime_alloc_flags + ); + if (result != 0 || allocation.ptr == nullptr) { + throw std::runtime_error( + allocation.registration_failure + + "; runtime pinned allocation failed (code " + + std::to_string(result) + ")" + ); + } + + allocation.runtime_allocated = true; + return allocation; +} + } // namespace instanttensor diff --git a/csrc/loader_common.cpp b/csrc/loader_common.cpp index 44f13cc..4c109d8 100644 --- a/csrc/loader_common.cpp +++ b/csrc/loader_common.cpp @@ -100,44 +100,52 @@ void Loader::init_buffer() { size_t host_buffer_size = inflight_host_buffer_size; this->host_buffer_entry = host_buffer_cache->get(host_buffer_size); if (this->host_buffer_entry.ptr == NULL) { - // aligned_alloc + cudaHostRegister is faster than cudaHostAlloc - this->host_buffer_entry.ptr = aligned_alloc(this->thread_alignment, host_buffer_size); - if (this->host_buffer_entry.ptr == NULL) { - throw std::runtime_error("Failed to allocate host buffer: " + std::string(strerror(errno))); - } - // Some CUDA driver/kernel combinations reject a multi-GiB // cudaHostRegister call even though the same storage can be pinned - // in smaller adjacent ranges. Keep the one-call fast path and only - // segment after that call fails. - HostRegistration registration; - try { - registration = register_host_buffer( - this->host_buffer_entry.ptr, - host_buffer_size, - this->cuda_host_register_flags, - HOST_REGISTER_SEGMENT_SIZE, - [](void* ptr, size_t size, unsigned int flags) { - return static_cast(cudaHostRegister(ptr, size, flags)); - }, - [](void* ptr) { - return static_cast(cudaHostUnregister(ptr)); - }, - []() { cudaGetLastError(); } + // in smaller adjacent ranges. A few driver/kernel combinations + // reject registration itself; cudaHostAlloc remains the portable + // pinned-memory fallback for those hosts. + const unsigned int runtime_alloc_flags = + this->cuda_host_register_flags + & (cudaHostAllocPortable | cudaHostAllocMapped); + HostBufferAllocation allocation = allocate_registered_host_buffer( + host_buffer_size, + this->thread_alignment, + this->cuda_host_register_flags, + runtime_alloc_flags, + HOST_REGISTER_SEGMENT_SIZE, + [](size_t alignment, size_t size) { + return aligned_alloc(alignment, size); + }, + [](void* ptr) { free(ptr); }, + [](void* ptr, size_t size, unsigned int flags) { + return static_cast(cudaHostRegister(ptr, size, flags)); + }, + [](void* ptr) { + return static_cast(cudaHostUnregister(ptr)); + }, + []() { cudaGetLastError(); }, + [](void** ptr, size_t size, unsigned int flags) { + return static_cast(cudaHostAlloc(ptr, size, flags)); + } + ); + this->host_buffer_entry.ptr = allocation.ptr; + if (allocation.runtime_allocated) { + fprintf( + stderr, + "[InstantTensor] cudaHostRegister paths failed for a %.2f " + "GiB buffer; using runtime-allocated pinned memory. %s\n", + host_buffer_size / static_cast(1ULL << 30), + allocation.registration_failure.c_str() ); - } catch (...) { - free(this->host_buffer_entry.ptr); - this->host_buffer_entry.ptr = nullptr; - throw; - } - if (registration.whole_buffer_error != 0) { + } else if (allocation.registration.whole_buffer_error != 0) { fprintf( stderr, "[InstantTensor] cudaHostRegister rejected a %.2f GiB buffer " "(code %d); registered it as %zu segments instead.\n", host_buffer_size / static_cast(1ULL << 30), - registration.whole_buffer_error, - registration.ranges.size() + allocation.registration.whole_buffer_error, + allocation.registration.ranges.size() ); } @@ -146,10 +154,18 @@ void Loader::init_buffer() { if (this->backend == Backend::URING || this->backend == Backend::URING_BUFFERED) { this->deregister_host_buffer_uring(); } - for (auto it = registration.ranges.rbegin(); it != registration.ranges.rend(); ++it) { - CUDA_CHECK(cudaHostUnregister(it->ptr)); + if (allocation.runtime_allocated) { + CUDA_CHECK(cudaFreeHost(ptr)); + } else { + for ( + auto it = allocation.registration.ranges.rbegin(); + it != allocation.registration.ranges.rend(); + ++it + ) { + CUDA_CHECK(cudaHostUnregister(it->ptr)); + } + free(ptr); } - free(ptr); }; } this->host_buffer = (char*)this->host_buffer_entry.ptr; diff --git a/tests/cpp/test_host_registration.cpp b/tests/cpp/test_host_registration.cpp index 85b9776..05f5f98 100644 --- a/tests/cpp/test_host_registration.cpp +++ b/tests/cpp/test_host_registration.cpp @@ -1,11 +1,13 @@ #include #include #include +#include #include #include using instanttensor::register_host_buffer; +using instanttensor::allocate_registered_host_buffer; int main() { std::vector storage(1024); @@ -86,4 +88,68 @@ int main() { assert(unregistered[0] == static_cast(base) + 256); assert(unregistered[1] == base); } + + { + std::vector runtime_storage(1024); + int ordinary_free_calls = 0; + int runtime_alloc_calls = 0; + auto allocation = allocate_registered_host_buffer( + storage.size(), + 256, + 0, + 0, + 256, + [&](size_t, size_t) { return base; }, + [&](void* ptr) { + assert(ptr == base); + ++ordinary_free_calls; + }, + [](void*, size_t, unsigned int) { return 1; }, + [](void*) { return 0; }, + []() {}, + [&](void** ptr, size_t size, unsigned int flags) { + ++runtime_alloc_calls; + assert(size == runtime_storage.size()); + assert(flags == 0); + *ptr = runtime_storage.data(); + return 0; + } + ); + assert(allocation.ptr == runtime_storage.data()); + assert(allocation.runtime_allocated); + assert(allocation.registration.ranges.empty()); + assert(!allocation.registration_failure.empty()); + assert(ordinary_free_calls == 1); + assert(runtime_alloc_calls == 1); + } + + { + int ordinary_free_calls = 0; + bool threw = false; + try { + allocate_registered_host_buffer( + storage.size(), + 256, + 0, + 0, + 256, + [&](size_t, size_t) { return base; }, + [&](void* ptr) { + assert(ptr == base); + ++ordinary_free_calls; + }, + [](void*, size_t, unsigned int) { return 1; }, + [](void*) { return 0; }, + []() {}, + [](void**, size_t, unsigned int) { return 2; } + ); + } catch (const std::runtime_error& error) { + threw = true; + assert(std::string(error.what()).find( + "runtime pinned allocation failed (code 2)" + ) != std::string::npos); + } + assert(threw); + assert(ordinary_free_calls == 1); + } }