diff --git a/backends/arm/CMakeLists.txt b/backends/arm/CMakeLists.txt index d91d8add6a8..d5d704b3920 100644 --- a/backends/arm/CMakeLists.txt +++ b/backends/arm/CMakeLists.txt @@ -273,6 +273,45 @@ if(EXECUTORCH_BUILD_VGF) # vgf backend list(TRANSFORM _vgf_backend_sources PREPEND "${EXECUTORCH_ROOT}/") add_library(vgf_backend ${_vgf_backend_sources}) + + # Should be absent in ordinary production builds. + option(EXECUTORCH_VGF_IO_STATS "Collect per-execution VGF copy statistics" + OFF + ) + if(EXECUTORCH_VGF_IO_STATS) + target_sources( + vgf_backend + PRIVATE ${EXECUTORCH_ROOT}/backends/arm/runtime/VGFExecutionStats.cpp + ) + target_compile_definitions(vgf_backend PUBLIC EXECUTORCH_VGF_IO_STATS=1) + endif() + if(EXECUTORCH_BUILD_TESTS) + find_package(Threads REQUIRED) + foreach(_stats_mode IN ITEMS enabled disabled) + set(_stats_target vgf_execution_stats_${_stats_mode}_test) + add_executable( + ${_stats_target} + ${EXECUTORCH_ROOT}/backends/arm/test/vgf_execution_stats_test.cpp + ${EXECUTORCH_ROOT}/backends/arm/runtime/VGFExecutionStats.cpp + ) + target_include_directories( + ${_stats_target} PRIVATE ${_common_include_directories} + ) + target_compile_features(${_stats_target} PRIVATE cxx_std_17) + target_link_libraries(${_stats_target} PRIVATE Threads::Threads) + if(_stats_mode STREQUAL "enabled") + target_compile_definitions( + ${_stats_target} PRIVATE EXECUTORCH_VGF_IO_STATS=1 + ) + else() + target_compile_definitions( + ${_stats_target} PRIVATE EXECUTORCH_VGF_IO_STATS=0 + ) + endif() + add_test(NAME ${_stats_target} COMMAND ${_stats_target}) + endforeach() + endif() + install(TARGETS vgf_backend EXPORT ExecuTorchTargets) target_include_directories( vgf_backend PRIVATE ${_common_include_directories} ${VULKAN_HEADERS_PATH} diff --git a/backends/arm/runtime/VGFBackend.cpp b/backends/arm/runtime/VGFBackend.cpp index dc2af5dbed7..d01b80c1773 100644 --- a/backends/arm/runtime/VGFBackend.cpp +++ b/backends/arm/runtime/VGFBackend.cpp @@ -65,6 +65,7 @@ using executorch::runtime::EventTracerEntry; #include #include +#include #include namespace executorch { @@ -448,6 +449,35 @@ class VGFBackend final : public ::executorch::runtime::BackendInterface { } is_initialized_ = true; + +#if defined(EXECUTORCH_VGF_IO_STATS) && EXECUTORCH_VGF_IO_STATS + // Selected device, queried once at initialization, not per + // execute. + VkPhysicalDeviceProperties properties{}; + vkGetPhysicalDeviceProperties(vk_physical_device, &properties); + VgfDeviceInfo info{}; + info.valid = true; + info.vendor_id = properties.vendorID; + info.device_id = properties.deviceID; + info.api_version = properties.apiVersion; + info.driver_version = properties.driverVersion; + std::memcpy( + info.device_name, properties.deviceName, sizeof(info.device_name)); + if (vkGetPhysicalDeviceProperties2 != nullptr && + properties.apiVersion >= VK_MAKE_VERSION(1, 2, 0)) { + VkPhysicalDeviceDriverProperties driver{}; + driver.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DRIVER_PROPERTIES; + VkPhysicalDeviceProperties2 properties2{}; + properties2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2; + properties2.pNext = &driver; + vkGetPhysicalDeviceProperties2(vk_physical_device, &properties2); + std::memcpy( + info.driver_name, driver.driverName, sizeof(info.driver_name)); + std::memcpy( + info.driver_info, driver.driverInfo, sizeof(info.driver_info)); + } + set_vgf_device_info(info); +#endif } // Vulkan teardown belongs to destroy(), not static destruction: the @@ -619,6 +649,8 @@ class VGFBackend final : public ::executorch::runtime::BackendInterface { } } + VGF_STATS_EXECUTION(handle); + #ifdef ET_EVENT_TRACER_ENABLED EventTracer* event_tracer = context.event_tracer(); @@ -685,7 +717,7 @@ class VGFBackend final : public ::executorch::runtime::BackendInterface { ET_LOG(Error, "Failed to map Vulkan IO memory"); return Error::Internal; } - memcpy(data, tensor->mutable_data_ptr(), io_size); + VGF_STATS_MEMCPY_IN(data, tensor->mutable_data_ptr(), io_size); repr->unmap_io(io); } @@ -798,7 +830,7 @@ class VGFBackend final : public ::executorch::runtime::BackendInterface { ET_LOG(Error, "Failed to map Vulkan IO memory"); return Error::Internal; } - memcpy(tensor->mutable_data_ptr(), data, io_size); + VGF_STATS_MEMCPY_OUT(tensor->mutable_data_ptr(), data, io_size); repr->unmap_io(io); } @@ -807,6 +839,7 @@ class VGFBackend final : public ::executorch::runtime::BackendInterface { event_tracer_end_profiling_delegate(event_tracer, vgf_execute_event); #endif + VGF_STATS_SUCCESS(); return Error::Ok; } diff --git a/backends/arm/runtime/VGFExecutionStats.cpp b/backends/arm/runtime/VGFExecutionStats.cpp new file mode 100644 index 00000000000..346e4de0bb1 --- /dev/null +++ b/backends/arm/runtime/VGFExecutionStats.cpp @@ -0,0 +1,57 @@ +/* + * Copyright 2026 Arm Limited and/or its affiliates. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ +#include + +#if defined(EXECUTORCH_VGF_IO_STATS) && EXECUTORCH_VGF_IO_STATS +#include + +namespace executorch::backends::vgf { +namespace { +thread_local VgfStatsBuffer* capture_buffer = nullptr; +thread_local VgfExecutionStats* execution_stats = nullptr; +std::mutex device_info_mutex; +VgfDeviceInfo device_info; +} // namespace + +// cppcheck-suppress unusedFunction +VgfStatsBuffer* set_vgf_stats_buffer(VgfStatsBuffer* buffer) noexcept { + auto* previous = capture_buffer; + capture_buffer = buffer; + return previous; +} +VgfExecutionStats* current_vgf_execution_stats() noexcept { + return execution_stats; +} +void set_vgf_device_info(const VgfDeviceInfo& info) { + std::lock_guard lock(device_info_mutex); + device_info = info; +} + +// cppcheck-suppress unusedFunction +VgfDeviceInfo get_vgf_device_info() { + std::lock_guard lock(device_info_mutex); + return device_info; +} + +ScopedVgfExecutionStats::ScopedVgfExecutionStats(const void* handle) noexcept + : previous_(execution_stats), start_(VgfStatsClock::now()) { + record_.delegate_handle = reinterpret_cast(handle); + execution_stats = &record_.stats; +} +ScopedVgfExecutionStats::~ScopedVgfExecutionStats() { + record_.stats.execute_ns = vgf_elapsed_ns(start_); + execution_stats = previous_; + if (capture_buffer) { + if (capture_buffer->size == capture_buffer->capacity) { + capture_buffer->overflow = true; + } else { + capture_buffer->records[capture_buffer->size++] = record_; + } + } +} +} // namespace executorch::backends::vgf +#endif diff --git a/backends/arm/runtime/VGFExecutionStats.h b/backends/arm/runtime/VGFExecutionStats.h new file mode 100644 index 00000000000..f776bfcaec2 --- /dev/null +++ b/backends/arm/runtime/VGFExecutionStats.h @@ -0,0 +1,159 @@ +/* + * Copyright 2026 Arm Limited and/or its affiliates. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ +#pragma once + +#include + +#if defined(EXECUTORCH_VGF_IO_STATS) && EXECUTORCH_VGF_IO_STATS +#include +#include +#include + +namespace executorch::backends::vgf { + +// One record per VGFBackend::execute(), not per whole-model invocation. +struct VgfExecutionStats { + uint64_t in_copy_bytes = 0; + uint64_t out_copy_bytes = 0; + uint64_t in_memcpy_ns = 0; + uint64_t out_memcpy_ns = 0; + uint64_t submit_wait_ns = 0; + uint64_t execute_ns = 0; + uint64_t imports = 0; + uint64_t import_ns = 0; + uint64_t binding_refresh_ns = 0; + uint64_t in_zero_copy_bytes = 0; + uint64_t out_zero_copy_bytes = 0; + // Future I/O zero-copy fallbacks, not undelegated CPU graph operators. + uint64_t fallback_count = 0; + uint64_t rejection_count = 0; + + void reset() { + *this = {}; + } +}; + +struct VgfExecutionRecord { + VgfExecutionStats stats; + uintptr_t delegate_handle = 0; + bool success = false; +}; + +// The caller owns this preallocated storage. No allocation, logging or file +// I/O is performed by execute() to publish a record. Overflow is an error for +// the benchmark runner, never silent truncation of measurements. +struct VgfStatsBuffer { + VgfExecutionRecord* records = nullptr; + size_t capacity = 0; + size_t size = 0; + bool overflow = false; +}; + +struct VgfDeviceInfo { + bool valid = false; + uint32_t vendor_id = 0; + uint32_t device_id = 0; + uint32_t api_version = 0; + uint32_t driver_version = 0; + char device_name[256] = {}; + char driver_name[256] = {}; + char driver_info[256] = {}; +}; + +VgfStatsBuffer* set_vgf_stats_buffer(VgfStatsBuffer* buffer) noexcept; +VgfExecutionStats* current_vgf_execution_stats() noexcept; +void set_vgf_device_info(const VgfDeviceInfo& info); +VgfDeviceInfo get_vgf_device_info(); + +using VgfStatsClock = std::chrono::steady_clock; +inline uint64_t vgf_elapsed_ns(VgfStatsClock::time_point start) noexcept { + return static_cast( + std::chrono::duration_cast( + VgfStatsClock::now() - start) + .count()); +} + +class ScopedVgfStatsTimer final { + public: + explicit ScopedVgfStatsTimer(uint64_t* destination) noexcept + : destination_(destination), start_(VgfStatsClock::now()) {} + ~ScopedVgfStatsTimer() { + if (destination_) + *destination_ += vgf_elapsed_ns(start_); + } + ScopedVgfStatsTimer(const ScopedVgfStatsTimer&) = delete; + ScopedVgfStatsTimer& operator=(const ScopedVgfStatsTimer&) = delete; + + private: + uint64_t* destination_; + VgfStatsClock::time_point start_; +}; + +class ScopedVgfExecutionStats final { + public: + explicit ScopedVgfExecutionStats(const void* handle) noexcept; + ~ScopedVgfExecutionStats(); + void mark_success() noexcept { + record_.success = true; + } + ScopedVgfExecutionStats(const ScopedVgfExecutionStats&) = delete; + ScopedVgfExecutionStats& operator=(const ScopedVgfExecutionStats&) = delete; + + private: + VgfExecutionRecord record_{}; + VgfExecutionStats* previous_; + VgfStatsClock::time_point start_; +}; + +inline void vgf_stats_memcpy(void* dst, const void* src, size_t n, bool input) { + auto* stats = current_vgf_execution_stats(); + if (!stats) { + std::memcpy(dst, src, n); + return; + } + { + ScopedVgfStatsTimer timer( + input ? &stats->in_memcpy_ns : &stats->out_memcpy_ns); + std::memcpy(dst, src, n); + } + // Count the exact memcpy length, after the timed region. This deliberately + // excludes map_io/unmap_io, which do not map/unmap Vulkan memory per call. + (input ? stats->in_copy_bytes : stats->out_copy_bytes) += n; +} + +} // namespace executorch::backends::vgf + +#define VGF_STATS_CAT_INNER(a, b) a##b +#define VGF_STATS_CAT(a, b) VGF_STATS_CAT_INNER(a, b) +#define VGF_STATS_EXECUTION(handle) \ + ::executorch::backends::vgf::ScopedVgfExecutionStats \ + vgf_execution_stats_scope(handle) +#define VGF_STATS_SUCCESS() vgf_execution_stats_scope.mark_success() +#define VGF_STATS_TIME(field) \ + ::executorch::backends::vgf::ScopedVgfStatsTimer VGF_STATS_CAT( \ + vgf_stats_timer_, __LINE__)( \ + ::executorch::backends::vgf::current_vgf_execution_stats() \ + ? &::executorch::backends::vgf::current_vgf_execution_stats() \ + -> field \ + : nullptr) +#define VGF_STATS_MEMCPY_IN(dst, src, n) \ + ::executorch::backends::vgf::vgf_stats_memcpy(dst, src, n, true) +#define VGF_STATS_MEMCPY_OUT(dst, src, n) \ + ::executorch::backends::vgf::vgf_stats_memcpy(dst, src, n, false) +#else +#define VGF_STATS_EXECUTION(handle) \ + do { \ + } while (false) +#define VGF_STATS_SUCCESS() \ + do { \ + } while (false) +#define VGF_STATS_TIME(field) \ + do { \ + } while (false) +#define VGF_STATS_MEMCPY_IN(dst, src, n) memcpy(dst, src, n) +#define VGF_STATS_MEMCPY_OUT(dst, src, n) memcpy(dst, src, n) +#endif diff --git a/backends/arm/runtime/VGFSetup.cpp b/backends/arm/runtime/VGFSetup.cpp index 2f5fe75da6b..3b627af2322 100644 --- a/backends/arm/runtime/VGFSetup.cpp +++ b/backends/arm/runtime/VGFSetup.cpp @@ -10,6 +10,7 @@ * appropriate vulkan structures. */ +#include #include #include @@ -4085,6 +4086,8 @@ bool VgfRepr::execute_vgf(executorch::runtime::EventTracer* event_tracer) { return false; } + // Submit + fence wait only; reset is above. + VGF_STATS_TIME(submit_wait_ns); result = vkQueueSubmit(vk_queue, 1, &submit, vk_execute_fence); if (result != VK_SUCCESS) { ET_LOG(Error, "VGF/VkFence wait failed, error %d", result); diff --git a/backends/arm/runtime/targets.bzl b/backends/arm/runtime/targets.bzl index 230e95e8fba..409091d29a5 100644 --- a/backends/arm/runtime/targets.bzl +++ b/backends/arm/runtime/targets.bzl @@ -40,6 +40,7 @@ def define_common_targets(): name = "vgf_backend", srcs = [ "VGFBackend.cpp", + "VGFExecutionStats.cpp", "VGFNeuralStatistics.cpp", "VGFSetup.cpp", # Volk must be compiled directly into this target so its global @@ -49,6 +50,7 @@ def define_common_targets(): "fbsource//third-party/vulkan-headers-1.4.343/v1.4.343/src:volk_arm_src", ], exported_headers = [ + "VGFExecutionStats.h", "VGFNeuralStatistics.h", "VGFSetup.h", "VGFVulkanFeatures.h", diff --git a/backends/arm/test/test_arm_backend.sh b/backends/arm/test/test_arm_backend.sh index 3d16faf713a..3004a9576ad 100755 --- a/backends/arm/test/test_arm_backend.sh +++ b/backends/arm/test/test_arm_backend.sh @@ -489,8 +489,12 @@ test_runtime_vgf() { # Build the VGF tests explicitly. This makes a missing/renamed target fail # the job instead of being hidden among unrelated C++ tests. cmake --build "${ctest_build_dir}" \ - --target vgf_neural_statistics_test vgf_vulkan_features_test \ - --parallel + --target \ + vgf_neural_statistics_test \ + vgf_vulkan_features_test \ + vgf_execution_stats_enabled_test \ + vgf_execution_stats_disabled_test \ + --parallel # --no-tests=error is intentional: discovering zero VGF tests must fail CI. ctest \ diff --git a/backends/arm/test/vgf_execution_stats_test.cpp b/backends/arm/test/vgf_execution_stats_test.cpp new file mode 100644 index 00000000000..2cab9ab6506 --- /dev/null +++ b/backends/arm/test/vgf_execution_stats_test.cpp @@ -0,0 +1,86 @@ +/* + * Copyright 2026 Arm Limited and/or its affiliates. + * + * This source code is licensed under the BSD-style license found in the + * LICENSE file in the root directory of this source tree. + */ +#include +#include +#include +#include +#include + +#define CHECK(expr) \ + do { \ + if (!(expr)) { \ + std::fprintf(stderr, "check failed at line %d: %s\n", __LINE__, #expr); \ + std::abort(); \ + } \ + } while (false) + +#if defined(EXECUTORCH_VGF_IO_STATS) && EXECUTORCH_VGF_IO_STATS +using namespace executorch::backends::vgf; + +void execute_for_test(size_t in_bytes, size_t out_bytes, bool success = true) { + VGF_STATS_EXECUTION(nullptr); + char input[128] = {}; + char output[128] = {}; + VGF_STATS_MEMCPY_IN(output, input, in_bytes); + VGF_STATS_MEMCPY_IN(output, input, 7); + { + // Two timers in the same scope test expansion of __LINE__. + VGF_STATS_TIME(submit_wait_ns); + VGF_STATS_TIME(binding_refresh_ns); + } + VGF_STATS_MEMCPY_OUT(output, input, out_bytes); + if (success) + VGF_STATS_SUCCESS(); +} + +int main() { + VgfExecutionRecord records[8]; + VgfStatsBuffer buffer{records, 8, 0, false}; + CHECK(set_vgf_stats_buffer(&buffer) == nullptr); + execute_for_test(11, 23); + execute_for_test(0, 5); + CHECK(buffer.size == 2); + CHECK(records[0].stats.in_copy_bytes == 18); + CHECK(records[0].stats.out_copy_bytes == 23); + CHECK(records[1].stats.in_copy_bytes == 7); + CHECK(records[1].stats.out_copy_bytes == 5); + CHECK(records[0].success && records[1].success); + CHECK(current_vgf_execution_stats() == nullptr); + // No nonzero timer assertion: a valid monotonic clock can have coarse ticks. + CHECK(records[0].stats.execute_ns >= records[0].stats.in_memcpy_ns); + execute_for_test(1, 2, false); + CHECK(!records[2].success); + records[0].stats.imports = 3; + records[0].stats.fallback_count = 4; + records[0].stats.reset(); + CHECK(records[0].stats.in_copy_bytes == 0); + CHECK(records[0].stats.imports == 0); + CHECK(records[0].stats.fallback_count == 0); + std::thread other([] { execute_for_test(100, 101); }); + other.join(); + CHECK(buffer.size == 3); // Capture does not cross threads. + for (int i = 0; i < 6; ++i) + execute_for_test(1, 1); + CHECK(buffer.size == 8 && buffer.overflow); + CHECK(set_vgf_stats_buffer(nullptr) == &buffer); + execute_for_test(1, 1); // No sink is also valid. + std::puts("VGF stats enabled: PASS"); +} +#else +int main() { + char src[] = "unchanged", dst[sizeof(src)] = {}; + int side_effect = 0; + VGF_STATS_EXECUTION(++side_effect); + VGF_STATS_TIME(no_such_field); + VGF_STATS_SUCCESS(); + VGF_STATS_MEMCPY_IN(dst, src, sizeof(src)); + CHECK(std::memcmp(dst, src, sizeof(src)) == 0); + VGF_STATS_MEMCPY_OUT(dst, src, sizeof(src)); + CHECK(side_effect == 0); + std::puts("VGF stats disabled: PASS"); +} +#endif