From c0ad6cdc24482ce91d2a2dcfa63f4c7afa8360ff Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 13:26:40 +0900 Subject: [PATCH 1/9] feat: GPU usage monitoring with selectable animation source (CPU/GPU/Both) - New 'Usage Source' tray menu (CPU / GPU / CPU + GPU), persisted per platform (Windows registry / macOS defaults / Linux settings.conf), default CPU so existing installs are unaffected. Tooltip shows the selected source's usage; sleep/idle logic follows effective usage. - New GpuMonitor trait (fn get_gpu_usage() -> io::Result) with per-platform implementations: - Windows: PDH \GPU Engine(*)\Utilization Percentage (Win10 1803+, all vendors); max across engine instances (Task Manager semantics). - macOS: IORegistry PerformanceStatistics 'Device Utilization %' (powermetrics data source, no root). IOKit owns the IOServiceMatching dictionary (CFReleasing it segfaults - bisected on real hardware), so the looked-up service handle is cached for the process lifetime with a one-shot re-lookup on failed reads. - Linux: /sys/class/drm/card*/device/gpu_busy_percent + nvidia-smi, max of available sources. - build.rs: winres import host-gated, resource embedding target-gated (winres is a target.'cfg(windows)' build-dep resolved against host). - Verified: Windows (RTX 4090, all 3 sources), Linux (nix build + smoke test on NixOS), macOS (real hardware, real IORegistry reads). --- Cargo.lock | 3 + Cargo.toml | 4 +- build.rs | 17 ++- src/app.rs | 170 ++++++++++++++++++++++++------ src/events.rs | 18 ++++ src/platform/linux/gpu_usage.rs | 74 +++++++++++++ src/platform/linux/mod.rs | 2 + src/platform/linux/settings.rs | 10 ++ src/platform/macos/gpu_usage.rs | 143 +++++++++++++++++++++++++ src/platform/macos/mod.rs | 2 + src/platform/macos/settings.rs | 10 ++ src/platform/mod.rs | 18 ++++ src/platform/windows/gpu_usage.rs | 146 +++++++++++++++++++++++++ src/platform/windows/mod.rs | 2 + src/platform/windows/settings.rs | 29 +++++ 15 files changed, 615 insertions(+), 33 deletions(-) create mode 100644 src/platform/linux/gpu_usage.rs create mode 100644 src/platform/macos/gpu_usage.rs create mode 100644 src/platform/windows/gpu_usage.rs diff --git a/Cargo.lock b/Cargo.lock index 9ab64ec..3060ce9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -643,7 +643,9 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536" dependencies = [ "bitflags 2.9.1", + "block2", "dispatch2", + "libc", "objc2", ] @@ -867,6 +869,7 @@ dependencies = [ "flate2", "objc2", "objc2-app-kit", + "objc2-core-foundation", "objc2-foundation", "trayicon", "windows", diff --git a/Cargo.toml b/Cargo.toml index a362c13..0c507b5 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -35,6 +35,7 @@ dirs = "6.0" [target.'cfg(target_os = "macos")'.dependencies] objc2 = "0.6" objc2-app-kit = { version = "0.3" } +objc2-core-foundation = { version = "0.3" } objc2-foundation = { version = "0.3" } dirs = "6.0" dispatch = "0.2.0" @@ -46,7 +47,8 @@ features = [ "Win32_UI_WindowsAndMessaging", "Win32_UI_Shell", "Win32_System_Threading", - "Win32_System_SystemInformation" + "Win32_System_SystemInformation", + "Win32_System_Performance" ] [profile.release] diff --git a/build.rs b/build.rs index e15bbea..6468344 100644 --- a/build.rs +++ b/build.rs @@ -1,7 +1,8 @@ use flate2::{write::GzEncoder, Compression}; use std::{collections::HashMap, io, path::Path}; -// only include winres if compiling for Windows +// winres is a target.'cfg(windows)'.build-dependency, which cargo resolves +// against the *build host* — so it is only importable from a Windows host. #[cfg(target_os = "windows")] use winres::WindowsResource; @@ -30,14 +31,26 @@ fn main() -> io::Result<()> { println!("cargo:rustc-cfg=release"); } + // The build script runs on the *host*, but winres can only embed + // resources when the *target* is Windows (it errors out otherwise). + // The import above is host-gated to match how cargo resolves the + // target.'cfg(windows)' build-dependency; the runtime check below + // gates the actual work on the target. + let target_os = std::env::var("CARGO_CFG_TARGET_OS").unwrap_or_default(); + // Generate Windows resource file if compiling for Windows #[cfg(target_os = "windows")] - { + if target_os == "windows" { let mut res = WindowsResource::new(); res.set_icon("assets/appIcon.ico"); res.compile()?; } + // Link the IOKit framework for GPU usage monitoring (PerformanceStatistics) + if target_os == "macos" { + println!("cargo:rustc-link-lib=framework=IOKit"); + } + generate_icon_resources()?; Ok(()) } diff --git a/src/app.rs b/src/app.rs index cfb14bc..33b388e 100644 --- a/src/app.rs +++ b/src/app.rs @@ -6,12 +6,96 @@ use std::time::Duration; use crate::events::{build_menu, Events}; use crate::icon_manager::{IconManager, Theme}; -use crate::platform::{CpuMonitor, SettingsManager, SystemIntegration}; -use crate::platform::{CpuMonitorImpl, SettingsManagerImpl, SystemIntegrationImpl}; +use crate::platform::{CpuMonitor, GpuMonitor, SettingsManager, SystemIntegration}; +use crate::platform::{ + CpuMonitorImpl, GpuMonitorImpl, SettingsManagerImpl, SystemIntegrationImpl, +}; use crate::debug; use trayicon::*; +/// The usage source that drives the animation speed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum AnimationSource { + Cpu, + Gpu, + Both, +} + +impl AnimationSource { + /// Parse from the persisted string form; unknown values default to CPU. + pub fn from_str(s: &str) -> Self { + match s.to_ascii_lowercase().as_str() { + "gpu" => AnimationSource::Gpu, + "both" => AnimationSource::Both, + _ => AnimationSource::Cpu, + } + } + + /// Persisted string form (registry value / defaults key / settings.conf). + pub fn as_str(self) -> &'static str { + match self { + AnimationSource::Cpu => "cpu", + AnimationSource::Gpu => "gpu", + AnimationSource::Both => "both", + } + } + + /// Human-readable label for menus and logs. + pub fn label(self) -> &'static str { + match self { + AnimationSource::Cpu => "CPU", + AnimationSource::Gpu => "GPU", + AnimationSource::Both => "CPU + GPU", + } + } +} + +/// Sample CPU and/or GPU usage depending on the selected source. +/// +/// A `None` component means it was not sampled (the source does not need it) +/// or the read failed. +fn sample_usage(source: AnimationSource) -> (Option, Option) { + let cpu = match source { + AnimationSource::Gpu => None, + _ => CpuMonitorImpl::get_cpu_usage() + .map_err(|e| eprintln!("Failed to get CPU usage: {}", e)) + .ok(), + }; + let gpu = match source { + AnimationSource::Cpu => None, + _ => GpuMonitorImpl::get_gpu_usage() + .map_err(|e| eprintln!("Failed to get GPU usage: {}", e)) + .ok(), + }; + (cpu, gpu) +} + +/// The usage value that drives the animation: CPU, GPU, or the max of both. +/// `None` means no usable data was available this sample. +fn effective_usage( + source: AnimationSource, + cpu: Option, + gpu: Option, +) -> Option { + match source { + AnimationSource::Cpu => cpu, + AnimationSource::Gpu => gpu, + AnimationSource::Both => match (cpu, gpu) { + (Some(c), Some(g)) => Some(c.max(g)), + (c, g) => c.or(g), + }, + } +} + +/// Format a usage value for tooltips; `None` renders as "n/a". +fn fmt_pct(usage: Option) -> String { + match usage { + Some(v) => format!("{:.2}%", v), + None => "n/a".to_string(), + } +} + fn is_sleep_time() -> bool { let hour = SystemIntegrationImpl::get_local_hour(); // Check if current hour is between 22:00 and 6:00 @@ -40,6 +124,7 @@ pub struct App { event_receiver: Option>, icon_name: Arc>, theme: Arc>, + animation_source: Arc>, } impl App { @@ -53,6 +138,7 @@ impl App { let exit_flag = Arc::new(AtomicBool::new(false)); let theme = initial_theme.unwrap_or_else(SettingsManagerImpl::get_current_theme); + let animation_source = SettingsManagerImpl::get_animation_source(); let initial_icons = icon_manager .get_icon_set(initial_icon, Some(theme)) .ok_or("Invalid initial icon name")?; @@ -62,7 +148,7 @@ impl App { let _ = sender.send(e.clone()); }) .icon(initial_icons[0].clone()) - .tooltip("~Nyan~ RustCat - CPU Usage Monitor") + .tooltip("~Nyan~ RustCat - CPU/GPU Usage Monitor") .menu(build_menu(&icon_manager)) .on_right_click(Events::ShowMenu) .on_double_click(Events::RunTaskmgr) @@ -75,6 +161,7 @@ impl App { event_receiver: Some(receiver), icon_name: Arc::new(Mutex::new(initial_icon.to_string())), theme: Arc::new(Mutex::new(theme)), + animation_source: Arc::new(Mutex::new(animation_source)), }) } @@ -84,6 +171,7 @@ impl App { let icon_manager = self.icon_manager.clone(); let icon_name = self.icon_name.clone(); let theme = self.theme.clone(); + let animation_source = self.animation_source.clone(); thread::spawn(move || { let sleep_interval = 10; @@ -137,35 +225,40 @@ impl App { if update_counter >= 1000 { update_counter = 0; - let usage = match CpuMonitorImpl::get_cpu_usage() { - Ok(usage) => usage, - Err(e) => { - eprintln!("Failed to get CPU usage: {}", e); - continue; - } - }; - speed = (200.0 / (usage / 5.0).clamp(1.0_f64, 20.0_f64)).round() as u64; - debug!("CPU Usage: {:.2}% speed: {}", usage, speed); + let source = *animation_source.lock().unwrap(); + let (cpu_usage, gpu_usage) = sample_usage(source); + let usage = effective_usage(source, cpu_usage, gpu_usage); - // Check if CPU is idle (less than 5% usage) and it's sleep time (22:00-6:00) - if usage < 5.0 && is_sleep_time() { - idle_counter += 1000; // Add the update interval - if idle_counter >= idle_threshold && !is_sleeping { - is_sleeping = true; - icon_index = 0; // Reset animation to start from first sleeping frame - debug!("CPU has been idle for 1 minutes during sleep hours, switching to sleeping cat"); - } - } else { - idle_counter = 0; - if is_sleeping { - is_sleeping = false; - icon_index = 0; // Reset animation - if usage >= 5.0 { - debug!("CPU activity detected, switching back to normal cat"); - } else { - debug!("Outside sleep hours, switching back to normal cat"); + if let Some(usage) = usage { + speed = (200.0 / (usage / 5.0).clamp(1.0_f64, 20.0_f64)).round() as u64; + debug!("{} Usage: {:.2}% speed: {}", source.label(), usage, speed); + + // Check if the machine is idle (less than 5% usage) and + // it's sleep time (22:00-6:00) + if usage < 5.0 && is_sleep_time() { + idle_counter += 1000; // Add the update interval + if idle_counter >= idle_threshold && !is_sleeping { + is_sleeping = true; + icon_index = 0; // Reset animation to start from first sleeping frame + debug!("Usage has been idle for 1 minute during sleep hours, switching to sleeping cat"); + } + } else { + idle_counter = 0; + if is_sleeping { + is_sleeping = false; + icon_index = 0; // Reset animation + if usage >= 5.0 { + debug!("Activity detected, switching back to normal cat"); + } else { + debug!("Outside sleep hours, switching back to normal cat"); + } } } + } else { + debug!( + "No {} usage data available, keeping previous speed", + source.label() + ); } { @@ -173,7 +266,19 @@ impl App { let tooltip = if is_sleeping && current_icon_name == "cat" { "Shhhh, Your CPU is sleeping...💤".to_string() } else { - format!("CPU Usage: {:.2}%", usage) + match source { + AnimationSource::Cpu => { + format!("CPU Usage: {}", fmt_pct(cpu_usage)) + } + AnimationSource::Gpu => { + format!("GPU Usage: {}", fmt_pct(gpu_usage)) + } + AnimationSource::Both => format!( + "CPU: {} | GPU: {}", + fmt_pct(cpu_usage), + fmt_pct(gpu_usage) + ), + } }; ui_update(move || { if let Ok(mut tray) = tray_icon_clone.lock() { @@ -223,6 +328,11 @@ impl App { *self.icon_name.lock().unwrap() = icon_name; self.update_menu(); } + Events::SetAnimationSource(source) => { + SettingsManagerImpl::set_animation_source(source); + *self.animation_source.lock().unwrap() = source; + self.update_menu(); + } Events::ToggleRunOnStart => { let current_state = SettingsManagerImpl::is_run_on_start_enabled(); SettingsManagerImpl::set_run_on_start(!current_state); diff --git a/src/events.rs b/src/events.rs index 6d5c739..9814fa2 100644 --- a/src/events.rs +++ b/src/events.rs @@ -1,3 +1,4 @@ +use crate::app::AnimationSource; use crate::icon_manager::{IconManager, Theme}; use crate::platform::{SettingsManager, SettingsManagerImpl}; use crate::debug; @@ -8,6 +9,7 @@ pub enum Events { Exit, SetTheme(Theme), SetIcon(String), + SetAnimationSource(AnimationSource), RunTaskmgr, ToggleRunOnStart, ShowAboutDialog, @@ -60,6 +62,22 @@ pub fn build_menu(icon_manager: &IconManager) -> MenuBuilder { menu = menu.submenu("Icon", icon_menu); } + // Build usage source submenu - what drives the animation speed + let current_source = SettingsManagerImpl::get_animation_source(); + let mut source_menu = MenuBuilder::new(); + for source in [ + AnimationSource::Cpu, + AnimationSource::Gpu, + AnimationSource::Both, + ] { + source_menu = source_menu.radio( + source.label(), + current_source == source, + Events::SetAnimationSource(source), + ); + } + menu = menu.submenu("Usage Source", source_menu); + menu.separator() .checkable( "Run on Start", diff --git a/src/platform/linux/gpu_usage.rs b/src/platform/linux/gpu_usage.rs new file mode 100644 index 0000000..79051b0 --- /dev/null +++ b/src/platform/linux/gpu_usage.rs @@ -0,0 +1,74 @@ +use crate::platform::GpuMonitor; +use std::fs; +use std::io; +use std::process::Command; + +pub struct LinuxGpuMonitor; + +impl GpuMonitor for LinuxGpuMonitor { + fn get_gpu_usage() -> io::Result { + let sysfs = sysfs_gpu_usage(); + let nvidia = nvidia_smi_gpu_usage(); + + match (sysfs, nvidia) { + (Some(a), Some(b)) => Ok(a.max(b)), + (Some(a), None) | (None, Some(a)) => Ok(a), + (None, None) => Err(io::Error::other( + "No GPU usage source found (no /sys/class/drm/*/device/gpu_busy_percent and nvidia-smi unavailable)", + )), + } + } +} + +/// amdgpu / i915 (and other DRM drivers) expose a per-card busy percentage +/// directly in sysfs — no helper tool needed. Returns the max across cards. +fn sysfs_gpu_usage() -> Option { + let entries = fs::read_dir("/sys/class/drm").ok()?; + let mut max: Option = None; + for entry in entries.flatten() { + let name = entry.file_name(); + let name = name.to_string_lossy(); + if !name.starts_with("card") { + continue; + } + let path = format!("/sys/class/drm/{name}/device/gpu_busy_percent"); + if let Ok(content) = fs::read_to_string(path) { + if let Ok(v) = content.trim().parse::() { + max = Some(max.map_or(v, |m| m.max(v))); + } + } + } + max +} + +/// NVIDIA GPUs have no sysfs busy percentage — query `nvidia-smi` instead. +/// Returns the max utilization across all NVIDIA GPUs. +fn nvidia_smi_gpu_usage() -> Option { + let output = Command::new("nvidia-smi") + .args(["--query-gpu=utilization.gpu", "--format=csv,noheader,nounits"]) + .output() + .ok()?; + if !output.status.success() { + return None; + } + String::from_utf8_lossy(&output.stdout) + .lines() + .filter_map(|line| line.trim().parse::().ok()) + .max_by(f64::total_cmp) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Smoke test: exercise the sampling path and print the reading so it + /// can be checked manually on a machine with a GPU. Deliberately + /// lenient — a machine with no GPU at all is not a failure. + #[test] + fn gpu_usage_smoke() { + match LinuxGpuMonitor::get_gpu_usage() { + Ok(v) => eprintln!("Linux GPU usage: {:.2}%", v), + Err(e) => eprintln!("Linux GPU usage unavailable: {}", e), + } + } +} diff --git a/src/platform/linux/mod.rs b/src/platform/linux/mod.rs index 301dac7..a807213 100644 --- a/src/platform/linux/mod.rs +++ b/src/platform/linux/mod.rs @@ -1,8 +1,10 @@ pub mod app; pub mod cpu_usage; +pub mod gpu_usage; pub mod settings; pub mod system_integration; pub use cpu_usage::LinuxCpuMonitor; +pub use gpu_usage::LinuxGpuMonitor; pub use settings::LinuxSettingsManager; pub use system_integration::LinuxSystemIntegration; \ No newline at end of file diff --git a/src/platform/linux/settings.rs b/src/platform/linux/settings.rs index d74f21c..76f44ff 100644 --- a/src/platform/linux/settings.rs +++ b/src/platform/linux/settings.rs @@ -34,6 +34,16 @@ impl SettingsManager for LinuxSettingsManager { } } + fn get_animation_source() -> crate::app::AnimationSource { + read_setting("AnimationSource") + .map(|s| crate::app::AnimationSource::from_str(&s)) + .unwrap_or(crate::app::AnimationSource::Cpu) + } + + fn set_animation_source(source: crate::app::AnimationSource) { + write_setting("AnimationSource", source.as_str()); + } + fn is_run_on_start_enabled() -> bool { autostart_desktop_path().exists() } diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs new file mode 100644 index 0000000..0257d1c --- /dev/null +++ b/src/platform/macos/gpu_usage.rs @@ -0,0 +1,143 @@ +use crate::platform::GpuMonitor; +use objc2_core_foundation::{CFDictionary, CFNumber, CFString, CFType, CFRetained}; +use std::ffi::{c_void, CString}; +use std::io; +use std::ptr::NonNull; +use std::sync::Mutex; + +type MachPort = u32; + +/// The default (current) Mach port; IOKit interprets 0 as "this process". +const K_IO_MAIN_PORT_DEFAULT: MachPort = 0; + +// The service handle returned by IOServiceGetMatchingService is cached for +// the process lifetime (see GPU_SERVICE), so it is never released. +extern "C" { + fn IOServiceMatching(name: *const i8) -> *mut c_void; + fn IOServiceGetMatchingService(main_port: MachPort, matching: *mut c_void) -> u32; + fn IORegistryEntryCreateCFProperty( + entry: u32, + key: *const c_void, + allocator: *const c_void, + options: u32, + ) -> *const c_void; +} + +/// Cached IORegistry service handle for the GPU. The service number is +/// stable for the process lifetime, so it is looked up once and reused. +static GPU_SERVICE: Mutex> = Mutex::new(None); + +pub struct MacosGpuMonitor; + +impl GpuMonitor for MacosGpuMonitor { + fn get_gpu_usage() -> io::Result { + let service = get_gpu_service(); + if service == 0 { + return Err(io::Error::other( + "No GPU performance statistics available (no IOAccelerator service found)", + )); + } + match read_device_utilization(service) { + Some(usage) => Ok(usage), + None => { + // The property read failed — the cached service may be + // stale (e.g. after a GPU hot-plug). Invalidate the cache + // and retry once. + *GPU_SERVICE.lock().unwrap() = None; + let service = get_gpu_service(); + if service == 0 { + return Err(io::Error::other( + "No GPU performance statistics available (no IOAccelerator service found)", + )); + } + match read_device_utilization(service) { + Some(usage) => Ok(usage), + None => Err(io::Error::other( + "GPU service found but no 'Device Utilization %' statistic", + )), + } + } + } + } +} + +/// Return the cached GPU service handle, looking it up on first use. +fn get_gpu_service() -> u32 { + let mut guard = GPU_SERVICE.lock().unwrap(); + if guard.is_none() { + *guard = Some(lookup_gpu_service()); + } + guard.unwrap_or(0) +} + +/// Find the first service exposing GPU performance statistics. +/// +/// NOTE: the dictionary returned by `IOServiceMatching` must NOT be +/// released by us after `IOServiceGetMatchingService` — IOKit takes +/// ownership of it during the lookup (observed 2026-09-11: CFReleasing it +/// afterwards segfaults, and the property dictionary is allocated at the +/// matching dictionary's freed address). +fn lookup_gpu_service() -> u32 { + for class in ["IOAccelerator", "AGXAccelerator"] { + let Ok(class_cstr) = CString::new(class) else { + continue; + }; + let matching_raw = unsafe { IOServiceMatching(class_cstr.as_ptr()) }; + if matching_raw.is_null() { + continue; + } + let service = unsafe { IOServiceGetMatchingService(K_IO_MAIN_PORT_DEFAULT, matching_raw) }; + if service != 0 && read_device_utilization(service).is_some() { + return service; + } + } + 0 +} + +/// Read "Device Utilization %" from the IORegistry `PerformanceStatistics` +/// property of the given service. +/// +/// This is the same data source `powermetrics` uses, and it is readable +/// without root privileges. Returns `None` when the service does not +/// expose the key. +fn read_device_utilization(service: u32) -> Option { + let prop_key = CFString::from_str("PerformanceStatistics"); + let prop_raw = unsafe { + IORegistryEntryCreateCFProperty( + service, + CFRetained::as_ptr(&prop_key).as_ptr() as *const c_void, + std::ptr::null(), + 0, + ) + }; + if prop_raw.is_null() { + return None; + } + // Take ownership of the property (create rule). + let prop = unsafe { CFRetained::from_raw(NonNull::new_unchecked(prop_raw as *mut CFType)) }; + + let dict = prop.downcast::().ok()?; + // Cast to the concrete key/value types we expect. + let dict = unsafe { CFRetained::cast_unchecked::>(dict) }; + + let stat_key = CFString::from_str("Device Utilization %"); + let value = dict.get(&stat_key)?; + let number = value.downcast::().ok()?; + number.as_f64() +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Smoke test: exercise the sampling path and print the reading so it + /// can be checked manually on a machine with a GPU. Deliberately + /// lenient — a machine with no GPU at all is not a failure. + #[test] + fn gpu_usage_smoke() { + match MacosGpuMonitor::get_gpu_usage() { + Ok(v) => eprintln!("macOS GPU usage: {:.2}%", v), + Err(e) => eprintln!("macOS GPU usage unavailable: {}", e), + } + } +} diff --git a/src/platform/macos/mod.rs b/src/platform/macos/mod.rs index 1e1b5c0..b30ee62 100644 --- a/src/platform/macos/mod.rs +++ b/src/platform/macos/mod.rs @@ -1,8 +1,10 @@ pub mod app; pub mod cpu_usage; +pub mod gpu_usage; pub mod settings; pub mod system_integration; pub use cpu_usage::MacosCpuMonitor; +pub use gpu_usage::MacosGpuMonitor; pub use settings::MacosSettingsManager; pub use system_integration::MacosSystemIntegration; diff --git a/src/platform/macos/settings.rs b/src/platform/macos/settings.rs index 5e9d836..d9d6635 100644 --- a/src/platform/macos/settings.rs +++ b/src/platform/macos/settings.rs @@ -36,6 +36,16 @@ impl SettingsManager for MacosSettingsManager { } } + fn get_animation_source() -> crate::app::AnimationSource { + get_preference("AnimationSource") + .map(|s| crate::app::AnimationSource::from_str(&s)) + .unwrap_or(crate::app::AnimationSource::Cpu) + } + + fn set_animation_source(source: crate::app::AnimationSource) { + set_preference("AnimationSource", source.as_str()); + } + fn is_run_on_start_enabled() -> bool { let plist_path = dirs::home_dir() .unwrap_or_else(|| PathBuf::from("/tmp")) diff --git a/src/platform/mod.rs b/src/platform/mod.rs index 0fd3c68..af8d327 100644 --- a/src/platform/mod.rs +++ b/src/platform/mod.rs @@ -13,12 +13,23 @@ pub trait CpuMonitor { fn get_cpu_usage() -> io::Result; } +/// Cross-platform GPU usage monitoring trait +pub trait GpuMonitor { + /// Returns GPU usage percentage as a float (0.0 to 100.0). + /// + /// Returns `Err` when no GPU usage source is available (no GPU, no + /// driver, no `nvidia-smi`, ...). + fn get_gpu_usage() -> io::Result; +} + /// Cross-platform settings management trait pub trait SettingsManager { fn get_current_icon() -> String; fn set_current_icon(icon_name: &str); fn get_current_theme() -> crate::icon_manager::Theme; fn set_current_theme(theme: Option); + fn get_animation_source() -> crate::app::AnimationSource; + fn set_animation_source(source: crate::app::AnimationSource); fn is_run_on_start_enabled() -> bool; fn set_run_on_start(enable: bool); fn is_dark_mode_enabled() -> bool; @@ -40,6 +51,13 @@ pub type CpuMonitorImpl = macos::MacosCpuMonitor; #[cfg(target_os = "linux")] pub type CpuMonitorImpl = linux::LinuxCpuMonitor; +#[cfg(windows)] +pub type GpuMonitorImpl = windows::WindowsGpuMonitor; +#[cfg(target_os = "macos")] +pub type GpuMonitorImpl = macos::MacosGpuMonitor; +#[cfg(target_os = "linux")] +pub type GpuMonitorImpl = linux::LinuxGpuMonitor; + #[cfg(windows)] pub type SettingsManagerImpl = windows::WindowsSettingsManager; #[cfg(target_os = "macos")] diff --git a/src/platform/windows/gpu_usage.rs b/src/platform/windows/gpu_usage.rs new file mode 100644 index 0000000..cc6d1e7 --- /dev/null +++ b/src/platform/windows/gpu_usage.rs @@ -0,0 +1,146 @@ +use crate::platform::GpuMonitor; +use std::io; +use std::sync::Mutex; +use windows::core::PCWSTR; +use windows::Win32::System::Performance::{ + PDH_FMT_COUNTERVALUE, PDH_FMT_COUNTERVALUE_ITEM_W, PDH_FMT_DOUBLE, PDH_HCOUNTER, PDH_HQUERY, + PDH_MORE_DATA, PdhAddEnglishCounterW, PdhCollectQueryData, PdhCloseQuery, + PdhGetFormattedCounterArrayW, PdhGetFormattedCounterValue, PdhOpenQueryW, +}; + +pub struct WindowsGpuMonitor; + +/// PDH query state, opened lazily on the first sample and reused for the +/// lifetime of the process (a tray app never tears it down). +struct GpuPdhState { + hquery: PDH_HQUERY, + hcounter: PDH_HCOUNTER, +} + +// The PDH handles are only ever touched from the animation thread, which +// owns the `Mutex` guard; the raw handles themselves are not Send. +unsafe impl Send for GpuPdhState {} + +static GPU_STATE: Mutex> = Mutex::new(None); + +/// Per-engine utilization counter, available since Windows 10 1803 for all +/// GPU vendors (NVIDIA, AMD, Intel). The `(*)` wildcard expands to one +/// instance per GPU engine (3D, Copy, Video Decode, ...); we report the max +/// across engines as the overall GPU utilization, mirroring how Task +/// Manager surfaces "GPU busy". +const GPU_UTIL_COUNTER: PCWSTR = + windows::core::w!("\\GPU Engine(*)\\Utilization Percentage"); + +impl GpuMonitor for WindowsGpuMonitor { + fn get_gpu_usage() -> io::Result { + let mut state = GPU_STATE.lock().unwrap(); + if state.is_none() { + let mut hquery = PDH_HQUERY::default(); + let ret = unsafe { PdhOpenQueryW(None::<&PCWSTR>, 0, &mut hquery) }; + if ret != 0 { + return Err(io::Error::other(format!( + "PdhOpenQueryW failed: 0x{:08X}", + ret + ))); + } + + let mut hcounter = PDH_HCOUNTER::default(); + let ret = unsafe { + PdhAddEnglishCounterW(hquery, GPU_UTIL_COUNTER, 0, &mut hcounter) + }; + if ret != 0 { + unsafe { + let _ = PdhCloseQuery(hquery); + } + return Err(io::Error::other(format!( + "Counter not available (GPU counter missing on this system): 0x{:08X}", + ret + ))); + } + + // Prime the query: the first collect initializes the counters, + // so the first real sample has a baseline. + unsafe { + let _ = PdhCollectQueryData(hquery); + }; + + *state = Some(GpuPdhState { hquery, hcounter }); + } + + let state = state.as_ref().expect("state checked above"); + let ret = unsafe { PdhCollectQueryData(state.hquery) }; + if ret == PDH_MORE_DATA { + return Err(io::Error::other("PDH data not ready yet")); + } + if ret != 0 { + return Err(io::Error::other(format!( + "PdhCollectQueryData failed: 0x{:08X}", + ret + ))); + } + + // Wildcard counters expand into one instance per engine; the array + // API returns them all in one call (two-pass buffer query). + let mut size: u32 = 0; + let mut count: u32 = 0; + let ret = unsafe { + PdhGetFormattedCounterArrayW(state.hcounter, PDH_FMT_DOUBLE, &mut size, &mut count, None) + }; + if ret == PDH_MORE_DATA && count > 0 { + let item_size = std::mem::size_of::(); + let item_count = (size as usize) / item_size; + let mut items: Vec = (0..item_count) + .map(|_| PDH_FMT_COUNTERVALUE_ITEM_W::default()) + .collect(); + let ret = unsafe { + PdhGetFormattedCounterArrayW( + state.hcounter, + PDH_FMT_DOUBLE, + &mut size, + &mut count, + Some(items.as_mut_ptr()), + ) + }; + if ret == 0 { + let mut max: Option = None; + for item in items.iter() { + // PDH_CSTATUS_SUCCESS == 0; skip engines with no data. + if item.FmtValue.CStatus == 0 { + let v = unsafe { item.FmtValue.Anonymous.doubleValue }; + max = Some(max.map_or(v, |m| m.max(v))); + } + } + return max + .ok_or_else(|| io::Error::other("No GPU engine utilization data available")); + } + Err(io::Error::other(format!( + "PdhGetFormattedCounterArrayW failed: 0x{:08X}", + ret + ))) + } else if ret == 0 { + // Single-instance counter (no wildcard expansion). + let mut status: u32 = 0; + let mut value = PDH_FMT_COUNTERVALUE::default(); + let ret = unsafe { + PdhGetFormattedCounterValue( + state.hcounter, + PDH_FMT_DOUBLE, + Some(&mut status), + &mut value, + ) + }; + if ret == 0 && status == 0 { + return Ok(unsafe { value.Anonymous.doubleValue }); + } + Err(io::Error::other(format!( + "PdhGetFormattedCounterValue failed: 0x{:08X} (status: {})", + ret, status + ))) + } else { + Err(io::Error::other(format!( + "PdhGetFormattedCounterArrayW unexpected status: 0x{:08X}", + ret + ))) + } + } +} diff --git a/src/platform/windows/mod.rs b/src/platform/windows/mod.rs index 6518e71..06247ba 100644 --- a/src/platform/windows/mod.rs +++ b/src/platform/windows/mod.rs @@ -1,8 +1,10 @@ pub mod app; pub mod cpu_usage; +pub mod gpu_usage; pub mod settings; pub mod system_integration; pub use cpu_usage::WindowsCpuMonitor; +pub use gpu_usage::WindowsGpuMonitor; pub use settings::WindowsSettingsManager; pub use system_integration::WindowsSystemIntegration; diff --git a/src/platform/windows/settings.rs b/src/platform/windows/settings.rs index 9bdc4e6..178ed49 100644 --- a/src/platform/windows/settings.rs +++ b/src/platform/windows/settings.rs @@ -76,6 +76,35 @@ impl SettingsManager for WindowsSettingsManager { } } + fn get_animation_source() -> crate::app::AnimationSource { + let key = RegKey::predef(HKEY_CURRENT_USER); + if let Ok(sub_key) = key.open_subkey_with_flags("Software\\RustCat", KEY_READ) { + if let Ok(source_str) = sub_key.get_value::("AnimationSource") { + return crate::app::AnimationSource::from_str(&source_str); + } + } + + // Default source: CPU (original behavior) + crate::app::AnimationSource::Cpu + } + + fn set_animation_source(source: crate::app::AnimationSource) { + let key = RegKey::predef(HKEY_CURRENT_USER); + let sub_key = if let Ok(sub_key) = + key.open_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) + { + sub_key + } else { + key.create_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) + .expect("create_subkey_with_flags") + .0 + }; + + sub_key + .set_value("AnimationSource", &source.as_str()) + .expect("set_value"); + } + fn is_run_on_start_enabled() -> bool { let hkcu = RegKey::predef(HKEY_CURRENT_USER); if let Ok(run_key) = hkcu.open_subkey_with_flags( From adec2afb611d3b6533b65b4790af1b21683e301b Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 14:11:54 +0900 Subject: [PATCH 2/9] feat: per-GPU selection and per-engine utilization aggregation - Windows: parse PDH GPU Engine instance names (pid/luid/phys/eng), sum per-process utilization per physical engine (capped at 100%), max across engines; fixes underreporting when multiple processes share an engine and a latent 318-byte buffer overflow (allocated size/24 items instead of size bytes / count items). - macOS: enumerate ALL IOAccelerator/AGXAccelerator services via IOServiceGetServices instead of only the first match; per-service model name for the menu. - Linux: per-card entries from nvidia-smi (index+name) and sysfs (amdgpu); i915 no longer claimed (needs perf/PMU, not viable for a tray app). - New tray menu group 'GPU Device' (All GPUs + per-device, shown only with >=2 GPUs); persisted GpuScope setting on all three platforms; auto-fallback to All GPUs when the selected device disappears. Addresses the multi-GPU aggregation review comments on PR #41. --- src/app.rs | 34 ++- src/events.rs | 20 +- src/platform/linux/gpu_usage.rs | 132 +++++++--- src/platform/linux/settings.rs | 11 + src/platform/macos/gpu_usage.rs | 192 ++++++++++---- src/platform/macos/settings.rs | 11 + src/platform/mod.rs | 24 +- src/platform/windows/gpu_usage.rs | 406 ++++++++++++++++++++++-------- src/platform/windows/settings.rs | 36 +++ 9 files changed, 681 insertions(+), 185 deletions(-) diff --git a/src/app.rs b/src/app.rs index 33b388e..152c506 100644 --- a/src/app.rs +++ b/src/app.rs @@ -54,8 +54,12 @@ impl AnimationSource { /// Sample CPU and/or GPU usage depending on the selected source. /// /// A `None` component means it was not sampled (the source does not need it) -/// or the read failed. -fn sample_usage(source: AnimationSource) -> (Option, Option) { +/// or the read failed. `gpu_scope` limits GPU sampling to one device +/// (`None` = all devices, max across devices). +fn sample_usage( + source: AnimationSource, + gpu_scope: Option<&str>, +) -> (Option, Option) { let cpu = match source { AnimationSource::Gpu => None, _ => CpuMonitorImpl::get_cpu_usage() @@ -64,7 +68,7 @@ fn sample_usage(source: AnimationSource) -> (Option, Option) { }; let gpu = match source { AnimationSource::Cpu => None, - _ => GpuMonitorImpl::get_gpu_usage() + _ => GpuMonitorImpl::get_gpu_usage(gpu_scope) .map_err(|e| eprintln!("Failed to get GPU usage: {}", e)) .ok(), }; @@ -125,6 +129,8 @@ pub struct App { icon_name: Arc>, theme: Arc>, animation_source: Arc>, + /// Selected GPU device id; `None` = all GPUs. + gpu_scope: Arc>>, } impl App { @@ -139,6 +145,7 @@ impl App { let theme = initial_theme.unwrap_or_else(SettingsManagerImpl::get_current_theme); let animation_source = SettingsManagerImpl::get_animation_source(); + let gpu_scope = SettingsManagerImpl::get_gpu_scope(); let initial_icons = icon_manager .get_icon_set(initial_icon, Some(theme)) .ok_or("Invalid initial icon name")?; @@ -162,6 +169,7 @@ impl App { icon_name: Arc::new(Mutex::new(initial_icon.to_string())), theme: Arc::new(Mutex::new(theme)), animation_source: Arc::new(Mutex::new(animation_source)), + gpu_scope: Arc::new(Mutex::new(gpu_scope)), }) } @@ -172,6 +180,7 @@ impl App { let icon_name = self.icon_name.clone(); let theme = self.theme.clone(); let animation_source = self.animation_source.clone(); + let gpu_scope = self.gpu_scope.clone(); thread::spawn(move || { let sleep_interval = 10; @@ -226,7 +235,19 @@ impl App { if update_counter >= 1000 { update_counter = 0; let source = *animation_source.lock().unwrap(); - let (cpu_usage, gpu_usage) = sample_usage(source); + let scope = gpu_scope.lock().unwrap().clone(); + let (cpu_usage, gpu_usage) = sample_usage(source, scope.as_deref()); + if gpu_usage.is_none() { + if let Some(scope) = &scope { + // The scoped device may have disappeared (e.g. an + // eGPU unplugged); fall back to all GPUs. Only + // heal when enumeration itself succeeded. + let gpus = GpuMonitorImpl::enumerate_gpus(); + if !gpus.is_empty() && !gpus.iter().any(|d| &d.id == scope) { + *gpu_scope.lock().unwrap() = None; + } + } + } let usage = effective_usage(source, cpu_usage, gpu_usage); if let Some(usage) = usage { @@ -333,6 +354,11 @@ impl App { *self.animation_source.lock().unwrap() = source; self.update_menu(); } + Events::SetGpuScope(scope) => { + SettingsManagerImpl::set_gpu_scope(scope.clone()); + *self.gpu_scope.lock().unwrap() = scope; + self.update_menu(); + } Events::ToggleRunOnStart => { let current_state = SettingsManagerImpl::is_run_on_start_enabled(); SettingsManagerImpl::set_run_on_start(!current_state); diff --git a/src/events.rs b/src/events.rs index 9814fa2..550c1ed 100644 --- a/src/events.rs +++ b/src/events.rs @@ -1,6 +1,6 @@ use crate::app::AnimationSource; use crate::icon_manager::{IconManager, Theme}; -use crate::platform::{SettingsManager, SettingsManagerImpl}; +use crate::platform::{GpuMonitor, GpuMonitorImpl, SettingsManager, SettingsManagerImpl}; use crate::debug; use trayicon::MenuBuilder; @@ -10,6 +10,8 @@ pub enum Events { SetTheme(Theme), SetIcon(String), SetAnimationSource(AnimationSource), + /// Select which GPU drives the animation; `None` = all GPUs. + SetGpuScope(Option), RunTaskmgr, ToggleRunOnStart, ShowAboutDialog, @@ -78,6 +80,22 @@ pub fn build_menu(icon_manager: &IconManager) -> MenuBuilder { } menu = menu.submenu("Usage Source", source_menu); + // Build GPU device submenu — only shown when the machine has more than + // one GPU that exposes utilization. + let current_scope = SettingsManagerImpl::get_gpu_scope(); + let gpus = GpuMonitorImpl::enumerate_gpus(); + if gpus.len() > 1 { + let mut gpu_menu = MenuBuilder::new(); + gpu_menu = gpu_menu + .radio("All GPUs", current_scope.is_none(), Events::SetGpuScope(None)); + for device in &gpus { + let is_current = current_scope.as_deref() == Some(device.id.as_str()); + gpu_menu = gpu_menu + .radio(&device.name, is_current, Events::SetGpuScope(Some(device.id.clone()))); + } + menu = menu.submenu("GPU Device", gpu_menu); + } + menu.separator() .checkable( "Run on Start", diff --git a/src/platform/linux/gpu_usage.rs b/src/platform/linux/gpu_usage.rs index 79051b0..369e014 100644 --- a/src/platform/linux/gpu_usage.rs +++ b/src/platform/linux/gpu_usage.rs @@ -1,60 +1,129 @@ -use crate::platform::GpuMonitor; +use crate::platform::{GpuDevice, GpuMonitor}; use std::fs; use std::io; use std::process::Command; pub struct LinuxGpuMonitor; +/// A device id + name + reading from one utilization source. +struct GpuEntry { + id: String, + name: String, + value: f64, +} + impl GpuMonitor for LinuxGpuMonitor { - fn get_gpu_usage() -> io::Result { - let sysfs = sysfs_gpu_usage(); - let nvidia = nvidia_smi_gpu_usage(); + fn enumerate_gpus() -> Vec { + let mut devices: Vec = Vec::new(); + for entry in sysfs_gpu_entries() { + devices.push(GpuDevice { + id: entry.id, + name: entry.name, + }); + } + for entry in nvidia_smi_entries() { + devices.push(GpuDevice { + id: entry.id, + name: entry.name, + }); + } + devices + } + + fn get_gpu_usage(scope: Option<&str>) -> io::Result { + let mut entries = Vec::new(); + entries.extend(sysfs_gpu_entries()); + entries.extend(nvidia_smi_entries()); - match (sysfs, nvidia) { - (Some(a), Some(b)) => Ok(a.max(b)), - (Some(a), None) | (None, Some(a)) => Ok(a), - (None, None) => Err(io::Error::other( + if entries.is_empty() { + return Err(io::Error::other( "No GPU usage source found (no /sys/class/drm/*/device/gpu_busy_percent and nvidia-smi unavailable)", - )), + )); + } + if scope.map_or(false, |s| !entries.iter().any(|e| e.id == s)) { + return Err(io::Error::other( + "Selected GPU device no longer exists (fall back to all GPUs)", + )); } + let mut max: Option = None; + for e in &entries { + if let Some(scope) = scope { + if e.id != scope { + continue; + } + } + max = Some(max.map_or(e.value, |m| m.max(e.value))); + } + max.ok_or_else(|| { + io::Error::other("No GPU usage source found (no /sys/class/drm/*/device/gpu_busy_percent and nvidia-smi unavailable)") + }) } } -/// amdgpu / i915 (and other DRM drivers) expose a per-card busy percentage -/// directly in sysfs — no helper tool needed. Returns the max across cards. -fn sysfs_gpu_usage() -> Option { - let entries = fs::read_dir("/sys/class/drm").ok()?; - let mut max: Option = None; - for entry in entries.flatten() { +/// amdgpu (and a few other DRM drivers) expose a per-card busy percentage +/// directly in sysfs — no helper tool needed. Note: i915 does NOT expose +/// `gpu_busy_percent`; Intel utilization requires the i915 perf/PMU +/// interface, which needs perf_event permissions a tray app cannot rely on, +/// so on Intel-only systems this source simply finds no cards. +/// Returns one entry per card. +fn sysfs_gpu_entries() -> Vec { + let mut entries = Vec::new(); + let Ok(dir) = fs::read_dir("/sys/class/drm") else { + return entries; + }; + for entry in dir.flatten() { let name = entry.file_name(); let name = name.to_string_lossy(); - if !name.starts_with("card") { + let Some(index) = name.strip_prefix("card") else { continue; - } + }; let path = format!("/sys/class/drm/{name}/device/gpu_busy_percent"); if let Ok(content) = fs::read_to_string(path) { if let Ok(v) = content.trim().parse::() { - max = Some(max.map_or(v, |m| m.max(v))); + entries.push(GpuEntry { + id: format!("card{index}"), + name: format!("GPU card{index}"), + value: v, + }); } } } - max + entries } -/// NVIDIA GPUs have no sysfs busy percentage — query `nvidia-smi` instead. -/// Returns the max utilization across all NVIDIA GPUs. -fn nvidia_smi_gpu_usage() -> Option { - let output = Command::new("nvidia-smi") - .args(["--query-gpu=utilization.gpu", "--format=csv,noheader,nounits"]) +/// NVIDIA GPUs have no sysfs busy percentage — query `nvidia-smi` instead, +/// asking for the per-GPU index, name and utilization in one CSV pass. +/// Returns one entry per NVIDIA GPU. +fn nvidia_smi_entries() -> Vec { + let mut entries = Vec::new(); + let Ok(output) = Command::new("nvidia-smi") + .args(["--query-gpu=index,name,utilization.gpu", "--format=csv,noheader,nounits"]) .output() - .ok()?; + else { + return entries; + }; if !output.status.success() { - return None; + return entries; + } + for line in String::from_utf8_lossy(&output.stdout).lines() { + // Rows look like: `0, NVIDIA GeForce RTX 3080, 12` + let mut parts = line.splitn(3, ','); + let (Some(index), Some(name), Some(util)) = + (parts.next(), parts.next(), parts.next()) + else { + continue; + }; + let Ok(v) = util.trim().parse::() else { + continue; + }; + let index = index.trim(); + entries.push(GpuEntry { + id: format!("nvidia-{index}"), + name: name.trim().to_string(), + value: v, + }); } - String::from_utf8_lossy(&output.stdout) - .lines() - .filter_map(|line| line.trim().parse::().ok()) - .max_by(f64::total_cmp) + entries } #[cfg(test)] @@ -66,7 +135,8 @@ mod tests { /// lenient — a machine with no GPU at all is not a failure. #[test] fn gpu_usage_smoke() { - match LinuxGpuMonitor::get_gpu_usage() { + eprintln!("Linux GPUs: {:?}", LinuxGpuMonitor::enumerate_gpus()); + match LinuxGpuMonitor::get_gpu_usage(None) { Ok(v) => eprintln!("Linux GPU usage: {:.2}%", v), Err(e) => eprintln!("Linux GPU usage unavailable: {}", e), } diff --git a/src/platform/linux/settings.rs b/src/platform/linux/settings.rs index 76f44ff..21cc0c9 100644 --- a/src/platform/linux/settings.rs +++ b/src/platform/linux/settings.rs @@ -44,6 +44,17 @@ impl SettingsManager for LinuxSettingsManager { write_setting("AnimationSource", source.as_str()); } + fn get_gpu_scope() -> Option { + read_setting("GpuScope").filter(|s| !s.is_empty()) + } + + fn set_gpu_scope(scope: Option) { + match scope { + Some(scope) => write_setting("GpuScope", &scope), + None => remove_setting("GpuScope"), + } + } + fn is_run_on_start_enabled() -> bool { autostart_desktop_path().exists() } diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs index 0257d1c..88eb643 100644 --- a/src/platform/macos/gpu_usage.rs +++ b/src/platform/macos/gpu_usage.rs @@ -1,5 +1,5 @@ -use crate::platform::GpuMonitor; -use objc2_core_foundation::{CFDictionary, CFNumber, CFString, CFType, CFRetained}; +use crate::platform::{GpuDevice, GpuMonitor}; +use objc2_core_foundation::{CFArray, CFDictionary, CFNumber, CFString, CFType, CFRetained}; use std::ffi::{c_void, CString}; use std::io; use std::ptr::NonNull; @@ -10,11 +10,20 @@ type MachPort = u32; /// The default (current) Mach port; IOKit interprets 0 as "this process". const K_IO_MAIN_PORT_DEFAULT: MachPort = 0; -// The service handle returned by IOServiceGetMatchingService is cached for -// the process lifetime (see GPU_SERVICE), so it is never released. +/// A GPU service we track: its IORegistry handle, a stable-per-boot id for +/// menu selection, and a human-readable name. +#[derive(Clone)] +struct GpuEntry { + service: u32, + id: String, + name: String, +} + +// Service handles returned by IOKit are cached for the process lifetime +// (see GPU_ENTRIES) and never released. extern "C" { fn IOServiceMatching(name: *const i8) -> *mut c_void; - fn IOServiceGetMatchingService(main_port: MachPort, matching: *mut c_void) -> u32; + fn IOServiceGetServices(main_port: MachPort, matching: *mut c_void) -> *mut c_void; fn IORegistryEntryCreateCFProperty( entry: u32, key: *const c_void, @@ -23,61 +32,98 @@ extern "C" { ) -> *const c_void; } -/// Cached IORegistry service handle for the GPU. The service number is -/// stable for the process lifetime, so it is looked up once and reused. -static GPU_SERVICE: Mutex> = Mutex::new(None); +/// Cached list of GPU services. IOAccelerator/AGXAccelerator services are +/// looked up once and reused; a failed property read invalidates the cache +/// so a hot-plugged GPU is picked up on the next sample. +static GPU_ENTRIES: Mutex>> = Mutex::new(None); pub struct MacosGpuMonitor; impl GpuMonitor for MacosGpuMonitor { - fn get_gpu_usage() -> io::Result { - let service = get_gpu_service(); - if service == 0 { + fn enumerate_gpus() -> Vec { + gpu_entries() + .into_iter() + .map(|e| GpuDevice { + id: e.id, + name: e.name, + }) + .collect() + } + + fn get_gpu_usage(scope: Option<&str>) -> io::Result { + let mut entries = gpu_entries(); + if entries.is_empty() { return Err(io::Error::other( "No GPU performance statistics available (no IOAccelerator service found)", )); } - match read_device_utilization(service) { - Some(usage) => Ok(usage), - None => { - // The property read failed — the cached service may be - // stale (e.g. after a GPU hot-plug). Invalidate the cache - // and retry once. - *GPU_SERVICE.lock().unwrap() = None; - let service = get_gpu_service(); - if service == 0 { - return Err(io::Error::other( - "No GPU performance statistics available (no IOAccelerator service found)", - )); - } - match read_device_utilization(service) { - Some(usage) => Ok(usage), - None => Err(io::Error::other( - "GPU service found but no 'Device Utilization %' statistic", - )), - } + if scope.map_or(false, |s| !entries.iter().any(|e| e.id == s)) { + return Err(io::Error::other( + "Selected GPU device no longer exists (fall back to all GPUs)", + )); + } + if let Some(v) = read_max(&entries, scope) { + return Ok(v); + } + + // All reads failed — the cached services may be stale (e.g. after a + // GPU hot-plug). Invalidate the cache and retry once. + *GPU_ENTRIES.lock().unwrap() = None; + entries = gpu_entries(); + if entries.is_empty() { + return Err(io::Error::other( + "No GPU performance statistics available (no IOAccelerator service found)", + )); + } + if scope.map_or(false, |s| !entries.iter().any(|e| e.id == s)) { + return Err(io::Error::other( + "Selected GPU device no longer exists (fall back to all GPUs)", + )); + } + match read_max(&entries, scope) { + Some(v) => Ok(v), + None => Err(io::Error::other( + "GPU service found but no 'Device Utilization %' statistic", + )), + } + } +} + +/// The maximum "Device Utilization %" across the entries in scope +/// (`scope` = `None` means all entries), or `None` when no entry exposes +/// the statistic. +fn read_max(entries: &[GpuEntry], scope: Option<&str>) -> Option { + let mut max: Option = None; + for e in entries { + if let Some(scope) = scope { + if e.id != scope { + continue; } } + if let Some(v) = read_device_utilization(e.service) { + max = Some(max.map(|m| m.max(v)).unwrap_or(v)); + } } + max } -/// Return the cached GPU service handle, looking it up on first use. -fn get_gpu_service() -> u32 { - let mut guard = GPU_SERVICE.lock().unwrap(); +/// Return the cached GPU service list, looking it up on first use. +fn gpu_entries() -> Vec { + let mut guard = GPU_ENTRIES.lock().unwrap(); if guard.is_none() { - *guard = Some(lookup_gpu_service()); + *guard = Some(lookup_gpu_entries()); } - guard.unwrap_or(0) + guard.clone().unwrap_or_default() } -/// Find the first service exposing GPU performance statistics. +/// Enumerate every GPU accelerator service. /// -/// NOTE: the dictionary returned by `IOServiceMatching` must NOT be -/// released by us after `IOServiceGetMatchingService` — IOKit takes -/// ownership of it during the lookup (observed 2026-09-11: CFReleasing it -/// afterwards segfaults, and the property dictionary is allocated at the -/// matching dictionary's freed address). -fn lookup_gpu_service() -> u32 { +/// NOTE: the dictionary returned by `IOServiceMatching` must NOT be released +/// by us — IOKit takes ownership of it during the lookup (observed +/// 2026-09-11: CFReleasing it afterwards segfaults). The CFArray returned +/// by `IOServiceGetServices` IS ours to release, so it is wrapped in a +/// `CFRetained` (create rule). +fn lookup_gpu_entries() -> Vec { for class in ["IOAccelerator", "AGXAccelerator"] { let Ok(class_cstr) = CString::new(class) else { continue; @@ -86,12 +132,44 @@ fn lookup_gpu_service() -> u32 { if matching_raw.is_null() { continue; } - let service = unsafe { IOServiceGetMatchingService(K_IO_MAIN_PORT_DEFAULT, matching_raw) }; - if service != 0 && read_device_utilization(service).is_some() { - return service; + let services_raw = unsafe { IOServiceGetServices(K_IO_MAIN_PORT_DEFAULT, matching_raw) }; + if services_raw.is_null() { + continue; + } + // Take ownership of the returned CFArray (create rule). + let arr = unsafe { + CFRetained::from_raw(NonNull::new_unchecked(services_raw as *mut CFType)) + }; + let arr = match arr.downcast::() { + Ok(a) => a, + Err(_) => continue, + }; + let arr = unsafe { CFRetained::cast_unchecked::>(arr) }; + + let mut entries = Vec::new(); + for (i, number) in arr.iter().enumerate() { + let service = match number.as_f64() { + Some(v) if (v as u32) != 0 => v as u32, + _ => continue, + }; + // Only keep services that actually expose the utilization + // statistic (matches the single-service behavior). + if read_device_utilization(service).is_none() { + continue; + } + let name = read_string_property(service, "model") + .or_else(|| read_string_property(service, "name")) + .unwrap_or_else(|| format!("GPU {}", i)); + let id = format!("macos-gpu-{}", i); + entries.push(GpuEntry { + service, + id, + name, + }); } + return entries; } - 0 + Vec::new() } /// Read "Device Utilization %" from the IORegistry `PerformanceStatistics` @@ -126,6 +204,25 @@ fn read_device_utilization(service: u32) -> Option { number.as_f64() } +/// Read a string property (e.g. `model`) from the given service. +fn read_string_property(service: u32, key: &str) -> Option { + let prop_key = CFString::from_str(key); + let prop_raw = unsafe { + IORegistryEntryCreateCFProperty( + service, + CFRetained::as_ptr(&prop_key).as_ptr() as *const c_void, + std::ptr::null(), + 0, + ) + }; + if prop_raw.is_null() { + return None; + } + let prop = unsafe { CFRetained::from_raw(NonNull::new_unchecked(prop_raw as *mut CFType)) }; + let s = prop.downcast::().ok()?; + Some(s.to_string()) +} + #[cfg(test)] mod tests { use super::*; @@ -135,7 +232,8 @@ mod tests { /// lenient — a machine with no GPU at all is not a failure. #[test] fn gpu_usage_smoke() { - match MacosGpuMonitor::get_gpu_usage() { + eprintln!("macOS GPUs: {:?}", MacosGpuMonitor::enumerate_gpus()); + match MacosGpuMonitor::get_gpu_usage(None) { Ok(v) => eprintln!("macOS GPU usage: {:.2}%", v), Err(e) => eprintln!("macOS GPU usage unavailable: {}", e), } diff --git a/src/platform/macos/settings.rs b/src/platform/macos/settings.rs index d9d6635..cb83633 100644 --- a/src/platform/macos/settings.rs +++ b/src/platform/macos/settings.rs @@ -46,6 +46,17 @@ impl SettingsManager for MacosSettingsManager { set_preference("AnimationSource", source.as_str()); } + fn get_gpu_scope() -> Option { + get_preference("GpuScope").filter(|s| !s.is_empty()) + } + + fn set_gpu_scope(scope: Option) { + match scope { + Some(scope) => set_preference("GpuScope", &scope), + None => remove_preference("GpuScope"), + } + } + fn is_run_on_start_enabled() -> bool { let plist_path = dirs::home_dir() .unwrap_or_else(|| PathBuf::from("/tmp")) diff --git a/src/platform/mod.rs b/src/platform/mod.rs index af8d327..0cac698 100644 --- a/src/platform/mod.rs +++ b/src/platform/mod.rs @@ -13,13 +13,30 @@ pub trait CpuMonitor { fn get_cpu_usage() -> io::Result; } +/// A GPU device discovered by the platform monitor. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GpuDevice { + /// Platform-specific device id, stable for the current boot. + pub id: String, + /// Human-readable name for the tray menu. + pub name: String, +} + /// Cross-platform GPU usage monitoring trait pub trait GpuMonitor { - /// Returns GPU usage percentage as a float (0.0 to 100.0). + /// Enumerate the GPU devices that expose utilization. + fn enumerate_gpus() -> Vec; + + /// Returns GPU usage percentage as a float (0.0 to 100.0) for a scope. + /// + /// `scope` = `None` covers all devices (max across devices); + /// `scope` = `Some(id)` covers only that device (max across its + /// engines). Returns `Err` when the scope names a device that no + /// longer exists, so callers can fall back to all devices. /// /// Returns `Err` when no GPU usage source is available (no GPU, no /// driver, no `nvidia-smi`, ...). - fn get_gpu_usage() -> io::Result; + fn get_gpu_usage(scope: Option<&str>) -> io::Result; } /// Cross-platform settings management trait @@ -30,6 +47,9 @@ pub trait SettingsManager { fn set_current_theme(theme: Option); fn get_animation_source() -> crate::app::AnimationSource; fn set_animation_source(source: crate::app::AnimationSource); + /// Selected GPU device id; `None` = all GPUs (the default). + fn get_gpu_scope() -> Option; + fn set_gpu_scope(scope: Option); fn is_run_on_start_enabled() -> bool; fn set_run_on_start(enable: bool); fn is_dark_mode_enabled() -> bool; diff --git a/src/platform/windows/gpu_usage.rs b/src/platform/windows/gpu_usage.rs index cc6d1e7..cc6b1e4 100644 --- a/src/platform/windows/gpu_usage.rs +++ b/src/platform/windows/gpu_usage.rs @@ -1,11 +1,12 @@ -use crate::platform::GpuMonitor; +use crate::platform::{GpuDevice, GpuMonitor}; +use std::collections::BTreeMap; use std::io; use std::sync::Mutex; -use windows::core::PCWSTR; +use windows::core::{PCWSTR, PWSTR}; use windows::Win32::System::Performance::{ - PDH_FMT_COUNTERVALUE, PDH_FMT_COUNTERVALUE_ITEM_W, PDH_FMT_DOUBLE, PDH_HCOUNTER, PDH_HQUERY, - PDH_MORE_DATA, PdhAddEnglishCounterW, PdhCollectQueryData, PdhCloseQuery, - PdhGetFormattedCounterArrayW, PdhGetFormattedCounterValue, PdhOpenQueryW, + PDH_FMT_COUNTERVALUE_ITEM_W, PDH_FMT_DOUBLE, PDH_HCOUNTER, PDH_HQUERY, PDH_MORE_DATA, + PdhAddEnglishCounterW, PdhCollectQueryData, PdhCloseQuery, PdhGetFormattedCounterArrayW, + PdhGetFormattedCounterValue, PdhOpenQueryW, }; pub struct WindowsGpuMonitor; @@ -25,122 +26,327 @@ static GPU_STATE: Mutex> = Mutex::new(None); /// Per-engine utilization counter, available since Windows 10 1803 for all /// GPU vendors (NVIDIA, AMD, Intel). The `(*)` wildcard expands to one -/// instance per GPU engine (3D, Copy, Video Decode, ...); we report the max -/// across engines as the overall GPU utilization, mirroring how Task -/// Manager surfaces "GPU busy". +/// instance per (process, engine) pair — instance names look like +/// `pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D` — so the +/// utilization of a physical engine is the SUM of its instances (one per +/// process context), and the overall GPU figure is the max across engines +/// (the busiest engine), mirroring how Task Manager surfaces "GPU busy". const GPU_UTIL_COUNTER: PCWSTR = windows::core::w!("\\GPU Engine(*)\\Utilization Percentage"); -impl GpuMonitor for WindowsGpuMonitor { - fn get_gpu_usage() -> io::Result { - let mut state = GPU_STATE.lock().unwrap(); - if state.is_none() { - let mut hquery = PDH_HQUERY::default(); - let ret = unsafe { PdhOpenQueryW(None::<&PCWSTR>, 0, &mut hquery) }; - if ret != 0 { - return Err(io::Error::other(format!( - "PdhOpenQueryW failed: 0x{:08X}", - ret - ))); - } +/// One (process, engine) instance sample. +#[derive(Clone)] +struct EngineInstance { + /// GPU identity from the instance name (`luid_...` token); "default" + /// when the name carries no luid. + gpu_id: String, + /// Menu label for the GPU (`GPU 0`, `GPU 1`, ...). + gpu_label: String, + /// Physical engine identity: `luid/phys/eng` (or the full instance + /// name when the fields are missing). + engine_key: String, + value: f64, +} - let mut hcounter = PDH_HCOUNTER::default(); - let ret = unsafe { - PdhAddEnglishCounterW(hquery, GPU_UTIL_COUNTER, 0, &mut hcounter) - }; - if ret != 0 { - unsafe { - let _ = PdhCloseQuery(hquery); - } - return Err(io::Error::other(format!( - "Counter not available (GPU counter missing on this system): 0x{:08X}", - ret - ))); +/// Parse an instance name like +/// `pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D`. +/// +/// Returns `(gpu_id, gpu_label, engine_key)`. Missing fields degrade +/// gracefully so unusual names still produce a usable (unique) key. +fn parse_instance_name(name: &str) -> (String, String, String) { + let parts: Vec<&str> = name.split('_').collect(); + let mut luid: Option = None; + let mut phys: Option = None; + let mut eng: Option = None; + let mut i = 0; + while i < parts.len() { + match parts[i] { + "luid" if i + 2 < parts.len() => { + luid = Some(format!("{}_{}", parts[i + 1], parts[i + 2])); + i += 2; + } + "phys" if i + 1 < parts.len() => { + phys = Some(parts[i + 1].to_string()); + i += 1; } + "eng" if i + 1 < parts.len() => { + eng = Some(parts[i + 1].to_string()); + i += 1; + } + "pid" | "engtype" => {} + _ => {} + } + i += 1; + } - // Prime the query: the first collect initializes the counters, - // so the first real sample has a baseline. - unsafe { - let _ = PdhCollectQueryData(hquery); - }; + let gpu_id = luid.clone().unwrap_or_else(|| "default".to_string()); + let gpu_label = phys + .as_ref() + .map(|p| format!("GPU {p}")) + .unwrap_or_else(|| "GPU".to_string()); + let engine_key = match (&luid, &phys, &eng) { + (Some(l), Some(p), Some(e)) => format!("{l}/{p}/{e}"), + _ => name.to_string(), + }; + (gpu_id, gpu_label, engine_key) +} - *state = Some(GpuPdhState { hquery, hcounter }); - } +/// Read a NUL-terminated UTF-16 string from a PDH item name pointer. +unsafe fn read_item_name(ptr: PWSTR) -> Option { + let start = ptr.as_ptr(); + if start.is_null() { + return None; + } + let len = (0..4096).take_while(|&n| unsafe { *start.add(n) } != 0).count(); + if len == 0 { + return None; + } + Some(unsafe { String::from_utf16_lossy(std::slice::from_raw_parts(start, len)) }) +} - let state = state.as_ref().expect("state checked above"); - let ret = unsafe { PdhCollectQueryData(state.hquery) }; - if ret == PDH_MORE_DATA { - return Err(io::Error::other("PDH data not ready yet")); +/// Open (or reuse) the PDH query and collect one sample of every +/// (process, engine) instance. +fn collect_instances() -> io::Result> { + let mut state = GPU_STATE.lock().unwrap(); + if state.is_none() { + let mut hquery = PDH_HQUERY::default(); + let ret = unsafe { PdhOpenQueryW(None::<&PCWSTR>, 0, &mut hquery) }; + if ret != 0 { + return Err(io::Error::other(format!( + "PdhOpenQueryW failed: 0x{:08X}", + ret + ))); } + + let mut hcounter = PDH_HCOUNTER::default(); + let ret = unsafe { PdhAddEnglishCounterW(hquery, GPU_UTIL_COUNTER, 0, &mut hcounter) }; if ret != 0 { + unsafe { + let _ = PdhCloseQuery(hquery); + } return Err(io::Error::other(format!( - "PdhCollectQueryData failed: 0x{:08X}", + "Counter not available (GPU counter missing on this system): 0x{:08X}", ret ))); } - // Wildcard counters expand into one instance per engine; the array - // API returns them all in one call (two-pass buffer query). - let mut size: u32 = 0; - let mut count: u32 = 0; + // Prime the query: the first collect initializes the counters, + // so the first real sample has a baseline. + unsafe { + let _ = PdhCollectQueryData(hquery); + }; + + *state = Some(GpuPdhState { hquery, hcounter }); + } + + let state = state.as_ref().expect("state checked above"); + let ret = unsafe { PdhCollectQueryData(state.hquery) }; + if ret == PDH_MORE_DATA { + return Err(io::Error::other("PDH data not ready yet")); + } + if ret != 0 { + return Err(io::Error::other(format!( + "PdhCollectQueryData failed: 0x{:08X}", + ret + ))); + } + + // Wildcard counters expand into one instance per (process, engine); + // the array API returns them all in one call. The first pass yields + // the exact buffer size (struct array + trailing name storage), so + // allocate `size` bytes and read exactly `count` items. + let mut size: u32 = 0; + let mut count: u32 = 0; + let ret = unsafe { + PdhGetFormattedCounterArrayW(state.hcounter, PDH_FMT_DOUBLE, &mut size, &mut count, None) + }; + if ret == PDH_MORE_DATA && count > 0 { + let mut buf: Vec = vec![0u8; size as usize]; let ret = unsafe { - PdhGetFormattedCounterArrayW(state.hcounter, PDH_FMT_DOUBLE, &mut size, &mut count, None) + PdhGetFormattedCounterArrayW( + state.hcounter, + PDH_FMT_DOUBLE, + &mut size, + &mut count, + Some(buf.as_mut_ptr().cast()), + ) }; - if ret == PDH_MORE_DATA && count > 0 { - let item_size = std::mem::size_of::(); - let item_count = (size as usize) / item_size; - let mut items: Vec = (0..item_count) - .map(|_| PDH_FMT_COUNTERVALUE_ITEM_W::default()) - .collect(); - let ret = unsafe { - PdhGetFormattedCounterArrayW( - state.hcounter, - PDH_FMT_DOUBLE, - &mut size, - &mut count, - Some(items.as_mut_ptr()), - ) + if ret == 0 { + let items = unsafe { + std::slice::from_raw_parts(buf.as_ptr().cast::(), count as usize) }; - if ret == 0 { - let mut max: Option = None; - for item in items.iter() { - // PDH_CSTATUS_SUCCESS == 0; skip engines with no data. - if item.FmtValue.CStatus == 0 { - let v = unsafe { item.FmtValue.Anonymous.doubleValue }; - max = Some(max.map_or(v, |m| m.max(v))); - } + let mut instances = Vec::new(); + for item in items { + // PDH_CSTATUS_SUCCESS == 0; skip instances with no data. + if item.FmtValue.CStatus != 0 { + continue; } - return max - .ok_or_else(|| io::Error::other("No GPU engine utilization data available")); + let name = match unsafe { read_item_name(item.szName) } { + Some(n) => n, + None => continue, + }; + let (gpu_id, gpu_label, engine_key) = parse_instance_name(&name); + instances.push(EngineInstance { + gpu_id, + gpu_label, + engine_key, + value: unsafe { item.FmtValue.Anonymous.doubleValue }, + }); } - Err(io::Error::other(format!( - "PdhGetFormattedCounterArrayW failed: 0x{:08X}", - ret - ))) - } else if ret == 0 { - // Single-instance counter (no wildcard expansion). - let mut status: u32 = 0; - let mut value = PDH_FMT_COUNTERVALUE::default(); - let ret = unsafe { - PdhGetFormattedCounterValue( - state.hcounter, - PDH_FMT_DOUBLE, - Some(&mut status), - &mut value, - ) - }; - if ret == 0 && status == 0 { - return Ok(unsafe { value.Anonymous.doubleValue }); + return Ok(instances); + } + Err(io::Error::other(format!( + "PdhGetFormattedCounterArrayW failed: 0x{:08X}", + ret + ))) + } else if ret == 0 { + // Single-instance counter (no wildcard expansion). + let mut status: u32 = 0; + let mut value = windows::Win32::System::Performance::PDH_FMT_COUNTERVALUE::default(); + let ret = unsafe { + PdhGetFormattedCounterValue( + state.hcounter, + PDH_FMT_DOUBLE, + Some(&mut status), + &mut value, + ) + }; + if ret == 0 && status == 0 { + return Ok(vec![EngineInstance { + gpu_id: "default".to_string(), + gpu_label: "GPU".to_string(), + engine_key: "default/default/0".to_string(), + value: unsafe { value.Anonymous.doubleValue }, + }]); + } + Err(io::Error::other(format!( + "PdhGetFormattedCounterValue failed: 0x{:08X} (status: {})", + ret, status + ))) + } else { + Err(io::Error::other(format!( + "PdhGetFormattedCounterArrayW unexpected status: 0x{:08X}", + ret + ))) + } +} + +/// Sum per-process instances per physical engine (capped at 100%) and +/// return the max across engines for the requested GPU scope. +fn aggregate(instances: &[EngineInstance], scope: Option<&str>) -> io::Result { + let mut engines: BTreeMap<(String, String), f64> = BTreeMap::new(); + let mut scoped = false; + for inst in instances { + if let Some(scope) = scope { + if inst.gpu_id != scope { + continue; } - Err(io::Error::other(format!( - "PdhGetFormattedCounterValue failed: 0x{:08X} (status: {})", - ret, status - ))) - } else { - Err(io::Error::other(format!( - "PdhGetFormattedCounterArrayW unexpected status: 0x{:08X}", - ret - ))) } + scoped = true; + let entry = engines + .entry((inst.gpu_id.clone(), inst.engine_key.clone())) + .or_insert(0.0); + *entry = (*entry + inst.value).min(100.0); + } + if !scoped { + return Err(io::Error::other( + "Selected GPU device no longer exists (fall back to all GPUs)", + )); + } + engines + .values() + .copied() + .max_by(f64::total_cmp) + .ok_or_else(|| io::Error::other("No GPU engine utilization data available")) +} + +impl GpuMonitor for WindowsGpuMonitor { + fn enumerate_gpus() -> Vec { + let instances = match collect_instances() { + Ok(i) => i, + Err(_) => return Vec::new(), + }; + let mut seen: BTreeMap = BTreeMap::new(); + for inst in &instances { + seen.entry(inst.gpu_id.clone()) + .or_insert_with(|| inst.gpu_label.clone()); + } + seen + .into_iter() + .map(|(id, name)| GpuDevice { id, name }) + .collect() + } + + fn get_gpu_usage(scope: Option<&str>) -> io::Result { + let instances = collect_instances()?; + aggregate(&instances, scope) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_name_full() { + let (gpu_id, gpu_label, engine_key) = + parse_instance_name("pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D"); + assert_eq!(gpu_id, "0x00000000_0x00017FEA"); + assert_eq!(gpu_label, "GPU 0"); + assert_eq!(engine_key, "0x00000000_0x00017FEA/0/0"); + } + + #[test] + fn parse_name_engtype_with_spaces() { + let (gpu_id, _label, engine_key) = + parse_instance_name("pid_1_luid_0x1_0x2_phys_0_eng_10_engtype_Video Codec 0"); + assert_eq!(gpu_id, "0x1_0x2"); + // engtype is free text after the "engtype_" marker; the key is + // luid/phys/eng and must not include it. + assert_eq!(engine_key, "0x1_0x2/0/10"); + } + + #[test] + fn parse_name_degenerate() { + let (gpu_id, _label, engine_key) = parse_instance_name("3D 0"); + assert_eq!(gpu_id, "default"); + assert_eq!(engine_key, "3D 0"); + } + + #[test] + fn aggregate_sums_per_engine_and_maxes() { + let inst = |_pid: &str, eng: &str, v: f64| EngineInstance { + gpu_id: "g".into(), + gpu_label: "GPU 0".into(), + engine_key: eng.into(), + value: v, + }; + let instances = vec![ + inst("1", "e0", 35.0), + inst("2", "e0", 45.0), // same engine, second process → 80 + inst("1", "e1", 10.0), // other engine, must not win + ]; + let v = aggregate(&instances, None).unwrap(); + assert_eq!(v, 80.0); + + // Sum is capped at 100. + let capped = vec![EngineInstance { + gpu_id: "g".into(), + gpu_label: "GPU 0".into(), + engine_key: "e0".into(), + value: 60.0, + }]; + let mut instances = capped.clone(); + instances.push(capped[0].clone()); + assert_eq!(aggregate(&instances, None).unwrap(), 100.0); + + // Scope filters by GPU id (the 60% instance of the other GPU is + // ignored). + let mut other = inst("1", "e9", 50.0); + other.gpu_id = "other".into(); + let mut instances = capped; + instances.push(other); + assert_eq!(aggregate(&instances, Some("g")).unwrap(), 60.0); + assert!(aggregate(&instances, Some("missing")).is_err()); } } diff --git a/src/platform/windows/settings.rs b/src/platform/windows/settings.rs index 178ed49..4a22f1a 100644 --- a/src/platform/windows/settings.rs +++ b/src/platform/windows/settings.rs @@ -105,6 +105,42 @@ impl SettingsManager for WindowsSettingsManager { .expect("set_value"); } + fn get_gpu_scope() -> Option { + let key = RegKey::predef(HKEY_CURRENT_USER); + if let Ok(sub_key) = key.open_subkey_with_flags("Software\\RustCat", KEY_READ) { + if let Ok(scope) = sub_key.get_value::("GpuScope") { + if !scope.is_empty() { + return Some(scope); + } + } + } + None + } + + fn set_gpu_scope(scope: Option) { + let key = RegKey::predef(HKEY_CURRENT_USER); + let sub_key = if let Ok(sub_key) = + key.open_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) + { + sub_key + } else { + key.create_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) + .expect("create_subkey_with_flags") + .0 + }; + + match scope { + Some(scope) => { + sub_key + .set_value("GpuScope", &scope) + .expect("set_value"); + } + None => { + let _ = sub_key.delete_value("GpuScope"); + } + } + } + fn is_run_on_start_enabled() -> bool { let hkcu = RegKey::predef(HKEY_CURRENT_USER); if let Ok(run_key) = hkcu.open_subkey_with_flags( From 863af915097d447e72f45b07e792a8600248805a Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 14:14:31 +0900 Subject: [PATCH 3/9] fix(macos): use IOServiceGetMatchingServices iterator (IOServiceGetServices does not exist) --- src/platform/macos/gpu_usage.rs | 80 ++++++++++++++++++--------------- 1 file changed, 43 insertions(+), 37 deletions(-) diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs index 88eb643..0e6be8a 100644 --- a/src/platform/macos/gpu_usage.rs +++ b/src/platform/macos/gpu_usage.rs @@ -1,5 +1,5 @@ use crate::platform::{GpuDevice, GpuMonitor}; -use objc2_core_foundation::{CFArray, CFDictionary, CFNumber, CFString, CFType, CFRetained}; +use objc2_core_foundation::{CFDictionary, CFNumber, CFString, CFType, CFRetained}; use std::ffi::{c_void, CString}; use std::io; use std::ptr::NonNull; @@ -19,11 +19,17 @@ struct GpuEntry { name: String, } -// Service handles returned by IOKit are cached for the process lifetime -// (see GPU_ENTRIES) and never released. +// Service handles kept in GPU_ENTRIES are cached for the process lifetime +// and never released; services we skip are released as we go. extern "C" { fn IOServiceMatching(name: *const i8) -> *mut c_void; - fn IOServiceGetServices(main_port: MachPort, matching: *mut c_void) -> *mut c_void; + fn IOServiceGetMatchingServices( + main_port: MachPort, + matching: *mut c_void, + existing: *mut u32, + ) -> u32; + fn IOIteratorNext(iterator: u32) -> u32; + fn IOObjectRelease(object: u32); fn IORegistryEntryCreateCFProperty( entry: u32, key: *const c_void, @@ -119,10 +125,9 @@ fn gpu_entries() -> Vec { /// Enumerate every GPU accelerator service. /// /// NOTE: the dictionary returned by `IOServiceMatching` must NOT be released -/// by us — IOKit takes ownership of it during the lookup (observed -/// 2026-09-11: CFReleasing it afterwards segfaults). The CFArray returned -/// by `IOServiceGetServices` IS ours to release, so it is wrapped in a -/// `CFRetained` (create rule). +/// by us — `IOServiceGetMatchingServices` is declared `CF_RELEASES_ARGUMENT` +/// and always consumes one reference of the matching dictionary (observed +/// 2026-09-11: CFReleasing it afterwards segfaults). fn lookup_gpu_entries() -> Vec { for class in ["IOAccelerator", "AGXAccelerator"] { let Ok(class_cstr) = CString::new(class) else { @@ -132,42 +137,43 @@ fn lookup_gpu_entries() -> Vec { if matching_raw.is_null() { continue; } - let services_raw = unsafe { IOServiceGetServices(K_IO_MAIN_PORT_DEFAULT, matching_raw) }; - if services_raw.is_null() { + let mut iterator: u32 = 0; + let ret = unsafe { + IOServiceGetMatchingServices(K_IO_MAIN_PORT_DEFAULT, matching_raw, &mut iterator) + }; + if ret != 0 || iterator == 0 { + // kIOReturnSuccess == 0; a NULL iterator means "no matches". continue; } - // Take ownership of the returned CFArray (create rule). - let arr = unsafe { - CFRetained::from_raw(NonNull::new_unchecked(services_raw as *mut CFType)) - }; - let arr = match arr.downcast::() { - Ok(a) => a, - Err(_) => continue, - }; - let arr = unsafe { CFRetained::cast_unchecked::>(arr) }; let mut entries = Vec::new(); - for (i, number) in arr.iter().enumerate() { - let service = match number.as_f64() { - Some(v) if (v as u32) != 0 => v as u32, - _ => continue, - }; + let mut service = unsafe { IOIteratorNext(iterator) }; + while service != 0 { // Only keep services that actually expose the utilization - // statistic (matches the single-service behavior). - if read_device_utilization(service).is_none() { - continue; + // statistic (matches the single-service behavior). Kept + // services are cached for the process lifetime and never + // released; the rest are released right away. + if read_device_utilization(service).is_some() { + let name = read_string_property(service, "model") + .or_else(|| read_string_property(service, "name")) + .unwrap_or_else(|| format!("GPU {}", entries.len())); + let id = format!("macos-gpu-{}", entries.len()); + entries.push(GpuEntry { + service, + id, + name, + }); + } else { + unsafe { IOObjectRelease(service) }; } - let name = read_string_property(service, "model") - .or_else(|| read_string_property(service, "name")) - .unwrap_or_else(|| format!("GPU {}", i)); - let id = format!("macos-gpu-{}", i); - entries.push(GpuEntry { - service, - id, - name, - }); + service = unsafe { IOIteratorNext(iterator) }; + } + // Release the iterator; the services we kept are cached. + unsafe { IOObjectRelease(iterator) }; + + if !entries.is_empty() { + return entries; } - return entries; } Vec::new() } From c7fbbeca42a75c4bfbce871e49965257b12ab8da Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 14:23:40 +0900 Subject: [PATCH 4/9] fix(windows): identify physical GPU by phys field, not luid MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PDH GPU Engine instance name's luid token is a per-GPU-context id — a single physical GPU (this 4090) exposes many luids (3 observed, 727 instances). Using luid as the device id split one card into several 'GPU 0' entries. The physical GPU is the phys field, and a physical engine is (phys, eng); per-engine utilization is the sum of all its (process, context) instances, max across engines. Verified on a live RTX 4090: 33 physical engines, per-engine sums stay <= 100 under load, matching Task Manager's GPU busy. Also add a Windows GPU smoke test (mirroring macOS/Linux). --- src/platform/windows/gpu_usage.rs | 79 +++++++++++++++++-------------- 1 file changed, 44 insertions(+), 35 deletions(-) diff --git a/src/platform/windows/gpu_usage.rs b/src/platform/windows/gpu_usage.rs index cc6b1e4..700e382 100644 --- a/src/platform/windows/gpu_usage.rs +++ b/src/platform/windows/gpu_usage.rs @@ -37,13 +37,16 @@ const GPU_UTIL_COUNTER: PCWSTR = /// One (process, engine) instance sample. #[derive(Clone)] struct EngineInstance { - /// GPU identity from the instance name (`luid_...` token); "default" - /// when the name carries no luid. + /// Physical GPU id from the instance name (`phys_`), e.g. + /// "win-gpu-0"; "default" when the name carries no phys field. Note: + /// the `luid_...` token is a per-GPU-context id (a single physical + /// GPU can have many luids), so it must NOT be used for the device + /// identity. gpu_id: String, /// Menu label for the GPU (`GPU 0`, `GPU 1`, ...). gpu_label: String, - /// Physical engine identity: `luid/phys/eng` (or the full instance - /// name when the fields are missing). + /// Physical engine identity: `win-gpu-/` (or the full + /// instance name when the fields are missing). engine_key: String, value: f64, } @@ -51,44 +54,38 @@ struct EngineInstance { /// Parse an instance name like /// `pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D`. /// -/// Returns `(gpu_id, gpu_label, engine_key)`. Missing fields degrade +/// Returns `(gpu_id, gpu_label, engine_key)`. The physical GPU is +/// identified by the `phys` field (the `luid` field varies per GPU +/// context even on a single physical GPU). Missing fields degrade /// gracefully so unusual names still produce a usable (unique) key. fn parse_instance_name(name: &str) -> (String, String, String) { let parts: Vec<&str> = name.split('_').collect(); - let mut luid: Option = None; - let mut phys: Option = None; - let mut eng: Option = None; + let mut phys: Option<&str> = None; + let mut eng: Option<&str> = None; let mut i = 0; while i < parts.len() { match parts[i] { - "luid" if i + 2 < parts.len() => { - luid = Some(format!("{}_{}", parts[i + 1], parts[i + 2])); - i += 2; - } "phys" if i + 1 < parts.len() => { - phys = Some(parts[i + 1].to_string()); + phys = Some(parts[i + 1]); i += 1; } "eng" if i + 1 < parts.len() => { - eng = Some(parts[i + 1].to_string()); + eng = Some(parts[i + 1]); i += 1; } - "pid" | "engtype" => {} _ => {} } i += 1; } - let gpu_id = luid.clone().unwrap_or_else(|| "default".to_string()); - let gpu_label = phys - .as_ref() - .map(|p| format!("GPU {p}")) - .unwrap_or_else(|| "GPU".to_string()); - let engine_key = match (&luid, &phys, &eng) { - (Some(l), Some(p), Some(e)) => format!("{l}/{p}/{e}"), - _ => name.to_string(), - }; - (gpu_id, gpu_label, engine_key) + match (phys, eng) { + (Some(p), Some(e)) => ( + format!("win-gpu-{p}"), + format!("GPU {p}"), + format!("win-gpu-{p}/{e}"), + ), + _ => ("default".to_string(), "GPU".to_string(), name.to_string()), + } } /// Read a NUL-terminated UTF-16 string from a PDH item name pointer. @@ -215,7 +212,7 @@ fn collect_instances() -> io::Result> { return Ok(vec![EngineInstance { gpu_id: "default".to_string(), gpu_label: "GPU".to_string(), - engine_key: "default/default/0".to_string(), + engine_key: "default".to_string(), value: unsafe { value.Anonymous.doubleValue }, }]); } @@ -291,19 +288,19 @@ mod tests { fn parse_name_full() { let (gpu_id, gpu_label, engine_key) = parse_instance_name("pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D"); - assert_eq!(gpu_id, "0x00000000_0x00017FEA"); + assert_eq!(gpu_id, "win-gpu-0"); assert_eq!(gpu_label, "GPU 0"); - assert_eq!(engine_key, "0x00000000_0x00017FEA/0/0"); + assert_eq!(engine_key, "win-gpu-0/0"); } #[test] - fn parse_name_engtype_with_spaces() { - let (gpu_id, _label, engine_key) = - parse_instance_name("pid_1_luid_0x1_0x2_phys_0_eng_10_engtype_Video Codec 0"); - assert_eq!(gpu_id, "0x1_0x2"); - // engtype is free text after the "engtype_" marker; the key is - // luid/phys/eng and must not include it. - assert_eq!(engine_key, "0x1_0x2/0/10"); + fn parse_name_multiple_gpus_and_engtypes() { + // Second physical GPU, high engine index, engtype with spaces. + let (gpu_id, gpu_label, engine_key) = + parse_instance_name("pid_1_luid_0x1_0x2_phys_1_eng_10_engtype_Video Codec 0"); + assert_eq!(gpu_id, "win-gpu-1"); + assert_eq!(gpu_label, "GPU 1"); + assert_eq!(engine_key, "win-gpu-1/10"); } #[test] @@ -313,6 +310,18 @@ mod tests { assert_eq!(engine_key, "3D 0"); } + /// Smoke test: exercise the sampling path and print the reading so it + /// can be checked manually on a machine with a GPU. Deliberately + /// lenient — a machine with no GPU at all is not a failure. + #[test] + fn gpu_usage_smoke() { + eprintln!("Windows GPUs: {:?}", WindowsGpuMonitor::enumerate_gpus()); + match WindowsGpuMonitor::get_gpu_usage(None) { + Ok(v) => eprintln!("Windows GPU usage: {:.2}%", v), + Err(e) => eprintln!("Windows GPU usage unavailable: {}", e), + } + } + #[test] fn aggregate_sums_per_engine_and_maxes() { let inst = |_pid: &str, eng: &str, v: f64| EngineInstance { From a014cd0fbdb56dbfa236eaa09bf1202468099d43 Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 14:28:42 +0900 Subject: [PATCH 5/9] fix: address multi-GPU review round 2 - macOS: re-enumerate accelerator services every ~30s (eGPU hot-plug) and never cache an empty result, so a later-attached GPU is discoverable. - Scope selection is now session-only (device ids are only stable for the current boot); drop the persisted GpuScope setting on all platforms. - app.rs: only run the stale-device recovery when a GPU sample was actually requested (source != CPU), so CPU mode no longer re-enumerates (and spawns nvidia-smi) every second. --- src/app.rs | 17 +++++++---- src/events.rs | 7 ++--- src/platform/linux/settings.rs | 11 -------- src/platform/macos/gpu_usage.rs | 48 +++++++++++++++++++++++++------- src/platform/macos/settings.rs | 11 -------- src/platform/mod.rs | 7 +++-- src/platform/windows/settings.rs | 36 ------------------------ 7 files changed, 57 insertions(+), 80 deletions(-) diff --git a/src/app.rs b/src/app.rs index 152c506..257068f 100644 --- a/src/app.rs +++ b/src/app.rs @@ -145,7 +145,10 @@ impl App { let theme = initial_theme.unwrap_or_else(SettingsManagerImpl::get_current_theme); let animation_source = SettingsManagerImpl::get_animation_source(); - let gpu_scope = SettingsManagerImpl::get_gpu_scope(); + // The GPU device selection is session-only: device ids are only + // stable for the current boot, so a persisted selection could + // point at a different adapter on a later boot. + let gpu_scope: Option = None; let initial_icons = icon_manager .get_icon_set(initial_icon, Some(theme)) .ok_or("Invalid initial icon name")?; @@ -156,7 +159,7 @@ impl App { }) .icon(initial_icons[0].clone()) .tooltip("~Nyan~ RustCat - CPU/GPU Usage Monitor") - .menu(build_menu(&icon_manager)) + .menu(build_menu(&icon_manager, None)) .on_right_click(Events::ShowMenu) .on_double_click(Events::RunTaskmgr) .build()?; @@ -237,7 +240,10 @@ impl App { let source = *animation_source.lock().unwrap(); let scope = gpu_scope.lock().unwrap().clone(); let (cpu_usage, gpu_usage) = sample_usage(source, scope.as_deref()); - if gpu_usage.is_none() { + // Only run the stale-device recovery when a GPU sample + // was actually requested (with source = CPU the GPU is + // deliberately not sampled, so `None` is not a failure). + if gpu_usage.is_none() && source != AnimationSource::Cpu { if let Some(scope) = &scope { // The scoped device may have disappeared (e.g. an // eGPU unplugged); fall back to all GPUs. Only @@ -355,7 +361,7 @@ impl App { self.update_menu(); } Events::SetGpuScope(scope) => { - SettingsManagerImpl::set_gpu_scope(scope.clone()); + // Session-only selection (see App::new). *self.gpu_scope.lock().unwrap() = scope; self.update_menu(); } @@ -394,9 +400,10 @@ impl App { fn update_menu(&self) { let tray_icon = self.tray_icon.clone(); let icon_manager = self.icon_manager.clone(); + let gpu_scope = self.gpu_scope.lock().unwrap().clone(); ui_update(move || { if let Ok(mut tray) = tray_icon.lock() { - if let Err(e) = tray.set_menu(&build_menu(&icon_manager)) { + if let Err(e) = tray.set_menu(&build_menu(&icon_manager, gpu_scope.as_deref())) { eprintln!("Failed to update menu: {}", e); } } diff --git a/src/events.rs b/src/events.rs index 550c1ed..bb0f2de 100644 --- a/src/events.rs +++ b/src/events.rs @@ -18,7 +18,7 @@ pub enum Events { ShowMenu, } -pub fn build_menu(icon_manager: &IconManager) -> MenuBuilder { +pub fn build_menu(icon_manager: &IconManager, gpu_scope: Option<&str>) -> MenuBuilder { let run_on_start_enabled = SettingsManagerImpl::is_run_on_start_enabled(); let current_icon = SettingsManagerImpl::get_current_icon(); let current_theme = SettingsManagerImpl::get_current_theme(); @@ -82,14 +82,13 @@ pub fn build_menu(icon_manager: &IconManager) -> MenuBuilder { // Build GPU device submenu — only shown when the machine has more than // one GPU that exposes utilization. - let current_scope = SettingsManagerImpl::get_gpu_scope(); let gpus = GpuMonitorImpl::enumerate_gpus(); if gpus.len() > 1 { let mut gpu_menu = MenuBuilder::new(); gpu_menu = gpu_menu - .radio("All GPUs", current_scope.is_none(), Events::SetGpuScope(None)); + .radio("All GPUs", gpu_scope.is_none(), Events::SetGpuScope(None)); for device in &gpus { - let is_current = current_scope.as_deref() == Some(device.id.as_str()); + let is_current = gpu_scope == Some(device.id.as_str()); gpu_menu = gpu_menu .radio(&device.name, is_current, Events::SetGpuScope(Some(device.id.clone()))); } diff --git a/src/platform/linux/settings.rs b/src/platform/linux/settings.rs index 21cc0c9..76f44ff 100644 --- a/src/platform/linux/settings.rs +++ b/src/platform/linux/settings.rs @@ -44,17 +44,6 @@ impl SettingsManager for LinuxSettingsManager { write_setting("AnimationSource", source.as_str()); } - fn get_gpu_scope() -> Option { - read_setting("GpuScope").filter(|s| !s.is_empty()) - } - - fn set_gpu_scope(scope: Option) { - match scope { - Some(scope) => write_setting("GpuScope", &scope), - None => remove_setting("GpuScope"), - } - } - fn is_run_on_start_enabled() -> bool { autostart_desktop_path().exists() } diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs index 0e6be8a..919b1a2 100644 --- a/src/platform/macos/gpu_usage.rs +++ b/src/platform/macos/gpu_usage.rs @@ -38,10 +38,20 @@ extern "C" { ) -> *const c_void; } -/// Cached list of GPU services. IOAccelerator/AGXAccelerator services are -/// looked up once and reused; a failed property read invalidates the cache -/// so a hot-plugged GPU is picked up on the next sample. -static GPU_ENTRIES: Mutex>> = Mutex::new(None); +/// Cached list of GPU services. IOAccelerator/AGXAccelerator services can +/// be hot-plugged (e.g. an eGPU attached after startup), so the list is +/// re-enumerated every `GPU_REFRESH_INTERVAL` samples and an empty result +/// is never cached (a later-attached GPU must be discoverable on the next +/// sample). A failed property read also invalidates the cache. +static GPU_CACHE: Mutex> = Mutex::new(None); + +struct GpuCache { + entries: Vec, + samples_since_refresh: u32, +} + +/// ~30 s at the app's 1 sample/s cadence. +const GPU_REFRESH_INTERVAL: u32 = 30; pub struct MacosGpuMonitor; @@ -74,7 +84,7 @@ impl GpuMonitor for MacosGpuMonitor { // All reads failed — the cached services may be stale (e.g. after a // GPU hot-plug). Invalidate the cache and retry once. - *GPU_ENTRIES.lock().unwrap() = None; + *GPU_CACHE.lock().unwrap() = None; entries = gpu_entries(); if entries.is_empty() { return Err(io::Error::other( @@ -113,13 +123,31 @@ fn read_max(entries: &[GpuEntry], scope: Option<&str>) -> Option { max } -/// Return the cached GPU service list, looking it up on first use. +/// Return the cached GPU service list, re-enumerating when due (first +/// use, every `GPU_REFRESH_INTERVAL` samples, or after an invalidation). fn gpu_entries() -> Vec { - let mut guard = GPU_ENTRIES.lock().unwrap(); - if guard.is_none() { - *guard = Some(lookup_gpu_entries()); + let mut guard = GPU_CACHE.lock().unwrap(); + let due = match guard.as_mut() { + None => true, + Some(cache) => { + cache.samples_since_refresh += 1; + cache.samples_since_refresh >= GPU_REFRESH_INTERVAL + } + }; + if due { + let found = lookup_gpu_entries(); + if found.is_empty() { + // Nothing found yet (or all detached) — do not cache the empty + // result; retry on the next call. + *guard = None; + } else { + *guard = Some(GpuCache { + entries: found, + samples_since_refresh: 0, + }); + } } - guard.clone().unwrap_or_default() + guard.as_ref().map(|c| c.entries.clone()).unwrap_or_default() } /// Enumerate every GPU accelerator service. diff --git a/src/platform/macos/settings.rs b/src/platform/macos/settings.rs index cb83633..d9d6635 100644 --- a/src/platform/macos/settings.rs +++ b/src/platform/macos/settings.rs @@ -46,17 +46,6 @@ impl SettingsManager for MacosSettingsManager { set_preference("AnimationSource", source.as_str()); } - fn get_gpu_scope() -> Option { - get_preference("GpuScope").filter(|s| !s.is_empty()) - } - - fn set_gpu_scope(scope: Option) { - match scope { - Some(scope) => set_preference("GpuScope", &scope), - None => remove_preference("GpuScope"), - } - } - fn is_run_on_start_enabled() -> bool { let plist_path = dirs::home_dir() .unwrap_or_else(|| PathBuf::from("/tmp")) diff --git a/src/platform/mod.rs b/src/platform/mod.rs index 0cac698..69f46c1 100644 --- a/src/platform/mod.rs +++ b/src/platform/mod.rs @@ -14,6 +14,10 @@ pub trait CpuMonitor { } /// A GPU device discovered by the platform monitor. +/// +/// The device id is stable for the current boot; the selection itself +/// (`App::gpu_scope`) is session-only, so a persisted id can never point +/// at a different adapter on a later boot. #[derive(Debug, Clone, PartialEq, Eq)] pub struct GpuDevice { /// Platform-specific device id, stable for the current boot. @@ -47,9 +51,6 @@ pub trait SettingsManager { fn set_current_theme(theme: Option); fn get_animation_source() -> crate::app::AnimationSource; fn set_animation_source(source: crate::app::AnimationSource); - /// Selected GPU device id; `None` = all GPUs (the default). - fn get_gpu_scope() -> Option; - fn set_gpu_scope(scope: Option); fn is_run_on_start_enabled() -> bool; fn set_run_on_start(enable: bool); fn is_dark_mode_enabled() -> bool; diff --git a/src/platform/windows/settings.rs b/src/platform/windows/settings.rs index 4a22f1a..178ed49 100644 --- a/src/platform/windows/settings.rs +++ b/src/platform/windows/settings.rs @@ -105,42 +105,6 @@ impl SettingsManager for WindowsSettingsManager { .expect("set_value"); } - fn get_gpu_scope() -> Option { - let key = RegKey::predef(HKEY_CURRENT_USER); - if let Ok(sub_key) = key.open_subkey_with_flags("Software\\RustCat", KEY_READ) { - if let Ok(scope) = sub_key.get_value::("GpuScope") { - if !scope.is_empty() { - return Some(scope); - } - } - } - None - } - - fn set_gpu_scope(scope: Option) { - let key = RegKey::predef(HKEY_CURRENT_USER); - let sub_key = if let Ok(sub_key) = - key.open_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) - { - sub_key - } else { - key.create_subkey_with_flags("Software\\RustCat", KEY_WRITE | KEY_READ) - .expect("create_subkey_with_flags") - .0 - }; - - match scope { - Some(scope) => { - sub_key - .set_value("GpuScope", &scope) - .expect("set_value"); - } - None => { - let _ = sub_key.delete_value("GpuScope"); - } - } - } - fn is_run_on_start_enabled() -> bool { let hkcu = RegKey::predef(HKEY_CURRENT_USER); if let Ok(run_key) = hkcu.open_subkey_with_flags( From 6fc9717eadc23a5eba5357c6f81f2c53067d3996 Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 14:46:33 +0900 Subject: [PATCH 6/9] test(macos): temp probe for service handle stability --- src/platform/macos/gpu_usage.rs | 46 +++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs index 919b1a2..a680332 100644 --- a/src/platform/macos/gpu_usage.rs +++ b/src/platform/macos/gpu_usage.rs @@ -261,6 +261,52 @@ fn read_string_property(service: u32, key: &str) -> Option { mod tests { use super::*; + /// TEMP probe: enumerate raw services twice (with a pause) to see + /// whether the io_service_t handles are stable across re-enumeration. + #[test] + fn probe_handle_stability() { + fn raw_services() -> Vec { + let mut out = Vec::new(); + for class in ["IOAccelerator", "AGXAccelerator"] { + let Ok(c) = CString::new(class) else { continue }; + let matching = unsafe { IOServiceMatching(c.as_ptr()) }; + if matching.is_null() { + continue; + } + let mut it: u32 = 0; + if unsafe { IOServiceGetMatchingServices(0, matching, &mut it) } != 0 || it == 0 { + continue; + } + let mut s = unsafe { IOIteratorNext(it) }; + while s != 0 { + out.push(s); + s = unsafe { IOIteratorNext(it) }; + } + unsafe { IOObjectRelease(it) }; + // release the extra refs we took for the probe (we keep one + // reference alive for the process; releasing all would be + // fine for a probe too, but keep it simple) + } + out + } + let first = raw_services(); + eprintln!("probe first: {:?}", first); + std::thread::sleep(std::time::Duration::from_millis(500)); + let second = raw_services(); + eprintln!("probe second: {:?}", second); + eprintln!( + "probe stable: {} (equal sets: {})", + first == second, + { + let mut a = first.clone(); + let mut b = second.clone(); + a.sort(); + b.sort(); + a == b + } + ); + } + /// Smoke test: exercise the sampling path and print the reading so it /// can be checked manually on a machine with a GPU. Deliberately /// lenient — a machine with no GPU at all is not a failure. From 3b4702136dc4225841528eaac9ee23b2a11e25a0 Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 15:42:54 +0900 Subject: [PATCH 7/9] fix: address multi-GPU review round 3 (per-GPU selection) Windows: - Enumerate installed adapters via DXGI (IDXGIFactory1::EnumAdapters1) so idle (no process-context) secondary GPUs appear in the device menu; software fallback adapters (e.g. Microsoft Basic Render Driver) are filtered out. - Identify physical GPUs by their adapter LUID, which appears in both the PDH instance names and DXGI's AdapterLuid, so menu ids and sampling ids match exactly. The phys field is not a reliable per-adapter index (it is phys_0 for every adapter on some multi-GPU systems). - Allocate the PDH array buffer with an explicit alignment (AlignedBuf) to avoid misaligned-access UB when PDH fills it with counter-value items. macOS: - Reference-count IOKit service handles (SERVICE_REFS) and release them via an RAII EntriesGuard, so superseded or detached device handles are IOObjectRelease'd exactly once when their last reference drops. Cargo.toml: add Win32_Graphics_Dxgi to the windows feature set. --- Cargo.toml | 1 + src/platform/macos/gpu_usage.rs | 210 +++++++++++++-------- src/platform/windows/gpu_usage.rs | 295 +++++++++++++++++++++++++----- 3 files changed, 386 insertions(+), 120 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index 0c507b5..2fa494a 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -44,6 +44,7 @@ dispatch = "0.2.0" version = "0.62" features = [ "Win32_Foundation", + "Win32_Graphics_Dxgi", "Win32_UI_WindowsAndMessaging", "Win32_UI_Shell", "Win32_System_Threading", diff --git a/src/platform/macos/gpu_usage.rs b/src/platform/macos/gpu_usage.rs index a680332..4509b16 100644 --- a/src/platform/macos/gpu_usage.rs +++ b/src/platform/macos/gpu_usage.rs @@ -1,17 +1,25 @@ use crate::platform::{GpuDevice, GpuMonitor}; use objc2_core_foundation::{CFDictionary, CFNumber, CFString, CFType, CFRetained}; +use std::collections::{HashMap, HashSet}; use std::ffi::{c_void, CString}; use std::io; use std::ptr::NonNull; -use std::sync::Mutex; +use std::sync::{LazyLock, Mutex}; type MachPort = u32; /// The default (current) Mach port; IOKit interprets 0 as "this process". const K_IO_MAIN_PORT_DEFAULT: MachPort = 0; -/// A GPU service we track: its IORegistry handle, a stable-per-boot id for +/// A GPU service we track: its IORegistry handle, a device-derived id for /// menu selection, and a human-readable name. +/// +/// The id is derived from the IORegistry service handle itself (which is +/// stable for the device's lifetime — verified 2026-09-11: re-enumeration +/// returns the same handle), NOT from enumeration position. This means a +/// hot-plug/detach that changes the enumeration order cannot make a scoped +/// selection silently point at a different device: a removed device's id +/// simply disappears and triggers the missing-device fallback. #[derive(Clone)] struct GpuEntry { service: u32, @@ -19,8 +27,68 @@ struct GpuEntry { name: String, } -// Service handles kept in GPU_ENTRIES are cached for the process lifetime -// and never released; services we skip are released as we go. +/// Reference counts for owned IOKit service handles. Every handle returned +/// by `IOIteratorNext` is retained here; a handle is `IOObjectRelease`d when +/// its count drops to zero. This is what lets us release superseded or +/// detached services on each periodic refresh (and on cache invalidation) +/// without leaking them for the life of the tray process, while still keeping +/// a handle alive for any in-flight reader that holds a reference to it. +static SERVICE_REFS: LazyLock>> = + LazyLock::new(|| Mutex::new(HashMap::new())); + +fn retain_service(service: u32) { + let mut map = SERVICE_REFS.lock().unwrap(); + *map.entry(service).or_insert(0) += 1; +} + +fn release_service(service: u32) { + let mut map = SERVICE_REFS.lock().unwrap(); + match map.get_mut(&service) { + Some(count) if *count > 1 => *count -= 1, + Some(_) => { + map.remove(&service); + unsafe { IOObjectRelease(service) }; + } + None => { + // Not tracked (defensive); nothing to release. + } + } +} + +fn retain_entries(entries: &[GpuEntry]) { + for e in entries { + retain_service(e.service); + } +} + +fn release_entries(entries: &[GpuEntry]) { + for e in entries { + release_service(e.service); + } +} + +/// RAII guard over a set of entries that owns one reference to each of their +/// service handles. `gpu_entries()` returns a clone with these references +/// already taken; the guard releases them on drop so callers never leak. +struct EntriesGuard { + entries: Vec, +} + +impl EntriesGuard { + fn new(entries: Vec) -> Self { + EntriesGuard { entries } + } + fn as_slice(&self) -> &[GpuEntry] { + &self.entries + } +} + +impl Drop for EntriesGuard { + fn drop(&mut self) { + release_entries(&self.entries); + } +} + extern "C" { fn IOServiceMatching(name: *const i8) -> *mut c_void; fn IOServiceGetMatchingServices( @@ -57,17 +125,20 @@ pub struct MacosGpuMonitor; impl GpuMonitor for MacosGpuMonitor { fn enumerate_gpus() -> Vec { - gpu_entries() - .into_iter() + let guard = EntriesGuard::new(gpu_entries()); + guard + .as_slice() + .iter() .map(|e| GpuDevice { - id: e.id, - name: e.name, + id: e.id.clone(), + name: e.name.clone(), }) .collect() } fn get_gpu_usage(scope: Option<&str>) -> io::Result { - let mut entries = gpu_entries(); + let guard = EntriesGuard::new(gpu_entries()); + let entries = guard.as_slice(); if entries.is_empty() { return Err(io::Error::other( "No GPU performance statistics available (no IOAccelerator service found)", @@ -78,25 +149,27 @@ impl GpuMonitor for MacosGpuMonitor { "Selected GPU device no longer exists (fall back to all GPUs)", )); } - if let Some(v) = read_max(&entries, scope) { + if let Some(v) = read_max(entries, scope) { return Ok(v); } // All reads failed — the cached services may be stale (e.g. after a - // GPU hot-plug). Invalidate the cache and retry once. - *GPU_CACHE.lock().unwrap() = None; - entries = gpu_entries(); - if entries.is_empty() { + // GPU hot-plug). Invalidate the cache (releasing its references) and + // retry once. + invalidate_gpu_cache(); + let guard2 = EntriesGuard::new(gpu_entries()); + let entries2 = guard2.as_slice(); + if entries2.is_empty() { return Err(io::Error::other( "No GPU performance statistics available (no IOAccelerator service found)", )); } - if scope.map_or(false, |s| !entries.iter().any(|e| e.id == s)) { + if scope.map_or(false, |s| !entries2.iter().any(|e| e.id == s)) { return Err(io::Error::other( "Selected GPU device no longer exists (fall back to all GPUs)", )); } - match read_max(&entries, scope) { + match read_max(entries2, scope) { Some(v) => Ok(v), None => Err(io::Error::other( "GPU service found but no 'Device Utilization %' statistic", @@ -105,6 +178,17 @@ impl GpuMonitor for MacosGpuMonitor { } } +/// Release the references held by the current cache's entries and clear the +/// cache. In-flight readers are unaffected because their `EntriesGuard` +/// clones hold independent references. +fn invalidate_gpu_cache() { + let mut guard = GPU_CACHE.lock().unwrap(); + if let Some(old) = guard.as_ref() { + release_entries(&old.entries); + } + *guard = None; +} + /// The maximum "Device Utilization %" across the entries in scope /// (`scope` = `None` means all entries), or `None` when no entry exposes /// the statistic. @@ -125,6 +209,9 @@ fn read_max(entries: &[GpuEntry], scope: Option<&str>) -> Option { /// Return the cached GPU service list, re-enumerating when due (first /// use, every `GPU_REFRESH_INTERVAL` samples, or after an invalidation). +/// +/// The returned clone owns one reference to each of its service handles; +/// wrap it in an [`EntriesGuard`] (as all callers do) to release them. fn gpu_entries() -> Vec { let mut guard = GPU_CACHE.lock().unwrap(); let due = match guard.as_mut() { @@ -135,7 +222,13 @@ fn gpu_entries() -> Vec { } }; if due { + // lookup_gpu_entries() retains every handle it keeps; the old cache + // still holds its own references, so release those now (superseded + // or detached devices drop to zero and are IOObjectRelease'd). let found = lookup_gpu_entries(); + if let Some(old) = guard.as_ref() { + release_entries(&old.entries); + } if found.is_empty() { // Nothing found yet (or all detached) — do not cache the empty // result; retry on the next call. @@ -147,7 +240,9 @@ fn gpu_entries() -> Vec { }); } } - guard.as_ref().map(|c| c.entries.clone()).unwrap_or_default() + let clone = guard.as_ref().map(|c| c.entries.clone()).unwrap_or_default(); + retain_entries(&clone); + clone } /// Enumerate every GPU accelerator service. @@ -156,7 +251,15 @@ fn gpu_entries() -> Vec { /// by us — `IOServiceGetMatchingServices` is declared `CF_RELEASES_ARGUMENT` /// and always consumes one reference of the matching dictionary (observed /// 2026-09-11: CFReleasing it afterwards segfaults). +/// +/// Ownership: every handle returned by `IOIteratorNext` is immediately +/// retained in `SERVICE_REFS`. Kept services keep that reference (the cache +/// now owns it); services we skip — or that are duplicates across classes, +/// since a single device can match both `IOAccelerator` and +/// `AGXAccelerator` — are released right away. fn lookup_gpu_entries() -> Vec { + let mut entries = Vec::new(); + let mut seen: HashSet = HashSet::new(); for class in ["IOAccelerator", "AGXAccelerator"] { let Ok(class_cstr) = CString::new(class) else { continue; @@ -174,36 +277,33 @@ fn lookup_gpu_entries() -> Vec { continue; } - let mut entries = Vec::new(); let mut service = unsafe { IOIteratorNext(iterator) }; while service != 0 { - // Only keep services that actually expose the utilization - // statistic (matches the single-service behavior). Kept - // services are cached for the process lifetime and never - // released; the rest are released right away. - if read_device_utilization(service).is_some() { + // We now own this reference. + retain_service(service); + // A device can match multiple classes; keep only the first. + if seen.insert(service) && read_device_utilization(service).is_some() { let name = read_string_property(service, "model") .or_else(|| read_string_property(service, "name")) - .unwrap_or_else(|| format!("GPU {}", entries.len())); - let id = format!("macos-gpu-{}", entries.len()); + .unwrap_or_else(|| "GPU".to_string()); + // Device-derived, position-independent id (the service + // handle is stable for the device's lifetime). + let id = format!("macos-gpu-{}", service); entries.push(GpuEntry { service, id, name, }); } else { - unsafe { IOObjectRelease(service) }; + // Skipped or duplicate — release the reference we just took. + release_service(service); } service = unsafe { IOIteratorNext(iterator) }; } - // Release the iterator; the services we kept are cached. + // Release the iterator; kept services are referenced separately. unsafe { IOObjectRelease(iterator) }; - - if !entries.is_empty() { - return entries; - } } - Vec::new() + entries } /// Read "Device Utilization %" from the IORegistry `PerformanceStatistics` @@ -261,52 +361,6 @@ fn read_string_property(service: u32, key: &str) -> Option { mod tests { use super::*; - /// TEMP probe: enumerate raw services twice (with a pause) to see - /// whether the io_service_t handles are stable across re-enumeration. - #[test] - fn probe_handle_stability() { - fn raw_services() -> Vec { - let mut out = Vec::new(); - for class in ["IOAccelerator", "AGXAccelerator"] { - let Ok(c) = CString::new(class) else { continue }; - let matching = unsafe { IOServiceMatching(c.as_ptr()) }; - if matching.is_null() { - continue; - } - let mut it: u32 = 0; - if unsafe { IOServiceGetMatchingServices(0, matching, &mut it) } != 0 || it == 0 { - continue; - } - let mut s = unsafe { IOIteratorNext(it) }; - while s != 0 { - out.push(s); - s = unsafe { IOIteratorNext(it) }; - } - unsafe { IOObjectRelease(it) }; - // release the extra refs we took for the probe (we keep one - // reference alive for the process; releasing all would be - // fine for a probe too, but keep it simple) - } - out - } - let first = raw_services(); - eprintln!("probe first: {:?}", first); - std::thread::sleep(std::time::Duration::from_millis(500)); - let second = raw_services(); - eprintln!("probe second: {:?}", second); - eprintln!( - "probe stable: {} (equal sets: {})", - first == second, - { - let mut a = first.clone(); - let mut b = second.clone(); - a.sort(); - b.sort(); - a == b - } - ); - } - /// Smoke test: exercise the sampling path and print the reading so it /// can be checked manually on a machine with a GPU. Deliberately /// lenient — a machine with no GPU at all is not a failure. diff --git a/src/platform/windows/gpu_usage.rs b/src/platform/windows/gpu_usage.rs index 700e382..cb66664 100644 --- a/src/platform/windows/gpu_usage.rs +++ b/src/platform/windows/gpu_usage.rs @@ -1,8 +1,13 @@ use crate::platform::{GpuDevice, GpuMonitor}; -use std::collections::BTreeMap; +use std::alloc::{alloc, dealloc, Layout}; +use std::collections::{BTreeMap, BTreeSet}; use std::io; +use std::ptr::NonNull; use std::sync::Mutex; use windows::core::{PCWSTR, PWSTR}; +use windows::Win32::Graphics::Dxgi::{ + CreateDXGIFactory1, IDXGIAdapter1, IDXGIFactory1, DXGI_ADAPTER_FLAG_SOFTWARE, +}; use windows::Win32::System::Performance::{ PDH_FMT_COUNTERVALUE_ITEM_W, PDH_FMT_DOUBLE, PDH_HCOUNTER, PDH_HQUERY, PDH_MORE_DATA, PdhAddEnglishCounterW, PdhCollectQueryData, PdhCloseQuery, PdhGetFormattedCounterArrayW, @@ -24,6 +29,59 @@ unsafe impl Send for GpuPdhState {} static GPU_STATE: Mutex> = Mutex::new(None); +/// A heap-allocated byte buffer with an explicit alignment, for passing to +/// `PdhGetFormattedCounterArrayW`. PDH fills the buffer with a leading array +/// of `PDH_FMT_COUNTERVALUE_ITEM_W` structs (each holding a `PWSTR` and an +/// `f64`), which require 8-byte alignment; a plain `Vec` only guarantees +/// byte alignment, so casting its pointer to that slice type would be +/// undefined behavior. Allocating through the global allocator with an +/// explicit `Layout` gives the buffer the alignment the item type needs. +struct AlignedBuf { + ptr: NonNull, + len: usize, +} + +impl AlignedBuf { + fn new(len: usize) -> io::Result { + let layout = Layout::from_size_align( + len, + std::mem::align_of::(), + ) + .map_err(|_| io::Error::other("invalid PDH array buffer layout"))?; + let ptr = unsafe { alloc(layout) }; + if ptr.is_null() { + return Err(io::Error::other( + "out of memory allocating PDH array buffer", + )); + } + Ok(AlignedBuf { + ptr: unsafe { NonNull::new_unchecked(ptr) }, + len, + }) + } + + fn as_mut_ptr(&mut self) -> *mut u8 { + self.ptr.as_ptr() + } + + fn as_ptr(&self) -> *const u8 { + self.ptr.as_ptr() + } +} + +impl Drop for AlignedBuf { + fn drop(&mut self) { + // The same size/alignment pair succeeded at allocation time, so this + // cannot fail. + if let Ok(layout) = Layout::from_size_align( + self.len, + std::mem::align_of::(), + ) { + unsafe { dealloc(self.ptr.as_ptr(), layout) }; + } + } +} + /// Per-engine utilization counter, available since Windows 10 1803 for all /// GPU vendors (NVIDIA, AMD, Intel). The `(*)` wildcard expands to one /// instance per (process, engine) pair — instance names look like @@ -45,29 +103,47 @@ struct EngineInstance { gpu_id: String, /// Menu label for the GPU (`GPU 0`, `GPU 1`, ...). gpu_label: String, - /// Physical engine identity: `win-gpu-/` (or the full + /// Physical engine identity: `win-gpu-/` (or the full /// instance name when the fields are missing). engine_key: String, value: f64, } +/// Parse a hex u32 like `0x00017FEA` (the `0x` prefix is optional). +fn parse_hex_u32(s: &str) -> Option { + let s = s + .strip_prefix("0x") + .or_else(|| s.strip_prefix("0X")) + .unwrap_or(s); + u32::from_str_radix(s, 16).ok() +} + /// Parse an instance name like /// `pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D`. /// /// Returns `(gpu_id, gpu_label, engine_key)`. The physical GPU is -/// identified by the `phys` field (the `luid` field varies per GPU -/// context even on a single physical GPU). Missing fields degrade -/// gracefully so unusual names still produce a usable (unique) key. +/// identified by the `luid` field, which is the adapter LUID — the same +/// value DXGI reports as `IDXGIAdapter::GetDesc1().AdapterLuid`, so the +/// id matches `enumerate_dxgi_adapters()` exactly (the `phys` field is NOT +/// a reliable per-adapter index: on multi-GPU systems it can be `phys_0` +/// for every adapter). Missing fields degrade gracefully so unusual names +/// still produce a usable (unique) key. fn parse_instance_name(name: &str) -> (String, String, String) { let parts: Vec<&str> = name.split('_').collect(); - let mut phys: Option<&str> = None; + let mut luid: Option<(u32, u32)> = None; // (HighPart, LowPart) let mut eng: Option<&str> = None; let mut i = 0; while i < parts.len() { match parts[i] { - "phys" if i + 1 < parts.len() => { - phys = Some(parts[i + 1]); - i += 1; + // `luid_0x_0x` — the two following tokens are + // the adapter LUID halves. + "luid" if i + 2 < parts.len() => { + if let (Some(h), Some(l)) = + (parse_hex_u32(parts[i + 1]), parse_hex_u32(parts[i + 2])) + { + luid = Some((h, l)); + } + i += 2; } "eng" if i + 1 < parts.len() => { eng = Some(parts[i + 1]); @@ -78,11 +154,16 @@ fn parse_instance_name(name: &str) -> (String, String, String) { i += 1; } - match (phys, eng) { - (Some(p), Some(e)) => ( - format!("win-gpu-{p}"), - format!("GPU {p}"), - format!("win-gpu-{p}/{e}"), + match (luid, eng) { + (Some((h, l)), Some(e)) => ( + format!("win-gpu-{h:08x}_{l:08x}"), + format!("GPU 0x{l:08x}"), + format!("win-gpu-{h:08x}_{l:08x}/{e}"), + ), + (Some((h, l)), None) => ( + format!("win-gpu-{h:08x}_{l:08x}"), + format!("GPU 0x{l:08x}"), + format!("win-gpu-{h:08x}_{l:08x}"), ), _ => ("default".to_string(), "GPU".to_string(), name.to_string()), } @@ -158,7 +239,7 @@ fn collect_instances() -> io::Result> { PdhGetFormattedCounterArrayW(state.hcounter, PDH_FMT_DOUBLE, &mut size, &mut count, None) }; if ret == PDH_MORE_DATA && count > 0 { - let mut buf: Vec = vec![0u8; size as usize]; + let mut buf = AlignedBuf::new(size as usize)?; let ret = unsafe { PdhGetFormattedCounterArrayW( state.hcounter, @@ -230,7 +311,16 @@ fn collect_instances() -> io::Result> { /// Sum per-process instances per physical engine (capped at 100%) and /// return the max across engines for the requested GPU scope. -fn aggregate(instances: &[EngineInstance], scope: Option<&str>) -> io::Result { +/// +/// `known` is the set of GPU ids that currently exist (installed display +/// adapters plus any with live instances). A scope that is in `known` but has +/// no live instances is an *idle* GPU and reports 0%, rather than being +/// treated as a missing device. +fn aggregate( + instances: &[EngineInstance], + scope: Option<&str>, + known: &BTreeSet, +) -> io::Result { let mut engines: BTreeMap<(String, String), f64> = BTreeMap::new(); let mut scoped = false; for inst in instances { @@ -246,9 +336,25 @@ fn aggregate(instances: &[EngineInstance], scope: Option<&str>) -> io::Result return Ok(0.0), + Some(_) => { + return Err(io::Error::other( + "Selected GPU device no longer exists (fall back to all GPUs)", + )) + } + None => { + // All GPUs idle (or none with live data). + if !known.is_empty() { + return Ok(0.0); + } + return Err(io::Error::other( + "No GPU engine utilization data available", + )); + } + } } engines .values() @@ -257,18 +363,84 @@ fn aggregate(instances: &[EngineInstance], scope: Option<&str>) -> io::Result_`), the same +/// value that appears in the PDH instance names, so the ids match exactly +/// for scoping. The adapter description is used as the display name. +/// +/// Returns the real (hardware) adapters plus the set of software adapter ids +/// (e.g. "Microsoft Basic Render Driver"), which the caller can use to keep +/// the software fallbacks out of the device menu. +fn enumerate_dxgi() -> (Vec<(String, String)>, BTreeSet) { + let mut real: Vec<(String, String)> = Vec::new(); + let mut software: BTreeSet = BTreeSet::new(); + let factory: IDXGIFactory1 = match unsafe { CreateDXGIFactory1() } { + Ok(f) => f, + Err(_) => return (real, software), + }; + let mut index = 0u32; + loop { + // `EnumAdapters1` returns `Err` (DXGI_ERROR_NOT_FOUND) past the last + // adapter. + let adapter: IDXGIAdapter1 = match unsafe { factory.EnumAdapters1(index) } { + Ok(a) => a, + Err(_) => break, + }; + if let Ok(d) = unsafe { adapter.GetDesc1() } { + let id = format!( + "win-gpu-{:08x}_{:08x}", + d.AdapterLuid.HighPart, d.AdapterLuid.LowPart + ); + // Software fallback adapters are tracked separately. + if d.Flags & (DXGI_ADAPTER_FLAG_SOFTWARE.0 as u32) != 0 { + software.insert(id); + } else { + let end = d + .Description + .iter() + .position(|&c| c == 0) + .unwrap_or(d.Description.len()); + let name = String::from_utf16_lossy(&d.Description[..end]); + let name = if name.trim().is_empty() { + format!("GPU 0x{:08x}", d.AdapterLuid.LowPart) + } else { + name + }; + real.push((id, name)); + } + } + index += 1; + } + (real, software) +} + impl GpuMonitor for WindowsGpuMonitor { fn enumerate_gpus() -> Vec { - let instances = match collect_instances() { - Ok(i) => i, - Err(_) => return Vec::new(), - }; - let mut seen: BTreeMap = BTreeMap::new(); - for inst in &instances { - seen.entry(inst.gpu_id.clone()) - .or_insert_with(|| inst.gpu_label.clone()); + let (real, software) = enumerate_dxgi(); + let mut devices: BTreeMap = BTreeMap::new(); + // Installed adapters first (includes idle ones). + for (id, name) in real { + devices.insert(id, name); + } + // Active PDH instances catch GPUs not in the display list (e.g. + // headless compute adapters); their label is used only if the + // display enumeration did not already provide a name. Software + // fallback adapters are skipped to keep the menu clean. + if let Ok(instances) = collect_instances() { + for inst in &instances { + if !software.contains(&inst.gpu_id) { + devices + .entry(inst.gpu_id.clone()) + .or_insert_with(|| inst.gpu_label.clone()); + } + } } - seen + devices .into_iter() .map(|(id, name)| GpuDevice { id, name }) .collect() @@ -276,7 +448,17 @@ impl GpuMonitor for WindowsGpuMonitor { fn get_gpu_usage(scope: Option<&str>) -> io::Result { let instances = collect_instances()?; - aggregate(&instances, scope) + // The set of known GPU ids: installed display adapters plus any that + // currently have PDH instances. A scoped id in this set but without + // instances is an *idle* GPU (report 0%), not a missing one. + let mut known: BTreeSet = BTreeSet::new(); + for (id, _) in enumerate_dxgi().0 { + known.insert(id); + } + for inst in &instances { + known.insert(inst.gpu_id.clone()); + } + aggregate(&instances, scope, &known) } } @@ -286,21 +468,23 @@ mod tests { #[test] fn parse_name_full() { + // The id is the adapter LUID (matches DXGI's AdapterLuid), NOT the + // `phys` field (which is `phys_0` for every adapter on some systems). let (gpu_id, gpu_label, engine_key) = parse_instance_name("pid_10584_luid_0x00000000_0x00017FEA_phys_0_eng_0_engtype_3D"); - assert_eq!(gpu_id, "win-gpu-0"); - assert_eq!(gpu_label, "GPU 0"); - assert_eq!(engine_key, "win-gpu-0/0"); + assert_eq!(gpu_id, "win-gpu-00000000_00017fea"); + assert_eq!(gpu_label, "GPU 0x00017fea"); + assert_eq!(engine_key, "win-gpu-00000000_00017fea/0"); } #[test] fn parse_name_multiple_gpus_and_engtypes() { - // Second physical GPU, high engine index, engtype with spaces. + // Non-zero luid halves, high engine index, engtype with spaces. let (gpu_id, gpu_label, engine_key) = parse_instance_name("pid_1_luid_0x1_0x2_phys_1_eng_10_engtype_Video Codec 0"); - assert_eq!(gpu_id, "win-gpu-1"); - assert_eq!(gpu_label, "GPU 1"); - assert_eq!(engine_key, "win-gpu-1/10"); + assert_eq!(gpu_id, "win-gpu-00000001_00000002"); + assert_eq!(gpu_label, "GPU 0x00000002"); + assert_eq!(engine_key, "win-gpu-00000001_00000002/10"); } #[test] @@ -315,11 +499,19 @@ mod tests { /// lenient — a machine with no GPU at all is not a failure. #[test] fn gpu_usage_smoke() { - eprintln!("Windows GPUs: {:?}", WindowsGpuMonitor::enumerate_gpus()); + let gpus = WindowsGpuMonitor::enumerate_gpus(); + eprintln!("Windows GPUs: {:?}", gpus); match WindowsGpuMonitor::get_gpu_usage(None) { - Ok(v) => eprintln!("Windows GPU usage: {:.2}%", v), + Ok(v) => eprintln!("Windows GPU usage (all): {:.2}%", v), Err(e) => eprintln!("Windows GPU usage unavailable: {}", e), } + // Scope to each enumerated GPU to verify the id matches PDH. + for g in &gpus { + match WindowsGpuMonitor::get_gpu_usage(Some(&g.id)) { + Ok(v) => eprintln!(" scoped {}: {:.2}%", g.name, v), + Err(e) => eprintln!(" scoped {}: error: {}", g.name, e), + } + } } #[test] @@ -330,12 +522,16 @@ mod tests { engine_key: eng.into(), value: v, }; + // The set of GPU ids that currently exist. + let known: BTreeSet = ["g".to_string(), "other".to_string()] + .into_iter() + .collect(); let instances = vec![ inst("1", "e0", 35.0), inst("2", "e0", 45.0), // same engine, second process → 80 inst("1", "e1", 10.0), // other engine, must not win ]; - let v = aggregate(&instances, None).unwrap(); + let v = aggregate(&instances, None, &known).unwrap(); assert_eq!(v, 80.0); // Sum is capped at 100. @@ -347,7 +543,7 @@ mod tests { }]; let mut instances = capped.clone(); instances.push(capped[0].clone()); - assert_eq!(aggregate(&instances, None).unwrap(), 100.0); + assert_eq!(aggregate(&instances, None, &known).unwrap(), 100.0); // Scope filters by GPU id (the 60% instance of the other GPU is // ignored). @@ -355,7 +551,22 @@ mod tests { other.gpu_id = "other".into(); let mut instances = capped; instances.push(other); - assert_eq!(aggregate(&instances, Some("g")).unwrap(), 60.0); - assert!(aggregate(&instances, Some("missing")).is_err()); + assert_eq!(aggregate(&instances, Some("g"), &known).unwrap(), 60.0); + // A scope that is not a known device is missing. + assert!(aggregate(&instances, Some("missing"), &known).is_err()); + + // A known device with no live instances is idle → 0%, not missing. + let idle = vec![EngineInstance { + gpu_id: "other".into(), + gpu_label: "GPU 1".into(), + engine_key: "other/e0".into(), + value: 42.0, + }]; + assert_eq!(aggregate(&idle, Some("g"), &known).unwrap(), 0.0); + // No scope and no live instances but known devices exist → 0%. + assert_eq!(aggregate(&[], None, &known).unwrap(), 0.0); + // No scope, no instances, no known devices → error. + let empty: BTreeSet = BTreeSet::new(); + assert!(aggregate(&[], None, &empty).is_err()); } } From 47ca72a6ded722c9be85173b072c6bf5568bd3e8 Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 16:01:57 +0900 Subject: [PATCH 8/9] fix: rebuild tray menu before showing it (GPU topology changes) The GPU device submenu was only rebuilt on unrelated setting events, so an eGPU attached (or detached) after startup stayed absent from the menu until the user changed another setting. Re-enumerate and install a fresh menu in the ShowMenu handler before show_menu(), synchronously (not via ui_update) so the fresh menu is installed before it is displayed. --- src/app.rs | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/src/app.rs b/src/app.rs index 257068f..f637472 100644 --- a/src/app.rs +++ b/src/app.rs @@ -386,7 +386,18 @@ impl App { } } Events::ShowMenu => { + // Rebuild the menu before showing it so a GPU that was + // attached or detached since the last build is + // reflected immediately — the menu is otherwise only + // rebuilt on unrelated setting events, so an eGPU + // plugged into a single-GPU machine would stay absent + // until the user changed another setting. Done + // synchronously here (not via `ui_update`) so the fresh + // menu is installed before `show_menu()`. + let gpu_scope = self.gpu_scope.lock().unwrap().clone(); if let Ok(mut tray) = self.tray_icon.lock() { + let _ = tray + .set_menu(&build_menu(&self.icon_manager, gpu_scope.as_deref())); if let Err(e) = tray.show_menu() { eprintln!("Failed to show menu: {}", e); } From e9f696138d9fcf7184115ee8d13f51d40a9ede05 Mon Sep 17 00:00:00 2001 From: Bearice Ren Date: Fri, 11 Sep 2026 16:48:49 +0900 Subject: [PATCH 9/9] fix: hide GPU device submenu when animation source is CPU The 'GPU Device' menu was shown whenever the machine has more than one GPU, even when the animation is driven by CPU where the device selection is irrelevant. Gate it on current_source != AnimationSource::Cpu so it only appears for the Gpu and Both sources. The menu is already rebuilt on each source change (SetAnimationSource) and on ShowMenu, so it appears/disappears correctly. --- src/events.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/events.rs b/src/events.rs index bb0f2de..a909a6b 100644 --- a/src/events.rs +++ b/src/events.rs @@ -81,9 +81,11 @@ pub fn build_menu(icon_manager: &IconManager, gpu_scope: Option<&str>) -> MenuBu menu = menu.submenu("Usage Source", source_menu); // Build GPU device submenu — only shown when the machine has more than - // one GPU that exposes utilization. + // one GPU that exposes utilization AND the animation is driven by the + // GPU (source is `Gpu` or `Both`). In `Cpu` mode the device selection is + // irrelevant, so it is hidden. let gpus = GpuMonitorImpl::enumerate_gpus(); - if gpus.len() > 1 { + if gpus.len() > 1 && current_source != AnimationSource::Cpu { let mut gpu_menu = MenuBuilder::new(); gpu_menu = gpu_menu .radio("All GPUs", gpu_scope.is_none(), Events::SetGpuScope(None));