From 309e37f1e1a8ad6408abd96cbe2ed5c4f09ccab1 Mon Sep 17 00:00:00 2001 From: enricobuehler Date: Tue, 21 Jul 2026 15:52:33 +0200 Subject: [PATCH] fix(host/linux): scope the GPU clock pin to live client sessions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PUNKTFUNK_PIN_CLOCKS clock pin (AMD power_dpm_force_performance_level=high / NVIDIA nvmlDeviceSetGpuLockedClocks) was armed once at host start and released only at host exit, so the box's clocks stayed pinned the whole time the host ran — even with no client connected. Move the pin behind a box-wide refcount (gpuclocks::session_pin()) shared across both streaming planes (native + GameStream): it arms on the first live client and releases on the last disconnect. on_host_start() now only installs the process-scoped NVIDIA P2-cap application profile. No behavior change when PUNKTFUNK_PIN_CLOCKS is unset. Co-Authored-By: Claude Fable 5 --- .../punktfunk-host/src/gamestream/stream.rs | 6 + crates/punktfunk-host/src/linux/gpuclocks.rs | 120 ++++++++++++++---- crates/punktfunk-host/src/main.rs | 17 ++- crates/punktfunk-host/src/native.rs | 7 + 4 files changed, 121 insertions(+), 29 deletions(-) diff --git a/crates/punktfunk-host/src/gamestream/stream.rs b/crates/punktfunk-host/src/gamestream/stream.rs index a43a6eb0..75d30fd3 100644 --- a/crates/punktfunk-host/src/gamestream/stream.rs +++ b/crates/punktfunk-host/src/gamestream/stream.rs @@ -86,6 +86,12 @@ pub fn start( crate::events::emit(crate::events::EventKind::ClientConnected { client: event_client.clone(), }); + // GPU clock pin (Linux, opt-in `PUNKTFUNK_PIN_CLOCKS`): hold the box-wide vendor clock + // floor while this compat-plane stream runs, refcounted with every other live session + // across both planes. Released when the closure exits (stream stopped) — so idle clocks + // aren't pinned between Moonlight sessions. No-op off Linux / when the flag is unset. + #[cfg(target_os = "linux")] + let _clock_pin = crate::gpuclocks::session_pin(); let result = run( cfg, app.as_ref(), diff --git a/crates/punktfunk-host/src/linux/gpuclocks.rs b/crates/punktfunk-host/src/linux/gpuclocks.rs index 35f13ebe..62c467da 100644 --- a/crates/punktfunk-host/src/linux/gpuclocks.rs +++ b/crates/punktfunk-host/src/linux/gpuclocks.rs @@ -2,11 +2,15 @@ //! plan Tier 1B): the driver's adaptive P-state ramps clocks down between bursty encode frames, //! so every frame re-pays a spin-up. This is NOT theoretical — measured on the 780M (VCN 4), //! a 1440p HEVC encode takes ~4.4 ms/frame with the clocks hot (120 fps pacing) but ~8 ms/frame -//! at a 60 fps duty cycle: the sag doubles per-frame encode latency. +//! at a 60 fps duty cycle: the sag doubles per-frame encode latency. So the pin is worth holding +//! only while a client is actively streaming: it is armed on the first live client and released +//! when the last one disconnects (refcounted across both streaming planes — see [`session_pin`]), +//! leaving the driver's idle power management alone the rest of the time. //! //! **AMD** (`PUNKTFUNK_PIN_CLOCKS=1`, root-gated by sysfs ownership): write `high` into each -//! amdgpu card's `power_dpm_force_performance_level` for the host lifetime, restoring the prior -//! value on exit. Non-root gets EACCES → logged once with the privilege recipe. Deliberately +//! amdgpu card's `power_dpm_force_performance_level` while a client is streaming, restoring the +//! prior value when the last client disconnects. Non-root gets EACCES → logged once with the +//! privilege recipe. Deliberately //! opt-in: it defeats power management box-wide and is wrong on battery (Steam Deck!). //! //! **NVIDIA** — two independent halves, both no-ops off NVIDIA: @@ -28,14 +32,16 @@ //! while leaving boost headroom — NVIDIA's own latency guidance is "raise the floor, don't pin //! the max" (locking above base just gets throttled; a max pin only burns idle watts). Non-root //! callers get `NVML_ERROR_NO_PERMISSION` — logged once with the privilege recipe, then the -//! host runs unpinned. The pin is undone on drop (host exit); after a crash it persists until -//! driver reload/reboot, which the reset-before-pin on the next start self-heals. Deliberately +//! host runs unpinned. The pin is undone on drop (when the last client disconnects); after a +//! crash it persists until driver reload/reboot, which the reset-before-pin on the next arm +//! self-heals. Deliberately //! NOT default-on: it defeats idle downclocking for the whole box and is wrong on //! battery-powered hosts. // Every `unsafe` block in this file carries a `// SAFETY:` proof; enforce it (unsafe-proof program). #![deny(clippy::undocumented_unsafe_blocks)] use std::os::raw::{c_char, c_int, c_uint, c_void}; +use std::sync::{Mutex, OnceLock}; /// `nvmlDevice_t` — an opaque driver handle. type NvmlDevice = *mut c_void; @@ -133,8 +139,9 @@ struct AmdPin { restore: String, } -/// Host-lifetime guard: holds the armed clock pins (NVML floor and/or amdgpu perf level) and -/// undoes them on drop. +/// Holds the armed clock pins (NVML floor and/or amdgpu perf level) and undoes them on drop. Owned +/// by the [`session_pin`] refcount: constructed when the first live client arms the pin, dropped +/// (clocks restored) when the last one disconnects. pub struct ClockGuard { nvml: Option, amd: Vec, @@ -142,9 +149,10 @@ pub struct ClockGuard { // SAFETY: `ClockGuard` holds opaque NVML device handles + resolved fn pointers from the loaded // driver library (plus plain sysfs paths/strings). NVML is documented thread-safe, the handles are -// plain driver tokens with no thread affinity, and the guard is only ever *moved* (held in `main`, -// dropped once at exit) and used through `&mut`/ownership — never shared. Transfer across threads -// is therefore sound. +// plain driver tokens with no thread affinity, and the guard is only ever *moved* (into the +// `pin_refcount` mutex when armed, taken back out and dropped when the last client disconnects) and +// used through exclusive ownership behind that mutex — never shared. Transfer across threads is +// therefore sound. unsafe impl Send for ClockGuard {} impl Drop for ClockGuard { @@ -182,22 +190,89 @@ impl Drop for ClockGuard { } /// Startup hook for the host subcommands (`serve` / `punktfunk1-host`): install the NVIDIA P2-cap -/// application profile and, when `PUNKTFUNK_PIN_CLOCKS` is set, arm the vendor clock pin (NVML -/// core-clock floor / amdgpu `high` performance level). Returns the guard keeping the pins for -/// the host lifetime. `None` when nothing was armed. -pub fn on_host_start() -> Option { +/// application profile — the process-scoped, no-root half of the NVIDIA lever, which the driver +/// only acts on once the host holds a live CUDA/NVENC context (i.e. during a session). +/// +/// The vendor clock *pin* is deliberately NOT armed here anymore: held for the whole host lifetime +/// it kept the box's clocks hot even with no client connected. It is now refcounted per live client +/// via [`session_pin`] (armed on both streaming planes), so idle clocks are left to the driver's +/// power management until someone actually streams. +pub fn on_host_start() { if nvidia_present() { ensure_cuda_perf_profile(); } +} + +/// The box-wide clock-pin refcount, shared across BOTH streaming planes (native + GameStream): the +/// vendor pin is a single global GPU setting, so N concurrent sessions share ONE pin — armed when +/// the first client goes live, released when the last one leaves. +struct PinRefcount { + /// Number of live [`SessionClockPin`] handles. + live: usize, + /// The armed pins, present iff `live > 0` and something was actually pinnable. + guard: Option, +} + +fn pin_refcount() -> &'static Mutex { + static STATE: OnceLock> = OnceLock::new(); + STATE.get_or_init(|| { + Mutex::new(PinRefcount { + live: 0, + guard: None, + }) + }) +} + +/// RAII handle that keeps the box-wide clock pin armed while it is alive. Obtain one per live client +/// session on either plane via [`session_pin`]; when the last outstanding handle drops, the pin is +/// released and the driver's idle downclocking resumes. A no-op handle when `PUNKTFUNK_PIN_CLOCKS` +/// is unset (the opt-in gate) — the refcount only ticks for sessions that asked for pinning. +pub struct SessionClockPin { + /// Whether this handle actually incremented the refcount (false = opt-in gate off → no-op). + counted: bool, +} + +/// Arm the box-wide clock pin for one live client session, refcounted across every plane. Returns +/// an RAII handle; the pin is released when the last handle drops. A no-op (returns immediately, +/// touches no GPU state) unless `PUNKTFUNK_PIN_CLOCKS` is set. +pub fn session_pin() -> SessionClockPin { if !flag_truthy("PUNKTFUNK_PIN_CLOCKS") { - return None; + return SessionClockPin { counted: false }; } - let nvml = if nvidia_present() { pin_nvidia() } else { None }; - let amd = pin_amdgpu(); - if nvml.is_none() && amd.is_empty() { - return None; + let mut state = pin_refcount().lock().unwrap(); + state.live += 1; + if state.live == 1 { + // 0 → 1: the first live client — arm the vendor pin (reset-before-pin inside `pin_nvidia` + // heals a stale pin from a crashed previous run). + let nvml = if nvidia_present() { pin_nvidia() } else { None }; + let amd = pin_amdgpu(); + state.guard = if nvml.is_none() && amd.is_empty() { + None // nothing pinnable (no perms / no supported GPU) — the session streams unpinned + } else { + Some(ClockGuard { nvml, amd }) + }; + } + SessionClockPin { counted: true } +} + +impl Drop for SessionClockPin { + fn drop(&mut self) { + if !self.counted { + return; + } + // Take the guard out under the lock but drop it *outside* — releasing the pin does NVML + + // sysfs I/O we don't want to hold the refcount lock across. + let release = { + let mut state = pin_refcount().lock().unwrap(); + state.live = state.live.saturating_sub(1); + if state.live == 0 { + state.guard.take() + } else { + None + } + }; + drop(release); } - Some(ClockGuard { nvml, amd }) } /// Force every amdgpu card's DPM performance level to `high` for the session — the encode-latency @@ -238,7 +313,7 @@ fn pin_amdgpu() -> Vec { card = %name, was = %prev, "amdgpu performance level pinned to high (encode clock sag removed) — \ - restored on host exit" + restored when the last client disconnects" ); pins.push(AmdPin { path, @@ -327,7 +402,8 @@ fn pin_nvidia() -> Option { } tracing::info!( devices = pinned.len(), - "NVIDIA core-clock floor armed (min=TDP/base, max=boost) — released on host exit" + "NVIDIA core-clock floor armed (min=TDP/base, max=boost) — released when the last \ + client disconnects" ); Some(NvmlPin { nvml, pinned }) } diff --git a/crates/punktfunk-host/src/main.rs b/crates/punktfunk-host/src/main.rs index 76e8906f..1654a6d8 100644 --- a/crates/punktfunk-host/src/main.rs +++ b/crates/punktfunk-host/src/main.rs @@ -246,14 +246,17 @@ fn real_main() -> Result<()> { crate::capture::dxgi::install_gpu_pref_hook(); } - // NVIDIA clock hygiene (Linux, host subcommands only): install the P2-cap driver profile and, - // under PUNKTFUNK_PIN_CLOCKS, hold the NVML core-clock floor for the host lifetime (reset on - // exit via the guard's Drop). No-op off NVIDIA / on the tool subcommands. + // NVIDIA clock hygiene (Linux, host subcommands only): install the P2-cap driver profile. The + // vendor clock *pin* (PUNKTFUNK_PIN_CLOCKS) is no longer held for the host lifetime — it is + // armed per live client via `gpuclocks::session_pin()` on both streaming planes, so idle clocks + // are left alone while nobody is connected. No-op off NVIDIA / on the tool subcommands. #[cfg(target_os = "linux")] - let _nv_clocks = match args.first().map(String::as_str) { - Some("serve") | Some("punktfunk1-host") => gpuclocks::on_host_start(), - _ => None, - }; + if matches!( + args.first().map(String::as_str), + Some("serve") | Some("punktfunk1-host") + ) { + gpuclocks::on_host_start(); + } match args.first().map(String::as_str) { // The host: the native punktfunk/1 plane + management API by default (secure), and — with diff --git a/crates/punktfunk-host/src/native.rs b/crates/punktfunk-host/src/native.rs index c4f3a0e0..d5ec9c02 100644 --- a/crates/punktfunk-host/src/native.rs +++ b/crates/punktfunk-host/src/native.rs @@ -1199,6 +1199,13 @@ async fn serve_session( launch: hello.launch.clone(), plane: crate::events::Plane::Native, }); + // GPU clock pin (Linux, opt-in `PUNKTFUNK_PIN_CLOCKS`): hold the box-wide vendor clock floor for + // as long as THIS session streams, refcounted with every other live session across both planes. + // RAII like the marker above — armed on the first live client, released when the last one + // disconnects, so idle clocks aren't pinned while nobody is connected. No-op off Linux / when + // the flag is unset. + #[cfg(target_os = "linux")] + let _clock_pin = crate::gpuclocks::session_pin(); // The session's launch, threaded into the data plane. Windows carries the store-qualified id // (spawned into the interactive user session once capture is live); other hosts resolve the id // to its shell command HERE against the host's own library — a client can only ever pick an