feat(encode/nvenc): SPIR-V cursor blend over Vulkan-allocated input slots — retire the PTX kernels
A vendored PTX blob is JIT'd against the driver's ISA ceiling, so the cursor-blend module silently dies on drivers older than the generating toolkit (CUDA_ERROR_UNSUPPORTED_PTX_VERSION/INVALID_PTX, 222/218 — the KWin leg's invisible composite cursor on driver 595/CUDA 13.2 vs a CUDA 13.3 blob). SPIR-V has no such coupling, and the in-tree precedent already exists twice (vulkan_video's CSC blend, VkBridge's exportable OPAQUE_FD → cuImportExternalMemory bridge). New pf_zerocopy::vkslot::VkSlotBlend: the direct-SDK NVENC encoder now allocates its input ring as exportable Vulkan buffers CUDA-imports (same contiguous InputSurface layouts, pitch = row bytes rounded to 256), and the cursor composite is a compute dispatch over the cursor's rectangle (cursor_blend.comp, vendored .spv; spec-constant selects ARGB/NV12/YUV444; BT.709 limited, matching the retired .cu). The surface SSBO is uint[] with every invocation owning whole words — no 8-bit-storage device dependency. Cursor-bearing frames force the existing CPU-synced submit path so the CUDA copy → Vulkan dispatch (fence-waited) → NVENC encode ordering is CPU-established; cursorless frames keep the stream-ordered fast path untouched. Any bring-up/alloc/registration failure falls back wholesale to plain pitched CUDA surfaces (never a mixed or short ring): sessions always encode, composite mode just loses the cursor, warned once. cursor_blend.cu / cursor_blend.ptx and the CursorBlend PTX loader are deleted. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -7,8 +7,12 @@
|
||||
//! and ffmpeg's `hevc_nvenc` (encode thread) — each thread makes it current before use;
|
||||
//! * device memory: pitched allocations, the reusable `BufferPool`/`DeviceBuffer`, IPC
|
||||
//! export/import, host readback, and the plane copies;
|
||||
//! * GL / external-memory interop (`RegisteredTexture`, `ExternalDmabuf`); and
|
||||
//! * the CUDA cursor-blend kernel (`CursorBlend`).
|
||||
//! * GL / external-memory interop (`RegisteredTexture`, `ExternalDmabuf`).
|
||||
//!
|
||||
//! (The CUDA cursor-blend PTX kernel that used to live here is retired: vendored PTX is JIT'd
|
||||
//! against the driver's ISA ceiling and silently dies on older drivers. The NVENC cursor blend
|
||||
//! is now the SPIR-V compute pass in [`super::vkslot`], dispatched over Vulkan-allocated,
|
||||
//! CUDA-imported input slots.)
|
||||
//!
|
||||
//! (We use GL interop, not EGL interop: `cuGraphicsEGLRegisterImage` is Tegra-only on the desktop
|
||||
//! driver — see [`super::egl`].)
|
||||
@@ -18,7 +22,6 @@
|
||||
#![deny(clippy::undocumented_unsafe_blocks)]
|
||||
|
||||
use anyhow::{bail, Result};
|
||||
use std::ffi::CStr;
|
||||
use std::os::raw::{c_uint, c_void};
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
|
||||
@@ -280,258 +283,6 @@ pub fn copy_stream_handle() -> *mut c_void {
|
||||
/// Max cursor-overlay bitmap edge (px) uploaded to the device blend buffer — matches the Vulkan path.
|
||||
pub const CURSOR_MAX: u32 = 256;
|
||||
|
||||
/// GPU cursor-overlay compositor for the NVENC path (cursor-as-metadata): loads the `cursor_blend`
|
||||
/// PTX module once and blends a straight-alpha RGBA cursor into an encoder-OWNED NVENC input surface
|
||||
/// (ARGB / NV12 / YUV444) with a small kernel launched over the cursor's rectangle — no full-frame
|
||||
/// pass, and the compositor's dmabuf is never touched. The cursor bitmap lives in a device buffer
|
||||
/// re-uploaded only when it changes. Requires `context()` to have succeeded (driver present).
|
||||
pub struct CursorBlend {
|
||||
module: CUmodule,
|
||||
f_argb: CUfunction,
|
||||
f_nv12: CUfunction,
|
||||
f_yuv444: CUfunction,
|
||||
cur_buf: CUdeviceptr, // device RGBA staging (CURSOR_MAX²·4, tight rows)
|
||||
}
|
||||
|
||||
// SAFETY: process-lifetime driver handles used only from the encode thread with the shared context
|
||||
// current — like [`DeviceBuffer`], moving the struct between threads cannot dangle or race.
|
||||
unsafe impl Send for CursorBlend {}
|
||||
|
||||
impl CursorBlend {
|
||||
/// Load the embedded PTX image and resolve the three blend kernels + a device cursor buffer.
|
||||
pub fn new(ptx: &[u8]) -> Result<CursorBlend> {
|
||||
// cuModuleLoadData reads a PTX image as a NUL-terminated string; the embedded .ptx is not,
|
||||
// so append a terminator.
|
||||
let mut image = ptx.to_vec();
|
||||
image.push(0);
|
||||
let mut module: CUmodule = std::ptr::null_mut();
|
||||
// SAFETY: `&mut module` is a live out-param the driver fills; `image` is a NUL-terminated PTX
|
||||
// byte image that outlives the synchronous load. `ck` bails on error before `module` is used.
|
||||
unsafe {
|
||||
ck(
|
||||
cuModuleLoadData(&mut module, image.as_ptr() as *const c_void),
|
||||
"cuModuleLoadData(cursor_blend)",
|
||||
)?;
|
||||
}
|
||||
let getf = |name: &CStr| -> Result<CUfunction> {
|
||||
let mut f: CUfunction = std::ptr::null_mut();
|
||||
// SAFETY: `module` loaded above; each name is a valid NUL-terminated symbol present in
|
||||
// the module (verified in the .ptx `.entry` list); `&mut f` is a live out-param.
|
||||
unsafe {
|
||||
ck(
|
||||
cuModuleGetFunction(&mut f, module, name.as_ptr()),
|
||||
"cuModuleGetFunction",
|
||||
)?;
|
||||
}
|
||||
Ok(f)
|
||||
};
|
||||
let f_argb = getf(c"blend_argb")?;
|
||||
let f_nv12 = getf(c"blend_nv12")?;
|
||||
let f_yuv444 = getf(c"blend_yuv444")?;
|
||||
let mut cur_buf: CUdeviceptr = 0;
|
||||
// SAFETY: `&mut cur_buf` is a live out-param; the size fits the CURSOR_MAX² RGBA buffer.
|
||||
unsafe {
|
||||
ck(
|
||||
cuMemAlloc_v2(&mut cur_buf, (CURSOR_MAX * CURSOR_MAX * 4) as usize),
|
||||
"cuMemAlloc(cursor)",
|
||||
)?;
|
||||
}
|
||||
Ok(CursorBlend {
|
||||
module,
|
||||
f_argb,
|
||||
f_nv12,
|
||||
f_yuv444,
|
||||
cur_buf,
|
||||
})
|
||||
}
|
||||
|
||||
/// Upload the cursor RGBA (`cw*ch*4`, tight rows) into the device blend buffer. Call only when
|
||||
/// the bitmap changes; position moves are just kernel args.
|
||||
pub fn upload(&self, rgba: &[u8], cw: u32, ch: u32) -> Result<()> {
|
||||
let cw = cw.min(CURSOR_MAX);
|
||||
let ch = ch.min(CURSOR_MAX);
|
||||
let row = cw as usize * 4;
|
||||
let copy = CUDA_MEMCPY2D {
|
||||
srcMemoryType: 1, // HOST
|
||||
srcHost: rgba.as_ptr() as *const c_void,
|
||||
srcPitch: row,
|
||||
dstMemoryType: CU_MEMORYTYPE_DEVICE,
|
||||
dstDevice: self.cur_buf,
|
||||
dstPitch: row,
|
||||
WidthInBytes: row,
|
||||
Height: ch as usize,
|
||||
..Default::default()
|
||||
};
|
||||
// SAFETY: HOST→DEVICE 2D copy of `row*ch` bytes; `rgba` covers at least that (caller passes
|
||||
// `cw*ch*4`), `cur_buf` is the CURSOR_MAX²·4 device alloc (row ≤ CURSOR_MAX·4, ch ≤ CURSOR_MAX).
|
||||
// Synchronous via `copy_blocking`. Requires the context current (caller's contract).
|
||||
unsafe { copy_blocking(©, "cursor HtoD") }
|
||||
}
|
||||
|
||||
/// Blend into a packed 4-byte (NVENC ARGB) owned surface at `(ox,oy)`.
|
||||
#[allow(clippy::too_many_arguments)] // surface geometry + cursor size + offset — a struct would just be unpacked at the call
|
||||
pub fn blend_argb(
|
||||
&self,
|
||||
surf: CUdeviceptr,
|
||||
pitch: usize,
|
||||
w: u32,
|
||||
h: u32,
|
||||
cw: u32,
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_surf, mut a_cur) = (surf, self.cur_buf);
|
||||
let (mut a_pitch, mut a_w, mut a_h) = (pitch as i32, w as i32, h as i32);
|
||||
let (mut a_cw, mut a_ch) = (cw.min(CURSOR_MAX) as i32, ch.min(CURSOR_MAX) as i32);
|
||||
let (mut a_ox, mut a_oy) = (ox, oy);
|
||||
let mut args: [*mut c_void; 9] = [
|
||||
&mut a_surf as *mut _ as *mut c_void,
|
||||
&mut a_pitch as *mut _ as *mut c_void,
|
||||
&mut a_w as *mut _ as *mut c_void,
|
||||
&mut a_h as *mut _ as *mut c_void,
|
||||
&mut a_cur as *mut _ as *mut c_void,
|
||||
&mut a_cw as *mut _ as *mut c_void,
|
||||
&mut a_ch as *mut _ as *mut c_void,
|
||||
&mut a_ox as *mut _ as *mut c_void,
|
||||
&mut a_oy as *mut _ as *mut c_void,
|
||||
];
|
||||
self.launch(self.f_argb, a_cw as u32, a_ch as u32, &mut args, sync)
|
||||
}
|
||||
|
||||
/// Blend into an owned planar YUV444 surface (3 stacked full-res planes) at `(ox,oy)`.
|
||||
#[allow(clippy::too_many_arguments)] // surface geometry + cursor size + offset — a struct would just be unpacked at the call
|
||||
pub fn blend_yuv444(
|
||||
&self,
|
||||
base: CUdeviceptr,
|
||||
pitch: usize,
|
||||
w: u32,
|
||||
h: u32,
|
||||
cw: u32,
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_base, mut a_cur) = (base, self.cur_buf);
|
||||
let (mut a_pitch, mut a_w, mut a_h) = (pitch as i32, w as i32, h as i32);
|
||||
let (mut a_cw, mut a_ch) = (cw.min(CURSOR_MAX) as i32, ch.min(CURSOR_MAX) as i32);
|
||||
let (mut a_ox, mut a_oy) = (ox, oy);
|
||||
let mut args: [*mut c_void; 9] = [
|
||||
&mut a_base as *mut _ as *mut c_void,
|
||||
&mut a_pitch as *mut _ as *mut c_void,
|
||||
&mut a_w as *mut _ as *mut c_void,
|
||||
&mut a_h as *mut _ as *mut c_void,
|
||||
&mut a_cur as *mut _ as *mut c_void,
|
||||
&mut a_cw as *mut _ as *mut c_void,
|
||||
&mut a_ch as *mut _ as *mut c_void,
|
||||
&mut a_ox as *mut _ as *mut c_void,
|
||||
&mut a_oy as *mut _ as *mut c_void,
|
||||
];
|
||||
self.launch(self.f_yuv444, a_cw as u32, a_ch as u32, &mut args, sync)
|
||||
}
|
||||
|
||||
/// Blend into an owned NV12 surface (Y plane at `base`, interleaved UV at `base + pitch*h`).
|
||||
#[allow(clippy::too_many_arguments)] // surface geometry + cursor size + offset — a struct would just be unpacked at the call
|
||||
pub fn blend_nv12(
|
||||
&self,
|
||||
base: CUdeviceptr,
|
||||
pitch: usize,
|
||||
w: u32,
|
||||
h: u32,
|
||||
cw: u32,
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_yb, mut a_uvb, mut a_cur) = (base, base + pitch as u64 * h as u64, self.cur_buf);
|
||||
let (mut a_yp, mut a_uvp) = (pitch as i32, pitch as i32);
|
||||
let (mut a_w, mut a_h) = (w as i32, h as i32);
|
||||
let (mut a_cw, mut a_ch) = (cw.min(CURSOR_MAX) as i32, ch.min(CURSOR_MAX) as i32);
|
||||
let (mut a_ox, mut a_oy) = (ox, oy);
|
||||
let mut args: [*mut c_void; 11] = [
|
||||
&mut a_yb as *mut _ as *mut c_void,
|
||||
&mut a_yp as *mut _ as *mut c_void,
|
||||
&mut a_uvb as *mut _ as *mut c_void,
|
||||
&mut a_uvp as *mut _ as *mut c_void,
|
||||
&mut a_w as *mut _ as *mut c_void,
|
||||
&mut a_h as *mut _ as *mut c_void,
|
||||
&mut a_cur as *mut _ as *mut c_void,
|
||||
&mut a_cw as *mut _ as *mut c_void,
|
||||
&mut a_ch as *mut _ as *mut c_void,
|
||||
&mut a_ox as *mut _ as *mut c_void,
|
||||
&mut a_oy as *mut _ as *mut c_void,
|
||||
];
|
||||
// One thread per 2x2 luma block → grid over ceil(cw/2) × ceil(ch/2).
|
||||
self.launch(
|
||||
self.f_nv12,
|
||||
(a_cw as u32).div_ceil(2),
|
||||
(a_ch as u32).div_ceil(2),
|
||||
&mut args,
|
||||
sync,
|
||||
)
|
||||
}
|
||||
|
||||
/// Launch `f` over a `work_w × work_h` grid (16×16 blocks) on the copy stream; `sync` waits
|
||||
/// for it, `!sync` leaves completion to the stream (stream-ordered consumers only — the
|
||||
/// kernel PARAMETERS are copied at launch time, so the arg locals need not outlive the call).
|
||||
fn launch(
|
||||
&self,
|
||||
f: CUfunction,
|
||||
work_w: u32,
|
||||
work_h: u32,
|
||||
args: &mut [*mut c_void],
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
if work_w == 0 || work_h == 0 {
|
||||
return Ok(());
|
||||
}
|
||||
const B: u32 = 16;
|
||||
let stream = copy_stream();
|
||||
// SAFETY: `f` is a resolved kernel from our loaded module; `args` holds pointers to live
|
||||
// locals whose types match the kernel's C parameters (per the call site above) — CUDA
|
||||
// copies the parameter values during `cuLaunchKernel` itself, so they need not outlive
|
||||
// the call. Grid/block dims are non-zero. Launched on the copy stream (ordered after the
|
||||
// input-surface copy issued on the same stream); `sync` waits, `!sync` leaves ordering to
|
||||
// the stream (the NVENC IO-stream binding). Requires the context current.
|
||||
unsafe {
|
||||
ck(
|
||||
cuLaunchKernel(
|
||||
f,
|
||||
work_w.div_ceil(B),
|
||||
work_h.div_ceil(B),
|
||||
1,
|
||||
B,
|
||||
B,
|
||||
1,
|
||||
0,
|
||||
stream,
|
||||
args.as_mut_ptr(),
|
||||
std::ptr::null_mut(),
|
||||
),
|
||||
"cuLaunchKernel(cursor)",
|
||||
)?;
|
||||
if sync {
|
||||
ck(cuStreamSynchronize(stream), "cuStreamSynchronize(cursor)")?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for CursorBlend {
|
||||
fn drop(&mut self) {
|
||||
// SAFETY: `cur_buf`/`module` are our own handles, freed exactly once here; the context is
|
||||
// current on the encode thread that drops the encoder. Errors are ignored on teardown.
|
||||
unsafe {
|
||||
let _ = cuMemFree_v2(self.cur_buf);
|
||||
let _ = cuModuleUnload(self.module);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Allocate one pitched device buffer for `width`x`height` 4-byte pixels; returns `(ptr, pitch)`.
|
||||
fn alloc_pitched(width: u32, height: u32) -> Result<(CUdeviceptr, usize)> {
|
||||
let mut ptr: CUdeviceptr = 0;
|
||||
|
||||
Reference in New Issue
Block a user