perf(linux/vulkan-encode): true-size headers make native NV12 zero-copy at every mode

Drop the padded-copy staging for producer-native NV12 and direct-import the
visible-size buffer at unaligned modes (1080p) too. Safe by the driver's own
contract rather than by allocation luck: native sessions now author the H265
SPS / AV1 sequence header at the RENDER size and pass the matching codedExtent
on every picture resource. RADV rounds the bitstream SPS up itself (with a
conformance window -- radv_video_patch_encode_session_parameters, per the
VK_KHR_video_encode_h265 proposal's "implementations may override" clause) and
programs the VCN session_init with the true extent plus nonzero firmware
padding, so the hardware edge-extends the alignment rows internally and never
fetches past the source's real extent -- the exact mechanism VAAPI/radeonsi has
always used to encode 1080p. The driver-emitted header NALs already carry the
patched SPS, so the wire format is unchanged.

The CSC and RGB-direct paths keep the app-aligned convention (their sources
genuinely cover the aligned extent); RGB padded-copy staging is untouched.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-21 13:02:09 +02:00
parent 49c6dd00c9
commit 004f0cacb8
+80 -134
View File
@@ -431,11 +431,18 @@ pub struct VulkanVideoEncoder {
/// the CSC; `None` ⇒ the compute-CSC path. Fixed per session (the picture format is baked /// the CSC; `None` ⇒ the compute-CSC path. Fixed per session (the picture format is baked
/// into the video session). /// into the video session).
rgb: Option<RgbDirect>, rgb: Option<RgbDirect>,
/// Producer supplied native NV12 rather than packed RGB. Aligned modes encode the imported /// Producer supplied native NV12 rather than packed RGB. EVERY mode encodes the imported
/// image directly; unaligned modes stage through the per-slot padded NV12 `Frame::pad_img` /// visible-size buffer directly — safely, because native sessions use TRUE-SIZE headers:
/// (transfer copy + edge duplication — see [`Self::nv12_padded`]), NEVER the producer buffer: /// the SPS/sequence header is authored at the render size and every picture resource's
/// the encoder reads the full 64x16-aligned coded extent, and an undersized direct source is /// `codedExtent` matches it, so RADV programs the VCN with `session_init` extent = the true
/// the exact OOB-read class behind the 2026-07-20 field GPU reset. /// size and a nonzero `padding_width/height`, and the FIRMWARE edge-extends the alignment
/// padding internally (radv_video_enc.c `radv_enc_session_init`; the driver also rounds the
/// bitstream SPS up itself and compensates with a conformance window —
/// `radv_video_patch_encode_session_parameters`, per the VK_KHR_video_encode_h265 proposal's
/// "implementations may override" clause). The source is never read past its extent — unlike
/// the app-aligned-SPS convention the CSC/RGB paths use, where the coded extent is 64x16-
/// aligned and an undersized direct source is the OOB-read class behind the 2026-07-20 field
/// GPU reset (those paths keep their aligned-size sources/staging).
native_nv12: bool, native_nv12: bool,
/// One-shot warning latch: a cursor bitmap arrived on an RGB-direct or native-NV12 session /// One-shot warning latch: a cursor bitmap arrived on an RGB-direct or native-NV12 session
@@ -644,24 +651,16 @@ impl VulkanVideoEncoder {
}), }),
_ => None, _ => None,
}; };
// Unaligned modes (1080p!) must NOT hand the visible-size producer buffer to the encoder
// directly — the VCN reads the full 64x16-aligned coded extent, sailing past the
// allocation (the 2026-07-20 field GPU reset). Stage through an aligned padded NV12 copy
// instead, exactly like RGB-direct's padded mode.
let nv12_pad = native_nv12 && !aligned;
if native_nv12 { if native_nv12 {
tracing::info!( tracing::info!(
native_nv12 = if aligned { native_nv12 = "active(direct-import)",
"active(direct-import)"
} else {
"active(padded-copy: mode is not 64x16-aligned — staging copy + edge \
duplication instead of the direct import)"
},
source_width = rw, source_width = rw,
source_height = rh, source_height = rh,
coded_width = w, fw_padding_width = w - rw,
coded_height = h, fw_padding_height = h - rh,
"vulkan-encode: producer-native NV12 encode source" "vulkan-encode: producer-native NV12 encode source (true-size headers: the \
driver aligns the bitstream SPS itself and the firmware edge-extends the \
padding — the source is never read past its extent)"
); );
} else { } else {
tracing::info!( tracing::info!(
@@ -908,13 +907,19 @@ impl VulkanVideoEncoder {
// ---- session parameters + header framing (HEVC: VPS/SPS/PPS on keyframes; AV1: a // ---- session parameters + header framing (HEVC: VPS/SPS/PPS on keyframes; AV1: a
// temporal-delimiter OBU per frame + a sequence-header OBU on keyframes) ---- // temporal-delimiter OBU per frame + a sequence-header OBU on keyframes) ----
// Native NV12 authors TRUE-SIZE headers (SPS/seq at the render size, no app-side
// conformance window): RADV rounds the bitstream SPS up itself and, keyed off the
// matching true-size codedExtent, programs the VCN with firmware padding so the source
// is never read past its extent. The CSC/RGB paths keep the app-aligned convention
// (their sources genuinely cover the aligned extent).
let (hdr_w, hdr_h) = if native_nv12 { (rw, rh) } else { (w, h) };
let (params, header, frame_prefix) = if av1 { let (params, header, frame_prefix) = if av1 {
build_parameters_av1( build_parameters_av1(
&device, &device,
&vq_dev, &vq_dev,
session, session,
w, hdr_w,
h, hdr_h,
rw, rw,
rh, rh,
av1_caps.max_level, av1_caps.max_level,
@@ -927,8 +932,8 @@ impl VulkanVideoEncoder {
&vq_dev, &vq_dev,
&venc_dev, &venc_dev,
session, session,
w, hdr_w,
h, hdr_h,
rw, rw,
rh, rh,
quality_level, quality_level,
@@ -1086,16 +1091,12 @@ impl VulkanVideoEncoder {
sampler, sampler,
ts_period_ns > 0.0 ts_period_ns > 0.0
&& ((rgb_cfg.is_none() && !native_nv12) && ((rgb_cfg.is_none() && !native_nv12)
|| rgb_cfg.as_ref().is_some_and(|c| c.padded) || rgb_cfg.as_ref().is_some_and(|c| c.padded)),
|| nv12_pad),
rgb_cfg.is_none() && !native_nv12, rgb_cfg.is_none() && !native_nv12,
if rgb_cfg.as_ref().is_some_and(|c| c.padded) { rgb_cfg
Some(vk::Format::B8G8R8A8_UNORM) .as_ref()
} else if nv12_pad { .is_some_and(|c| c.padded)
Some(NV12) .then_some(vk::Format::B8G8R8A8_UNORM),
} else {
None
},
guard.frames.last_mut().expect("frame just pushed"), guard.frames.last_mut().expect("frame just pushed"),
)?; )?;
} }
@@ -1298,17 +1299,9 @@ impl VulkanVideoEncoder {
} }
} }
/// The native-NV12 session runs in padded-copy mode: the mode is not 64x16-aligned, so the /// Import a DMA-BUF VkImage with usage/profile matching this session's source mode. Native
/// producer's visible-size buffer cannot be the encode source (same OOB-read class as /// NV12 and aligned RGB-direct are profiled `VIDEO_ENCODE_SRC` images. Padded RGB-direct
/// [`RgbDirect::padded`]) — each frame is transfer-copied into the aligned `Frame::pad_img` /// imports the producer allocation as transfer-source only.
/// with edges duplicated, and encoded from there.
fn nv12_padded(&self) -> bool {
self.native_nv12 && (self.render_w != self.width || self.render_h != self.height)
}
/// Import a DMA-BUF VkImage with usage/profile matching this session's source mode. Aligned
/// native NV12 and aligned RGB-direct are profiled `VIDEO_ENCODE_SRC` images. The padded
/// modes import the producer allocation as transfer-source only.
unsafe fn import_dmabuf( unsafe fn import_dmabuf(
&self, &self,
d: &pf_frame::DmabufFrame, d: &pf_frame::DmabufFrame,
@@ -1316,18 +1309,6 @@ impl VulkanVideoEncoder {
ch: u32, ch: u32,
) -> Result<(vk::Image, vk::DeviceMemory, vk::ImageView)> { ) -> Result<(vk::Image, vk::DeviceMemory, vk::ImageView)> {
if self.native_nv12 { if self.native_nv12 {
if self.nv12_padded() {
return super::vk_util::import_rgb_dmabuf_as(
&self.device,
&self.ext_fd,
&self.mem_props,
d,
cw,
ch,
vk::ImageUsageFlags::TRANSFER_SRC,
None,
);
}
let mut ps = NativeProfileStack::new(codec_op_for(self.codec == Codec::Av1)); let mut ps = NativeProfileStack::new(codec_op_for(self.codec == Codec::Av1));
let profile = *ps.wire(self.codec == Codec::Av1); let profile = *ps.wire(self.codec == Codec::Av1);
let arr = [profile]; let arr = [profile];
@@ -2137,10 +2118,10 @@ impl VulkanVideoEncoder {
Ok(()) Ok(())
} }
/// Producer-native NV12 submit: an aligned mode imports the producer's buffer directly as /// Producer-native NV12 submit: import the producer's visible-size buffer directly as the
/// the encode source (it covers the coded extent exactly); an unaligned mode stages it /// encode source. Safe at every mode because native sessions run true-size headers — the
/// through the padded aligned copy — NEVER the producer buffer, whose visible size the /// picture resources' codedExtent equals the source extent and the VCN firmware edge-extends
/// encoder's coded-extent reads would sail past (see [`Self::nv12_padded`]). /// the alignment padding internally (see the [`Self::native_nv12`] field docs).
#[allow(clippy::too_many_arguments)] #[allow(clippy::too_many_arguments)]
unsafe fn record_submit_nv12( unsafe fn record_submit_nv12(
&mut self, &mut self,
@@ -2197,42 +2178,17 @@ impl VulkanVideoEncoder {
); );
} }
let dev = self.device.clone(); let dev = self.device.clone();
let compute_cmd = self.frames[slot].compute_cmd;
let cmd = self.frames[slot].cmd; let cmd = self.frames[slot].cmd;
let csc_sem = self.frames[slot].csc_sem;
let fence = self.frames[slot].fence; let fence = self.frames[slot].fence;
let query_pool = self.frames[slot].query_pool; let query_pool = self.frames[slot].query_pool;
let bs_buf = self.frames[slot].bs_buf; let bs_buf = self.frames[slot].bs_buf;
let ts_pool = self.frames[slot].ts_pool; // The frame-size check above proved the buffer covers the (true-size) coded extent —
let (src_img, src_view, acquire, compute_active) = if self.nv12_padded() { // the direct import is the encode source at every mode.
let (img, _view, fresh) = self.import_cached(d, frame.width, frame.height)?; let (src_img, src_view, fresh) = self.import_cached(d, frame.width, frame.height)?;
let pad_img = self.frames[slot].pad_img; let acquire = if fresh {
let pad_view = self.frames[slot].pad_view; SrcAcquire::DmabufFresh
self.record_pad_blit(
&dev,
compute_cmd,
img,
fresh,
pad_img,
ts_pool,
&[
(vk::ImageAspectFlags::PLANE_0, 1),
(vk::ImageAspectFlags::PLANE_1, 2),
],
)?;
// The staging image ends the blit in GENERAL with the csc_sem ordering the hand-off
// — exactly the CscGeneral acquire contract.
(pad_img, pad_view, SrcAcquire::CscGeneral, true)
} else { } else {
// Aligned mode: render == coded, so the frame-size check above already proved the SrcAcquire::DmabufCached
// buffer covers the full coded extent — the direct import is safe.
let (img, view, fresh) = self.import_cached(d, frame.width, frame.height)?;
let acq = if fresh {
SrcAcquire::DmabufFresh
} else {
SrcAcquire::DmabufCached
};
(img, view, acq, false)
}; };
if self.codec == Codec::Av1 { if self.codec == Codec::Av1 {
self.record_coding_av1( self.record_coding_av1(
@@ -2246,34 +2202,13 @@ impl VulkanVideoEncoder {
)?; )?;
} }
dev.reset_fences(&[fence])?; dev.reset_fences(&[fence])?;
// The whole frame is one submit: the encoder reads the imported NV12 directly.
let ecmds = [cmd]; let ecmds = [cmd];
if compute_active { dev.queue_submit(
let ccmds = [compute_cmd]; self.encode_queue,
let sems = [csc_sem]; &[vk::SubmitInfo::default().command_buffers(&ecmds)],
dev.queue_submit( fence,
self.compute_queue, )?;
&[vk::SubmitInfo::default()
.command_buffers(&ccmds)
.signal_semaphores(&sems)],
vk::Fence::null(),
)?;
let wait_stages = [vk::PipelineStageFlags::ALL_COMMANDS];
dev.queue_submit(
self.encode_queue,
&[vk::SubmitInfo::default()
.command_buffers(&ecmds)
.wait_semaphores(&sems)
.wait_dst_stage_mask(&wait_stages)],
fence,
)?;
} else {
// The whole frame is one submit: the encoder reads the imported NV12 directly.
dev.queue_submit(
self.encode_queue,
&[vk::SubmitInfo::default().command_buffers(&ecmds)],
fence,
)?;
}
Ok(()) Ok(())
} }
@@ -2612,11 +2547,20 @@ impl VulkanVideoEncoder {
poc: i32, poc: i32,
) -> Result<()> { ) -> Result<()> {
use ash::vk::native as h; use ash::vk::native as h;
// Setup/reference AND source share the aligned coded extent: every encode source — CSC // Setup/reference AND source share ONE coded extent. App-aligned sessions (CSC, RGB)
// output, direct import (aligned mode only), padded staging — covers it by construction. // use the aligned size — their sources cover it by construction. Native NV12 runs
let ext2d = vk::Extent2D { // true-size headers: the render-size extent matches the SPS/sequence header, and RADV
width: self.width, // keys the firmware's internal alignment padding off it (see `native_nv12`).
height: self.height, let ext2d = if self.native_nv12 {
vk::Extent2D {
width: self.render_w,
height: self.render_h,
}
} else {
vk::Extent2D {
width: self.width,
height: self.height,
}
}; };
let ref_poc = if is_idr { 0 } else { self.slot_poc[ref_slot] }; let ref_poc = if is_idr { 0 } else { self.slot_poc[ref_slot] };
@@ -2832,11 +2776,20 @@ impl VulkanVideoEncoder {
) -> Result<()> { ) -> Result<()> {
use super::vk_av1_encode as av1; use super::vk_av1_encode as av1;
use ash::vk::native as h; use ash::vk::native as h;
// Setup/reference AND source share the aligned coded extent: every encode source — CSC // Setup/reference AND source share ONE coded extent. App-aligned sessions (CSC, RGB)
// output, direct import (aligned mode only), padded staging — covers it by construction. // use the aligned size — their sources cover it by construction. Native NV12 runs
let ext2d = vk::Extent2D { // true-size headers: the render-size extent matches the SPS/sequence header, and RADV
width: self.width, // keys the firmware's internal alignment padding off it (see `native_nv12`).
height: self.height, let ext2d = if self.native_nv12 {
vk::Extent2D {
width: self.render_w,
height: self.render_h,
}
} else {
vk::Extent2D {
width: self.width,
height: self.height,
}
}; };
// ---- required AV1 frame sub-structs (single tile; no CDEF/LR/segmentation/global-motion) ---- // ---- required AV1 frame sub-structs (single tile; no CDEF/LR/segmentation/global-motion) ----
@@ -3144,13 +3097,6 @@ impl VulkanVideoEncoder {
remaining fence wait still includes queue synchronization + RGB→YUV EFC \ remaining fence wait still includes queue synchronization + RGB→YUV EFC \
+ video encode" + video encode"
); );
} else if self.nv12_padded() {
tracing::info!(
nv12_copy_us = format!("{pre_encode_us:.0}"),
au_bytes = len,
"vulkan-encode split (sampled): padded NV12 staging-copy device time; \
remaining fence wait still includes queue synchronization + video encode"
);
} else { } else {
tracing::info!( tracing::info!(
csc_us = format!("{pre_encode_us:.0}"), csc_us = format!("{pre_encode_us:.0}"),