Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4bc7eecf05 |
@@ -819,77 +819,46 @@ impl DriverAttach {
|
||||
|
||||
/// One-shot WARN with everything the host can find out about WHY the driver isn't attached:
|
||||
/// driver-store presence, the devnode's PnP status/problem code, and where to look next.
|
||||
///
|
||||
/// Runs on its own thread and returns immediately. The caller is the session's pad service
|
||||
/// thread — the one feeding input and rumble — and everything below is slow: the driver-store
|
||||
/// check waits up to [`INVENTORY_WAIT`] for a `pnputil` enumeration that can take tens of
|
||||
/// seconds, and the devnode lookup is a synchronous PnP call. Blocking there stalled input for
|
||||
/// up to two seconds *per unattached pad* (the wait is a deadline, not a one-off: while the
|
||||
/// enumeration is still outstanding every pad pays it again), at exactly the moment a session
|
||||
/// is already going wrong. Diagnostics must never be able to hurt the thing they diagnose.
|
||||
///
|
||||
/// Off the hot path the wait also stops being a compromise — it can afford to be patient and
|
||||
/// report what it actually found rather than "still enumerating".
|
||||
fn diagnose(&self) {
|
||||
let (driver, inf, driver_log) = (self.driver, self.inf, self.driver_log);
|
||||
let shm_name = self.shm_name.clone();
|
||||
let instance_id = self.instance_id.clone();
|
||||
std::thread::Builder::new()
|
||||
.name("pf-driver-diagnose".into())
|
||||
.spawn(move || diagnose_blocking(driver, inf, driver_log, &shm_name, instance_id))
|
||||
.ok();
|
||||
let store = match driver_store_has(self.inf) {
|
||||
Some(true) => "driver package present in the driver store",
|
||||
Some(false) => {
|
||||
"driver package NOT in the driver store — run: punktfunk-host.exe driver install --gamepad"
|
||||
}
|
||||
None => "driver store could not be queried (pnputil failed or still enumerating)",
|
||||
};
|
||||
let devnode = match &self.instance_id {
|
||||
Some(id) => devnode_status_line(id),
|
||||
None => {
|
||||
"no per-session devnode (SwDeviceCreate failed earlier — see the warning above)"
|
||||
.to_string()
|
||||
}
|
||||
};
|
||||
tracing::warn!(
|
||||
driver = self.driver,
|
||||
shm = %self.shm_name,
|
||||
grace_secs = ATTACH_GRACE.as_secs(),
|
||||
store,
|
||||
devnode = %devnode,
|
||||
driver_log = self.driver_log,
|
||||
"gamepad driver has not attached to the shared section — the virtual pad exists but no \
|
||||
driver is serving it (games will not see it); an old (pre-sealed-channel) driver also \
|
||||
reads as not-attached: update with punktfunk-host.exe driver install --gamepad \
|
||||
(driver_log is only written by debug driver builds, or with the PFXUSB_DEBUG_LOG / \
|
||||
PFGAMEPAD_DEBUG_LOG / PFMOUSE_DEBUG_LOG system env var set + the device restarted)"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The body of [`DriverAttach::diagnose`], on its own thread. Split out rather than inlined into
|
||||
/// the closure so the blocking calls stay visible as blocking.
|
||||
fn diagnose_blocking(
|
||||
driver: &'static str,
|
||||
inf: &'static str,
|
||||
driver_log: &'static str,
|
||||
shm_name: &str,
|
||||
instance_id: Option<String>,
|
||||
) {
|
||||
let store = match driver_store_has(inf) {
|
||||
Some(true) => "driver package present in the driver store",
|
||||
Some(false) => {
|
||||
"driver package NOT in the driver store — run: punktfunk-host.exe driver install --gamepad"
|
||||
}
|
||||
None => "driver store could not be queried (pnputil failed or still enumerating)",
|
||||
};
|
||||
let devnode = match &instance_id {
|
||||
Some(id) => devnode_status_line(id),
|
||||
None => "no per-session devnode (SwDeviceCreate failed earlier — see the warning above)"
|
||||
.to_string(),
|
||||
};
|
||||
tracing::warn!(
|
||||
driver,
|
||||
shm = %shm_name,
|
||||
grace_secs = ATTACH_GRACE.as_secs(),
|
||||
store,
|
||||
devnode = %devnode,
|
||||
driver_log,
|
||||
"gamepad driver has not attached to the shared section — the virtual pad exists but no \
|
||||
driver is serving it (games will not see it); an old (pre-sealed-channel) driver also \
|
||||
reads as not-attached: update with punktfunk-host.exe driver install --gamepad \
|
||||
(driver_log is only written by debug driver builds, or with the PFXUSB_DEBUG_LOG / \
|
||||
PFGAMEPAD_DEBUG_LOG / PFMOUSE_DEBUG_LOG system env var set + the device restarted)"
|
||||
);
|
||||
}
|
||||
|
||||
/// How long [`driver_store_inventory`] waits for the background pnputil query before reporting
|
||||
/// without it. Only [`diagnose_blocking`] waits, and that has a thread to itself, so this is
|
||||
/// generous: pnputil routinely takes longer than a couple of seconds on a busy driver store, and
|
||||
/// the old two-second budget — chosen to limit the damage while this ran on the pad service thread
|
||||
/// — meant the diagnosis usually gave up and printed "still enumerating", which is the one answer
|
||||
/// that helps nobody. Nothing waits on this thread, so patience costs only a late log line.
|
||||
const INVENTORY_WAIT: Duration = Duration::from_secs(30);
|
||||
/// How long [`driver_store_inventory`] lets the caller wait for the background pnputil query
|
||||
/// before reporting without it — [`observe`] runs on the pad service thread, which must keep
|
||||
/// draining pad slots even when the driver store is wedged.
|
||||
const INVENTORY_WAIT: Duration = Duration::from_secs(2);
|
||||
|
||||
/// Driver-store inventory (`pnputil /enum-drivers`), lower-cased, fetched once per process — only
|
||||
/// consulted on the failure path, so the subprocess cost never hits a healthy session. The query
|
||||
/// runs on its OWN thread: pnputil can block for tens of seconds on a busy/wedged driver store,
|
||||
/// and this keeps one wedged query from being re-run per pad. `None` = not available yet (query
|
||||
/// still running past [`INVENTORY_WAIT`]) or
|
||||
/// and the caller is the pad service thread. `None` = not available yet (query still running) or
|
||||
/// failed; a query that outlives [`INVENTORY_WAIT`] still lands in the cache for later reports.
|
||||
fn driver_store_inventory() -> Option<&'static str> {
|
||||
static INV: OnceLock<String> = OnceLock::new();
|
||||
|
||||
@@ -341,6 +341,50 @@ pub fn mtu1500_shard_payload_for(peer: core::net::IpAddr) -> usize {
|
||||
}
|
||||
}
|
||||
|
||||
/// Floor for a negotiated `shard_payload` (even, well under every real path). A path whose UDP
|
||||
/// budget lands below this can't carry the QUIC control plane either (QUIC's own minimum is a
|
||||
/// 1200-byte UDP payload), so shrinking video shards further buys nothing — the clamp helpers
|
||||
/// bottom out here instead of producing degenerate confetti-sized shards.
|
||||
pub const MIN_SHARD_PAYLOAD: usize = 512;
|
||||
|
||||
/// The sealed wire size of a video datagram carrying `shard_payload` bytes of shard — what
|
||||
/// actually leaves the socket as UDP payload (punktfunk header + shard + crypto overhead).
|
||||
pub const fn sealed_datagram_bytes(shard_payload: usize) -> usize {
|
||||
HEADER_LEN + shard_payload + CRYPTO_OVERHEAD
|
||||
}
|
||||
|
||||
/// The UDP-payload size a path must carry for full-size IPv4 video datagrams: the sealed size
|
||||
/// of the [`mtu1500_shard_payload`] default (= 1472, the exact 1500-MTU IPv4 ceiling). Doubles
|
||||
/// as the QUIC MTU-discovery probe ceiling (`quic/endpoint.rs`): with the ceiling set to
|
||||
/// exactly this value, a control connection whose discovery settles AT the ceiling has proven
|
||||
/// the path carries full-size video datagrams, and one that settles BELOW it has proven the
|
||||
/// path cannot — a discrimination quinn's stock 1452 ceiling can't make in either direction.
|
||||
pub const fn video_datagram_udp_ceiling() -> usize {
|
||||
sealed_datagram_bytes(mtu1500_shard_payload())
|
||||
}
|
||||
|
||||
/// Largest even shard payload whose sealed datagram fits in `udp_budget` bytes of UDP payload
|
||||
/// (the quantity QUIC MTU discovery measures — [`video_datagram_udp_ceiling`] is its probe
|
||||
/// ceiling). Clamped to the peer's family default ([`mtu1500_shard_payload_for`]) so a generous
|
||||
/// budget never grows packets past today's wire, and floored at [`MIN_SHARD_PAYLOAD`].
|
||||
pub fn shard_payload_for_udp_budget(udp_budget: usize, peer: core::net::IpAddr) -> usize {
|
||||
let p = udp_budget.saturating_sub(HEADER_LEN + CRYPTO_OVERHEAD);
|
||||
let p = p - p % 2; // FEC requires even shards
|
||||
p.clamp(MIN_SHARD_PAYLOAD, mtu1500_shard_payload_for(peer))
|
||||
}
|
||||
|
||||
/// [`shard_payload_for_udp_budget`] for an operator-supplied ON-WIRE IP MTU (the number
|
||||
/// `netsh interface ipv4 show subinterfaces` / `ip link` shows): subtracts the family's IP+UDP
|
||||
/// headers first — 28 for IPv4 (and IPv4-mapped), 48 for IPv6.
|
||||
pub fn shard_payload_for_wire_mtu(wire_mtu: usize, peer: core::net::IpAddr) -> usize {
|
||||
let ip_udp = match peer {
|
||||
core::net::IpAddr::V4(_) => 28,
|
||||
core::net::IpAddr::V6(v6) if v6.to_ipv4_mapped().is_some() => 28,
|
||||
core::net::IpAddr::V6(_) => 48,
|
||||
};
|
||||
shard_payload_for_udp_budget(wire_mtu.saturating_sub(ip_udp), peer)
|
||||
}
|
||||
|
||||
/// Everything needed to construct a [`Session`](crate::session::Session).
|
||||
///
|
||||
/// `Debug` is implemented by hand to redact `key`/`salt`, and `key`/`salt` are zeroized
|
||||
@@ -514,6 +558,74 @@ mod tests {
|
||||
assert!(HEADER_LEN + (p + 2) + CRYPTO_OVERHEAD > 1452, "not maximal");
|
||||
}
|
||||
|
||||
/// The video-datagram ceiling IS the exact v4 sealed size — the QUIC MTU-discovery probe
|
||||
/// ceiling (endpoint.rs) relies on this equality for its settled-at-vs-below verdict.
|
||||
#[test]
|
||||
fn video_datagram_ceiling_is_the_sealed_default() {
|
||||
assert_eq!(
|
||||
video_datagram_udp_ceiling(),
|
||||
HEADER_LEN + mtu1500_shard_payload() + CRYPTO_OVERHEAD
|
||||
);
|
||||
assert_eq!(video_datagram_udp_ceiling(), 1472);
|
||||
}
|
||||
|
||||
/// Budget-derived sizing: even, sealed-fits-the-budget, clamped to the family default
|
||||
/// above and [`MIN_SHARD_PAYLOAD`] below.
|
||||
#[test]
|
||||
fn shard_payload_for_udp_budget_math() {
|
||||
use core::net::IpAddr;
|
||||
let v4: IpAddr = "192.168.1.50".parse().unwrap();
|
||||
let v6: IpAddr = "fd00::50".parse().unwrap();
|
||||
// The full ceiling reproduces the default exactly.
|
||||
assert_eq!(
|
||||
shard_payload_for_udp_budget(video_datagram_udp_ceiling(), v4),
|
||||
mtu1500_shard_payload()
|
||||
);
|
||||
// A WARP/Tailscale-shaped 1280 budget: sealed result must fit the budget, stay even.
|
||||
let p = shard_payload_for_udp_budget(1280, v4);
|
||||
assert_eq!(p % 2, 0);
|
||||
assert!(sealed_datagram_bytes(p) <= 1280);
|
||||
assert!(sealed_datagram_bytes(p + 2) > 1280, "not maximal");
|
||||
// Odd budgets round down to even shards.
|
||||
assert_eq!(shard_payload_for_udp_budget(1281, v4) % 2, 0);
|
||||
// A generous budget never grows past the family default (either family).
|
||||
assert_eq!(
|
||||
shard_payload_for_udp_budget(9000, v4),
|
||||
mtu1500_shard_payload()
|
||||
);
|
||||
assert_eq!(
|
||||
shard_payload_for_udp_budget(9000, v6),
|
||||
mtu1500_shard_payload_v6()
|
||||
);
|
||||
// Degenerate budgets bottom out at the floor instead of confetti.
|
||||
assert_eq!(shard_payload_for_udp_budget(100, v4), MIN_SHARD_PAYLOAD);
|
||||
}
|
||||
|
||||
/// Operator-facing wire-MTU sizing subtracts the right IP+UDP header per family, and 1500
|
||||
/// reproduces today's defaults exactly.
|
||||
#[test]
|
||||
fn shard_payload_for_wire_mtu_math() {
|
||||
use core::net::IpAddr;
|
||||
let v4: IpAddr = "192.168.1.50".parse().unwrap();
|
||||
let v6: IpAddr = "fd00::50".parse().unwrap();
|
||||
let mapped: IpAddr = "::ffff:192.168.1.50".parse().unwrap();
|
||||
assert_eq!(
|
||||
shard_payload_for_wire_mtu(1500, v4),
|
||||
mtu1500_shard_payload()
|
||||
);
|
||||
assert_eq!(
|
||||
shard_payload_for_wire_mtu(1500, mapped),
|
||||
mtu1500_shard_payload()
|
||||
);
|
||||
assert_eq!(
|
||||
shard_payload_for_wire_mtu(1500, v6),
|
||||
mtu1500_shard_payload_v6()
|
||||
);
|
||||
// 1280 wire − 28 − 64 = 1188 (v4); − 48 − 64 = 1168 (v6).
|
||||
assert_eq!(shard_payload_for_wire_mtu(1280, v4), 1188);
|
||||
assert_eq!(shard_payload_for_wire_mtu(1280, v6), 1168);
|
||||
}
|
||||
|
||||
/// Family selection: genuine v6 remotes get the v6 size; v4 — including the IPv4-mapped v6
|
||||
/// form a dual-stack `[::]` socket reports for a v4 client — keeps the v4 size.
|
||||
#[test]
|
||||
|
||||
@@ -47,6 +47,20 @@ fn stream_transport_idle(idle: std::time::Duration) -> Arc<quinn::TransportConfi
|
||||
// plane latest-wins at the source — ~200 ms of stereo Opus (proportionally less at
|
||||
// surround bitrates), so sustained congestion costs concealable drops, never lag.
|
||||
t.datagram_send_buffer_size(4 * 1024);
|
||||
// MTU discovery probes up to EXACTLY the sealed size of a full IPv4 video datagram (1472)
|
||||
// instead of quinn's stock 1452. Two reasons: (a) on a clean 1500-MTU path QUIC gets the
|
||||
// last 20 bytes per packet; (b) the ceiling turns discovery into a video-path verdict the
|
||||
// host's wire-MTU watcher reads (`punktfunk-host` `native/wire_mtu.rs`) — settled == ceiling
|
||||
// proves the path carries full-size video datagrams, settled BELOW it proves it cannot (a
|
||||
// VPN/overlay adapter at MTU ~1280 blackholes every video packet while all the small flows
|
||||
// pass: the "connects fine, black screen forever" field shape). With the stock 1452 ceiling
|
||||
// a healthy path and a constrained one are indistinguishable at the top. This is the ONLY
|
||||
// behavioral change on healthy paths, and it's confined to discovery: probes are padded
|
||||
// PINGs quinn already expects to lose above a constrained hop — a lost probe settles the
|
||||
// search lower, exactly as it did before.
|
||||
let mut mtud = quinn::MtuDiscoveryConfig::default();
|
||||
mtud.upper_bound(crate::config::video_datagram_udp_ceiling() as u16);
|
||||
t.mtu_discovery_config(Some(mtud));
|
||||
Arc::new(t)
|
||||
}
|
||||
|
||||
|
||||
@@ -26,9 +26,7 @@
|
||||
#![deny(clippy::undocumented_unsafe_blocks)]
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use punktfunk_core::config::{
|
||||
mtu1500_shard_payload_for, CompositorPref, FecConfig, FecScheme, GamepadPref, Role,
|
||||
};
|
||||
use punktfunk_core::config::{CompositorPref, FecConfig, FecScheme, GamepadPref, Role};
|
||||
use punktfunk_core::input::{InputEvent, InputKind};
|
||||
use punktfunk_core::packet::{FLAG_PIC, FLAG_PROBE, FLAG_SOF};
|
||||
use punktfunk_core::quic::{
|
||||
@@ -72,6 +70,9 @@ use input::{input_thread, ClientInput};
|
||||
/// The Hello→Welcome→Start negotiation (plan §W1); `serve_session` calls `handshake::negotiate`
|
||||
/// after the pairing gate.
|
||||
mod handshake;
|
||||
/// MTU resilience for the video data plane: `PUNKTFUNK_WIRE_MTU` override, the per-session
|
||||
/// path-MTU watch on the control connection, and the per-peer learned shard-payload clamp.
|
||||
mod wire_mtu;
|
||||
|
||||
/// The mid-stream control task (plan §W1); `serve_session` spawns `control::run` after the
|
||||
/// handshake to multiplex renegotiation / speed-test control messages onto the data-plane channels.
|
||||
|
||||
@@ -491,7 +491,12 @@ pub(super) async fn negotiate(
|
||||
// per-datagram loss on Wi-Fi — the "100 Mbps badly fails on the phone" root cause.
|
||||
// Negotiated, so the client follows. Jumbo (≈8900) is a future negotiated bump (needs
|
||||
// MAX_DATAGRAM_BYTES raised + end-to-end 9000 MTU).
|
||||
shard_payload: mtu1500_shard_payload_for(peer.ip()) as u16,
|
||||
// Resolution order (wire_mtu.rs): `PUNKTFUNK_WIRE_MTU` operator override, then a path
|
||||
// budget learned from a prior session whose QUIC MTU discovery settled below the
|
||||
// video-datagram ceiling (the "VPN on the host blackholes every video packet" field
|
||||
// shape — small flows pass, the stream is an endless black screen), then this family
|
||||
// default. Healthy paths take the default branch and are byte-identical to before.
|
||||
shard_payload: wire_mtu::negotiated_shard_payload(peer.ip()) as u16,
|
||||
encrypt: true,
|
||||
key,
|
||||
salt,
|
||||
@@ -658,6 +663,10 @@ pub(super) async fn negotiate(
|
||||
let start =
|
||||
Start::decode(&io::read_msg(recv).await?).map_err(|e| anyhow!("Start decode: {e:?}"))?;
|
||||
bringup.mark("start");
|
||||
// The session is real: watch this connection's MTU discovery settle and turn it into a
|
||||
// path verdict (WARN + learned clamp for the next session on a constrained path; clears a
|
||||
// stale clamp on a healthy one). Bounded ~10 s task, ends by itself.
|
||||
wire_mtu::spawn_watch(conn.clone(), welcome.shard_payload as usize);
|
||||
Ok::<_, anyhow::Error>((
|
||||
hello,
|
||||
welcome,
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
//! MTU resilience for the video data plane (the "connects fine, black screen forever" field
|
||||
//! shape).
|
||||
//!
|
||||
//! Video datagrams are sealed at a per-session `shard_payload` sized for a clean 1500-byte MTU
|
||||
//! (1472-byte UDP payloads). A host whose route to the client runs through a smaller-MTU hop —
|
||||
//! a VPN/overlay adapter (Tailscale/WARP/ZeroTier default to 1280) claiming the LAN route, or a
|
||||
//! lowered NIC MTU — delivers every SMALL flow (QUIC control, hole punch, input, audio) while
|
||||
//! 100 % of video datagrams die by fragmentation or local `WSAEMSGSIZE`: the client sits on a
|
||||
//! black screen reporting `loss_ppm=0` (it can't see gaps in packets it never saw any of) and
|
||||
//! the host streams into the void with every gauge green. Neither side observes the failure
|
||||
//! directly — but the control connection CAN: its MTU discovery probes up to exactly the sealed
|
||||
//! video-datagram size ([`video_datagram_udp_ceiling`], set in `quic/endpoint.rs`), so its
|
||||
//! settled MTU is a verdict on the path.
|
||||
//!
|
||||
//! Three legs, none of which changes a session on a healthy path:
|
||||
//! - **`PUNKTFUNK_WIRE_MTU=<bytes>`** — operator override; the shard payload is derived from
|
||||
//! the given on-wire IP MTU. Wire-compatible with every deployed client:
|
||||
//! `Welcome::shard_payload` is already negotiated per session (the v4/v6 split ships two
|
||||
//! values today) and clients follow the negotiated value.
|
||||
//! - **Watch** — a per-session task samples the control connection's discovered MTU once the
|
||||
//! search has had time to finish. A connection still alive that settled BELOW the ceiling is
|
||||
//! proof the path can't carry full-size video: log an actionable WARN and record the measured
|
||||
//! budget for the peer.
|
||||
//! - **Heal** — the next handshake from that peer clamps `shard_payload` to the recorded
|
||||
//! budget, so a reconnect fixes the stream. A later session that reaches the ceiling erases
|
||||
//! the record (the learn/heal loop is self-correcting in both directions).
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::net::IpAddr;
|
||||
use std::sync::{Mutex, OnceLock};
|
||||
|
||||
use punktfunk_core::config::{
|
||||
mtu1500_shard_payload_for, sealed_datagram_bytes, shard_payload_for_udp_budget,
|
||||
shard_payload_for_wire_mtu, video_datagram_udp_ceiling,
|
||||
};
|
||||
|
||||
/// Measured UDP-payload budget per peer IP, learned from live control connections whose MTU
|
||||
/// discovery settled below the video-datagram ceiling. In-memory only: a host restart
|
||||
/// re-learns in one session, and entries self-correct (a later ceiling-hit erases, a lower
|
||||
/// re-measure overwrites).
|
||||
fn learned() -> &'static Mutex<HashMap<IpAddr, u16>> {
|
||||
static LEARNED: OnceLock<Mutex<HashMap<IpAddr, u16>>> = OnceLock::new();
|
||||
LEARNED.get_or_init(|| Mutex::new(HashMap::new()))
|
||||
}
|
||||
|
||||
/// The shard payload for a new session to `peer`: `PUNKTFUNK_WIRE_MTU` override, else the
|
||||
/// peer's learned path budget, else the family default (today's exact behavior). Logs whenever
|
||||
/// the result differs from the default.
|
||||
pub(super) fn negotiated_shard_payload(peer: IpAddr) -> usize {
|
||||
let env = match std::env::var("PUNKTFUNK_WIRE_MTU") {
|
||||
Ok(v) => match v.trim().parse::<usize>() {
|
||||
Ok(mtu) => Some(mtu),
|
||||
Err(_) => {
|
||||
tracing::warn!(value = %v, "PUNKTFUNK_WIRE_MTU is not a number — ignoring it");
|
||||
None
|
||||
}
|
||||
},
|
||||
Err(_) => None,
|
||||
};
|
||||
let learned_budget = learned().lock().unwrap().get(&peer).copied();
|
||||
resolve(env, learned_budget, peer)
|
||||
}
|
||||
|
||||
/// Pure resolution (env override > learned budget > family default) — the tested core of
|
||||
/// [`negotiated_shard_payload`].
|
||||
fn resolve(env_wire_mtu: Option<usize>, learned_udp_budget: Option<u16>, peer: IpAddr) -> usize {
|
||||
let default = mtu1500_shard_payload_for(peer);
|
||||
if let Some(mtu) = env_wire_mtu {
|
||||
let p = shard_payload_for_wire_mtu(mtu, peer);
|
||||
if p != default {
|
||||
tracing::info!(
|
||||
wire_mtu = mtu,
|
||||
shard_payload = p,
|
||||
default,
|
||||
"wire MTU: shard payload set from PUNKTFUNK_WIRE_MTU"
|
||||
);
|
||||
}
|
||||
return p;
|
||||
}
|
||||
if let Some(budget) = learned_udp_budget {
|
||||
let p = shard_payload_for_udp_budget(budget as usize, peer);
|
||||
if p != default {
|
||||
tracing::info!(
|
||||
peer = %peer,
|
||||
udp_budget = budget,
|
||||
shard_payload = p,
|
||||
default,
|
||||
"wire MTU: shard payload clamped to this peer's measured path MTU (learned \
|
||||
from a prior session's QUIC MTU discovery) — video datagrams now fit the \
|
||||
constrained hop"
|
||||
);
|
||||
return p;
|
||||
}
|
||||
}
|
||||
default
|
||||
}
|
||||
|
||||
/// Sample the control connection's discovered MTU after the search has settled and turn it
|
||||
/// into a verdict. Spawned once per negotiated session; the task ends by itself after the
|
||||
/// final sample (bounded ~10 s lifetime, holding only a cheap `Connection` handle).
|
||||
pub(super) fn spawn_watch(conn: quinn::Connection, session_shard_payload: usize) {
|
||||
tokio::spawn(async move {
|
||||
let peer = conn.remote_address().ip();
|
||||
let ceiling = video_datagram_udp_ceiling() as u16;
|
||||
// Discovery finishes in a handful of RTTs on a LAN (well under the first sample) but
|
||||
// needs a loss timeout per failed probe on a constrained path — the second sample
|
||||
// covers that with margin. Max, because discovery only ever raises `current_mtu`.
|
||||
let mut settled = 0u16;
|
||||
for wait_s in [3u64, 7] {
|
||||
tokio::time::sleep(std::time::Duration::from_secs(wait_s)).await;
|
||||
settled = settled.max(conn.stats().path.current_mtu);
|
||||
if settled >= ceiling {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if settled >= ceiling {
|
||||
// The path carries full-size video datagrams — erase any stale learned clamp so
|
||||
// the next session returns to the default wire.
|
||||
if learned().lock().unwrap().remove(&peer).is_some() {
|
||||
tracing::info!(peer = %peer,
|
||||
"wire MTU: path re-measured at full size — learned clamp cleared");
|
||||
}
|
||||
return;
|
||||
}
|
||||
// A closed connection stops discovering, so a session that ended before the final
|
||||
// sample proves nothing (a healthy high-RTT path could still be mid-search): learn
|
||||
// only from a connection that stayed alive through the whole window.
|
||||
if conn.close_reason().is_some() {
|
||||
return;
|
||||
}
|
||||
learned().lock().unwrap().insert(peer, settled);
|
||||
if sealed_datagram_bytes(session_shard_payload) <= settled as usize {
|
||||
// This session was already clamped small enough — the path is still constrained
|
||||
// (keep the record fresh) but video fits, so no alarm.
|
||||
tracing::info!(peer = %peer, discovered_udp_mtu = settled,
|
||||
"wire MTU: constrained path re-measured; this session's video is sized to fit");
|
||||
} else {
|
||||
tracing::warn!(
|
||||
peer = %peer,
|
||||
discovered_udp_mtu = settled,
|
||||
needed_udp_mtu = ceiling,
|
||||
"wire MTU: this path CANNOT carry full-size video datagrams — the control \
|
||||
plane works but every video packet is oversized for a hop, which streams as \
|
||||
an endless black screen with zero reported loss. Typical cause: a VPN/overlay \
|
||||
adapter (Tailscale / Cloudflare WARP / ZeroTier) claiming the LAN route, or a \
|
||||
lowered NIC MTU — compare `ping <client> -f -l 1450` vs `-l 1200` and check \
|
||||
`netsh interface ipv4 show subinterfaces` (Windows) / `ip link` (Linux). The \
|
||||
measured budget is recorded: the NEXT session from this client sizes video to \
|
||||
fit automatically. To pin it for all sessions set PUNKTFUNK_WIRE_MTU."
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::net::{IpAddr, Ipv4Addr, Ipv6Addr};
|
||||
|
||||
const V4: IpAddr = IpAddr::V4(Ipv4Addr::new(192, 168, 1, 2));
|
||||
const V6: IpAddr = IpAddr::V6(Ipv6Addr::new(0x2001, 0xdb8, 0, 0, 0, 0, 0, 1));
|
||||
|
||||
#[test]
|
||||
fn default_when_nothing_known() {
|
||||
assert_eq!(resolve(None, None, V4), mtu1500_shard_payload_for(V4));
|
||||
assert_eq!(resolve(None, None, V6), mtu1500_shard_payload_for(V6));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn env_override_beats_learned() {
|
||||
// 1280 wire − 28 IP/UDP − 64 header/crypto = 1188.
|
||||
assert_eq!(resolve(Some(1280), Some(1472), V4), 1188);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn learned_budget_clamps() {
|
||||
// A WARP-shaped path: 1280-byte UDP budget → 1280 − 64 = 1216.
|
||||
assert_eq!(resolve(None, Some(1280), V4), 1216);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn learned_at_or_above_ceiling_is_the_default_wire() {
|
||||
assert_eq!(resolve(None, Some(1472), V4), mtu1500_shard_payload_for(V4));
|
||||
assert_eq!(resolve(None, Some(2000), V4), mtu1500_shard_payload_for(V4));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn env_full_mtu_is_the_default_wire_both_families() {
|
||||
assert_eq!(resolve(Some(1500), None, V4), mtu1500_shard_payload_for(V4));
|
||||
assert_eq!(resolve(Some(1500), None, V6), mtu1500_shard_payload_for(V6));
|
||||
}
|
||||
}
|
||||
@@ -333,6 +333,12 @@
|
||||
#define INBOUND_REQ_FLAG 2147483648
|
||||
#endif
|
||||
|
||||
// Floor for a negotiated `shard_payload` (even, well under every real path). A path whose UDP
|
||||
// budget lands below this can't carry the QUIC control plane either (QUIC's own minimum is a
|
||||
// 1200-byte UDP payload), so shrinking video shards further buys nothing — the clamp helpers
|
||||
// bottom out here instead of producing degenerate confetti-sized shards.
|
||||
#define MIN_SHARD_PAYLOAD 512
|
||||
|
||||
// 16-byte AEAD authentication tag appended by either session cipher.
|
||||
#define TAG_LEN 16
|
||||
|
||||
|
||||
@@ -360,32 +360,9 @@ fn ring_len(view: &pf_umdf_util::section::MappedView) -> u32 {
|
||||
/// from being coalesced away by a following LED/trigger report inside one host poll window (the
|
||||
/// confirmed stuck-rumble path).
|
||||
fn publish_output(view: &pf_umdf_util::section::MappedView, bytes: &[u8]) {
|
||||
// Serialized: the whole publish is a read-modify-write (read the cursor, write the slot it
|
||||
// names, then advance it) and the framework dispatches output callbacks in PARALLEL, so two
|
||||
// can be inside this at once. Unsynchronized, both read the same `ring_head`, both write the
|
||||
// SAME slot — tearing one report's bytes across the other's — and both store head+1, so the
|
||||
// cursor advances once for two reports and the host sees a single torn entry.
|
||||
//
|
||||
// An atomic `fetch_add` on the head does not fix it. That hands each writer a distinct slot,
|
||||
// but it advances the cursor BEFORE the slot bytes exist, so the host can read a slot that is
|
||||
// still being filled — trading a torn slot for a torn slot the host is invited to read. Making
|
||||
// the head-advance mean "the slot below is complete" is exactly what the lock buys.
|
||||
//
|
||||
// Poison-tolerant on purpose. Poison is sticky, so the repo's usual `if let Ok(g) = lock()`
|
||||
// would skip the publish for the REST OF THE PROCESS after a single panic elsewhere — silently
|
||||
// ending game output. Recovering the guard is safe here: the protected state is bytes in a
|
||||
// shared section, not an invariant a panic could have broken.
|
||||
let _publish = RING_PUBLISH
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
view.write_bytes(OFF_OUTPUT, bytes);
|
||||
let seq = view.read_u32(OFF_OUT_SEQ).wrapping_add(1);
|
||||
// Release, not a plain write: the host loads `out_seq` with Acquire specifically to order its
|
||||
// copy of the report bytes after it (`dualsense_windows.rs`, "Acquire pairs with the driver's
|
||||
// publish-then-bump store order"). An Acquire load pairs with a Release store and nothing
|
||||
// else, so as a plain write this promised the host an ordering it never actually established —
|
||||
// on a weakly-ordered core (ARM64) the fresh seq could arrive ahead of the bytes it announces.
|
||||
view.store_u32(OFF_OUT_SEQ, seq, Ordering::Release);
|
||||
view.write_u32(OFF_OUT_SEQ, seq);
|
||||
let len = ring_len(view);
|
||||
if len != 0 {
|
||||
let head = view.read_u32(OFF_RING_HEAD);
|
||||
@@ -398,11 +375,6 @@ fn publish_output(view: &pf_umdf_util::section::MappedView, bytes: &[u8]) {
|
||||
}
|
||||
}
|
||||
|
||||
/// Serializes [`publish_output`] against itself — see the note there for why an atomic cursor is
|
||||
/// not enough. Uncontended in the common case: one output report at a time is the norm, and the
|
||||
/// critical section is a few dozen bytes of memcpy into an already-mapped view.
|
||||
static RING_PUBLISH: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
|
||||
/// The sealed-channel client (per-pad: `ProcessSharingDisabled` gives each pad its own WUDFHost, so
|
||||
/// this static is per-pad). The handshake/adoption/validation state machine lives in `pf_umdf_util`.
|
||||
static CHANNEL: ChannelClient = ChannelClient::new();
|
||||
|
||||
@@ -358,48 +358,20 @@ fn read_state(data: Option<&MappedView>) -> (u32, u16, u8, u8, i16, i16, i16, i1
|
||||
/// host can tell "driver bound and alive" apart from "driver package missing/failed to bind" and see
|
||||
/// the game-visible polling path advance.
|
||||
fn touch_driver_marks(data: &MappedView) {
|
||||
let _marks = SECTION_PUBLISH
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
data.write_u32(OFF_DRIVER_PROTO, GAMEPAD_PROTO_VERSION);
|
||||
let hb = data.read_u32(OFF_DRIVER_HEARTBEAT).wrapping_add(1);
|
||||
data.write_u32(OFF_DRIVER_HEARTBEAT, hb);
|
||||
}
|
||||
|
||||
/// Publish a game's rumble (from SET_STATE) into the DATA section for the host to forward.
|
||||
///
|
||||
/// Serialized and Release-published, because IOCTLs arrive concurrently and neither property held
|
||||
/// before. `seq` was a read-modify-write across the two motor bytes: two `SET_STATE` calls could
|
||||
/// both read the same value and both write back `seq + 1`, so the host — which treats an unchanged
|
||||
/// seq as "nothing new" — saw one bump for two writes and skipped a level entirely. A skipped
|
||||
/// **stop** is the one that hurts: the pad keeps buzzing until the host's ~2.5 s idle force-off
|
||||
/// notices the game went quiet, which is where the bound on this bug comes from.
|
||||
///
|
||||
/// The seq store is Release for the same reason as `pf-gamepad`'s `out_seq`: the host loads it with
|
||||
/// Acquire and documents that as ordering its read of the motor bytes ("the driver bumps
|
||||
/// `rumble_seq` AFTER writing the rumble bytes", `gamepad_windows.rs`). A plain write gives that
|
||||
/// Acquire nothing to pair with, so the guarantee the host's comment claims did not exist in either
|
||||
/// direction — the host could read a fresh seq against stale motor levels on a weakly-ordered core.
|
||||
fn publish_rumble(data: Option<&MappedView>, large: u8, small: u8) {
|
||||
let Some(v) = data else { return };
|
||||
let _publish = SECTION_PUBLISH
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
v.write_u8(OFF_RUMBLE_LARGE, large);
|
||||
v.write_u8(OFF_RUMBLE_SMALL, small);
|
||||
let seq = v.read_u32(OFF_RUMBLE_SEQ).wrapping_add(1);
|
||||
v.store_u32(OFF_RUMBLE_SEQ, seq, Ordering::Release);
|
||||
v.write_u32(OFF_RUMBLE_SEQ, seq);
|
||||
}
|
||||
|
||||
/// Serializes the section's read-modify-write publishes ([`publish_rumble`], [`touch_driver_marks`])
|
||||
/// against each other. One lock rather than one per field: they are all short byte writes into the
|
||||
/// same mapped view, and the contention is nil compared to the IOCTL round trip that reaches them.
|
||||
///
|
||||
/// Poison-tolerant deliberately — poison is sticky, so bailing out on it would silently stop
|
||||
/// forwarding rumble for the rest of the process. The protected state is bytes in a shared section,
|
||||
/// not an invariant a panic elsewhere could have violated.
|
||||
static SECTION_PUBLISH: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||
|
||||
// Build the 29-byte GET_STATE buffer (the layout xinput1_4 parses).
|
||||
fn build_get_state(data: Option<&MappedView>) -> [u8; 29] {
|
||||
let (packet, buttons, lt, rt, lx, ly, rx, ry) = read_state(data);
|
||||
|
||||
Reference in New Issue
Block a user