Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
96f75f4e52 | ||
|
|
4c5b97cfe4 | ||
|
|
4beee17953 | ||
|
|
4ad0055416 | ||
|
|
db9cd40079 | ||
|
|
c63e8cee39 | ||
|
|
b670b5d844 | ||
|
|
064ea3de7d | ||
|
|
ec278c0478 | ||
|
|
5d91176500 | ||
|
|
551d0c3294 | ||
|
|
7f77fa68af | ||
|
|
ece8b16a78 | ||
|
|
2b91339cb8 | ||
|
|
ffa4577793 | ||
|
|
9bb8d84f12 | ||
|
|
f42aca690f | ||
|
|
34a02fdac5 | ||
|
|
8670b412c7 | ||
|
|
430499bdab | ||
|
|
92578803c2 |
@@ -50,7 +50,10 @@ on:
|
||||
- 'crates/pf-vaadec/**'
|
||||
- 'packaging/flatpak/**'
|
||||
- 'Cargo.lock'
|
||||
# Both halves of this job's correctness, not of the bundle's content: a change to either
|
||||
# can only be proven by a real run, and there is no other trigger that would give it one.
|
||||
- '.gitea/workflows/flatpak.yml'
|
||||
- 'scripts/ci/flatpak-deps-present.sh'
|
||||
tags: ['v*']
|
||||
workflow_dispatch:
|
||||
|
||||
@@ -270,12 +273,13 @@ jobs:
|
||||
|
||||
- name: Prefetch deps + sources (retried — the network phase, split off the build)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# All of the job's heavy network I/O happens HERE, retried, so a dropped DNS lookup
|
||||
# or TCP dial costs a backoff-retry instead of the whole (long) compile:
|
||||
# 1) --install-deps-only pulls everything the manifest declares from Flathub: the
|
||||
# GNOME 50 runtime/SDK + the rust-stable (//25.08, rustc 1.96) and llvm20 SDK
|
||||
# extensions. (No codec extension: the client links no FFmpeg — see the
|
||||
# manifest header.)
|
||||
# 1) the Flathub deps the manifest declares — the GNOME 50 runtime/SDK + the
|
||||
# rust-stable (//25.08, rustc 1.96) and llvm20 SDK extensions — but ONLY the ones
|
||||
# genuinely MISSING; see the block below. (No codec extension: the client links no
|
||||
# FFmpeg — see the manifest header.)
|
||||
# 2) --download-only fetches every source (all crates in cargo-sources.json) into
|
||||
# the .flatpak-builder state dir. Both are resumable/idempotent, so re-running
|
||||
# after a partial failure is safe and cheap.
|
||||
@@ -288,9 +292,40 @@ jobs:
|
||||
# for the mechanism.
|
||||
# 10 attempts (~9min budget), matching the remote-add bootstrap above — same shared,
|
||||
# load-sensitive runner, same flathub.org resolution path.
|
||||
bash scripts/ci/retry.sh 10 flatpak-builder --user --force-clean --disable-rofiles-fuse \
|
||||
--install-deps-from=flathub --install-deps-only \
|
||||
"$PWD/build-dir" "$MANIFEST"
|
||||
#
|
||||
# WHY THIS IS NOT AN UNCONDITIONAL `--install-deps-only` ANY MORE (2026-08-22):
|
||||
# that flag does not install what is missing, it UPDATES what is present.
|
||||
# builder_manifest_install_dep() branches on `flatpak info --show-commit <ref>` succeeding
|
||||
# and runs `flatpak update` for every dep already installed — with no fallback to a
|
||||
# plain install when that update fails — and ci/flatpak-ci.Dockerfile bakes
|
||||
# the entire runtime set, so on a healthy run it was a pure no-op that nonetheless made
|
||||
# every build depend on Flathub being healthy at that minute. It bit on 2026-08-22:
|
||||
# Updating runtime/org.freedesktop.Sdk.Extension.rust-stable/x86_64/25.08
|
||||
# Error: Failed to update org.freedesktop.Sdk.Extension.rust-stable: While pulling …
|
||||
# .filez: Server returned HTTP 404
|
||||
# dl.flathub.org served a 404 for one object of the then-current rust-stable//25.08
|
||||
# commit, deterministically — all 10 retry.sh attempts died on the SAME object over
|
||||
# ~9 min — and flatpak-builder SEGFAULTED on its own error path (rc=139), so retry.sh
|
||||
# saw a crash rather than a clean "this will never work" either. The build never wanted
|
||||
# that newer commit: the manifest pins a runtime VERSION, not a commit, and the baked
|
||||
# one satisfies it. Updating bought nothing and imported an upstream outage.
|
||||
#
|
||||
# So: assert what the image already has, and reach for Flathub only on a real miss —
|
||||
# the same "guard, don't install on top of a stale image" doctrine as the Tooling step.
|
||||
# The check lives in scripts/ci/flatpak-deps-present.sh (run its --self-test after
|
||||
# touching it): a bug in it that reports "satisfied" when it is not would build against
|
||||
# whatever runtime happened to be lying around, which is worth more than an inline
|
||||
# if-statement. It deliberately fails OPEN — anything it cannot parse takes the slow
|
||||
# install path below.
|
||||
if bash scripts/ci/flatpak-deps-present.sh "$MANIFEST"; then
|
||||
echo "deps satisfied by the baked image — not touching Flathub"
|
||||
flatpak list --user --columns=ref
|
||||
else
|
||||
echo "::warning::$MANIFEST declares deps punktfunk-flatpak-ci does not have — pulling from Flathub (~1.5 GB). Bump GNOME_VERSION/FREEDESKTOP_VERSION in ci/flatpak-ci.Dockerfile so this stays off the hot path."
|
||||
bash scripts/ci/retry.sh 10 flatpak-builder --user --force-clean --disable-rofiles-fuse \
|
||||
--install-deps-from=flathub --install-deps-only \
|
||||
"$PWD/build-dir" "$MANIFEST"
|
||||
fi
|
||||
bash scripts/ci/retry.sh 10 flatpak-builder --user --force-clean --disable-rofiles-fuse \
|
||||
--download-only --disable-updates \
|
||||
"$PWD/build-dir" "$MANIFEST"
|
||||
@@ -298,7 +333,17 @@ jobs:
|
||||
- name: Build the flatpak (offline — deps + sources prefetched above)
|
||||
run: |
|
||||
# Everything is already local (state dir warmed by the prefetch step), so this long
|
||||
# step needs no network; --install-deps-from stays as a no-op safety net.
|
||||
# step needs no network.
|
||||
#
|
||||
# --install-deps-from=flathub USED to sit here, commented as "a no-op safety net". It
|
||||
# was neither. builder-main.c calls builder_manifest_install_deps() whenever that flag
|
||||
# is set — --install-deps-only only decides whether it EXITS afterwards — so this step
|
||||
# re-ran the same `flatpak update` of the runtimes that killed the prefetch step on
|
||||
# 2026-08-22 (Flathub HTTP 404 on a rust-stable//25.08 object; see there). A live pull
|
||||
# of multi-GB runtimes is a strange thing to call a safety net in the step whose whole
|
||||
# design is to be offline, and it could only ever fire if the prefetch step above had
|
||||
# already failed the job. Dropped: the prefetch step is the one place that talks to
|
||||
# Flathub, and it is the one place with retries.
|
||||
#
|
||||
# --disable-updates is LOAD-BEARING, not tidiness: without it this step was never
|
||||
# actually offline. flatpak-builder runs the DOWNLOAD PHASE again as part of every
|
||||
@@ -326,7 +371,6 @@ jobs:
|
||||
flatpak-builder --user --force-clean --disable-rofiles-fuse \
|
||||
--default-branch="$FLATPAK_BRANCH" \
|
||||
--disable-updates \
|
||||
--install-deps-from=flathub \
|
||||
--repo="$PWD/repo" \
|
||||
"$PWD/build-dir" "$MANIFEST"
|
||||
|
||||
|
||||
@@ -515,8 +515,21 @@ class MainActivity : ComponentActivity() {
|
||||
else -> KeyEvent.KEYCODE_DPAD_RIGHT
|
||||
}
|
||||
|
||||
/** Resolve the panel's highest-refresh mode (same resolution) once, for [setConsoleHighRefreshRate]. */
|
||||
/**
|
||||
* Resolve the panel's highest-refresh mode (same resolution) once, for [setConsoleHighRefreshRate].
|
||||
*
|
||||
* NEVER on a TV, which leaves the id at `0` and makes every [setConsoleHighRefreshRate] call a
|
||||
* no-op. The pin exists for phone refresh governors that cap third-party apps at 60 Hz; a TV has
|
||||
* no such governor, and there it does active harm. `display.mode` is what [nativeDisplayMode]
|
||||
* reads to resolve "Native" refresh at connect, so a menu-time pin makes the session negotiate
|
||||
* the PINNED rate rather than the TV's real HDMI output — and [StreamScreen] then releases the
|
||||
* pin on TV (the decoder's own mode switch governs there), dropping the panel back to 60 while
|
||||
* the host is already serving 120. Every frame then waits out that mismatch, which is the
|
||||
* "latency explodes unless I set the refresh by hand" field report: picking a refresh explicitly
|
||||
* is precisely what bypasses the corrupted `nativeDisplayMode` answer.
|
||||
*/
|
||||
private fun resolveHighRefreshMode() {
|
||||
if (isTvDevice(this)) return
|
||||
@Suppress("DEPRECATION")
|
||||
val disp = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.R) display else windowManager.defaultDisplay
|
||||
highRefreshModeId = disp?.supportedModes?.maxWithOrNull(
|
||||
|
||||
@@ -478,7 +478,12 @@ fun nativeDisplayMode(context: Context): Triple<Int, Int, Int> {
|
||||
val mode = display.mode
|
||||
val w = mode.physicalWidth
|
||||
val h = mode.physicalHeight
|
||||
val hz = mode.refreshRate.toInt().coerceAtLeast(1)
|
||||
// ROUNDED, not truncated: TVs report the fractional NTSC rates over HDMI (59.94, 29.97,
|
||||
// 23.976), and `toInt()` turns 59.94 into 59 — a rate no display mode anywhere has, which the
|
||||
// host then serves by clamping DOWN to the highest mode it advertises at or below it. Rounding
|
||||
// also keeps this agreeing with `MainActivity.streamPanelFps`, which already rounds; the two
|
||||
// describe the same panel and must not disagree.
|
||||
val hz = kotlin.math.round(mode.refreshRate).toInt().coerceAtLeast(1)
|
||||
return Triple(maxOf(w, h), minOf(w, h), hz)
|
||||
}
|
||||
|
||||
|
||||
@@ -367,6 +367,38 @@ fn takeover_state_is_live(state: &TakeoverState) -> bool {
|
||||
|| state.forced_screen_env
|
||||
}
|
||||
|
||||
/// Restart the box's own autologin gaming session(s) after a leftover idle drop-in was swept off
|
||||
/// a host that died holding one ([`restore_takeover_on_startup`]).
|
||||
///
|
||||
/// Gated on the box actually being dark ([`box_session_live`]): if the user is already in game mode
|
||||
/// or on a desktop, the drop-in we removed was inert and bouncing their session would be the bug.
|
||||
/// Only an ACTIVE instance is restarted — under a just-removed idle drop-in, active means "running
|
||||
/// the sleep"; an inactive one is a leftover the display manager will handle on its own.
|
||||
fn hand_back_idled_units_after_crash() {
|
||||
if box_session_live() {
|
||||
return; // something is already drawing — the drop-in was inert
|
||||
}
|
||||
let units: Vec<String> = listed_autologin_units()
|
||||
.into_iter()
|
||||
.filter(|(_, active)| active == "active")
|
||||
.map(|(unit, _)| unit)
|
||||
.collect();
|
||||
if units.is_empty() {
|
||||
return;
|
||||
}
|
||||
tracing::warn!(
|
||||
?units,
|
||||
"gamescope: the box's Game Mode is running the dead host's idle placeholder and its panel \
|
||||
is dark — restarting it"
|
||||
);
|
||||
for unit in &units {
|
||||
if let RestoreVerb::Failed(why) = issue_restore_verb(&["restart", unit]) {
|
||||
tracing::error!(unit, status = %why, "gamescope: could not restart it");
|
||||
}
|
||||
}
|
||||
ensure_box_session_or_escalate(&units);
|
||||
}
|
||||
|
||||
/// On host startup, restore the TV's gaming session if a previous host instance took it over and
|
||||
/// crashed before restoring (`design/gamemode-and-dedicated-sessions.md` A3). Loads the persisted
|
||||
/// [`TakeoverState`] into the statics and schedules a restore after a short reconnect grace (so a
|
||||
@@ -399,6 +431,13 @@ pub fn restore_takeover_on_startup() {
|
||||
"gamescope: removed a leftover idle drop-in from a previous host instance — the box's \
|
||||
own Game Mode session would have started and then done nothing"
|
||||
);
|
||||
// Removing the FILE does not touch the unit RUNNING under it. That unit's `ExecStart` was
|
||||
// replaced with a sleep, so it is `active` and drawing nothing, and nothing below will
|
||||
// restart it: the takeover file may be absent, unparseable, or not `takeover_state_is_live`
|
||||
// — and all three of those exits used to leave the box sitting on a dark panel with its
|
||||
// Game Mode "running". A host killed mid-stream (SIGKILL, OOM, a yanked update) lands
|
||||
// exactly there, and on glass it is indistinguishable from broken hardware. Hand it back.
|
||||
hand_back_idled_units_after_crash();
|
||||
}
|
||||
let Ok(bytes) = std::fs::read(takeover_state_path()) else {
|
||||
return; // no takeover file — clean start
|
||||
@@ -2984,6 +3023,48 @@ fn replay_switch_under_restored_dm(dm: &str) {
|
||||
}
|
||||
}
|
||||
|
||||
/// The box's autologin gaming instances and their ACTIVE state, as `(unit, active)` pairs — the
|
||||
/// `--plain` columns are UNIT LOAD ACTIVE SUB DESCRIPTION, so the state is the third.
|
||||
///
|
||||
/// An unanswered query reads as "none listed", which is the safe direction for both callers: the
|
||||
/// takeover then frees nothing rather than killing a session it could not see properly, and the
|
||||
/// crash hand-back restarts nothing rather than bouncing one.
|
||||
fn listed_autologin_units() -> Vec<(String, String)> {
|
||||
let Ok(out) = crate::proc::output_within(
|
||||
Command::new("systemctl").args([
|
||||
"--user",
|
||||
"list-units",
|
||||
"--type=service",
|
||||
"--all",
|
||||
"--no-legend",
|
||||
"--plain",
|
||||
"gamescope-session-plus@*.service",
|
||||
]),
|
||||
UNIT_QUERY_BUDGET,
|
||||
) else {
|
||||
return Vec::new();
|
||||
};
|
||||
parse_listed_units(&String::from_utf8_lossy(&out.stdout))
|
||||
}
|
||||
|
||||
/// [`listed_autologin_units`]'s parser (the unit-testable core). Which column the ACTIVE state is
|
||||
/// in decides whether the takeover can tell a live gaming session from a dead leftover, and
|
||||
/// getting that wrong is silent in both directions — a live session read as dead leaves Steam
|
||||
/// holding the instance our own launch then collides with, and a dead one read as live idles a
|
||||
/// session nobody was in.
|
||||
fn parse_listed_units(stdout: &str) -> Vec<(String, String)> {
|
||||
stdout
|
||||
.lines()
|
||||
.filter_map(|l| {
|
||||
let mut cols = l.split_whitespace();
|
||||
let unit = cols.next()?;
|
||||
let active = cols.nth(1).unwrap_or("");
|
||||
(unit.starts_with("gamescope-session-plus@") && unit.ends_with(".service"))
|
||||
.then(|| (unit.to_string(), active.to_string()))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Stop every autologin gaming-mode session (`gamescope-session-plus@*.service`) so its
|
||||
/// single-instance Steam is free for our own host-managed session. Records the units so
|
||||
/// [`schedule_restore_tv_session`] can restart them on disconnect. Our own session is the transient
|
||||
@@ -3011,33 +3092,9 @@ fn replay_switch_under_restored_dm(dm: &str) {
|
||||
/// The ORDER is therefore load-bearing and not a style choice: stop the DM, bail if it did not
|
||||
/// land, and only then mask. A mask laid before a stop that never arrives is the storm.
|
||||
fn stop_autologin_sessions() -> Result<()> {
|
||||
let Ok(out) = crate::proc::output_within(
|
||||
Command::new("systemctl").args([
|
||||
"--user",
|
||||
"list-units",
|
||||
"--type=service",
|
||||
"--all",
|
||||
"--no-legend",
|
||||
"--plain",
|
||||
"gamescope-session-plus@*.service",
|
||||
]),
|
||||
UNIT_QUERY_BUDGET,
|
||||
) else {
|
||||
return Ok(());
|
||||
};
|
||||
// `(unit, ACTIVE state)` — the `--plain` columns are UNIT LOAD ACTIVE SUB DESCRIPTION.
|
||||
let listed: Vec<(String, String)> = String::from_utf8_lossy(&out.stdout)
|
||||
.lines()
|
||||
.filter_map(|l| {
|
||||
let mut cols = l.split_whitespace();
|
||||
let unit = cols.next()?;
|
||||
let active = cols.nth(1).unwrap_or("");
|
||||
(unit.starts_with("gamescope-session-plus@") && unit.ends_with(".service"))
|
||||
.then(|| (unit.to_string(), active.to_string()))
|
||||
})
|
||||
.collect();
|
||||
let listed = listed_autologin_units();
|
||||
if listed.is_empty() {
|
||||
return Ok(()); // nothing autologged in — Steam is already free
|
||||
return Ok(()); // nothing autologged in (or the query failed) — Steam is already free
|
||||
}
|
||||
let dm = display_manager_unit();
|
||||
// Only a LIVE instance holds Steam / justifies touching the DM. A loaded-but-inactive
|
||||
@@ -3439,7 +3496,13 @@ pub fn restore_takeover_now() {
|
||||
}
|
||||
*PENDING_RESTORE.lock().unwrap_or_else(|e| e.into_inner()) = None; // doing it right here
|
||||
tracing::info!("gamescope: host is shutting down — restoring the box's own session first");
|
||||
do_restore_tv_session();
|
||||
// `verify: false` — the escalation ladder waits up to a minute, and this runs inside
|
||||
// `native.rs`'s 20 s `SHUTDOWN_RESTORE_GRACE`, after which `exit(0)` runs no destructors.
|
||||
// Spending that grace watching instead of restoring would COST the hand-back, not check it.
|
||||
// The next host start is what covers a shutdown that left the box dark
|
||||
// ([`restore_takeover_on_startup`], which now hands the box back rather than only sweeping the
|
||||
// drop-in off it).
|
||||
do_restore_tv_session(false);
|
||||
}
|
||||
|
||||
/// What a bounded `systemctl --user` lifecycle verb on the RESTORE path actually did. Three states,
|
||||
@@ -3503,11 +3566,168 @@ fn connected_connector_under(base: &std::path::Path) -> bool {
|
||||
})
|
||||
}
|
||||
|
||||
/// How long a hand-back waits for the box to show something on its own panel before it starts
|
||||
/// escalating. Generous on purpose: the unit's `ExecStart` is a whole gamescope + Steam start, and
|
||||
/// on a cold box that is not quick — while a false escalation costs the user a session bounce.
|
||||
const HANDBACK_GRACE: Duration = Duration::from_secs(25);
|
||||
|
||||
/// How long each escalation rung gets. Shorter than [`HANDBACK_GRACE`]: by the time a rung runs,
|
||||
/// the ordinary start has already had its full grace and not delivered.
|
||||
const HANDBACK_RUNG_GRACE: Duration = Duration::from_secs(15);
|
||||
|
||||
/// Poll slice for the two waits above.
|
||||
const HANDBACK_POLL: Duration = Duration::from_millis(500);
|
||||
|
||||
/// Is ANYTHING driving the box's own panel right now — its game mode, or a desktop it switched to?
|
||||
///
|
||||
/// [`super::detect_active_session`] answers precisely the question the symptom asks: it reports the
|
||||
/// running compositor of our uid, and [`super::ActiveKind::None`] means nothing is drawing
|
||||
/// anywhere. Only sound AFTER `stop_session(SESSION_UNIT)` has killed our own managed session —
|
||||
/// that kill is a synchronous SIGKILL ([`kill_unit`]), so by the restore's escalation point our
|
||||
/// gamescope cannot still be answering for the box.
|
||||
fn box_session_live() -> bool {
|
||||
super::detect_active_session().kind != super::ActiveKind::None
|
||||
}
|
||||
|
||||
/// Poll [`box_session_live`] until it is true or `grace` runs out. [`HandbackWait::Superseded`]
|
||||
/// means a client reconnected and took the box over again — the hand-back we were checking is moot,
|
||||
/// and every remedy below would now be fighting a live stream for the box's session.
|
||||
enum HandbackWait {
|
||||
Live,
|
||||
Superseded,
|
||||
TimedOut,
|
||||
}
|
||||
|
||||
fn wait_for_box_session(grace: Duration) -> HandbackWait {
|
||||
let deadline = Instant::now() + grace;
|
||||
loop {
|
||||
if takeover_live() {
|
||||
return HandbackWait::Superseded;
|
||||
}
|
||||
if box_session_live() {
|
||||
return HandbackWait::Live;
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return HandbackWait::TimedOut;
|
||||
}
|
||||
std::thread::sleep(HANDBACK_POLL);
|
||||
}
|
||||
}
|
||||
|
||||
/// **The hand-back's last line of defence for a dark panel**, and the only part of this file that
|
||||
/// checks whether the restore it just performed actually WORKED.
|
||||
///
|
||||
/// Everything above issues a lifecycle verb and reports what systemd said about the JOB. That is
|
||||
/// not the same question as "does the box show a picture again", and the gap between the two is
|
||||
/// where every "my screen stays black after disconnecting" report lives — including ones whose
|
||||
/// trigger nobody has reproduced. So stop inferring the outcome and measure it: if nothing is
|
||||
/// driving the panel a full [`HANDBACK_GRACE`] after the hand-back, climb a ladder of remedies,
|
||||
/// each of which is a mechanism measured on both distro families (Bazzite `44.20260818`, Nobara
|
||||
/// f44, 2026-08-22), and say loudly at every rung what is happening.
|
||||
///
|
||||
/// 1. **`stop` the autologin unit.** Its login session's script is parked on
|
||||
/// `systemctl --user --wait start <unit>` (verified on both images), so stopping the unit
|
||||
/// releases that wait, the session exits, and `Relogin=true` logs straight back in — starting
|
||||
/// the unit inside a fresh login session with a seat. `stop`, never `restart`: a restart does
|
||||
/// NOT release the parked waiter (measured), which is exactly why it cannot rescue a box the
|
||||
/// ordinary restart already failed to bring back.
|
||||
/// 2. **Restart the display manager.** What the pre-0.31.0 takeover did on every disconnect, and
|
||||
/// proven to return the box to game mode. Needs privilege, so it can honestly fail.
|
||||
/// 3. **`PUNKTFUNK_RECOVER_SESSION_CMD`**, then an ERROR naming the command a human must run.
|
||||
///
|
||||
/// **Detached**, and that is not incidental. The restore runs under [`RESTORE_FLIGHT`], which a
|
||||
/// reconnecting client must take before it can re-take the box; watching for up to a minute while
|
||||
/// holding it would put that whole wait in front of every reconnect. So the caller fires this and
|
||||
/// returns, and the watcher stands down by itself the moment [`takeover_live`] says a new takeover
|
||||
/// armed — the box belongs to that stream now, and a remedy fired into it would be the bug.
|
||||
/// Call it AFTER `clear_takeover()`, or the very first poll reads our own finished takeover as a
|
||||
/// new one and stands down immediately.
|
||||
///
|
||||
/// A box that was already fine costs one [`box_session_live`] call and the thread exits.
|
||||
fn ensure_box_session_or_escalate(units: &[String]) {
|
||||
let units: Vec<String> = units.to_vec();
|
||||
std::thread::spawn(move || handback_watch(&units));
|
||||
}
|
||||
|
||||
fn handback_watch(units: &[String]) {
|
||||
match wait_for_box_session(HANDBACK_GRACE) {
|
||||
HandbackWait::Live => {
|
||||
tracing::info!(
|
||||
"gamescope: the box is driving its own panel again — hand-back complete"
|
||||
);
|
||||
return;
|
||||
}
|
||||
HandbackWait::Superseded => return,
|
||||
HandbackWait::TimedOut => {}
|
||||
}
|
||||
tracing::warn!(
|
||||
secs = HANDBACK_GRACE.as_secs(),
|
||||
units = ?units,
|
||||
"gamescope: NOTHING is driving the box's panel {}s after the hand-back — its screen is \
|
||||
dark. Escalating: stopping the autologin unit so the display manager relogins into a \
|
||||
session with a seat",
|
||||
HANDBACK_GRACE.as_secs()
|
||||
);
|
||||
// Rung 1 — release the login session's parked `--wait start` and let the DM relogin.
|
||||
for unit in units {
|
||||
if let RestoreVerb::Failed(why) = issue_restore_verb(&["stop", unit]) {
|
||||
tracing::warn!(unit, status = %why, "gamescope: could not stop the autologin unit");
|
||||
}
|
||||
}
|
||||
match wait_for_box_session(HANDBACK_RUNG_GRACE) {
|
||||
HandbackWait::Live => {
|
||||
tracing::info!(
|
||||
"gamescope: the display manager relogged the box into its own session — panel back"
|
||||
);
|
||||
return;
|
||||
}
|
||||
HandbackWait::Superseded => return,
|
||||
HandbackWait::TimedOut => {}
|
||||
}
|
||||
// Rung 2 — put the display manager itself through a restart.
|
||||
if let Some(dm) = display_manager_unit() {
|
||||
tracing::warn!(
|
||||
%dm,
|
||||
"gamescope: the box is still dark — restarting its display manager"
|
||||
);
|
||||
match restore_display_manager(&dm) {
|
||||
Ok(()) => match wait_for_box_session(HANDBACK_RUNG_GRACE) {
|
||||
HandbackWait::Live => {
|
||||
tracing::info!(%dm, "gamescope: the display manager brought the box back");
|
||||
return;
|
||||
}
|
||||
HandbackWait::Superseded => return,
|
||||
HandbackWait::TimedOut => {}
|
||||
},
|
||||
Err(why) => tracing::warn!(
|
||||
%dm,
|
||||
shape = why.shape(),
|
||||
reason = %why,
|
||||
"gamescope: could not restart the display manager"
|
||||
),
|
||||
}
|
||||
}
|
||||
// Rung 3 — the operator's own escape hatch, then say what is left to do by hand.
|
||||
if crate::try_recover_session() {
|
||||
tracing::warn!(
|
||||
"gamescope: fired PUNKTFUNK_RECOVER_SESSION_CMD to bring the box's session back"
|
||||
);
|
||||
return;
|
||||
}
|
||||
tracing::error!(
|
||||
units = ?units,
|
||||
"gamescope: the box has NO session driving its panel and every automatic remedy failed — \
|
||||
its screen stays dark until someone runs `systemctl --user restart <unit>` for one of \
|
||||
these, or `sudo systemctl restart display-manager.service`. Set \
|
||||
PUNKTFUNK_RECOVER_SESSION_CMD to let the host do this itself"
|
||||
);
|
||||
}
|
||||
|
||||
/// Tear down our host-managed session (freeing Steam) and restart the autologin gaming session(s)
|
||||
/// we stopped on connect — so the TV returns to gaming mode when no one is streaming. Invoked by
|
||||
/// [`start_restore_worker`] once the debounce deadline passes; takes the stopped-unit list so a
|
||||
/// cancelled+reconnected window keeps the list for a later real restore.
|
||||
fn do_restore_tv_session() {
|
||||
fn do_restore_tv_session(verify: bool) {
|
||||
// SteamOS: we reconfigured `gamescope-session.target` headless via a drop-in. Restore = remove
|
||||
// the drop-in + restart the target (back to the physical panel) — unless the user switched to a
|
||||
// desktop session meanwhile, in which case drop the override and leave the desktop alone.
|
||||
@@ -3574,6 +3794,9 @@ fn do_restore_tv_session() {
|
||||
),
|
||||
}
|
||||
clear_takeover(); // A3: consumed — after the restart, not before it
|
||||
if verify {
|
||||
ensure_box_session_or_escalate(&[STEAMOS_SESSION_TARGET.to_string()]);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
@@ -3699,14 +3922,14 @@ fn do_restore_tv_session() {
|
||||
}
|
||||
// (The idle drop-in is already gone — removed above every early return, so the restarts
|
||||
// below bring the box's real session back rather than another idle one.)
|
||||
for unit in units {
|
||||
for unit in &units {
|
||||
// Checked, not discarded: this call and the SteamOS `restart` above were the two places
|
||||
// that logged an unconditional success over a thrown-away exit status. A `--user start`
|
||||
// fails for reasons an operator can act on (the unit is masked, its start limit tripped),
|
||||
// and the DM branch thirty lines up already shows the shape — say what happened.
|
||||
// `restart`, not `start`: the idle takeover leaves the unit ACTIVE, and `start` on an
|
||||
// active unit is a no-op that would report success over a session still running nothing.
|
||||
match issue_restore_verb(&["restart", &unit]) {
|
||||
match issue_restore_verb(&["restart", unit]) {
|
||||
RestoreVerb::Done => tracing::info!(
|
||||
unit,
|
||||
"restored the TV's autologin gaming session (debounce elapsed, no client)"
|
||||
@@ -3731,6 +3954,12 @@ fn do_restore_tv_session() {
|
||||
}
|
||||
}
|
||||
clear_takeover(); // A3: consumed — and only now, with the restarts actually issued
|
||||
// …and CHECK that the restart above actually put a picture back on the box's panel, rather
|
||||
// than trusting the job status to mean that. AFTER `clear_takeover`, which is what makes a
|
||||
// later `takeover_live()` mean "a client reconnected" — see [`ensure_box_session_or_escalate`].
|
||||
if verify {
|
||||
ensure_box_session_or_escalate(&units);
|
||||
}
|
||||
}
|
||||
|
||||
/// Host-lifetime worker that fires a pending [`schedule_restore_tv_session`] once its debounce
|
||||
@@ -3767,7 +3996,10 @@ pub fn start_restore_worker() -> std::sync::Arc<()> {
|
||||
}
|
||||
};
|
||||
if still_due {
|
||||
do_restore_tv_session();
|
||||
// The disconnect restore: verified. This is the path the field reports
|
||||
// are about, it is on a worker thread with no deadline over it, and a box
|
||||
// left dark here stays dark until someone walks up to it.
|
||||
do_restore_tv_session(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -5334,12 +5566,12 @@ mod tests {
|
||||
classify_output_size, connected_connector_under, display_manager_unit_under, dm_plan,
|
||||
game_hz, gamescope_output_size, hdr_args, idle_dropin_body, idle_dropin_path,
|
||||
install_idle_dropin, is_steam_launch, mask_unit, missing_flags, mode_mismatch,
|
||||
nested_wrapper_script, our_wsi_layer_dir, plan_bind, release_autologin_mask,
|
||||
remove_idle_dropin, script_hardcodes_gamescope, sentinel_advanced, shape_dedicated_command,
|
||||
switch_ends_mask_window, takeover_state_is_live, unmask_unit, xwayland_refusal_marker,
|
||||
BindOff, BindPlan, BoxOutputSize, DmHelperError, SessionBind, TakeoverState, WsiPlan,
|
||||
AUTOLOGIN_MASKED, DISTRO_GAMESCOPE_PATH, PENDING_RESTORE, RESTORE_FLIGHT,
|
||||
STOPPED_AUTOLOGIN, WSI_OFF_ENV, X11_SOCKET_DIR,
|
||||
nested_wrapper_script, our_wsi_layer_dir, parse_listed_units, plan_bind,
|
||||
release_autologin_mask, remove_idle_dropin, script_hardcodes_gamescope, sentinel_advanced,
|
||||
shape_dedicated_command, switch_ends_mask_window, takeover_state_is_live, unmask_unit,
|
||||
xwayland_refusal_marker, BindOff, BindPlan, BoxOutputSize, DmHelperError, SessionBind,
|
||||
TakeoverState, WsiPlan, AUTOLOGIN_MASKED, DISTRO_GAMESCOPE_PATH, PENDING_RESTORE,
|
||||
RESTORE_FLIGHT, STOPPED_AUTOLOGIN, WSI_OFF_ENV, X11_SOCKET_DIR,
|
||||
};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
@@ -5538,6 +5770,39 @@ mod tests {
|
||||
/// drop-in APPENDS the sleep to the box's own session command and both run — the takeover
|
||||
/// would then be fighting the very Steam it set out to free, and nothing on the box would say
|
||||
/// why. Pins the reset, its order, and that the resolved `sleep` is the one that gets run.
|
||||
/// The `--plain` column the ACTIVE state lives in, pinned against real `systemctl --user
|
||||
/// list-units` output from both distro families. Read the wrong column and a live gaming
|
||||
/// session looks dead (Steam stays held, and our launch collides with it) or a dead leftover
|
||||
/// looks live (the takeover idles a session nobody was in) — both silent on glass.
|
||||
#[test]
|
||||
fn listed_units_take_the_active_column_not_the_load_column() {
|
||||
// Bazzite 44.20260818 and Nobara f44, verbatim (unit / LOAD / ACTIVE / SUB / description).
|
||||
let out = "gamescope-session-plus@ogui-steam.service loaded active running Gamescope Session Plus\n\
|
||||
gamescope-session-plus@steam.service loaded inactive dead Gamescope Session Plus\n";
|
||||
assert_eq!(
|
||||
parse_listed_units(out),
|
||||
vec![
|
||||
(
|
||||
"gamescope-session-plus@ogui-steam.service".to_string(),
|
||||
"active".to_string()
|
||||
),
|
||||
(
|
||||
"gamescope-session-plus@steam.service".to_string(),
|
||||
"inactive".to_string()
|
||||
),
|
||||
]
|
||||
);
|
||||
// `loaded` is the LOAD column and must never be mistaken for the state — that is the
|
||||
// off-by-one this pins.
|
||||
assert!(parse_listed_units(out).iter().all(|(_, a)| a != "loaded"));
|
||||
// Anything that is not one of our template's instances is not ours to touch.
|
||||
assert!(
|
||||
parse_listed_units("plasma-plasmashell.service loaded active running Shell\n")
|
||||
.is_empty()
|
||||
);
|
||||
assert!(parse_listed_units("").is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn idle_dropin_replaces_exec_start_rather_than_appending() {
|
||||
let body = idle_dropin_body("/usr/bin/sleep");
|
||||
|
||||
@@ -26,17 +26,26 @@
|
||||
|
||||
use super::{audio_control, audio_probe, minted, pad_endpoint as pe};
|
||||
use anyhow::Result;
|
||||
use windows::Win32::Devices::DeviceAndDriverInstallation::SetupDiEnumDeviceInfo;
|
||||
use windows::Win32::Devices::DeviceAndDriverInstallation::{
|
||||
SetupDiEnumDeviceInfo, SPDRP_HARDWAREID,
|
||||
};
|
||||
|
||||
/// The `Device Parameters` REG_DWORD each punktfunk-minted devnode family stamps on itself. The
|
||||
/// VALUE is what differs per family; presence of the NAME is "this one is ours", which is all a
|
||||
/// sweep needs.
|
||||
const OWNER_MARKERS: [&str; 3] = [
|
||||
pub(crate) const OWNER_MARKERS: [&str; 3] = [
|
||||
pe::PAD_INDEX_VALUE,
|
||||
minted::ROLE_MARKER,
|
||||
audio_probe::PROBE_MARKER,
|
||||
];
|
||||
|
||||
/// The Steam streaming hardware ids every audio devnode this product mints is created with —
|
||||
/// the second half of the ABANDONED-devnode test in [`owned_devnodes`].
|
||||
const MINTED_HWIDS: [&str; 2] = [
|
||||
"ROOT\\SteamStreamingSpeakers",
|
||||
"ROOT\\SteamStreamingMicrophone",
|
||||
];
|
||||
|
||||
/// What one sweep removed. `endpoint_records` is counted separately from `devnodes` because the
|
||||
/// registry half is best-effort by design — see [`delete_endpoint_record`].
|
||||
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||
@@ -117,11 +126,91 @@ fn owned_devnodes() -> Result<Vec<String>> {
|
||||
.any(|m| pe::read_devparam_dword(&set, &did, m).is_some())
|
||||
{
|
||||
out.push(inst);
|
||||
continue;
|
||||
}
|
||||
// ABANDONED: `ROOT\MEDIA\NNNN` carrying one of our minting hardware ids but no marker at
|
||||
// all — a devnode registered by a host that died before the marker write landed. It is
|
||||
// still bound and still serving endpoints, so leaving it behind is the "uninstalling
|
||||
// punktfunk left Sound settings full of Punktfunk devices forever" report all over again.
|
||||
//
|
||||
// The instance prefix is what makes this safe, and it is NOT redundant with
|
||||
// [`is_removable_instance`]: Steam's own devnodes carry these very hardware ids and are
|
||||
// ROOT-enumerated too, but live under `ROOT\SteamStreamingSpeakers\*` /
|
||||
// `ROOT\SteamStreamingMicrophone\*`. Only `ROOT\MEDIA\*` can have come from our
|
||||
// `SetupDiCreateDeviceInfoW(… DICD_GENERATE_ID)`.
|
||||
if is_abandoned_mint(
|
||||
&inst,
|
||||
&pe::devnode_multi_sz_prop(&set, &did, SPDRP_HARDWAREID),
|
||||
) {
|
||||
out.push(inst);
|
||||
}
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// The ABANDONED-devnode test, split out from the PnP enumeration so the rule that keeps this
|
||||
/// sweep off VALVE'S OWN devices is checkable without a live devinfo set. See [`owned_devnodes`].
|
||||
fn is_abandoned_mint(instance_id: &str, hwids: &[String]) -> bool {
|
||||
instance_id
|
||||
.to_ascii_uppercase()
|
||||
.starts_with("ROOT\\MEDIA\\")
|
||||
&& MINTED_HWIDS
|
||||
.iter()
|
||||
.any(|want| hwids.iter().any(|h| h.eq_ignore_ascii_case(want)))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod abandoned_tests {
|
||||
use super::is_abandoned_mint;
|
||||
|
||||
fn hw(s: &str) -> Vec<String> {
|
||||
vec![s.to_string()]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn adopts_our_own_unmarked_devnodes() {
|
||||
// What a host that died mid-mint leaves behind, either role.
|
||||
assert!(is_abandoned_mint(
|
||||
r"ROOT\MEDIA\0004",
|
||||
&hw(r"ROOT\SteamStreamingMicrophone")
|
||||
));
|
||||
assert!(is_abandoned_mint(
|
||||
r"ROOT\MEDIA\0002",
|
||||
&hw(r"ROOT\SteamStreamingSpeakers")
|
||||
));
|
||||
// PnP casing is not guaranteed on either half.
|
||||
assert!(is_abandoned_mint(
|
||||
r"root\media\0009",
|
||||
&hw(r"root\steamstreamingspeakers")
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn never_matches_valves_own_devices() {
|
||||
// THE safety rule: Steam's devnodes carry the very same hardware ids and are ROOT-
|
||||
// enumerated too — only the instance prefix separates them from ours.
|
||||
assert!(!is_abandoned_mint(
|
||||
r"ROOT\STEAMSTREAMINGMICROPHONE\0000",
|
||||
&hw(r"ROOT\SteamStreamingMicrophone")
|
||||
));
|
||||
assert!(!is_abandoned_mint(
|
||||
r"ROOT\STEAMSTREAMINGSPEAKERS\0000",
|
||||
&hw(r"ROOT\SteamStreamingSpeakers")
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn never_matches_other_vendors_or_real_hardware() {
|
||||
// VB-Cable mints ROOT\MEDIA devnodes too — a different hardware id is all that saves it.
|
||||
assert!(!is_abandoned_mint(r"ROOT\MEDIA\0000", &hw("VBAudioVACWDM")));
|
||||
assert!(!is_abandoned_mint(
|
||||
r"HDAUDIO\FUNC_01&VEN_10EC&DEV_0897",
|
||||
&hw(r"ROOT\SteamStreamingSpeakers")
|
||||
));
|
||||
assert!(!is_abandoned_mint(r"ROOT\MEDIA\0001", &[]));
|
||||
}
|
||||
}
|
||||
|
||||
/// A devnode this sweep is allowed to remove: ROOT-enumerated, i.e. software-created.
|
||||
///
|
||||
/// Every devnode we mint comes from `SetupDiCreateDeviceInfoW(… DICD_GENERATE_ID)` on the MEDIA
|
||||
|
||||
@@ -275,13 +275,22 @@ fn ensure_role(role: Role) -> Result<(String, String, Option<String>)> {
|
||||
let (hwid, inf) = discover_driver(role.needle(), role.inf_name())?;
|
||||
let devnode = match find_role_devnode(role)? {
|
||||
Some(inst) => inst,
|
||||
None => {
|
||||
let inst = pe::create_media_devnode(role.desc(), &hwid, |set, did| {
|
||||
pe::write_devparam_dword(set, did, ROLE_MARKER, role.value())
|
||||
})?;
|
||||
tracing::info!(role = role.label(), devnode = %inst, "minted an audio devnode");
|
||||
inst
|
||||
}
|
||||
// Before minting a SECOND devnode, reclaim an abandoned one. Minting is two PnP steps
|
||||
// (register, then mark), and a host that dies between them — the 0.30.0 teardown abort
|
||||
// did exactly this, five times on one box — leaves a registered, driver-bound, endpoint-
|
||||
// serving devnode that carries no marker. Nothing then resolves it: the next pass mints
|
||||
// a fresh one and the orphan lingers as a duplicate "Punktfunk Speakers"/"Punktfunk
|
||||
// Microphone" in the Sound zoo, invisible to the marker-matched uninstall sweep.
|
||||
None => match adopt_orphan_devnode(role, &hwid)? {
|
||||
Some(inst) => inst,
|
||||
None => {
|
||||
let inst = pe::create_media_devnode(role.desc(), &hwid, |set, did| {
|
||||
pe::write_devparam_dword(set, did, ROLE_MARKER, role.value())
|
||||
})?;
|
||||
tracing::info!(role = role.label(), devnode = %inst, "minted an audio devnode");
|
||||
inst
|
||||
}
|
||||
},
|
||||
};
|
||||
pe::bind_driver(&hwid, &inf)?;
|
||||
|
||||
@@ -531,6 +540,61 @@ fn find_role_devnode(role: Role) -> Result<Option<String>> {
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
/// Reclaim an ABANDONED punktfunk devnode for `role`, re-marking it so it resolves normally from
|
||||
/// here on; `None` when there is nothing to adopt (the ordinary first-mint path).
|
||||
///
|
||||
/// The shape adopted is `ROOT\MEDIA\NNNN` + the role's Steam hardware id + NO owner marker.
|
||||
/// That triple can only be ours: `ROOT\MEDIA\NNNN` is what
|
||||
/// `SetupDiCreateDeviceInfoW(… DICD_GENERATE_ID)` on the MEDIA class yields, and STEAM'S OWN
|
||||
/// devnodes are enumerated under `ROOT\SteamStreamingSpeakers\*` /
|
||||
/// `ROOT\SteamStreamingMicrophone\*` — they carry the same hardware id but never that instance
|
||||
/// prefix, which is precisely what keeps this from adopting (and later sweeping) Steam's devices.
|
||||
/// A marker of ANY family is left alone: it is a live devnode, ours but spoken for.
|
||||
///
|
||||
/// Which family the orphan came from does not matter. Every one is a plain instance of the same
|
||||
/// Valve driver; roles are ours to assign, and re-marking it here is what makes the assignment
|
||||
/// stick across restarts.
|
||||
fn adopt_orphan_devnode(role: Role, hwid: &str) -> Result<Option<String>> {
|
||||
use windows::Win32::Devices::DeviceAndDriverInstallation::{
|
||||
SetupDiEnumDeviceInfo, SPDRP_HARDWAREID,
|
||||
};
|
||||
let set = pe::media_class_devs()?;
|
||||
for i in 0.. {
|
||||
let mut did = pe::devinfo_data();
|
||||
// SAFETY: live set; `did` is a live out-param with cbSize set.
|
||||
if unsafe { SetupDiEnumDeviceInfo(set.0, i, &mut did) }.is_err() {
|
||||
break; // ERROR_NO_MORE_ITEMS
|
||||
}
|
||||
let Some(inst) = pe::instance_id(&set, &did) else {
|
||||
continue;
|
||||
};
|
||||
if !inst.to_ascii_uppercase().starts_with("ROOT\\MEDIA\\") {
|
||||
continue;
|
||||
}
|
||||
if !pe::devnode_multi_sz_prop(&set, &did, SPDRP_HARDWAREID)
|
||||
.iter()
|
||||
.any(|h| h.eq_ignore_ascii_case(hwid))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if super::devnode_cleanup::OWNER_MARKERS
|
||||
.iter()
|
||||
.any(|m| pe::read_devparam_dword(&set, &did, m).is_some())
|
||||
{
|
||||
continue;
|
||||
}
|
||||
pe::write_devparam_dword(&set, &mut did, ROLE_MARKER, role.value())?;
|
||||
tracing::warn!(
|
||||
role = role.label(),
|
||||
devnode = %inst,
|
||||
"adopted an abandoned audio devnode — one of ours whose owner marker never landed \
|
||||
(a host that died mid-mint). Re-marked and reused instead of minting a duplicate"
|
||||
);
|
||||
return Ok(Some(inst));
|
||||
}
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
/// Find the (exact hardware id, INF path) for one of Steam's streaming drivers: prefer any
|
||||
/// installed devnode whose hardware-id list contains `needle` (its `oemNN.inf` is the driver
|
||||
/// Windows already trusts), else fall back to Steam's driver directory. Shared with the
|
||||
|
||||
@@ -1192,6 +1192,17 @@ fn grant_system_full_control(subkey_path: &str) -> Result<()> {
|
||||
result
|
||||
}
|
||||
|
||||
/// The MMDevices hive an endpoint's record lives in, chosen by the direction its id encodes
|
||||
/// (`{0.0.1.…}` = capture, anything else = render). Render is the safe default: it is what every
|
||||
/// non-capture id resolves to, and the pad program only ever has render endpoints.
|
||||
fn mmdev_path_for(endpoint_id: &str) -> &'static str {
|
||||
if endpoint_id.starts_with(CAPTURE_ENDPOINT_ID_PREFIX) {
|
||||
MMDEV_CAPTURE_PATH
|
||||
} else {
|
||||
MMDEV_RENDER_PATH
|
||||
}
|
||||
}
|
||||
|
||||
/// The raw-registry stamp route: repair the Properties key ACL, then write the serialized
|
||||
/// values (see [`reg_registry_value`]). Values written here are STORED but possibly not
|
||||
/// SERVED until an AudioEndpointBuilder restart — the caller's read-back decides.
|
||||
@@ -1199,7 +1210,14 @@ fn registry_stamp(endpoint_id: &str, stamps: &[&Stamp]) -> Result<()> {
|
||||
use winreg::enums::HKEY_LOCAL_MACHINE;
|
||||
use winreg::RegKey;
|
||||
let guid = endpoint_guid_part(endpoint_id)?;
|
||||
let path = format!(r"{MMDEV_RENDER_PATH}\{guid}\Properties");
|
||||
// The hive follows the endpoint's DIRECTION. This was hardcoded to Render, which is
|
||||
// invisible for the pad program (its endpoints are render-only) but wrong for the minted
|
||||
// provider, which stamps the virtual microphone's CAPTURE endpoint through the same
|
||||
// writer: the fallback then reached for `…\Render\{capture-guid}\Properties`, a key that
|
||||
// cannot exist, so every registry-route stamp of a capture endpoint failed on a box where
|
||||
// the property store was denied — silently, since the caller degrades to "keeps the
|
||||
// driver's default name".
|
||||
let path = format!(r"{}\{guid}\Properties", mmdev_path_for(endpoint_id));
|
||||
grant_system_full_control(&path)
|
||||
.with_context(|| format!("make {path} writable (registry stamp route)"))?;
|
||||
let key = RegKey::predef(HKEY_LOCAL_MACHINE)
|
||||
@@ -2109,6 +2127,23 @@ fn pad_capture_thread(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The registry stamp route must reach for the hive matching the endpoint's DIRECTION —
|
||||
/// it was hardcoded to Render, so a capture endpoint's fallback stamp could never land.
|
||||
#[test]
|
||||
fn registry_stamp_hive_follows_the_endpoint_direction() {
|
||||
assert_eq!(
|
||||
mmdev_path_for("{0.0.1.00000000}.{2753f927-2093-4ab4-aa90-9d880e959128}"),
|
||||
MMDEV_CAPTURE_PATH,
|
||||
"the minted microphone's capture endpoint records under Capture"
|
||||
);
|
||||
assert_eq!(
|
||||
mmdev_path_for("{0.0.0.00000000}.{5da9b5c9-8a10-4b54-8cf6-ce02b8354f16}"),
|
||||
MMDEV_RENDER_PATH,
|
||||
);
|
||||
// Anything unrecognised keeps the old behaviour rather than inventing a hive.
|
||||
assert_eq!(mmdev_path_for("nonsense"), MMDEV_RENDER_PATH);
|
||||
}
|
||||
|
||||
/// The serialized container blob for pad 0 must be byte-for-byte the on-glass-measured
|
||||
/// value, and byte 23 must be the pad index.
|
||||
#[test]
|
||||
|
||||
@@ -590,6 +590,9 @@ fn watch(
|
||||
|
||||
// ---- Phase 1: wait for the game to show up. ----
|
||||
let start_deadline = spawned_at + START_GRACE;
|
||||
// How long the scan has *continuously* seen something for this title — the scan-side twin of
|
||||
// [`SHIM_WINDOW`]. See `scan_settled` below for what it is protecting against.
|
||||
let mut seen_since: Option<Instant> = None;
|
||||
loop {
|
||||
if cancelled() {
|
||||
return;
|
||||
@@ -713,12 +716,39 @@ fn watch(
|
||||
&& (child.is_some() || spawned.is_some())
|
||||
&& spawned_at.elapsed() >= SHIM_WINDOW;
|
||||
let live = scanner.find(&shared.spec, shared.launch_stamp);
|
||||
// The same rule for what the *scan* finds, and for the same reason. A store's launch is a
|
||||
// chain of process trees, and the ones that run before the game carry the signals the game
|
||||
// carries: Steam wraps its shader pre-caching and its Proton prefix work in the very
|
||||
// `reaper SteamLaunch AppId=<appid>` the game gets, so the first poll of a launch can match
|
||||
// a tree that was never the game.
|
||||
//
|
||||
// Latching on one poll is what costs, because the two phases are patient in opposite ways.
|
||||
// This one waits [`START_GRACE`] — five minutes — and ending it never ends the session.
|
||||
// Phase 2 waits [`EXIT_CONFIRM`] — three seconds — and ending it *does*. A single sighting
|
||||
// flips the lease from the first to the second, permanently; when that tree then exits with
|
||||
// the real game not yet started, the stream drops mid-launch. On Linux that ended a Rocket
|
||||
// League session 10 s after launch, while Steam was still compiling its shaders, and the
|
||||
// player had to launch a second time to get one that stayed up (field report 2026-08-22).
|
||||
//
|
||||
// Requiring the sighting to persist buys that back for a few seconds of `GameRunning`
|
||||
// latency and nothing else — exit detection is untouched. ⚠ It is a window, not a proof: a
|
||||
// pre-launch tree that outlives the window still latches. Signals sharp enough to tell one
|
||||
// from the other belong in [`crate::procscan`] (where Steam's shader job is already excluded
|
||||
// by name); this bounds what no signal caught.
|
||||
let scan_settled = if live.is_empty() {
|
||||
seen_since = None;
|
||||
false
|
||||
} else {
|
||||
seen_since.get_or_insert_with(Instant::now).elapsed() >= SHIM_WINDOW
|
||||
};
|
||||
// A provider saying so is as good as seeing it — better, for a title there is nothing to
|
||||
// see: it is the launcher that started the game telling us it did. This is the only way a
|
||||
// [`LeaseKind::Reported`] lease ever leaves this phase, and for a `Matched` one it just
|
||||
// gets there sooner than the scan would.
|
||||
// gets there sooner than the scan would. Not gated by the window above: a report is the
|
||||
// launcher's own statement about the game, not an inference from a process that resembles
|
||||
// it, so there is nothing to wait out.
|
||||
let said_running = reported().is_some_and(|l| l.running);
|
||||
if !live.is_empty() || child_alive || said_running {
|
||||
if scan_settled || child_alive || said_running {
|
||||
known = live.clone();
|
||||
publish(&live);
|
||||
shared.was_running.store(true, Ordering::Relaxed);
|
||||
@@ -731,6 +761,8 @@ fn watch(
|
||||
title = %shared.game.title,
|
||||
kind = kind.as_str(),
|
||||
procs = live.len(),
|
||||
// Which processes, not just how many: see [`crate::procscan::names`].
|
||||
names = ?crate::procscan::names(&live),
|
||||
"the launched game is running"
|
||||
);
|
||||
break;
|
||||
@@ -2019,6 +2051,78 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// 🛑 The 2026-08-22 field report: a **pre-launch** process tree must not be mistaken for the
|
||||
/// game.
|
||||
///
|
||||
/// Steam wraps its shader pre-caching in the same `SteamLaunch AppId=` reaper the game itself
|
||||
/// gets, so the first poll of a launch matches a tree that was never the game. What shipped
|
||||
/// latched on that single sighting: the lease left the start phase immediately, and when the
|
||||
/// compile finished and that tree exited — with Rocket League still starting — the exit watch
|
||||
/// called it the game exiting and closed the session with `APP_EXITED`, 10 s after launch. On
|
||||
/// the player's screen the stream dropped mid-"Processing Vulkan shaders"; their workaround was
|
||||
/// to launch the game twice.
|
||||
///
|
||||
/// The scanner now knows Steam's replayer by name ([`crate::procscan`]). This pins the bound
|
||||
/// behind that: a matched process that does not outlive [`SHIM_WINDOW`] never arms the exit
|
||||
/// watch, whatever it was — which is what covers the pre-launch trees nobody has named yet.
|
||||
///
|
||||
/// Ignored by default: it outlives the shim window and then waits out [`EXIT_CONFIRM`], ~11 s.
|
||||
#[cfg(target_os = "linux")]
|
||||
#[test]
|
||||
#[ignore = "drives a real process for ~11s (shim window + exit confirmation)"]
|
||||
fn a_pre_launch_tree_that_exits_never_ends_the_session() {
|
||||
use std::sync::atomic::AtomicUsize;
|
||||
|
||||
// The stand-in has to keep the name `sleep`: coreutils is a multi-call binary that
|
||||
// dispatches on `argv[0]`, and under any other name it exits instantly — which would pass
|
||||
// this test for entirely the wrong reason. (Same trap as the live matcher test in
|
||||
// [`crate::procscan`].)
|
||||
let td = tempfile::tempdir().expect("tempdir");
|
||||
let stand_in = td.path().join("sleep");
|
||||
std::fs::copy("/bin/sleep", &stand_in).expect("copy a stand-in pre-launch binary");
|
||||
let launch_stamp = launch_clock();
|
||||
|
||||
// Alive for less than the shim window — Steam's shader job, in miniature.
|
||||
let mut child = std::process::Command::new(&stand_in)
|
||||
.arg("3")
|
||||
.spawn()
|
||||
.expect("spawn the fake pre-launch tree");
|
||||
// Reaped on its own thread: a zombie keeps its `/proc` entry with an unchanged start time,
|
||||
// so the scan would call it alive forever and the exit under test never happen.
|
||||
std::thread::spawn(move || {
|
||||
let _ = child.wait();
|
||||
});
|
||||
|
||||
static PRE_EXITS: AtomicUsize = AtomicUsize::new(0);
|
||||
PRE_EXITS.store(0, Ordering::SeqCst);
|
||||
let lease = open(
|
||||
LeaseRequest {
|
||||
launch_stamp,
|
||||
// No child and no pid: the scan is the only signal, which is the field-report shape
|
||||
// (`steam steam://rungameid/…` had already handed off and exited).
|
||||
..req("steam:pre-launch", DetectSpec::dir(td.path()), false)
|
||||
},
|
||||
Box::new(|| {
|
||||
PRE_EXITS.fetch_add(1, Ordering::SeqCst);
|
||||
}),
|
||||
);
|
||||
let shared = lease.shared();
|
||||
assert!(matches!(shared.kind(), LeaseKind::Matched));
|
||||
|
||||
std::thread::sleep(SHIM_WINDOW + EXIT_CONFIRM + Duration::from_secs(3));
|
||||
assert_eq!(
|
||||
PRE_EXITS.load(Ordering::SeqCst),
|
||||
0,
|
||||
"a tree that ran before the game must not end the session when it exits — this is the \
|
||||
field report"
|
||||
);
|
||||
assert_ne!(
|
||||
shared.state(),
|
||||
GameState::Exited,
|
||||
"the game never started, so nothing of it can have exited"
|
||||
);
|
||||
}
|
||||
|
||||
/// The whole point of the module, against a real process: a `Child` lease sees its game running,
|
||||
/// notices when it exits, and reports that exit exactly once.
|
||||
///
|
||||
|
||||
@@ -1111,6 +1111,21 @@ fn spawn_sender(
|
||||
|
||||
use crate::send_pacing::percentile;
|
||||
|
||||
/// How long to ignore further keyframe requests after emitting one.
|
||||
///
|
||||
/// The window bounds IDR emission in TIME, so it needs an absolute floor rather than a frame
|
||||
/// count: it has to outlast the round trip in which the client receives and decodes the IDR it
|
||||
/// already asked for. The original `frame_interval * 2` closes long before that at high refresh —
|
||||
/// 16.7 ms at 120 fps, while a Moonlight client under loss re-asks every ~30 ms — so every request
|
||||
/// passed the gate and the stream became ~32 full IDRs/s, whose bulk causes the very loss that
|
||||
/// prompts the next request. That storm sustains itself and reads as stutter at a flat latency
|
||||
/// (field log, AMD RX 7800 XT / Bazzite 44 HEVC, 2026-08-22: 1118 requests, 1115 honoured, 3
|
||||
/// coalesced). 100 ms matches the encoder-reset backoff below and is about one IDR's service time
|
||||
/// on a saturated link.
|
||||
fn keyframe_coalesce_window(frame_interval: Duration) -> Duration {
|
||||
(frame_interval * 2).max(Duration::from_millis(100))
|
||||
}
|
||||
|
||||
/// The encode → packetize loop, over a borrowed capturer. Sending runs on a dedicated thread
|
||||
/// (see [`spawn_sender`]) so a send spike can never stall capture/encode.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
@@ -1194,6 +1209,11 @@ fn stream_body(
|
||||
// also fails safe when nobody tells it, but pass the REAL depth: `idd_depth` is configurable
|
||||
// and a deeper ring is free pipelining the fallback would forfeit.
|
||||
enc.set_input_ring_depth(capturer.pipeline_depth().max(1));
|
||||
// What `enc` was opened against. The capture source can change size/format UNDER this loop with
|
||||
// nothing negotiating it (see the follow-the-source guard below); tracked so the loop can notice.
|
||||
// Both sites that swap `enc` re-bind `frame` with it, so this is always
|
||||
// `(frame.format, frame.width, frame.height)` right after one.
|
||||
let mut enc_src = (frame.format, frame.width, frame.height);
|
||||
// FEC overhead percent (Sunshine default 20). Override with PUNKTFUNK_FEC_PCT (0 = data-only).
|
||||
let fec_pct: u8 = std::env::var("PUNKTFUNK_FEC_PCT")
|
||||
.ok()
|
||||
@@ -1273,9 +1293,9 @@ fn stream_body(
|
||||
// RFI (VAAPI/AMD — `supports_rfi=false`) each one becomes a full IDR, so an un-coalesced request
|
||||
// stream turns EVERY frame into a 4K IDR, saturates the send path, and collapses the session
|
||||
// instead of recovering. One fresh IDR already resolves all pending loss, so after emitting one
|
||||
// we ignore further keyframe requests for a short in-flight window (~2 frames). NVENC
|
||||
// ref-invalidation (cheap, no IDR spike) is never rate-limited — only full keyframes are.
|
||||
let keyframe_coalesce = frame_interval * 2;
|
||||
// we ignore further keyframe requests for the in-flight window below. NVENC ref-invalidation
|
||||
// (cheap, no IDR spike) is never rate-limited — only full keyframes are.
|
||||
let keyframe_coalesce = keyframe_coalesce_window(frame_interval);
|
||||
let mut last_keyframe: Option<Instant> = None;
|
||||
// A frame dropped at the pipeline head (below) breaks the reference chain for the following
|
||||
// P-frames: the client never receives it, but the encoder advanced its references past it, and —
|
||||
@@ -1362,6 +1382,7 @@ fn stream_body(
|
||||
.context("reopen encoder after rebuild")?;
|
||||
// A rebuilt encoder starts unconfigured — same reason as the first open above.
|
||||
enc.set_input_ring_depth(capturer.pipeline_depth().max(1));
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
supports_rfi = enc.caps().supports_rfi;
|
||||
enc.request_keyframe();
|
||||
last_keyframe = Some(Instant::now());
|
||||
@@ -1375,6 +1396,82 @@ fn stream_body(
|
||||
}
|
||||
}
|
||||
let t_cap = tick.elapsed();
|
||||
// Follow an AUTONOMOUS source mode change — one nothing negotiated. The IDD-push capturer
|
||||
// re-opens its ring on a confirmed display-descriptor change (a fullscreen game mode-setting
|
||||
// the virtual display, or an HDR flip changing the format), and the encoder is the one
|
||||
// component that cannot follow a resolution change in place. Every `submit` below then
|
||||
// refuses the frame ("captured WxH != encoder AxB"), and the submit ladder only rebuilds the
|
||||
// encoder IN PLACE — at the SAME configured size — which cannot fix a size the source has
|
||||
// already left, so all five resets burn on it and the stream ends (native/stream.rs carried
|
||||
// the identical gap; a 2026-08-22 field report hit it there at 4K→1080p).
|
||||
//
|
||||
// GameStream has no mid-stream mode-change message, so the client is NOT told: Moonlight
|
||||
// decodes a bitstream that disagrees with the resolution it configured its decoder from.
|
||||
// That is the same bargain the first open above already takes whenever the captured size
|
||||
// differs from the negotiated one (the monitor-mirror case) — tolerant decoders re-init off
|
||||
// the SPS and scale, a strict one (Media Foundation on Xbox) may stall and drop the session.
|
||||
// Taking it here too is strictly better than the alternative, which is ending every stream
|
||||
// the moment a game changes mode.
|
||||
if enc_src != (frame.format, frame.width, frame.height) {
|
||||
match encode::open_video(
|
||||
cfg.codec,
|
||||
frame.format,
|
||||
frame.width,
|
||||
frame.height,
|
||||
cfg.fps,
|
||||
cfg.bitrate_kbps as u64 * 1000,
|
||||
frame.is_cuda(),
|
||||
// Derived from the delivered format, so an HDR flip re-opens at the right depth.
|
||||
gs_bit_depth(frame.format),
|
||||
encode::ChromaFormat::Yuv420, // GameStream stays 4:2:0 — see the first open
|
||||
cursor_blend, // same capture cursor mode — see the first open
|
||||
cfg.slices, // client slicing ceiling — see the first open
|
||||
) {
|
||||
Ok(e) => {
|
||||
tracing::info!(
|
||||
from = %format!("{}x{} {:?}", enc_src.1, enc_src.2, enc_src.0),
|
||||
to = %format!("{}x{} {:?}", frame.width, frame.height, frame.format),
|
||||
negotiated = ?(cfg.width, cfg.height),
|
||||
"gamestream: the capture source changed mode mid-stream — reopened the \
|
||||
encoder at the delivered size (the client is not told; a strict decoder \
|
||||
may not follow — see the note at this guard)"
|
||||
);
|
||||
enc = e;
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
// A rebuilt encoder starts unconfigured — same reasons as the first open.
|
||||
enc.set_input_ring_depth(capturer.pipeline_depth().max(1));
|
||||
supports_rfi = enc.caps().supports_rfi;
|
||||
enc.request_keyframe();
|
||||
last_keyframe = Some(Instant::now());
|
||||
// The old encoder died with its in-flight submissions — their AUs will never
|
||||
// arrive, so the numbering prediction restarts at `au_seq` (same reasoning as
|
||||
// the capture rebuild above). Restart the stall clock for the fresh encoder and
|
||||
// give it the full reset budget.
|
||||
enc_inflight = 0;
|
||||
encoder_resets = 0;
|
||||
last_au_at = Instant::now();
|
||||
}
|
||||
Err(e) => {
|
||||
// Don't spend the stream on the FIRST failed open: the mode-set that triggered
|
||||
// this is exactly the kind of event that leaves the driver settling, which is
|
||||
// what the submit ladder's backoff exists for. Spend the shared reset budget at
|
||||
// the same exponential pace, re-entering this guard each round — the old encoder
|
||||
// stays installed and mismatched meanwhile, so it simply keeps failing submit.
|
||||
encoder_resets += 1;
|
||||
if encoder_resets > MAX_ENCODER_RESETS {
|
||||
return Err(e).context("reopen encoder at the source's new mode");
|
||||
}
|
||||
let backoff = frame_interval
|
||||
.max(Duration::from_millis(100u64 << (encoder_resets - 1).min(4)));
|
||||
tracing::warn!(error = %format!("{e:#}"), reset = encoder_resets,
|
||||
max = MAX_ENCODER_RESETS,
|
||||
"gamestream: reopening the encoder at the source's new mode failed — retrying");
|
||||
next_frame = Instant::now() + backoff;
|
||||
std::thread::sleep(backoff);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Honor a client recovery request. Prefer reference-frame invalidation (the encoder
|
||||
// re-references an older still-valid frame — no costly IDR spike); if the encoder can't
|
||||
// invalidate (range too old, or no NVENC RFI) it returns false and we force a keyframe.
|
||||
@@ -1716,6 +1813,27 @@ mod tests {
|
||||
assert_eq!(t.game.title, "/opt/game/run");
|
||||
}
|
||||
|
||||
/// The coalesce window must bound forced IDRs in time, not in frames. A frame-scaled window
|
||||
/// vanishes exactly where it matters most — at high refresh, where a client's recovery spam
|
||||
/// arrives far slower than two frame intervals and so passes the gate every time.
|
||||
#[test]
|
||||
fn keyframe_coalesce_window_outlasts_a_clients_request_cadence() {
|
||||
// The observed storm: a 120 fps session against a client re-asking every ~30 ms. The
|
||||
// pre-floor window was 16.7 ms, so every request became a full IDR.
|
||||
let at_120 = keyframe_coalesce_window(Duration::from_secs_f64(1.0 / 120.0));
|
||||
assert!(
|
||||
at_120 >= Duration::from_millis(100),
|
||||
"120 fps window {at_120:?} does not outlast a ~30 ms request cadence"
|
||||
);
|
||||
// 60 fps was under the floor too (33.3 ms), which is why this is not a 120-only fix.
|
||||
assert!(keyframe_coalesce_window(Duration::from_secs_f64(1.0 / 60.0)) >= at_120);
|
||||
// A slow stream keeps the frame-scaled window — the floor only ever raises it.
|
||||
assert_eq!(
|
||||
keyframe_coalesce_window(Duration::from_millis(200)),
|
||||
Duration::from_millis(400)
|
||||
);
|
||||
}
|
||||
|
||||
/// End-to-end check of the send thread: batches pushed on the channel arrive, complete and
|
||||
/// byte-identical, at a peer socket via the paced sendmmsg path.
|
||||
#[test]
|
||||
|
||||
@@ -55,7 +55,14 @@ pub struct DetectSpec {
|
||||
/// Steam appid, for titles Steam itself installed (never for non-Steam shortcuts, whose reaper
|
||||
/// appid semantics differ — those carry an [`exe`](Self::exe) instead). On Linux this is the
|
||||
/// sharpest signal available: Steam wraps every launch — native or Proton — in
|
||||
/// `reaper SteamLaunch AppId=<appid>`, whose lifetime is exactly the game's.
|
||||
/// `reaper SteamLaunch AppId=<appid>`.
|
||||
///
|
||||
/// ⚠ That reaper is the *appid's*, not the game's. Steam wraps its **pre-launch** work for a
|
||||
/// title in one too — shader pre-caching most visibly — so a launch is a chain of reaper trees
|
||||
/// and only the last of them is the game. Reading the first as the game is what dropped a
|
||||
/// stream 10 s into a Rocket League launch, mid-shader-compile (field report 2026-08-22); the
|
||||
/// shader job is excluded by name in [`crate::procscan`], and [`crate::gamelease`] waits out a
|
||||
/// window before believing any of them.
|
||||
pub steam_appid: Option<u32>,
|
||||
/// A launcher-stamped environment marker.
|
||||
pub env_marker: Option<EnvMarker>,
|
||||
|
||||
@@ -1875,6 +1875,14 @@ pub(super) fn virtual_stream(ctx: SessionContext, prepared: Option<PreparedDispl
|
||||
mut cur_display_gen,
|
||||
built_bitrate,
|
||||
) = pipe;
|
||||
// What `enc` was opened against. The capture source can change format/size UNDER this loop with
|
||||
// no client `Reconfigure` at all — the IDD-push capturer re-opens its ring on a confirmed
|
||||
// display-descriptor change (a fullscreen game mode-setting the virtual display, an HDR flip) —
|
||||
// and every backend's `submit` then refuses the frame. Tracked so the loop can FOLLOW the
|
||||
// source (see the guard in the submit path) instead of dying against an error no in-place
|
||||
// encoder reset can fix. Every site below that swaps `enc` re-binds `frame` with it, so this is
|
||||
// always `(frame.format, frame.width, frame.height)` immediately after one.
|
||||
let mut enc_src = (frame.format, frame.width, frame.height);
|
||||
// The display exists now, so the portal has answered: settle the cursor plan against what it
|
||||
// actually negotiated rather than what this session asked for (see `settle_portal_cursor`).
|
||||
// `mut`: every capture-loss rebuild re-runs `create`, hence re-negotiates.
|
||||
@@ -2613,6 +2621,7 @@ pub(super) fn virtual_stream(ctx: SessionContext, prepared: Option<PreparedDispl
|
||||
);
|
||||
cur_mode = new_mode;
|
||||
next = std::time::Instant::now();
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
// H2/H3: the backend may have honored a different mode than requested — KWin caps
|
||||
// a virtual output's refresh, or Windows pf-vdisplay rejects a resolution its
|
||||
// running monitor doesn't advertise and the host falls back to the actual display
|
||||
@@ -2695,6 +2704,7 @@ pub(super) fn virtual_stream(ctx: SessionContext, prepared: Option<PreparedDispl
|
||||
trace.as_ref(),
|
||||
true,
|
||||
) {
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
// The owed AUs died with the old encoder — same bookkeeping as a resize.
|
||||
inflight.clear();
|
||||
last_au_at = std::time::Instant::now();
|
||||
@@ -3388,6 +3398,7 @@ pub(super) fn virtual_stream(ctx: SessionContext, prepared: Option<PreparedDispl
|
||||
interval = new_interval;
|
||||
cur_node_id = new_node_id;
|
||||
cur_display_gen = new_display_gen;
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
// The rebuild re-ran `create`, so the portal answered again — possibly a different
|
||||
// backend's portal (the retarget above), possibly with a different verdict. Settle
|
||||
// the cursor plan against THIS display, exactly as bring-up did: the retarget arm
|
||||
@@ -3650,6 +3661,106 @@ pub(super) fn virtual_stream(ctx: SessionContext, prepared: Option<PreparedDispl
|
||||
// exactly that volume, so host apps already tone-mapped the content into it and the honest
|
||||
// mastering description IS the client's panel. (The IDD capturer only knows the generic
|
||||
// baseline; if the driver ever forwards per-content IDDCX_HDR10_METADATA, prefer that here.)
|
||||
// Follow an AUTONOMOUS source change — one no client `Reconfigure` announced. The IDD-push
|
||||
// capturer re-opens its ring on a confirmed display-descriptor change: a fullscreen game
|
||||
// mode-setting the virtual display (2026-08-22 field report: a 4K60 HEVC session, the game
|
||||
// switched the display to 1080p mid-play), or an HDR flip changing the frame format. The
|
||||
// encoder is the one component that cannot follow that in place (same note as
|
||||
// `try_inplace_resize`), so every `submit` below refuses the frame — and the submit-error
|
||||
// path only rebuilds the encoder IN PLACE, at the SAME configured size, which cannot fix a
|
||||
// size the source has already left. All five resets burn on it and the session ends while
|
||||
// audio keeps running. Reopen at what the source actually delivers instead; the client
|
||||
// learns the new mode from the `Reconfigured` below and its decoder from the opening IDR.
|
||||
if enc_src != (frame.format, frame.width, frame.height) {
|
||||
let actual = delivered_mode(frame.width, frame.height, interval);
|
||||
// Same per-mode pin the client-initiated resize re-resolves: PyroWave's Automatic rate
|
||||
// IS a function of the mode, so carrying the old one across a source-driven mode change
|
||||
// hands it the wrong operating point. H.26x rates are mode-independent (ABR owns them),
|
||||
// and an explicit client rate is never second-guessed.
|
||||
let src_kbps = if bitrate_auto && plan.codec == crate::encode::Codec::PyroWave {
|
||||
resolve_bitrate_kbps_for(plan.codec, 0, &actual, plan.chroma, plan.bit_depth)
|
||||
} else {
|
||||
bitrate_kbps
|
||||
};
|
||||
let opened = crate::encode::open_video(
|
||||
plan.codec,
|
||||
frame.format,
|
||||
frame.width,
|
||||
frame.height,
|
||||
actual.refresh_hz,
|
||||
src_kbps as u64 * 1000,
|
||||
frame.is_cuda(),
|
||||
bit_depth,
|
||||
plan.chroma,
|
||||
plan.cursor_blend,
|
||||
plan.max_slices,
|
||||
)
|
||||
.with_context(|| {
|
||||
format!(
|
||||
"the capture source changed to {}x{} {:?} mid-session and the encoder could not \
|
||||
be reopened at it",
|
||||
frame.width, frame.height, frame.format
|
||||
)
|
||||
});
|
||||
let mut new_enc = match opened {
|
||||
Ok(e) => e,
|
||||
Err(e) => {
|
||||
// Don't spend the session on the FIRST failed open. The mode-set that triggered
|
||||
// this is exactly the kind of event that leaves the driver settling — the same
|
||||
// transient the submit path's backoff exists for ("NVENC session open failing
|
||||
// after a codec switch", 2026-07) — so spend the shared reset budget on it at
|
||||
// the same exponential pace, re-entering this guard each round. The old encoder
|
||||
// is still installed and still mismatched; it simply keeps failing submit until
|
||||
// an open succeeds or the budget runs out.
|
||||
encoder_resets += 1;
|
||||
if encoder_resets > MAX_ENCODER_RESETS {
|
||||
return Err(e).context("encoder reopen at the source's new mode");
|
||||
}
|
||||
let backoff = std::cmp::max(
|
||||
interval,
|
||||
std::time::Duration::from_millis(100u64 << (encoder_resets - 1).min(4)),
|
||||
);
|
||||
tracing::warn!(error = %format!("{e:#}"), reset = encoder_resets,
|
||||
max = MAX_ENCODER_RESETS,
|
||||
"reopening the encoder at the source's new mode failed — retrying");
|
||||
next = std::time::Instant::now() + backoff;
|
||||
std::thread::sleep(backoff);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
if let Some(c) = plan.wire_chunk {
|
||||
new_enc.set_wire_chunking(c);
|
||||
}
|
||||
// A rebuilt encoder starts with the ring bound unset — re-report it, as every other
|
||||
// rebuild site does, or an in-place backend can encode a texture the capturer has
|
||||
// already rotated and overwritten.
|
||||
new_enc.set_input_ring_depth(capturer.pipeline_depth().max(1));
|
||||
tracing::info!(
|
||||
from = %format!("{}x{} {:?}", enc_src.1, enc_src.2, enc_src.0),
|
||||
to = %format!("{}x{} {:?}", frame.width, frame.height, frame.format),
|
||||
"the capture source changed mode mid-session with no client reconfigure — reopened \
|
||||
the encoder at the delivered size"
|
||||
);
|
||||
enc = new_enc;
|
||||
enc_src = (frame.format, frame.width, frame.height);
|
||||
adopt_built_bitrate(&mut bitrate_kbps, src_kbps, &live_bitrate, &retarget_tx);
|
||||
// The owed AUs died with the old encoder — same bookkeeping as a resize.
|
||||
inflight.clear();
|
||||
last_au_at = std::time::Instant::now();
|
||||
encoder_resets = 0;
|
||||
// A fresh encoder opens on an IDR — anchor the cooldown.
|
||||
last_forced_idr = Some(std::time::Instant::now());
|
||||
// The client's mode slot still says the old size, and its stats/aspect follow it.
|
||||
// Publish what it is really decoding now, exactly as an accepted resize does.
|
||||
live_mode.store(
|
||||
pack_mode(actual.width, actual.height, actual.refresh_hz),
|
||||
Ordering::Relaxed,
|
||||
);
|
||||
let _ = reconfig_result_tx.send(Reconfigured {
|
||||
accepted: true,
|
||||
mode: actual,
|
||||
});
|
||||
}
|
||||
let hdr_meta = capturer.hdr_meta().map(|m| client_hdr.unwrap_or(m));
|
||||
enc.set_hdr_meta(hdr_meta);
|
||||
let mut resend_meta = hdr_meta != last_hdr_meta;
|
||||
|
||||
@@ -87,6 +87,26 @@ pub fn resolve(pid: u32) -> Option<ProcRef> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Short names for the processes a lease adopted, in `procs` order.
|
||||
///
|
||||
/// Diagnostics only — nothing decides anything on these, and they are deliberately not part of
|
||||
/// [`ProcRef`], which is compared for equality. They exist because a launch that adopted the game
|
||||
/// and a launch that adopted a *pre-launch* tree logged identically (`procs=1`), which is what left
|
||||
/// the 2026-08-22 field report unclosable from its log: the one question worth asking of that line
|
||||
/// is which process the lease latched onto.
|
||||
pub fn names(procs: &[ProcRef]) -> Vec<String> {
|
||||
#[cfg(any(target_os = "linux", windows))]
|
||||
{
|
||||
let scanner = Scanner::system();
|
||||
procs.iter().map(|p| scanner.name_of(*p)).collect()
|
||||
}
|
||||
#[cfg(not(any(target_os = "linux", windows)))]
|
||||
{
|
||||
let _ = procs;
|
||||
Vec::new()
|
||||
}
|
||||
}
|
||||
|
||||
/// An out-of-band opinion on whether a spec's game is still running, independent of the process scan.
|
||||
///
|
||||
/// Consulted **only to veto** declaring a game gone — never to declare it running, and never as the
|
||||
|
||||
@@ -126,6 +126,14 @@ impl Scanner {
|
||||
Some(ProcRef { pid, start })
|
||||
}
|
||||
|
||||
/// This process's `comm` — its short name, as `ps` shows it. Diagnostics only (see
|
||||
/// [`super::names`]); `?` for a process that has already gone, which is routine.
|
||||
pub fn name_of(&self, p: ProcRef) -> String {
|
||||
std::fs::read_to_string(self.root.join(p.pid.to_string()).join("comm"))
|
||||
.map(|s| s.trim().to_string())
|
||||
.unwrap_or_else(|_| "?".into())
|
||||
}
|
||||
|
||||
/// Which of `procs` are still the same live processes — pid present **and** start time unchanged,
|
||||
/// so a recycled pid is never reported alive (rule 2).
|
||||
pub fn alive(&self, procs: &[ProcRef]) -> Vec<ProcRef> {
|
||||
@@ -180,16 +188,29 @@ impl Scanner {
|
||||
if let Some(tok) = steam_tok {
|
||||
// Both tokens together, exact-matched, so `AppId=57` never satisfies appid 570 and
|
||||
// Steam's own (non-reaper) helper steps aren't mistaken for the game.
|
||||
//
|
||||
// …with one exception, because the reaper is *not* only the game's: Steam wraps its
|
||||
// shader pre-caching for a title in the same `SteamLaunch AppId=<appid>` reaper it
|
||||
// wraps the game in, so that job satisfies this recipe exactly while the game has
|
||||
// not started yet. Adopting it points the lease at a tree that exits when the
|
||||
// compile finishes, which reads as the game exiting — on Linux that dropped a
|
||||
// Rocket League stream 10 s into a launch, mid-"Processing Vulkan shaders", and the
|
||||
// player had to launch a second time to get a session that stayed up (field report
|
||||
// 2026-08-22). The payload names itself: `fossilize_replay` is Steam's replayer and
|
||||
// is never a game.
|
||||
let mut launch = false;
|
||||
let mut appid = false;
|
||||
let mut shader = false;
|
||||
for arg in cmdline.split(|&b| b == 0) {
|
||||
if arg == b"SteamLaunch" {
|
||||
launch = true;
|
||||
} else if arg == tok.as_bytes() {
|
||||
appid = true;
|
||||
} else if program_name(arg) == b"fossilize_replay" {
|
||||
shader = true;
|
||||
}
|
||||
}
|
||||
if launch && appid {
|
||||
if launch && appid && !shader {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -247,6 +268,15 @@ impl Scanner {
|
||||
}
|
||||
}
|
||||
|
||||
/// The last `/`-separated component of an argv entry — the program's own name, when the entry is a
|
||||
/// path to one. Bytes rather than `str` because an argv entry is not required to be UTF-8.
|
||||
fn program_name(arg: &[u8]) -> &[u8] {
|
||||
match arg.iter().rposition(|&b| b == b'/') {
|
||||
Some(i) => &arg[i + 1..],
|
||||
None => arg,
|
||||
}
|
||||
}
|
||||
|
||||
/// Read a `/proc` blob with a hard size cap (see [`MAX_PROC_BLOB`]). `None` when the process vanished
|
||||
/// or the file is unreadable — both routine during a scan.
|
||||
fn read_capped(path: &Path) -> Option<Vec<u8>> {
|
||||
@@ -472,6 +502,42 @@ mod tests {
|
||||
assert_eq!(pids(s.find(&DetectSpec::steam(57), None)), vec![31]);
|
||||
}
|
||||
|
||||
/// The 2026-08-22 field report: Steam's **shader pre-caching** runs under the game's own
|
||||
/// `SteamLaunch AppId=` reaper, so it satisfies the appid recipe while the game has not started.
|
||||
///
|
||||
/// Adopting it is what dropped a Rocket League stream 10 s into a launch — the lease called that
|
||||
/// tree the game, and its exit (the compile finishing) the game exiting. The reaper's payload is
|
||||
/// the whole tell, and it is only ever Steam's replayer.
|
||||
#[test]
|
||||
fn steam_shader_pre_caching_is_not_the_game() {
|
||||
let td = fake_proc_root(
|
||||
1000.0,
|
||||
&[
|
||||
// The shader job for this very appid — the game is still being brought up.
|
||||
FakeProc::new(35, 50_000).cmdline(&[
|
||||
"/home/p/.steam/ubuntu12_32/reaper",
|
||||
"SteamLaunch",
|
||||
"AppId=252950",
|
||||
"--",
|
||||
"/home/p/.steam/steamapps/common/SteamLinuxRuntime/fossilize_replay",
|
||||
"/home/p/.steam/steamapps/shadercache/252950/fozpipelinesv6/steamapprun_pipeline_cache.foz",
|
||||
]),
|
||||
// The game itself, same appid, same reaper. This one IS the game.
|
||||
FakeProc::new(36, 50_000).cmdline(&[
|
||||
"/home/p/.steam/ubuntu12_32/reaper",
|
||||
"SteamLaunch",
|
||||
"AppId=252950",
|
||||
"--",
|
||||
"/home/p/.steam/steamapps/common/Proton/proton",
|
||||
"waitforexitandrun",
|
||||
"/home/p/.steam/steamapps/common/rocketleague/RocketLeague.exe",
|
||||
]),
|
||||
],
|
||||
);
|
||||
let s = scanner(td.path());
|
||||
assert_eq!(pids(s.find(&DetectSpec::steam(252_950), None)), vec![36]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn matches_env_marker_by_exact_value_or_presence() {
|
||||
let td = fake_proc_root(
|
||||
|
||||
@@ -120,6 +120,14 @@ impl Scanner {
|
||||
Some(ProcRef { pid, start })
|
||||
}
|
||||
|
||||
/// This process's image file name. Diagnostics only (see [`super::names`]); `?` for a process
|
||||
/// that has already gone or cannot be opened, which is routine.
|
||||
pub fn name_of(&self, p: ProcRef) -> String {
|
||||
process_start_and_image(p.pid)
|
||||
.and_then(|(_, image)| image.file_name().map(|n| n.to_string_lossy().into_owned()))
|
||||
.unwrap_or_else(|| "?".into())
|
||||
}
|
||||
|
||||
/// Which of `procs` are still the same live processes — pid present **and** creation time
|
||||
/// unchanged, so a recycled pid is never reported alive (rule 2). Windows reuses pids briskly, so
|
||||
/// this check is what makes signalling a remembered pid safe at all.
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
# shellcheck shell=bash
|
||||
# Does this box already have every Flathub dep a flatpak manifest declares?
|
||||
# bash scripts/ci/flatpak-deps-present.sh <manifest.yml> -> exit 0 = yes, 1 = no
|
||||
# bash scripts/ci/flatpak-deps-present.sh --self-test -> run the asserts below
|
||||
#
|
||||
# WHY THIS EXISTS: flatpak.yml used to prefetch deps with `flatpak-builder --install-deps-only`,
|
||||
# which does NOT mean "install what is missing". builder_manifest_install_dep() branches on
|
||||
# `flatpak info --show-commit <ref>` succeeding and runs `flatpak update` for every dep that IS
|
||||
# installed (a failed update is fatal there — it never falls back to install) — and
|
||||
# ci/flatpak-ci.Dockerfile bakes the whole runtime set, so on a healthy run that flag did nothing
|
||||
# except make the build depend on Flathub being up at that minute. On 2026-08-22 it took the job
|
||||
# down: dl.flathub.org returned HTTP 404 for one .filez object of the then-current
|
||||
# rust-stable//25.08 commit, identically on all 10 retry.sh attempts (~9 min), and flatpak-builder
|
||||
# segfaulted on its own error path (rc=139) so the retry wrapper could not tell a dead end from a
|
||||
# blip. Nothing about the build wanted that newer commit: the manifest pins a runtime VERSION, not
|
||||
# a commit, and the baked one satisfies it.
|
||||
#
|
||||
# So the workflow asks this first and only reaches for Flathub on a real miss.
|
||||
#
|
||||
# FAILS OPEN, deliberately: an unreadable/unexpected manifest reports "not present" (1), so the
|
||||
# caller does the full install. Silently skipping the install on a manifest we stopped
|
||||
# understanding is how you build against the wrong runtime.
|
||||
set -uo pipefail
|
||||
|
||||
deps_present() {
|
||||
local manifest="$1" runtime rt_ver sdk exts e
|
||||
|
||||
runtime=$(sed -n 's/^runtime: *//p' "$manifest" | head -1)
|
||||
rt_ver=$(sed -n 's/^runtime-version: *//p' "$manifest" | tr -d "\"'" | head -1)
|
||||
sdk=$(sed -n 's/^sdk: *//p' "$manifest" | head -1)
|
||||
exts=$(sed -n '/^sdk-extensions:/,/^[^ #-]/p' "$manifest" | sed -n 's/^ *- *//p')
|
||||
|
||||
[ -n "$runtime" ] && [ -n "$rt_ver" ] && [ -n "$sdk" ] && [ -n "$exts" ] || return 1
|
||||
|
||||
flatpak info --user "$runtime//$rt_ver" >/dev/null 2>&1 || return 1
|
||||
flatpak info --user "$sdk//$rt_ver" >/dev/null 2>&1 || return 1
|
||||
# Extensions are checked for PRESENCE, not version: flatpak-builder resolves their version from
|
||||
# the SDK's own metadata (it prints "Dependency Extension: … 25.08"), never from the manifest.
|
||||
# Any bump that moves them moves runtime-version too, which the two checks above already catch.
|
||||
for e in $exts; do
|
||||
flatpak info --user "$e" >/dev/null 2>&1 || return 1
|
||||
done
|
||||
}
|
||||
|
||||
self_test() {
|
||||
local rc fails=0 full
|
||||
# NOT `local`: the EXIT trap fires after this function has returned.
|
||||
SELFTEST_TMP=$(mktemp -d) || return 1
|
||||
trap 'rm -rf "$SELFTEST_TMP"' EXIT
|
||||
local tmp="$SELFTEST_TMP"
|
||||
|
||||
cat > "$tmp/ok.yml" <<'YML'
|
||||
runtime: org.gnome.Platform
|
||||
runtime-version: '50'
|
||||
sdk: org.gnome.Sdk
|
||||
sdk-extensions:
|
||||
- org.freedesktop.Sdk.Extension.rust-stable
|
||||
- org.freedesktop.Sdk.Extension.llvm20
|
||||
command: punktfunk-client
|
||||
YML
|
||||
# A manifest this script cannot read (the fail-open case).
|
||||
printf 'app-id: io.unom.Punktfunk\n' > "$tmp/unparseable.yml"
|
||||
|
||||
# Stub `flatpak`: $INSTALLED is the newline-separated set of refs it admits to having.
|
||||
mkdir -p "$tmp/bin"
|
||||
cat > "$tmp/bin/flatpak" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
# only `flatpak info --user <ref>` is exercised here
|
||||
# args are: info --user <ref>
|
||||
[ "$1" = info ] || exit 0
|
||||
printf '%s\n' "$INSTALLED" | grep -qxF "$3"
|
||||
STUB
|
||||
chmod +x "$tmp/bin/flatpak"
|
||||
PATH="$tmp/bin:$PATH"
|
||||
|
||||
check() { # <expected rc> <label> <installed set> <manifest>
|
||||
INSTALLED="$3" deps_present "$4"; rc=$?
|
||||
if [ "$rc" != "$1" ]; then
|
||||
echo "FAIL: $2 (expected rc=$1, got $rc)" >&2; fails=$((fails + 1))
|
||||
else
|
||||
echo "ok: $2"
|
||||
fi
|
||||
}
|
||||
|
||||
full='org.gnome.Platform//50
|
||||
org.gnome.Sdk//50
|
||||
org.freedesktop.Sdk.Extension.rust-stable
|
||||
org.freedesktop.Sdk.Extension.llvm20'
|
||||
|
||||
check 0 "everything baked -> skip Flathub" "$full" "$tmp/ok.yml"
|
||||
check 1 "cold box -> install" "" "$tmp/ok.yml"
|
||||
check 1 "runtime missing -> install" "${full/org.gnome.Platform\/\/50/x}" "$tmp/ok.yml"
|
||||
check 1 "sdk missing -> install" "${full/org.gnome.Sdk\/\/50/x}" "$tmp/ok.yml"
|
||||
# The regression that started all this: llvm20 fine, rust-stable not.
|
||||
check 1 "one sdk-extension missing -> install" "${full/*.rust-stable/x}" "$tmp/ok.yml"
|
||||
# A runtime installed at ANOTHER version must not pass just because the name matches.
|
||||
check 1 "runtime at the wrong version" 'org.gnome.Platform//51
|
||||
org.gnome.Sdk//51
|
||||
org.freedesktop.Sdk.Extension.rust-stable
|
||||
org.freedesktop.Sdk.Extension.llvm20' "$tmp/ok.yml"
|
||||
check 1 "unreadable manifest -> fail open" "$full" "$tmp/unparseable.yml"
|
||||
|
||||
[ "$fails" = 0 ] || { echo "$fails check(s) failed" >&2; return 1; }
|
||||
echo "all checks passed"
|
||||
}
|
||||
|
||||
case "${1:---help}" in
|
||||
--self-test) self_test ;;
|
||||
--help|-h) sed -n '2,4p' "$0"; exit 2 ;;
|
||||
*) deps_present "$1" ;;
|
||||
esac
|
||||
Reference in New Issue
Block a user