Compare commits
113
Commits
+24
-30
@@ -6,17 +6,12 @@
|
||||
#
|
||||
# The plugin backend is PURE PYTHON (clients/decky/main.py — no compiled binary), so we do NOT
|
||||
# need the Decky CLI (which requires Docker + rust-nightly only to compile native backends).
|
||||
# We build the frontend with pnpm and assemble the store-layout zip by hand:
|
||||
#
|
||||
# punktfunk.zip
|
||||
# punktfunk/ <- single top-level dir == plugin.json "name"
|
||||
# plugin.json [required]
|
||||
# package.json [required; CI stamps "version" — Decky reads the installed version here]
|
||||
# main.py [required: python backend]
|
||||
# dist/index.js [required: rollup output]
|
||||
# update.json [CI-baked {channel, manifest}: where the plugin's self-update check polls]
|
||||
# README.md (recommended)
|
||||
# LICENSE [required by the plugin store]
|
||||
# We build the frontend with pnpm and stage the store-layout tree with the SAME script local
|
||||
# builds use (clients/decky/scripts/package.sh) — the plugin's file list lives in exactly ONE
|
||||
# place, so a file added there (bin/, assets/, controller_config/, …) can never be silently
|
||||
# missing from the published build. (Hand-assembling the zip here is how the shipped plugin
|
||||
# lost the shortcut artwork + Steam Input layout for a while.) CI only adds `update.json` on
|
||||
# top: the {channel, manifest} pointer the plugin's self-update check polls.
|
||||
#
|
||||
# SELF-UPDATE (no Decky store): alongside the zip we also publish a tiny per-channel
|
||||
# `manifest.json` ({version, artifact=<immutable per-version zip URL>, sha256}). The installed
|
||||
@@ -90,28 +85,27 @@ jobs:
|
||||
- name: Assemble store-layout zip
|
||||
working-directory: ${{ gitea.workspace }}
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends zip >/dev/null
|
||||
STAGE="$RUNNER_TEMP/decky"
|
||||
DEST="$STAGE/$PLUGIN"
|
||||
rm -rf "$STAGE"; mkdir -p "$DEST/dist" "$DEST/bin"
|
||||
cp clients/decky/plugin.json "$DEST/"
|
||||
cp clients/decky/package.json "$DEST/"
|
||||
cp clients/decky/main.py "$DEST/"
|
||||
cp clients/decky/dist/index.js "$DEST/dist/"
|
||||
cp clients/decky/README.md "$DEST/"
|
||||
# The stream-launch wrapper (target of the Steam shortcut); keep it executable
|
||||
# (runner_info() also re-chmods at runtime in case the zip/extract drops the bit).
|
||||
cp clients/decky/bin/punktfunkrun.sh "$DEST/bin/"
|
||||
chmod 0755 "$DEST/bin/punktfunkrun.sh"
|
||||
# Store requires a LICENSE in the plugin root; the project is MIT OR Apache-2.0.
|
||||
cp LICENSE-MIT "$DEST/LICENSE"
|
||||
# Self-update channel pointer the backend reads (main.py check_update). It points at
|
||||
# THIS channel's manifest.json (published below); that manifest in turn points at the
|
||||
# immutable per-version zip, so its sha256 stays valid across future alias re-uploads.
|
||||
# node:22-bookworm ships python3 (a package.sh dep) but not zip; install both anyway
|
||||
# so an image change can't silently break the build.
|
||||
apt-get update && apt-get install -y --no-install-recommends zip python3 >/dev/null
|
||||
# Stage the canonical plugin tree (dist/, main.py, bin/, assets/, controller_config/,
|
||||
# LICENSE, …) with the same script local/sideload builds use — see the header comment.
|
||||
# Runs AFTER the version stamp, so the staged package.json carries $VERSION.
|
||||
bash clients/decky/scripts/package.sh
|
||||
DEST="clients/decky/out/$PLUGIN"
|
||||
# CI-only addition: the self-update channel pointer the backend reads (main.py
|
||||
# check_update). It points at THIS channel's manifest.json (published below); that
|
||||
# manifest in turn points at the immutable per-version zip, so its sha256 stays valid
|
||||
# across future alias re-uploads.
|
||||
printf '{"channel":"%s","manifest":"%s/%s/manifest.json"}\n' "$ALIAS" "$BASE" "$ALIAS" > "$DEST/update.json"
|
||||
( cd "$STAGE" && zip -r "$RUNNER_TEMP/punktfunk.zip" "$PLUGIN" )
|
||||
( cd clients/decky/out && zip -r "$RUNNER_TEMP/punktfunk.zip" "$PLUGIN" )
|
||||
ls -lh "$RUNNER_TEMP/punktfunk.zip"
|
||||
unzip -l "$RUNNER_TEMP/punktfunk.zip"
|
||||
# Backstop against packaging drift: the runtime-loaded pieces MUST be in the zip.
|
||||
for f in main.py dist/index.js bin/punktfunkrun.sh assets/grid.png \
|
||||
controller_config/punktfunk.vdf update.json; do
|
||||
unzip -l "$RUNNER_TEMP/punktfunk.zip" "$PLUGIN/$f" >/dev/null || { echo "MISSING $f" >&2; exit 1; }
|
||||
done
|
||||
# The update manifest the plugin polls: the immutable per-version artifact + its
|
||||
# sha256 (Decky's installer verifies the download against this hash, aborting on
|
||||
# mismatch — so it MUST be the per-version URL, never the mutable alias).
|
||||
|
||||
@@ -73,8 +73,34 @@ jobs:
|
||||
# sufficient — the Tooling step's dnf install pulls a systemd package upgrade whose RPM
|
||||
# trigger re-runs authselect and regenerates this file, undoing the fix. It's reapplied
|
||||
# there, right before the first `flatpak` network call.
|
||||
- name: Fix container DNS (drop nss-resolve — no systemd-resolved in CI)
|
||||
run: sed -i 's/resolve \[!UNAVAIL=return\] //' /etc/nsswitch.conf
|
||||
- name: Fix container DNS (drop nss-resolve, resolve over TCP)
|
||||
run: |
|
||||
sed -i 's/resolve \[!UNAVAIL=return\] //' /etc/nsswitch.conf
|
||||
# Resolve over TCP instead of UDP. The documented root cause of the flathub
|
||||
# bootstrap failures (investigated 2026-07-11, see the Tooling step) is this box's
|
||||
# Docker embedded resolver at 127.0.0.11 DROPPING UDP lookups while the shared
|
||||
# runner fleet is saturated — a datagram nobody retransmits, so the lookup just
|
||||
# times out. The answer then was to widen retry.sh's budget to 10 attempts (~9 min),
|
||||
# which is enough to outlast a main push's ~8-workflow fan-out but NOT a TAG push's
|
||||
# 13: v0.15.0 (twice) and v0.16.0 each burned all 10 attempts and failed the job,
|
||||
# each needing a manual re-run.
|
||||
#
|
||||
# `use-vc` makes glibc use TCP, where the kernel retransmits and the query cannot be
|
||||
# silently lost under load. Same resolver, same search path — only the transport
|
||||
# changes, so internal names (git.unom.io) resolve exactly as before; deliberately
|
||||
# NO extra nameservers, which would risk answering an internal name from a public
|
||||
# resolver. Docker's embedded DNS serves TCP on 127.0.0.11:53 as well as UDP.
|
||||
# retry.sh stays as the backstop for genuine upstream blips.
|
||||
#
|
||||
# Non-fatal: Docker bind-mounts /etc/resolv.conf and can present it read-only, and a
|
||||
# DNS tuning that cannot be applied must not be what fails the release build — that
|
||||
# would trade an occasional re-run for a hard stop. Falling back to UDP just restores
|
||||
# today's behaviour, which retry.sh already covers.
|
||||
if ! grep -q '^options .*use-vc' /etc/resolv.conf 2>/dev/null; then
|
||||
echo 'options use-vc timeout:3 attempts:3' >> /etc/resolv.conf \
|
||||
|| echo "::warning::could not set use-vc (read-only resolv.conf?); staying on UDP"
|
||||
fi
|
||||
cat /etc/resolv.conf || true
|
||||
|
||||
# fedora:43 has no node, but actions/checkout (a JS action) needs it. A plain `run:` step
|
||||
# executes via the container shell (no node needed), so install node BEFORE checkout.
|
||||
|
||||
@@ -0,0 +1,69 @@
|
||||
# Publish the plugin framework (@punktfunk/plugin-kit) to the Gitea npm registry
|
||||
# (https://git.unom.io/api/packages/unom/npm/).
|
||||
#
|
||||
# Trigger: push a tag `plugin-kit-vX.Y.Z` (must equal plugin-kit/package.json "version"),
|
||||
# or run manually. Versions independently of the app's `v*` and the SDK's `sdk-v*` tags.
|
||||
#
|
||||
# The kit's devDependency on @punktfunk/host is `file:../sdk`, so the SDK's dist must be
|
||||
# built BEFORE the kit's `bun install` copies it.
|
||||
#
|
||||
# Auth: REGISTRY_TOKEN — the same repo Actions secret sdk-publish.yml uses.
|
||||
name: plugin-kit-publish
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ['plugin-kit-v*']
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-24.04
|
||||
container:
|
||||
image: oven/bun:1
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
# oven/bun's slim base ships neither git, a CA bundle, nor node — actions/checkout's HTTPS
|
||||
# fetch needs git + ca-certificates, and the version-guard step below uses node.
|
||||
- name: Install git + node + CA certs
|
||||
run: apt-get update && apt-get install -y --no-install-recommends ca-certificates git nodejs
|
||||
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Build the SDK (file:../sdk dependency source)
|
||||
working-directory: sdk
|
||||
run: |
|
||||
bun install --frozen-lockfile --ignore-scripts
|
||||
bun run build
|
||||
|
||||
- name: Install dependencies
|
||||
working-directory: plugin-kit
|
||||
run: bun install --frozen-lockfile --ignore-scripts
|
||||
|
||||
- name: Typecheck
|
||||
working-directory: plugin-kit
|
||||
run: bun run typecheck
|
||||
|
||||
- name: Test
|
||||
working-directory: plugin-kit
|
||||
run: bun test
|
||||
|
||||
- name: Build (dist/ JS + .d.ts + theme.css)
|
||||
working-directory: plugin-kit
|
||||
run: bun run build
|
||||
|
||||
- name: Tag matches package version
|
||||
if: startsWith(github.ref, 'refs/tags/')
|
||||
working-directory: plugin-kit
|
||||
run: |
|
||||
TAG="${GITHUB_REF_NAME#plugin-kit-v}"
|
||||
PKG="$(node -p "require('./package.json').version")"
|
||||
test "$TAG" = "$PKG" || { echo "tag $GITHUB_REF_NAME does not match package version $PKG"; exit 1; }
|
||||
|
||||
- name: Publish to Gitea registry
|
||||
working-directory: plugin-kit
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.REGISTRY_TOKEN }}
|
||||
run: |
|
||||
test -n "$NODE_AUTH_TOKEN" || { echo "REGISTRY_TOKEN secret is empty"; exit 1; }
|
||||
printf '//git.unom.io/api/packages/unom/npm/:_authToken=%s\n' "$NODE_AUTH_TOKEN" >> .npmrc
|
||||
bun publish
|
||||
@@ -149,6 +149,26 @@ jobs:
|
||||
# inherits this from the env during the xcframework build).
|
||||
echo "CMAKE_POLICY_VERSION_MINIMUM=3.5" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Pin + prune Xcode DerivedData
|
||||
# Without -derivedDataPath, xcodebuild derives its DerivedData directory name from the
|
||||
# PROJECT'S ABSOLUTE PATH — and act_runner rotates its workspace
|
||||
# (~/.cache/act/<hash>/hostexecutor), so each rotation minted a brand new ~760 MB tree
|
||||
# under ~/Library that nothing ever collected. 31 of them piled up in three days
|
||||
# (~32 GB with the shared ModuleCache), filled the runner's boot volume, and failed
|
||||
# v0.16.0's xcframework build with "No space left on device". Pinning one path makes the
|
||||
# tree REUSED instead of multiplied — it also keeps the module cache warm between runs.
|
||||
run: |
|
||||
DD="$HOME/ci/derived-data/release"
|
||||
mkdir -p "$DD"
|
||||
echo "DERIVED_DATA=$DD" >> "$GITHUB_ENV"
|
||||
# Safety net for trees the pin does not own: the legacy per-path ones from before this
|
||||
# change, and anything another job leaves in the default root. Untouched for a week ⇒ gone.
|
||||
if [ -d "$HOME/Library/Developer/Xcode/DerivedData" ]; then
|
||||
find "$HOME/Library/Developer/Xcode/DerivedData" -mindepth 1 -maxdepth 1 \
|
||||
-mtime +7 -exec rm -rf {} + 2>/dev/null || true
|
||||
fi
|
||||
echo "disk after prune:"; df -h /System/Volumes/Data | tail -1
|
||||
|
||||
- name: Build PunktfunkCore.xcframework (mac + iOS + tvOS)
|
||||
# tvOS is a tier-3 target (nightly -Zbuild-std): slow on the first build, then cached on
|
||||
# the self-hosted runner. Built on canary too so the tvOS archive/upload below runs on the
|
||||
@@ -176,6 +196,7 @@ jobs:
|
||||
-project "$PROJECT" -scheme Punktfunk \
|
||||
-destination 'generic/platform=macOS' \
|
||||
-archivePath "$RUNNER_TEMP/Punktfunk-macos.xcarchive" \
|
||||
-derivedDataPath "$DERIVED_DATA" \
|
||||
-skipMacroValidation -skipPackagePluginValidation \
|
||||
MARKETING_VERSION="$VERSION" CURRENT_PROJECT_VERSION="$BUILD_NUM" \
|
||||
CODE_SIGNING_ALLOWED=NO
|
||||
@@ -273,6 +294,7 @@ jobs:
|
||||
-project "$PROJECT" -scheme Punktfunk \
|
||||
-destination 'generic/platform=macOS' \
|
||||
-archivePath "$RUNNER_TEMP/Punktfunk-macos-appstore.xcarchive" \
|
||||
-derivedDataPath "$DERIVED_DATA" \
|
||||
-skipMacroValidation -skipPackagePluginValidation \
|
||||
-allowProvisioningUpdates \
|
||||
-authenticationKeyPath "$RUNNER_TEMP/asc.p8" \
|
||||
@@ -336,6 +358,7 @@ jobs:
|
||||
-project "$PROJECT" -scheme Punktfunk-iOS \
|
||||
-destination 'generic/platform=iOS' \
|
||||
-archivePath "$RUNNER_TEMP/Punktfunk-ios.xcarchive" \
|
||||
-derivedDataPath "$DERIVED_DATA" \
|
||||
-skipMacroValidation -skipPackagePluginValidation \
|
||||
-allowProvisioningUpdates \
|
||||
-authenticationKeyPath "$RUNNER_TEMP/asc.p8" \
|
||||
@@ -394,6 +417,7 @@ jobs:
|
||||
-project "$PROJECT" -scheme Punktfunk-tvOS \
|
||||
-destination 'generic/platform=tvOS' \
|
||||
-archivePath "$RUNNER_TEMP/Punktfunk-tvos.xcarchive" \
|
||||
-derivedDataPath "$DERIVED_DATA" \
|
||||
-skipMacroValidation -skipPackagePluginValidation \
|
||||
-allowProvisioningUpdates \
|
||||
-authenticationKeyPath "$RUNNER_TEMP/asc.p8" \
|
||||
|
||||
Generated
+67
-27
@@ -656,6 +656,30 @@ version = "0.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724"
|
||||
|
||||
[[package]]
|
||||
name = "chacha20"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c3613f74bd2eac03dad61bd53dbe620703d4371614fe0bc3b9f04dd36fe4e818"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"cipher",
|
||||
"cpufeatures",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "chacha20poly1305"
|
||||
version = "0.10.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "10cd79432192d1c0f4e1a0fef9527696cc039165d729fb41b3f4f4f354c2dc35"
|
||||
dependencies = [
|
||||
"aead",
|
||||
"chacha20",
|
||||
"cipher",
|
||||
"poly1305",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ciborium"
|
||||
version = "0.2.2"
|
||||
@@ -691,6 +715,7 @@ checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
|
||||
dependencies = [
|
||||
"crypto-common",
|
||||
"inout",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2159,7 +2184,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "latency-probe"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
|
||||
[[package]]
|
||||
name = "lazy_static"
|
||||
@@ -2264,7 +2289,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "libvpl-sys"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"bindgen",
|
||||
"cmake",
|
||||
@@ -2299,7 +2324,7 @@ checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
||||
|
||||
[[package]]
|
||||
name = "loss-harness"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"punktfunk-core",
|
||||
]
|
||||
@@ -2788,7 +2813,7 @@ checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220"
|
||||
|
||||
[[package]]
|
||||
name = "pf-capture"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ashpd",
|
||||
@@ -2808,7 +2833,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-client-core"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ash",
|
||||
@@ -2832,7 +2857,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-clipboard"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ashpd",
|
||||
@@ -2850,7 +2875,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-console-ui"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ash",
|
||||
@@ -2871,7 +2896,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-encode"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ash",
|
||||
@@ -2881,6 +2906,7 @@ dependencies = [
|
||||
"libvpl-sys",
|
||||
"nvidia-video-codec-sdk",
|
||||
"openh264",
|
||||
"pf-capture",
|
||||
"pf-frame",
|
||||
"pf-gpu",
|
||||
"pf-host-config",
|
||||
@@ -2894,7 +2920,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-ffvk"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"ash",
|
||||
"bindgen",
|
||||
@@ -2903,7 +2929,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-frame"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"libc",
|
||||
@@ -2915,7 +2941,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-gpu"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"pf-host-config",
|
||||
@@ -2929,11 +2955,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-host-config"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
|
||||
[[package]]
|
||||
name = "pf-inject"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ashpd",
|
||||
@@ -2961,14 +2987,14 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-paths"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pf-presenter"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ash",
|
||||
@@ -2983,7 +3009,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-vdisplay"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ashpd",
|
||||
@@ -3013,7 +3039,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-win-display"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"pf-paths",
|
||||
@@ -3025,7 +3051,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pf-zerocopy"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ash",
|
||||
@@ -3136,6 +3162,17 @@ dependencies = [
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "poly1305"
|
||||
version = "0.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8159bd90725d2df49889a078b54f4f79e87f1f8a8444194cdca81d38f5393abf"
|
||||
dependencies = [
|
||||
"cpufeatures",
|
||||
"opaque-debug",
|
||||
"universal-hash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "polyval"
|
||||
version = "0.6.2"
|
||||
@@ -3221,7 +3258,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-client-android"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"android_logger",
|
||||
"jni",
|
||||
@@ -3237,7 +3274,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-client-linux"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"async-channel",
|
||||
@@ -3253,7 +3290,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-client-session"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"pf-client-core",
|
||||
@@ -3268,7 +3305,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-client-windows"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"async-channel",
|
||||
"ffmpeg-next",
|
||||
@@ -3287,11 +3324,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-core"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"bytes",
|
||||
"cbindgen",
|
||||
"chacha20poly1305",
|
||||
"criterion",
|
||||
"fec-rs",
|
||||
"hmac",
|
||||
@@ -3318,7 +3356,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-host"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"aes-gcm",
|
||||
@@ -3363,11 +3401,13 @@ dependencies = [
|
||||
"rand 0.8.6",
|
||||
"rcgen",
|
||||
"reis",
|
||||
"ring",
|
||||
"roxmltree",
|
||||
"rsa",
|
||||
"rusqlite",
|
||||
"rustls",
|
||||
"rusty_enet",
|
||||
"semver",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"sha2",
|
||||
@@ -3400,7 +3440,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-probe"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"mdns-sd",
|
||||
@@ -3414,7 +3454,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "punktfunk-tray"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ksni",
|
||||
@@ -3437,7 +3477,7 @@ checksum = "d55d956fa96f5ec02be2e13af0e20391a5aa83d6a074e3ad368959d0fab299ea"
|
||||
|
||||
[[package]]
|
||||
name = "pyrowave-sys"
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
dependencies = [
|
||||
"bindgen",
|
||||
"cmake",
|
||||
|
||||
+1
-1
@@ -48,7 +48,7 @@ exclude = [
|
||||
ndk = { path = "clients/android/native/vendor/ndk" }
|
||||
|
||||
[workspace.package]
|
||||
version = "0.15.0"
|
||||
version = "0.17.2"
|
||||
edition = "2021"
|
||||
rust-version = "1.82"
|
||||
license = "MIT OR Apache-2.0"
|
||||
|
||||
+1133
-1
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,107 @@
|
||||
package io.unom.punktfunk
|
||||
|
||||
import android.content.ClipData
|
||||
import android.content.ClipboardManager
|
||||
import android.content.Context
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import io.unom.punktfunk.kit.NativeBridge
|
||||
|
||||
/**
|
||||
* Text clipboard sync for the active session (the desktop-client model, text-only v1):
|
||||
* * **Device → host**: a local copy (the primary-clip listener, plus one probe at start) is
|
||||
* announced as a lazy offer — the text crosses only when the host actually pastes (a
|
||||
* `fetch:` event, answered with the clipboard's current content).
|
||||
* * **Host → device**: a host copy arrives as an `offer:` event and is fetched eagerly into
|
||||
* the system clipboard (Android apps can't lazily materialize a paste from the network
|
||||
* without a content-provider round-trip that isn't worth it here).
|
||||
*
|
||||
* Loop guard: text set from a host fetch is remembered ([lastFromHost]) so the resulting
|
||||
* primary-clip-changed callback doesn't bounce it straight back as a new offer. Clipboard reads
|
||||
* happen while the stream is foreground (Android only allows focused-app reads). The native
|
||||
* events are drained on a dedicated thread and applied on the main thread; [stop] joins it.
|
||||
*/
|
||||
class ClipboardSync(
|
||||
private val context: Context,
|
||||
private val handle: Long,
|
||||
) {
|
||||
private val main = Handler(Looper.getMainLooper())
|
||||
private val cm = context.getSystemService(Context.CLIPBOARD_SERVICE) as ClipboardManager
|
||||
|
||||
@Volatile private var running = true
|
||||
private var seq = 0
|
||||
private var lastOffered: String? = null
|
||||
private var lastFromHost: String? = null
|
||||
private var pendingFetch = -1
|
||||
private var thread: Thread? = null
|
||||
|
||||
private val clipListener = ClipboardManager.OnPrimaryClipChangedListener { offerLocal() }
|
||||
|
||||
fun start() {
|
||||
NativeBridge.nativeClipControl(handle, true)
|
||||
cm.addPrimaryClipChangedListener(clipListener)
|
||||
thread = Thread({ pollLoop() }, "pf-clipboard").also { it.start() }
|
||||
offerLocal() // whatever is already on the clipboard is pasteable host-side right away
|
||||
}
|
||||
|
||||
fun stop() {
|
||||
running = false
|
||||
cm.removePrimaryClipChangedListener(clipListener)
|
||||
thread?.join(600) // one poll timeout (250 ms) + slack
|
||||
thread = null
|
||||
}
|
||||
|
||||
/** Announce the current local text (if it's new and not an echo of a host copy). */
|
||||
private fun offerLocal() {
|
||||
if (!running) return
|
||||
val text = currentClipText() ?: return
|
||||
if (text == lastOffered || text == lastFromHost) return
|
||||
lastOffered = text
|
||||
seq += 1
|
||||
NativeBridge.nativeClipOfferText(handle, seq)
|
||||
}
|
||||
|
||||
private fun currentClipText(): String? = runCatching {
|
||||
cm.primaryClip?.takeIf { it.itemCount > 0 }?.getItemAt(0)
|
||||
?.coerceToText(context)?.toString()?.takeIf { it.isNotEmpty() }
|
||||
}.getOrNull()
|
||||
|
||||
private fun pollLoop() {
|
||||
while (running) {
|
||||
val ev = NativeBridge.nativeNextClip(handle) ?: continue
|
||||
if (ev == "closed") return
|
||||
main.post { handleEvent(ev) }
|
||||
}
|
||||
}
|
||||
|
||||
private fun handleEvent(ev: String) {
|
||||
if (!running) return
|
||||
val parts = ev.split(":", limit = 3)
|
||||
when (parts[0]) {
|
||||
"offer" -> {
|
||||
val offerSeq = parts.getOrNull(1)?.toIntOrNull() ?: return
|
||||
if (parts.getOrNull(2) == "1") {
|
||||
pendingFetch = NativeBridge.nativeClipFetchText(handle, offerSeq)
|
||||
}
|
||||
}
|
||||
"fetch" -> {
|
||||
val req = parts.getOrNull(1)?.toIntOrNull() ?: return
|
||||
val text = currentClipText()
|
||||
if (text != null) {
|
||||
NativeBridge.nativeClipServeText(handle, req, text)
|
||||
} else {
|
||||
NativeBridge.nativeClipCancel(handle, req)
|
||||
}
|
||||
}
|
||||
"data" -> {
|
||||
val xfer = parts.getOrNull(1)?.toIntOrNull() ?: return
|
||||
if (xfer != pendingFetch) return // stale/unknown transfer
|
||||
pendingFetch = -1
|
||||
val text = parts.getOrNull(2)?.takeIf { it.isNotEmpty() } ?: return
|
||||
lastFromHost = text
|
||||
runCatching { cm.setPrimaryClip(ClipData.newPlainText("Punktfunk", text)) }
|
||||
}
|
||||
// "state"/"cancel"/"error": nothing to drive in the text-only v1.
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -54,6 +54,21 @@ class MainActivity : ComponentActivity() {
|
||||
var padKeyProbe: ((KeyEvent) -> Boolean)? = null
|
||||
var padMotionProbe: ((MotionEvent) -> Boolean)? = null
|
||||
|
||||
/**
|
||||
* Physical-mouse forwarder for the active session (built/released by StreamScreen, like
|
||||
* [gamepadRouter]): uncaptured hover/click/wheel forwards as absolute cursor input, captured
|
||||
* ([android.view.View.requestPointerCapture]) raw deltas as relative mouse-look. The dispatch
|
||||
* overrides below route every SOURCE_MOUSE event here while streaming. Null while not streaming.
|
||||
*/
|
||||
var mouseForwarder: MouseForwarder? = null
|
||||
|
||||
/**
|
||||
* TV remote-as-pointer for the active session (StreamScreen builds it on TV devices only):
|
||||
* hold SELECT to toggle, then the D-pad glides the host cursor. Consulted first for
|
||||
* non-gamepad keys while streaming. Null while not streaming or not a TV.
|
||||
*/
|
||||
var remotePointer: RemotePointer? = null
|
||||
|
||||
/**
|
||||
* Set by [StreamScreen] to its disconnect action. The emergency-exit chord (below) invokes it so a
|
||||
* couch user with no keyboard/Back can always leave a stream.
|
||||
@@ -324,9 +339,29 @@ class MainActivity : ComponentActivity() {
|
||||
return true // consumed
|
||||
}
|
||||
}
|
||||
// TV remote-as-pointer sees non-gamepad keys first (SELECT long-press toggles it;
|
||||
// while active it owns the D-pad/SELECT/PLAY-PAUSE/BACK).
|
||||
if (!event.isFromSource(InputDevice.SOURCE_GAMEPAD)) {
|
||||
remotePointer?.let { if (it.onKey(event)) return true }
|
||||
}
|
||||
// Ctrl+Alt+Shift+Q — the cross-client pointer-capture toggle chord. Swallow both
|
||||
// edges of the Q (the modifiers already went over the wire, exactly like desktop).
|
||||
if (event.keyCode == KeyEvent.KEYCODE_Q &&
|
||||
event.isCtrlPressed && event.isAltPressed && event.isShiftPressed
|
||||
) {
|
||||
if (event.action == KeyEvent.ACTION_DOWN && event.repeatCount == 0) {
|
||||
mouseForwarder?.toggleCapture()
|
||||
}
|
||||
return true
|
||||
}
|
||||
when (event.keyCode) {
|
||||
// A mouse back/forward button whose BUTTON_* press went unconsumed makes the
|
||||
// framework synthesize a FALLBACK BACK — the button already went over the wire
|
||||
// as X1/X2, and it must never yank the user out of the stream.
|
||||
KeyEvent.KEYCODE_BACK ->
|
||||
if (event.flags and KeyEvent.FLAG_FALLBACK != 0) return true
|
||||
// Leave these to the system even while streaming.
|
||||
KeyEvent.KEYCODE_BACK, // → BackHandler leaves the stream
|
||||
// (BACK above → BackHandler leaves the stream.)
|
||||
KeyEvent.KEYCODE_VOLUME_UP,
|
||||
KeyEvent.KEYCODE_VOLUME_DOWN,
|
||||
KeyEvent.KEYCODE_VOLUME_MUTE,
|
||||
@@ -394,6 +429,10 @@ class MainActivity : ComponentActivity() {
|
||||
override fun dispatchGenericMotionEvent(event: MotionEvent): Boolean {
|
||||
if (streamHandle != 0L) {
|
||||
if (gamepadRouter?.onMotion(event) == true) return true
|
||||
// Physical mouse (uncaptured): hover motion, wheel, button edges.
|
||||
if (event.isFromSource(InputDevice.SOURCE_MOUSE)) {
|
||||
mouseForwarder?.let { if (it.onGenericMotion(event)) return true }
|
||||
}
|
||||
return super.dispatchGenericMotionEvent(event)
|
||||
}
|
||||
// The Controllers debug screen sees pad motion before the stick→D-pad synthesis below.
|
||||
@@ -431,6 +470,24 @@ class MainActivity : ComponentActivity() {
|
||||
return super.dispatchGenericMotionEvent(event)
|
||||
}
|
||||
|
||||
/**
|
||||
* Mouse clicks/drags ride the TOUCH stream (the pointer is "down"). While streaming they
|
||||
* belong to the mouse forwarder, never to the Compose touch-gesture layer — a physical
|
||||
* mouse click must be a real click at the cursor, not a synthesized trackpad tap.
|
||||
*/
|
||||
override fun dispatchTouchEvent(ev: MotionEvent): Boolean {
|
||||
if (streamHandle != 0L && ev.isFromSource(InputDevice.SOURCE_MOUSE)) {
|
||||
mouseForwarder?.let { if (it.onTouchEvent(ev)) return true }
|
||||
}
|
||||
return super.dispatchTouchEvent(ev)
|
||||
}
|
||||
|
||||
/** The OS is the source of truth for pointer capture (it releases on focus loss). */
|
||||
override fun onPointerCaptureChanged(hasCapture: Boolean) {
|
||||
super.onPointerCaptureChanged(hasCapture)
|
||||
mouseForwarder?.onCaptureChanged(hasCapture)
|
||||
}
|
||||
|
||||
/** Keys that drive the console UI — D-pad + face buttons; used to classify the last input source. */
|
||||
private fun isConsoleNavKey(kc: Int): Boolean = when (kc) {
|
||||
KeyEvent.KEYCODE_DPAD_UP, KeyEvent.KEYCODE_DPAD_DOWN, KeyEvent.KEYCODE_DPAD_LEFT,
|
||||
|
||||
@@ -0,0 +1,206 @@
|
||||
package io.unom.punktfunk
|
||||
|
||||
import android.view.InputDevice
|
||||
import android.view.MotionEvent
|
||||
import io.unom.punktfunk.kit.NativeBridge
|
||||
import kotlin.math.roundToInt
|
||||
|
||||
/** True when any connected input device is a pointer (USB/BT mouse, or a touchpad driving one). */
|
||||
fun hasPhysicalMouse(): Boolean = InputDevice.getDeviceIds().any { id ->
|
||||
InputDevice.getDevice(id)?.supportsSource(InputDevice.SOURCE_MOUSE) == true
|
||||
}
|
||||
|
||||
/**
|
||||
* Physical mouse → wire, in two modes (the iPadOS/desktop model):
|
||||
* * **uncaptured** (default): hover/drag positions forward as absolute cursor moves
|
||||
* (`MouseMoveAbs`, host-normalized against the window size) — desktop-style pointing. The
|
||||
* local cursor is hidden over the stream (StreamScreen sets a TYPE_NULL pointer icon); the
|
||||
* host's own cursor, composited into the video, is the one you see.
|
||||
* * **captured**: the OS pointer is grabbed ([android.view.View.requestPointerCapture]) and raw
|
||||
* relative deltas forward as `MouseMove` — FPS mouse-look. Engaged at stream start / by
|
||||
* clicking into the stream when the "Capture pointer for games" setting is on, and toggled
|
||||
* any time by Ctrl+Alt+Shift+Q (the cross-client chord). Focus loss releases it (the OS
|
||||
* guarantees that); a click re-engages.
|
||||
*
|
||||
* Buttons ride [MotionEvent.ACTION_BUTTON_PRESS]/RELEASE edges (left/middle/right/back/forward →
|
||||
* wire 1/2/3/4/5), the wheel rides [MotionEvent.ACTION_SCROLL] with fractional accumulation so
|
||||
* high-resolution wheels don't lose sub-notch travel. Held buttons are tracked and flushed on
|
||||
* capture loss / stream exit so nothing sticks on the host. Events reach this class from
|
||||
* MainActivity's dispatch overrides (uncaptured) and the capture view's captured-pointer listener.
|
||||
*/
|
||||
class MouseForwarder(
|
||||
private val handle: Long,
|
||||
private val invertScroll: Boolean,
|
||||
private val captureWanted: Boolean,
|
||||
private val surfaceSize: () -> Pair<Int, Int>,
|
||||
) {
|
||||
/** Capture plumbing, owned by StreamScreen (the focusable capture view). */
|
||||
var onRequestCapture: (() -> Unit)? = null
|
||||
var onReleaseCapture: (() -> Unit)? = null
|
||||
|
||||
/** Live capture state, updated from [android.app.Activity.onPointerCaptureChanged]. */
|
||||
var captured = false
|
||||
private set
|
||||
|
||||
/** Chord-released: no auto re-engage (start / click) until the user opts back in. */
|
||||
private var userReleased = false
|
||||
|
||||
private val heldButtons = mutableSetOf<Int>()
|
||||
private var scrollAccV = 0f
|
||||
private var scrollAccH = 0f
|
||||
private var moveAccX = 0f
|
||||
private var moveAccY = 0f
|
||||
|
||||
/** Uncaptured mouse events on the TOUCH stream (position while a button is down). */
|
||||
fun onTouchEvent(ev: MotionEvent): Boolean {
|
||||
when (ev.actionMasked) {
|
||||
MotionEvent.ACTION_DOWN -> {
|
||||
if (captureWanted && !captured && !userReleased) {
|
||||
// The engaging click: grab the pointer and swallow the click (desktop
|
||||
// parity — the click that captures never reaches the host). The paired
|
||||
// BUTTON_RELEASE is dropped by the held-set guard in [button].
|
||||
onRequestCapture?.invoke()
|
||||
return true
|
||||
}
|
||||
sendAbs(ev)
|
||||
}
|
||||
MotionEvent.ACTION_MOVE -> sendAbs(ev)
|
||||
// Button edges are documented on the generic stream, but be robust to either.
|
||||
MotionEvent.ACTION_BUTTON_PRESS -> button(ev.actionButton, true)
|
||||
MotionEvent.ACTION_BUTTON_RELEASE -> button(ev.actionButton, false)
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
/** Uncaptured mouse events on the GENERIC stream (hover motion, wheel, button edges). */
|
||||
fun onGenericMotion(ev: MotionEvent): Boolean {
|
||||
when (ev.actionMasked) {
|
||||
MotionEvent.ACTION_HOVER_MOVE -> sendAbs(ev)
|
||||
MotionEvent.ACTION_SCROLL -> wheel(ev)
|
||||
MotionEvent.ACTION_BUTTON_PRESS -> button(ev.actionButton, true)
|
||||
MotionEvent.ACTION_BUTTON_RELEASE -> button(ev.actionButton, false)
|
||||
MotionEvent.ACTION_HOVER_ENTER, MotionEvent.ACTION_HOVER_EXIT -> {}
|
||||
else -> return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Captured-pointer events (the view holds [android.view.View.requestPointerCapture]): x/y ARE
|
||||
* the relative deltas ([InputDevice.SOURCE_MOUSE_RELATIVE]), batched samples included. A
|
||||
* captured touchpad reports absolute finger coordinates instead — not handled (the touch
|
||||
* gesture layer is the touchpad story); returning false leaves those to the framework.
|
||||
*/
|
||||
fun onCapturedPointer(ev: MotionEvent): Boolean {
|
||||
if (!ev.isFromSource(InputDevice.SOURCE_MOUSE_RELATIVE)) return false
|
||||
when (ev.actionMasked) {
|
||||
MotionEvent.ACTION_MOVE -> {
|
||||
var dx = 0f
|
||||
var dy = 0f
|
||||
for (i in 0 until ev.historySize) {
|
||||
dx += ev.getHistoricalX(i)
|
||||
dy += ev.getHistoricalY(i)
|
||||
}
|
||||
dx += ev.x
|
||||
dy += ev.y
|
||||
moveAccX += dx
|
||||
moveAccY += dy
|
||||
val ox = moveAccX.toInt() // truncate toward zero — sub-pixel remainder kept w/ sign
|
||||
val oy = moveAccY.toInt()
|
||||
if (ox != 0 || oy != 0) {
|
||||
NativeBridge.nativeSendPointerMove(handle, ox, oy)
|
||||
moveAccX -= ox
|
||||
moveAccY -= oy
|
||||
}
|
||||
}
|
||||
MotionEvent.ACTION_BUTTON_PRESS -> button(ev.actionButton, true)
|
||||
MotionEvent.ACTION_BUTTON_RELEASE -> button(ev.actionButton, false)
|
||||
MotionEvent.ACTION_SCROLL -> wheel(ev)
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
/** Ctrl+Alt+Shift+Q: release the grab, or (re-)engage it — works even when auto-capture is off. */
|
||||
fun toggleCapture() {
|
||||
if (captured) {
|
||||
userReleased = true
|
||||
onReleaseCapture?.invoke()
|
||||
} else {
|
||||
userReleased = false
|
||||
onRequestCapture?.invoke()
|
||||
}
|
||||
}
|
||||
|
||||
/** Auto-engage at stream start (setting on + a mouse actually present). */
|
||||
fun engageFromStart() {
|
||||
if (captureWanted && !captured && !userReleased && hasPhysicalMouse()) {
|
||||
onRequestCapture?.invoke()
|
||||
}
|
||||
}
|
||||
|
||||
/** From [android.app.Activity.onPointerCaptureChanged] — the OS is the source of truth. */
|
||||
fun onCaptureChanged(has: Boolean) {
|
||||
captured = has
|
||||
// Losing the grab (focus loss, chord) must not leave buttons held on the host.
|
||||
if (!has) flushButtons()
|
||||
}
|
||||
|
||||
/** Stream teardown: lift anything held and let the grab go. */
|
||||
fun release() {
|
||||
flushButtons()
|
||||
if (captured) onReleaseCapture?.invoke()
|
||||
}
|
||||
|
||||
private fun sendAbs(ev: MotionEvent) {
|
||||
val (w, h) = surfaceSize()
|
||||
if (w <= 0 || h <= 0) return
|
||||
NativeBridge.nativeSendPointerAbs(
|
||||
handle,
|
||||
ev.x.roundToInt().coerceIn(0, w - 1),
|
||||
ev.y.roundToInt().coerceIn(0, h - 1),
|
||||
w,
|
||||
h,
|
||||
)
|
||||
}
|
||||
|
||||
private fun wheel(ev: MotionEvent) {
|
||||
val dir = if (invertScroll) -1f else 1f
|
||||
// Android: AXIS_VSCROLL + = up/away, AXIS_HSCROLL + = right — the wire's convention too.
|
||||
scrollAccV += ev.getAxisValue(MotionEvent.AXIS_VSCROLL) * 120f * dir
|
||||
scrollAccH += ev.getAxisValue(MotionEvent.AXIS_HSCROLL) * 120f * dir
|
||||
val v = scrollAccV.toInt()
|
||||
if (v != 0) {
|
||||
NativeBridge.nativeSendScroll(handle, 0, v)
|
||||
scrollAccV -= v
|
||||
}
|
||||
val h = scrollAccH.toInt()
|
||||
if (h != 0) {
|
||||
NativeBridge.nativeSendScroll(handle, 1, h)
|
||||
scrollAccH -= h
|
||||
}
|
||||
}
|
||||
|
||||
private fun button(actionButton: Int, down: Boolean) {
|
||||
val b = when (actionButton) {
|
||||
MotionEvent.BUTTON_PRIMARY -> 1
|
||||
MotionEvent.BUTTON_TERTIARY -> 2
|
||||
MotionEvent.BUTTON_SECONDARY -> 3
|
||||
MotionEvent.BUTTON_BACK -> 4
|
||||
MotionEvent.BUTTON_FORWARD -> 5
|
||||
else -> return
|
||||
}
|
||||
if (down) {
|
||||
heldButtons.add(b)
|
||||
NativeBridge.nativeSendPointerButton(handle, b, true)
|
||||
} else if (heldButtons.remove(b)) {
|
||||
// Only release what we pressed — drops the release of a swallowed engaging click
|
||||
// and anything that raced a capture transition.
|
||||
NativeBridge.nativeSendPointerButton(handle, b, false)
|
||||
}
|
||||
}
|
||||
|
||||
private fun flushButtons() {
|
||||
heldButtons.forEach { NativeBridge.nativeSendPointerButton(handle, it, false) }
|
||||
heldButtons.clear()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
package io.unom.punktfunk
|
||||
|
||||
import android.os.Handler
|
||||
import android.os.Looper
|
||||
import android.view.Choreographer
|
||||
import android.view.KeyEvent
|
||||
import io.unom.punktfunk.kit.NativeBridge
|
||||
import kotlin.math.hypot
|
||||
|
||||
// Hold this long on SELECT (pointer-mode toggle) / PLAY-PAUSE (keyboard toggle) for the long-press
|
||||
// action instead of the tap action.
|
||||
private const val LONG_PRESS_MS = 800L
|
||||
|
||||
// D-pad glide ballistics, in screen-widths per second: start slow enough to hit a close button,
|
||||
// ramp over RAMP_S seconds of continuous hold so crossing the desktop doesn't take all day.
|
||||
private const val SPEED_MIN = 0.14f
|
||||
private const val SPEED_MAX = 0.70f
|
||||
private const val RAMP_S = 1.2f
|
||||
|
||||
/**
|
||||
* Android TV remote as a pointer — the Android analogue of the Apple client's Siri-remote pointer,
|
||||
* adapted for D-pad-only remotes (most Android TV remotes have no touch surface). For the
|
||||
* "TV as a desktop client" use case, where a plain remote is often the only thing in hand.
|
||||
*
|
||||
* While streaming on a TV, **hold SELECT ≈ 0.8 s** to toggle pointer mode. While active:
|
||||
* * D-pad (held) glides the host cursor with ramping acceleration (relative `MouseMove`,
|
||||
* Choreographer-paced, diagonal-normalized);
|
||||
* * SELECT tap = left click; PLAY/PAUSE tap = right click (Siri-remote parity);
|
||||
* * PLAY/PAUSE held = toggle the on-screen keyboard; BACK = leave pointer mode
|
||||
* (a second BACK then leaves the stream as usual).
|
||||
* While inactive, everything except the SELECT long-press passes through untouched (D-pad =
|
||||
* arrow keys, SELECT tap = Enter — synthesized on release, since the down was held back to
|
||||
* disambiguate the long-press).
|
||||
*
|
||||
* Only consulted for non-gamepad key events on TV devices (MainActivity gates the calls); all
|
||||
* state lives on the main thread.
|
||||
*/
|
||||
class RemotePointer(
|
||||
private val handle: Long,
|
||||
private val surfaceWidth: () -> Int,
|
||||
private val onActiveChanged: (Boolean) -> Unit,
|
||||
private val onKeyboardToggle: () -> Unit,
|
||||
) {
|
||||
var active = false
|
||||
private set
|
||||
|
||||
private val handler = Handler(Looper.getMainLooper())
|
||||
private val held = mutableSetOf<Int>() // D-pad keycodes currently down
|
||||
private var moveAccX = 0f
|
||||
private var moveAccY = 0f
|
||||
private var lastFrameNs = 0L
|
||||
private var rampSec = 0f
|
||||
private var tickerRunning = false
|
||||
private var centerLongFired = false
|
||||
private var playLongFired = false
|
||||
|
||||
private val centerLong = Runnable {
|
||||
centerLongFired = true
|
||||
toggle()
|
||||
}
|
||||
private val playLong = Runnable {
|
||||
playLongFired = true
|
||||
onKeyboardToggle()
|
||||
}
|
||||
|
||||
private val frame = object : Choreographer.FrameCallback {
|
||||
override fun doFrame(nowNs: Long) {
|
||||
if (!tickerRunning) return
|
||||
if (held.isEmpty() || !active) {
|
||||
tickerRunning = false
|
||||
return
|
||||
}
|
||||
val dt = if (lastFrameNs == 0L) {
|
||||
1f / 60f
|
||||
} else {
|
||||
((nowNs - lastFrameNs) / 1e9f).coerceIn(0.001f, 0.1f)
|
||||
}
|
||||
lastFrameNs = nowNs
|
||||
rampSec += dt
|
||||
var vx = 0f
|
||||
var vy = 0f
|
||||
if (KeyEvent.KEYCODE_DPAD_LEFT in held) vx -= 1f
|
||||
if (KeyEvent.KEYCODE_DPAD_RIGHT in held) vx += 1f
|
||||
if (KeyEvent.KEYCODE_DPAD_UP in held) vy -= 1f
|
||||
if (KeyEvent.KEYCODE_DPAD_DOWN in held) vy += 1f
|
||||
val mag = hypot(vx, vy)
|
||||
if (mag > 0f) {
|
||||
val w = surfaceWidth().coerceAtLeast(640)
|
||||
val speed = w * (SPEED_MIN + (SPEED_MAX - SPEED_MIN) * (rampSec / RAMP_S).coerceAtMost(1f))
|
||||
moveAccX += vx / mag * speed * dt
|
||||
moveAccY += vy / mag * speed * dt
|
||||
val ox = moveAccX.toInt() // truncate toward zero — sub-pixel remainder kept
|
||||
val oy = moveAccY.toInt()
|
||||
if (ox != 0 || oy != 0) {
|
||||
NativeBridge.nativeSendPointerMove(handle, ox, oy)
|
||||
moveAccX -= ox
|
||||
moveAccY -= oy
|
||||
}
|
||||
}
|
||||
Choreographer.getInstance().postFrameCallback(this)
|
||||
}
|
||||
}
|
||||
|
||||
/** One remote key event; true = consumed. Ignore key repeats — the ticker owns motion. */
|
||||
fun onKey(event: KeyEvent): Boolean {
|
||||
val down = event.action == KeyEvent.ACTION_DOWN
|
||||
when (event.keyCode) {
|
||||
KeyEvent.KEYCODE_DPAD_CENTER -> {
|
||||
if (down) {
|
||||
if (event.repeatCount == 0) {
|
||||
centerLongFired = false
|
||||
handler.postDelayed(centerLong, LONG_PRESS_MS)
|
||||
}
|
||||
} else {
|
||||
handler.removeCallbacks(centerLong)
|
||||
if (!centerLongFired) {
|
||||
if (active) {
|
||||
click(1)
|
||||
} else {
|
||||
// The down was held back to disambiguate the long-press, so the
|
||||
// normal path never saw it — synthesize the Enter here instead.
|
||||
NativeBridge.nativeSendKey(handle, 0x0D, true, 0)
|
||||
NativeBridge.nativeSendKey(handle, 0x0D, false, 0)
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
KeyEvent.KEYCODE_DPAD_UP, KeyEvent.KEYCODE_DPAD_DOWN,
|
||||
KeyEvent.KEYCODE_DPAD_LEFT, KeyEvent.KEYCODE_DPAD_RIGHT,
|
||||
-> {
|
||||
if (!active) return false
|
||||
if (down) {
|
||||
if (held.add(event.keyCode) && held.size == 1) startTicker()
|
||||
} else {
|
||||
held.remove(event.keyCode)
|
||||
}
|
||||
return true
|
||||
}
|
||||
KeyEvent.KEYCODE_MEDIA_PLAY_PAUSE -> {
|
||||
if (!active) return false // inactive: the media-key VK path owns it
|
||||
if (down) {
|
||||
if (event.repeatCount == 0) {
|
||||
playLongFired = false
|
||||
handler.postDelayed(playLong, LONG_PRESS_MS)
|
||||
}
|
||||
} else {
|
||||
handler.removeCallbacks(playLong)
|
||||
if (!playLongFired) click(3)
|
||||
}
|
||||
return true
|
||||
}
|
||||
KeyEvent.KEYCODE_BACK -> {
|
||||
if (!active) return false
|
||||
if (!down) toggle() // leave pointer mode; the next BACK leaves the stream
|
||||
return true
|
||||
}
|
||||
else -> return false
|
||||
}
|
||||
}
|
||||
|
||||
/** Stream teardown: stop timers/ticker; nothing wire-held to flush (clicks are edges). */
|
||||
fun release() {
|
||||
handler.removeCallbacks(centerLong)
|
||||
handler.removeCallbacks(playLong)
|
||||
active = false
|
||||
held.clear()
|
||||
tickerRunning = false
|
||||
}
|
||||
|
||||
private fun toggle() {
|
||||
active = !active
|
||||
if (!active) {
|
||||
held.clear()
|
||||
tickerRunning = false
|
||||
}
|
||||
onActiveChanged(active)
|
||||
}
|
||||
|
||||
private fun startTicker() {
|
||||
rampSec = 0f
|
||||
lastFrameNs = 0L
|
||||
if (!tickerRunning) {
|
||||
tickerRunning = true
|
||||
Choreographer.getInstance().postFrameCallback(frame)
|
||||
}
|
||||
}
|
||||
|
||||
private fun click(button: Int) {
|
||||
NativeBridge.nativeSendPointerButton(handle, button, true)
|
||||
NativeBridge.nativeSendPointerButton(handle, button, false)
|
||||
}
|
||||
}
|
||||
@@ -109,6 +109,27 @@ data class Settings(
|
||||
* setup where the OS-level pad (lizard mode) is preferred.
|
||||
*/
|
||||
val sc2Capture: Boolean = true,
|
||||
|
||||
/**
|
||||
* Lock a physical mouse to the stream ([android.view.View.requestPointerCapture]) and forward
|
||||
* raw relative motion — FPS mouse-look, the iPad "Capture pointer for games" twin. Engages at
|
||||
* stream start and on a click into the stream; Ctrl+Alt+Shift+Q toggles it live (the chord
|
||||
* works even with this off). Off (default): a mouse points absolutely, desktop-style.
|
||||
*/
|
||||
val pointerCapture: Boolean = false,
|
||||
|
||||
/**
|
||||
* Flip scroll direction — the mouse wheel and the two-finger touch scroll both. Parity with
|
||||
* the Apple/GTK clients' "Invert scroll direction".
|
||||
*/
|
||||
val invertScroll: Boolean = false,
|
||||
|
||||
/**
|
||||
* Sync text copied on this device to the host and vice versa while streaming (the desktop
|
||||
* clients' shared clipboard, text-only here). Only effective when the host advertises the
|
||||
* clipboard capability; the protocol is opt-in per session either way.
|
||||
*/
|
||||
val clipboardSync: Boolean = true,
|
||||
)
|
||||
|
||||
/** [Settings.touchMode] values; persisted by name. */
|
||||
@@ -172,6 +193,9 @@ class SettingsStore(context: Context) {
|
||||
autoWakeEnabled = prefs.getBoolean(K_AUTO_WAKE, true),
|
||||
rumbleOnPhone = prefs.getBoolean(K_RUMBLE_ON_PHONE, false),
|
||||
sc2Capture = prefs.getBoolean(K_SC2_CAPTURE, true),
|
||||
pointerCapture = prefs.getBoolean(K_POINTER_CAPTURE, false),
|
||||
invertScroll = prefs.getBoolean(K_INVERT_SCROLL, false),
|
||||
clipboardSync = prefs.getBoolean(K_CLIPBOARD_SYNC, true),
|
||||
)
|
||||
|
||||
fun save(s: Settings) {
|
||||
@@ -195,6 +219,9 @@ class SettingsStore(context: Context) {
|
||||
.putBoolean(K_AUTO_WAKE, s.autoWakeEnabled)
|
||||
.putBoolean(K_RUMBLE_ON_PHONE, s.rumbleOnPhone)
|
||||
.putBoolean(K_SC2_CAPTURE, s.sc2Capture)
|
||||
.putBoolean(K_POINTER_CAPTURE, s.pointerCapture)
|
||||
.putBoolean(K_INVERT_SCROLL, s.invertScroll)
|
||||
.putBoolean(K_CLIPBOARD_SYNC, s.clipboardSync)
|
||||
.apply()
|
||||
}
|
||||
|
||||
@@ -233,6 +260,9 @@ class SettingsStore(context: Context) {
|
||||
const val K_AUTO_WAKE = "auto_wake_enabled"
|
||||
const val K_RUMBLE_ON_PHONE = "rumble_on_phone"
|
||||
const val K_SC2_CAPTURE = "sc2_capture"
|
||||
const val K_POINTER_CAPTURE = "pointer_capture"
|
||||
const val K_INVERT_SCROLL = "invert_scroll"
|
||||
const val K_CLIPBOARD_SYNC = "clipboard_sync"
|
||||
|
||||
/** Legacy Boolean the enum replaced — read once as the migration default, never written. */
|
||||
const val K_TRACKPAD = "trackpad_mode"
|
||||
|
||||
@@ -412,6 +412,27 @@ private fun ControlsSettings(s: Settings, update: (Settings) -> Unit, onOpenCont
|
||||
style = MaterialTheme.typography.bodySmall,
|
||||
color = MaterialTheme.colorScheme.onSurfaceVariant,
|
||||
)
|
||||
ToggleRow(
|
||||
title = "Capture pointer for games",
|
||||
subtitle = "Lock a connected mouse to the stream and send raw relative motion " +
|
||||
"(mouse-look). Ctrl+Alt+Shift+Q toggles it live; click the stream to re-capture. " +
|
||||
"Off: the mouse points at the desktop directly",
|
||||
checked = s.pointerCapture,
|
||||
onCheckedChange = { on -> update(s.copy(pointerCapture = on)) },
|
||||
)
|
||||
ToggleRow(
|
||||
title = "Invert scroll direction",
|
||||
subtitle = "Flip the mouse wheel and two-finger touch scrolling",
|
||||
checked = s.invertScroll,
|
||||
onCheckedChange = { on -> update(s.copy(invertScroll = on)) },
|
||||
)
|
||||
ToggleRow(
|
||||
title = "Shared clipboard",
|
||||
subtitle = "Text copied here pastes on the host and vice versa (hosts with " +
|
||||
"clipboard sharing enabled)",
|
||||
checked = s.clipboardSync,
|
||||
onCheckedChange = { on -> update(s.copy(clipboardSync = on)) },
|
||||
)
|
||||
}
|
||||
SettingsCard {
|
||||
SettingDropdown(
|
||||
|
||||
@@ -176,6 +176,14 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
// "hold to quit" hint overlay. Set from the router's onExitArmed (main thread).
|
||||
var exitArming by remember { mutableStateOf(false) }
|
||||
|
||||
// True while the TV remote is acting as a pointer (hold SELECT toggles) — drives the mode hint.
|
||||
var remotePointerOn by remember { mutableStateOf(false) }
|
||||
|
||||
// Focus anchor the soft keyboard is summoned onto AND the pointer-capture grab target (a grab
|
||||
// needs a focusable view; captured-pointer events land on it). Declared before the effect
|
||||
// below so the capture callbacks can reach the view once it exists.
|
||||
var keyCapture by remember { mutableStateOf<KeyCaptureView?>(null) }
|
||||
|
||||
DisposableEffect(handle) {
|
||||
window?.addFlags(WindowManager.LayoutParams.FLAG_KEEP_SCREEN_ON)
|
||||
wifiLocks.forEach { lock ->
|
||||
@@ -221,6 +229,54 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
// Show a "hold to quit" hint the moment the chord completes (the router debounces the actual
|
||||
// exit); it clears when the buttons release early or the hold elapses. Runs on the main thread.
|
||||
router.onExitArmed = { armed -> exitArming = armed }
|
||||
// Physical mouse: uncaptured hover/click/wheel forwards as absolute pointing; captured
|
||||
// (setting or the Ctrl+Alt+Shift+Q chord) raw deltas forward as relative mouse-look.
|
||||
// The local cursor is hidden over the stream — the host's own cursor, composited into
|
||||
// the video, is the one the user sees (twin of the desktop clients' hidden cursor).
|
||||
val decor = window?.decorView
|
||||
val priorPointerIcon = decor?.pointerIcon
|
||||
decor?.pointerIcon = android.view.PointerIcon.getSystemIcon(
|
||||
context,
|
||||
android.view.PointerIcon.TYPE_NULL,
|
||||
)
|
||||
val mouse = MouseForwarder(
|
||||
handle,
|
||||
invertScroll = initialSettings.invertScroll,
|
||||
captureWanted = initialSettings.pointerCapture,
|
||||
surfaceSize = { (decor?.width ?: 0) to (decor?.height ?: 0) },
|
||||
)
|
||||
mouse.onRequestCapture = {
|
||||
// The grab needs the (focusable) capture view: focus it, then ask. Posted so a
|
||||
// request racing view attach/focus settles on the next frame.
|
||||
keyCapture?.let { v ->
|
||||
v.post {
|
||||
v.requestFocus()
|
||||
v.requestPointerCapture()
|
||||
}
|
||||
}
|
||||
}
|
||||
mouse.onReleaseCapture = { keyCapture?.releasePointerCapture() }
|
||||
activity?.mouseForwarder = mouse
|
||||
// TV remote-as-pointer: hold SELECT ≈ 0.8 s to toggle; the D-pad then glides the host
|
||||
// cursor (see RemotePointer). TV only — a phone's remote-less keys stay on the VK path.
|
||||
val remote = if (isTv) {
|
||||
RemotePointer(
|
||||
handle,
|
||||
surfaceWidth = { decor?.width ?: 1920 },
|
||||
onActiveChanged = { on -> remotePointerOn = on },
|
||||
onKeyboardToggle = { keyCapture?.let { it.setImeVisible(!it.imeShown) } },
|
||||
)
|
||||
} else {
|
||||
null
|
||||
}
|
||||
activity?.remotePointer = remote
|
||||
// Shared clipboard (text v1): only when the user setting is on AND the host has a
|
||||
// working clipboard service. Protocol-level opt-in + the poll thread live in the sync.
|
||||
val clip = if (initialSettings.clipboardSync && NativeBridge.nativeClipSupported(handle)) {
|
||||
ClipboardSync(context, handle).also { it.start() }
|
||||
} else {
|
||||
null
|
||||
}
|
||||
activity?.setConsoleHighRefreshRate(false) // let the decoder's setFrameRate pick the panel rate
|
||||
// Host→client feedback (rumble + DualSense lightbar/LEDs), routed to each controller by pad
|
||||
// index via the router; poll threads stopped + joined before the router is released and the
|
||||
@@ -286,6 +342,7 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
}
|
||||
onDispose {
|
||||
closed.set(true) // from here the handle gets freed; surfaceDestroyed must not touch it
|
||||
clip?.stop() // stop + join the clipboard poll thread BEFORE the handle is freed
|
||||
feedback.onHidRaw = null
|
||||
feedback.stop() // stop + join the poll threads BEFORE the router is released / handle freed
|
||||
sc2UsbReceiver?.let { runCatching { context.unregisterReceiver(it) } }
|
||||
@@ -293,6 +350,12 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
router.onExitArmed = null // don't poke Compose state from release()'s disarm while tearing down
|
||||
router.release() // flush every slot (nothing sticks host-side) + drop the hot-plug listener
|
||||
activity?.gamepadRouter = null
|
||||
// Mouse/remote-pointer teardown: lift held buttons, drop the grab, restore the cursor.
|
||||
mouse.release()
|
||||
activity?.mouseForwarder = null
|
||||
remote?.release()
|
||||
activity?.remotePointer = null
|
||||
decor?.pointerIcon = priorPointerIcon
|
||||
activity?.streamHandle = 0L
|
||||
activity?.requestStreamExit = null
|
||||
// Back in the menus: the SC2 (if present) resumes driving the console UI.
|
||||
@@ -320,8 +383,12 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
// Back gesture = a deliberate exit → signal the quit so the host tears down now (no linger).
|
||||
BackHandler { NativeBridge.nativeDisconnectQuit(handle); onDisconnect() }
|
||||
|
||||
// Focus anchor the three-finger keyboard swipe summons the IME onto (see KeyCaptureView).
|
||||
var keyCapture by remember { mutableStateOf<KeyCaptureView?>(null) }
|
||||
// Auto-engage pointer capture at stream start (setting on + a mouse actually present).
|
||||
// Delayed a beat: the grab needs window focus and the capture view attached.
|
||||
LaunchedEffect(handle) {
|
||||
delay(400)
|
||||
activity?.mouseForwarder?.engageFromStart()
|
||||
}
|
||||
|
||||
Box(modifier = Modifier.fillMaxSize()) {
|
||||
AndroidView(
|
||||
@@ -379,11 +446,26 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
if (exitArming) {
|
||||
ExitChordHint(Modifier.align(Alignment.TopCenter).padding(top = 16.dp))
|
||||
}
|
||||
// Invisible 1-px focus anchor for the host-typing soft keyboard (three-finger swipe
|
||||
// up in the mouse modes) — it never draws or takes touches, it just owns IME focus.
|
||||
// Remote-pointer mode hint — the remote's keys are remapped while it's on, so say so.
|
||||
if (remotePointerOn) {
|
||||
RemotePointerHint(Modifier.align(Alignment.TopCenter).padding(top = 16.dp))
|
||||
}
|
||||
// Invisible 1-px focus anchor for the host-typing soft keyboard (three-finger swipe up
|
||||
// in the mouse modes) AND the pointer-capture grab target — it never draws or takes
|
||||
// touches, it just owns IME focus and receives captured-pointer events.
|
||||
AndroidView(
|
||||
modifier = Modifier.size(1.dp),
|
||||
factory = { ctx -> KeyCaptureView(ctx).also { keyCapture = it } },
|
||||
factory = { ctx ->
|
||||
KeyCaptureView(ctx).also { v ->
|
||||
keyCapture = v
|
||||
// Real IME text path when the host types committed text (see KeyCaptureView).
|
||||
v.textHandle =
|
||||
if (NativeBridge.nativeTextInputSupported(handle)) handle else 0L
|
||||
v.setOnCapturedPointerListener { _, ev ->
|
||||
(ctx as? MainActivity)?.mouseForwarder?.onCapturedPointer(ev) ?: false
|
||||
}
|
||||
}
|
||||
},
|
||||
)
|
||||
// Touch input per the Settings model: trackpad/direct-pointer mouse (the shared gesture
|
||||
// vocabulary) or real multi-touch passthrough — see TouchInput.kt. Passthrough gets no
|
||||
@@ -396,6 +478,7 @@ fun StreamScreen(handle: Long, micEnabled: Boolean, onDisconnect: () -> Unit) {
|
||||
else -> streamTouchInput(
|
||||
handle,
|
||||
trackpad = touchMode == TouchMode.TRACKPAD,
|
||||
invertScroll = initialSettings.invertScroll,
|
||||
onCycleStats = { statsVerbosity = statsVerbosity.next() },
|
||||
onKeyboard = { show -> keyCapture?.setImeVisible(show) },
|
||||
)
|
||||
@@ -423,14 +506,35 @@ private fun ExitChordHint(modifier: Modifier = Modifier) {
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The remote-pointer mode cue: while active the remote's keys are remapped (D-pad glides the host
|
||||
* cursor, SELECT clicks), so the overlay both confirms the toggle and teaches the vocabulary.
|
||||
*/
|
||||
@Composable
|
||||
private fun RemotePointerHint(modifier: Modifier = Modifier) {
|
||||
Text(
|
||||
"Remote pointer — SELECT click · play/pause right-click · hold SELECT to exit",
|
||||
modifier = modifier
|
||||
.background(Color.Black.copy(alpha = 0.55f), RoundedCornerShape(8.dp))
|
||||
.padding(horizontal = 14.dp, vertical = 8.dp),
|
||||
color = Color.White,
|
||||
fontSize = 15.sp,
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Invisible focus anchor for typing on the host: the three-finger swipe summons the device IME
|
||||
* onto this view. `TYPE_NULL` puts the IME in "dumb keyboard" mode — it delivers raw [KeyEvent]s
|
||||
* (no composing text, no autocorrect), which flow through `MainActivity.dispatchKeyEvent` →
|
||||
* `Keymap.toVk` → the host, the exact path a hardware keyboard takes. Text an IME insists on
|
||||
* committing instead still arrives: the non-editable [BaseInputConnection] synthesizes KeyEvents
|
||||
* for it via `KeyCharacterMap` (with Shift carried as meta state — see the IME-shift wrap in
|
||||
* `MainActivity.dispatchKeyEvent`).
|
||||
* onto this view. Two IME models, picked by the host's capabilities:
|
||||
* * **Text path** ([textHandle] set — the host advertised `HOST_CAP_TEXT_INPUT`): a real
|
||||
* editable [HostTextConnection], so the IME gives autocorrect, gesture typing, non-Latin
|
||||
* composition and emoji, all mirrored to the host as committed text + diffs.
|
||||
* * **Fallback** (older host): `TYPE_NULL` puts the IME in "dumb keyboard" mode — raw
|
||||
* [KeyEvent]s flow through `MainActivity.dispatchKeyEvent` → `Keymap.toVk` → the host, the
|
||||
* exact path a hardware keyboard takes (with the IME-shift wrap documented there).
|
||||
*
|
||||
* Doubles as the pointer-capture grab target: a grab needs a focusable view, and captured-pointer
|
||||
* events are delivered to it (routed to [MouseForwarder.onCapturedPointer] via the listener the
|
||||
* stream screen installs).
|
||||
*/
|
||||
private class KeyCaptureView(context: Context) : View(context) {
|
||||
init {
|
||||
@@ -438,17 +542,32 @@ private class KeyCaptureView(context: Context) : View(context) {
|
||||
isFocusableInTouchMode = true
|
||||
}
|
||||
|
||||
/** The session handle when the host types committed text; `0` = VK-only fallback. */
|
||||
var textHandle: Long = 0L
|
||||
|
||||
/** Whether [setImeVisible] last showed the IME — for toggle-style callers (remote pointer). */
|
||||
var imeShown = false
|
||||
private set
|
||||
|
||||
override fun onCheckIsTextEditor(): Boolean = true
|
||||
|
||||
override fun onCreateInputConnection(outAttrs: EditorInfo): InputConnection {
|
||||
outAttrs.imeOptions = EditorInfo.IME_FLAG_NO_EXTRACT_UI or
|
||||
EditorInfo.IME_FLAG_NO_FULLSCREEN or EditorInfo.IME_FLAG_NO_ENTER_ACTION
|
||||
return if (textHandle != 0L) {
|
||||
outAttrs.inputType = InputType.TYPE_CLASS_TEXT or
|
||||
InputType.TYPE_TEXT_FLAG_AUTO_CORRECT or InputType.TYPE_TEXT_FLAG_MULTI_LINE
|
||||
HostTextConnection(this, textHandle)
|
||||
} else {
|
||||
outAttrs.inputType = InputType.TYPE_NULL
|
||||
outAttrs.imeOptions = EditorInfo.IME_FLAG_NO_EXTRACT_UI or EditorInfo.IME_FLAG_NO_FULLSCREEN
|
||||
return BaseInputConnection(this, false)
|
||||
BaseInputConnection(this, false)
|
||||
}
|
||||
}
|
||||
|
||||
fun setImeVisible(show: Boolean) {
|
||||
val imm = context.getSystemService(Context.INPUT_METHOD_SERVICE) as? InputMethodManager
|
||||
?: return
|
||||
imeShown = show
|
||||
if (show) {
|
||||
requestFocus()
|
||||
imm.showSoftInput(this, 0)
|
||||
@@ -457,3 +576,113 @@ private class KeyCaptureView(context: Context) : View(context) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* IME → host text bridge (the `HOST_CAP_TEXT_INPUT` path): a real **editable** connection, so
|
||||
* the IME runs its full machinery (autocorrect, gesture typing, non-Latin composition), mirrored
|
||||
* to the host as it happens. The one piece of host-side state tracked is *what the host currently
|
||||
* shows of the active composition* ([sentComposition]): composing updates send a common-prefix
|
||||
* diff (backspaces + the new suffix) so corrections materialize live on the host; a commit
|
||||
* settles it. [setComposingRegion] adopts already-committed text as the active composition
|
||||
* (autocorrect-revert / backspace-into-word flows), so the next update diffs against it instead
|
||||
* of retyping. Newlines become Enter taps; [deleteSurroundingText] becomes Backspace/Delete taps.
|
||||
*
|
||||
* Known approximation: diff lengths are counted in Unicode scalars, assuming one host Backspace
|
||||
* deletes one scalar — true for the composition text IMEs actually produce (emoji and other
|
||||
* multi-unit graphemes commit directly rather than composing).
|
||||
*/
|
||||
private class HostTextConnection(
|
||||
view: KeyCaptureView,
|
||||
private val handle: Long,
|
||||
) : BaseInputConnection(view, true) {
|
||||
/** What the host currently shows of the active composition ("" = none). */
|
||||
private var sentComposition = ""
|
||||
|
||||
override fun commitText(text: CharSequence, newCursorPosition: Int): Boolean {
|
||||
retype(text.toString())
|
||||
sentComposition = ""
|
||||
val ok = super.commitText(text, newCursorPosition)
|
||||
trimEditable()
|
||||
return ok
|
||||
}
|
||||
|
||||
override fun setComposingText(text: CharSequence, newCursorPosition: Int): Boolean {
|
||||
retype(text.toString())
|
||||
return super.setComposingText(text, newCursorPosition)
|
||||
}
|
||||
|
||||
override fun finishComposingText(): Boolean {
|
||||
// The composition text stands as committed — the host already shows it verbatim.
|
||||
sentComposition = ""
|
||||
return super.finishComposingText()
|
||||
}
|
||||
|
||||
override fun setComposingRegion(start: Int, end: Int): Boolean {
|
||||
val e = editable
|
||||
if (e != null) {
|
||||
val a = start.coerceIn(0, e.length)
|
||||
val b = end.coerceIn(0, e.length)
|
||||
sentComposition = e.subSequence(minOf(a, b), maxOf(a, b)).toString()
|
||||
}
|
||||
return super.setComposingRegion(start, end)
|
||||
}
|
||||
|
||||
override fun deleteSurroundingText(beforeLength: Int, afterLength: Int): Boolean {
|
||||
repeat(beforeLength.coerceIn(0, MAX_TAPS)) { tapVk(VK_BACK) }
|
||||
repeat(afterLength.coerceIn(0, MAX_TAPS)) { tapVk(VK_DELETE) }
|
||||
return super.deleteSurroundingText(beforeLength, afterLength)
|
||||
}
|
||||
|
||||
override fun performEditorAction(actionCode: Int): Boolean {
|
||||
tapVk(VK_RETURN)
|
||||
return true
|
||||
}
|
||||
|
||||
/** Replace the host's view of the composition with [text] via a common-prefix diff. */
|
||||
private fun retype(text: String) {
|
||||
var common = sentComposition.commonPrefixWith(text)
|
||||
// Never split a surrogate pair mid-diff — back off to the pair boundary.
|
||||
if (common.isNotEmpty() && common.last().isHighSurrogate()) {
|
||||
common = common.dropLast(1)
|
||||
}
|
||||
val stale = sentComposition.substring(common.length)
|
||||
repeat(stale.codePointCount(0, stale.length).coerceAtMost(MAX_TAPS)) { tapVk(VK_BACK) }
|
||||
sendText(text.substring(common.length))
|
||||
sentComposition = text
|
||||
}
|
||||
|
||||
/** Forward literal text, turning newlines into Enter taps (control chars never ride text). */
|
||||
private fun sendText(s: String) {
|
||||
var chunk = StringBuilder()
|
||||
for (ch in s) {
|
||||
if (ch == '\n') {
|
||||
if (chunk.isNotEmpty()) {
|
||||
NativeBridge.nativeSendText(handle, chunk.toString())
|
||||
chunk = StringBuilder()
|
||||
}
|
||||
tapVk(VK_RETURN)
|
||||
} else {
|
||||
chunk.append(ch)
|
||||
}
|
||||
}
|
||||
if (chunk.isNotEmpty()) NativeBridge.nativeSendText(handle, chunk.toString())
|
||||
}
|
||||
|
||||
private fun tapVk(vk: Int) {
|
||||
NativeBridge.nativeSendKey(handle, vk, true, 0)
|
||||
NativeBridge.nativeSendKey(handle, vk, false, 0)
|
||||
}
|
||||
|
||||
/** Bound the mirror buffer: once nothing is composing, old text serves no purpose. */
|
||||
private fun trimEditable() {
|
||||
val e = editable ?: return
|
||||
if (getComposingSpanStart(e) == -1 && e.length > 4000) e.clear()
|
||||
}
|
||||
|
||||
private companion object {
|
||||
const val VK_BACK = 0x08
|
||||
const val VK_RETURN = 0x0D
|
||||
const val VK_DELETE = 0x2E
|
||||
const val MAX_TAPS = 256
|
||||
}
|
||||
}
|
||||
|
||||
@@ -99,9 +99,11 @@ internal suspend fun PointerInputScope.streamTouchPassthrough(handle: Long) {
|
||||
internal suspend fun PointerInputScope.streamTouchInput(
|
||||
handle: Long,
|
||||
trackpad: Boolean,
|
||||
invertScroll: Boolean,
|
||||
onCycleStats: () -> Unit,
|
||||
onKeyboard: (show: Boolean) -> Unit,
|
||||
) {
|
||||
val scrollDir = if (invertScroll) -1 else 1
|
||||
var lastTapUp = 0L
|
||||
var lastTapX = 0f
|
||||
var lastTapY = 0f
|
||||
@@ -184,12 +186,12 @@ internal suspend fun PointerInputScope.streamTouchInput(
|
||||
val sy = ((prevCy - cy) / SCROLL_DIV).toInt() // finger up → wheel up
|
||||
val sx = ((cx - prevCx) / SCROLL_DIV).toInt()
|
||||
if (sy != 0) {
|
||||
NativeBridge.nativeSendScroll(handle, 0, sy * 120)
|
||||
NativeBridge.nativeSendScroll(handle, 0, sy * 120 * scrollDir)
|
||||
prevCy = cy
|
||||
moved = true
|
||||
}
|
||||
if (sx != 0) {
|
||||
NativeBridge.nativeSendScroll(handle, 1, sx * 120)
|
||||
NativeBridge.nativeSendScroll(handle, 1, sx * 120 * scrollDir)
|
||||
prevCx = cx
|
||||
moved = true
|
||||
}
|
||||
|
||||
@@ -106,6 +106,17 @@ object Keymap {
|
||||
KeyEvent.KEYCODE_DPAD_UP -> 0x26
|
||||
KeyEvent.KEYCODE_DPAD_RIGHT -> 0x27
|
||||
KeyEvent.KEYCODE_DPAD_DOWN -> 0x28
|
||||
// TV-remote SELECT = Enter (a gamepad's press routes via SOURCE_GAMEPAD before this).
|
||||
KeyEvent.KEYCODE_DPAD_CENTER -> 0x0D
|
||||
|
||||
// Consumer/media keys — forwarded to the host while streaming (volume stays local:
|
||||
// MainActivity's pass-through list wins before the map is consulted).
|
||||
KeyEvent.KEYCODE_MEDIA_PLAY_PAUSE,
|
||||
KeyEvent.KEYCODE_MEDIA_PLAY,
|
||||
KeyEvent.KEYCODE_MEDIA_PAUSE -> 0xB3 // VK_MEDIA_PLAY_PAUSE
|
||||
KeyEvent.KEYCODE_MEDIA_NEXT -> 0xB0 // VK_MEDIA_NEXT_TRACK
|
||||
KeyEvent.KEYCODE_MEDIA_PREVIOUS -> 0xB1 // VK_MEDIA_PREV_TRACK
|
||||
KeyEvent.KEYCODE_MEDIA_STOP -> 0xB2 // VK_MEDIA_STOP
|
||||
|
||||
// Modifiers (L/R-specific VKs; the host folds the generic ones onto the left variant)
|
||||
KeyEvent.KEYCODE_SHIFT_LEFT -> 0xA0
|
||||
|
||||
@@ -287,6 +287,50 @@ object NativeBridge {
|
||||
/** One key transition. vk: Windows VK (0 = dropped by Rust). mods: VK modifier mask (0 for now). */
|
||||
external fun nativeSendKey(handle: Long, vk: Int, down: Boolean, mods: Int)
|
||||
|
||||
/**
|
||||
* Whether the host advertised committed-text injection (`HOST_CAP_TEXT_INPUT`) — its inject
|
||||
* backend can type Unicode text directly. Picks the real IME `InputConnection` (autocorrect,
|
||||
* gesture typing, non-Latin scripts) over the TYPE_NULL raw-key fallback. False on `0`.
|
||||
*/
|
||||
external fun nativeTextInputSupported(handle: Long): Boolean
|
||||
|
||||
/**
|
||||
* Committed IME text → one `TextInput` wire event per Unicode scalar, in order. Control
|
||||
* characters are skipped natively (Enter/Backspace ride [nativeSendKey]). Only meaningful
|
||||
* when [nativeTextInputSupported] returned true — older hosts ignore the events.
|
||||
*/
|
||||
external fun nativeSendText(handle: Long, text: String)
|
||||
|
||||
// ---- Shared clipboard (text v1): Kotlin drives ClipboardManager, Rust the protocol ----
|
||||
// Opt-in per session (nativeClipControl). Local copies are announced as lazy offers; bytes
|
||||
// cross only when the host pastes (a "fetch:" event answered by nativeClipServeText). Host
|
||||
// copies arrive as "offer:" events, fetched eagerly into the system clipboard.
|
||||
|
||||
/** Whether the host advertised a working shared-clipboard service (HOST_CAP_CLIPBOARD). */
|
||||
external fun nativeClipSupported(handle: Long): Boolean
|
||||
|
||||
/** Session-level clipboard opt-in/out; nothing happens until enabled=true crosses. */
|
||||
external fun nativeClipControl(handle: Long, enabled: Boolean)
|
||||
|
||||
/** Announce "this device's clipboard now holds text". [seq]: monotonic, newest wins. */
|
||||
external fun nativeClipOfferText(handle: Long, seq: Int)
|
||||
|
||||
/** Pull the text of the host's offer [seq] → transfer id echoed on "data:"/"error:", or -1. */
|
||||
external fun nativeClipFetchText(handle: Long, seq: Int): Int
|
||||
|
||||
/** Answer a "fetch:" event with the clipboard's current text (the host is pasting). */
|
||||
external fun nativeClipServeText(handle: Long, reqId: Int, text: String)
|
||||
|
||||
/** Abort a clipboard transfer by id (either direction). */
|
||||
external fun nativeClipCancel(handle: Long, id: Int)
|
||||
|
||||
/**
|
||||
* Block ≤250 ms for the next clipboard event, as a compact string: `state:<0|1>` ·
|
||||
* `offer:<seq>:<hasText>` · `fetch:<reqId>` · `data:<xferId>:<text>` · `cancel:<id>` ·
|
||||
* `error:<id>:<code>` · `closed` (session gone) — null on timeout. Dedicated poll thread.
|
||||
*/
|
||||
external fun nativeNextClip(handle: Long): String?
|
||||
|
||||
// ---- Gamepad: each controller forwarded on its own wire pad index (0..15, low byte of flags) ----
|
||||
// The pad index is assigned per Android device by GamepadRouter; a single controller lands on 0,
|
||||
// so its wire is byte-identical to the old single-pad path. The core folds the per-transition
|
||||
|
||||
@@ -404,7 +404,14 @@ fn feeder_loop(
|
||||
// stage is consumed: the HUD, or the ABR decode signal (`measure_decode`). The
|
||||
// HUD-only `received` point + host/network split stay gated on the overlay.
|
||||
if stats.enabled() || measure_decode {
|
||||
let received_ns = now_realtime_ns();
|
||||
// Core reassembly-completion stamp (ABI v9), NOT the pull instant: stamping
|
||||
// here would fold the hand-off queue wait into the network latency figure
|
||||
// (a client-side standing backlog masquerading as network). 0 = older core.
|
||||
let received_ns = if frame.received_ns > 0 {
|
||||
frame.received_ns as i128
|
||||
} else {
|
||||
now_realtime_ns()
|
||||
};
|
||||
{
|
||||
let mut g = in_flight
|
||||
.lock()
|
||||
|
||||
@@ -221,7 +221,13 @@ pub(super) fn run_sync(
|
||||
// samplers (`received` point, host/network split) stay gated on the overlay so
|
||||
// the hidden steady state adds only a wall-clock read + the receipt push.
|
||||
if stats.enabled() || measure_decode {
|
||||
let received_ns = now_realtime_ns();
|
||||
// Core reassembly-completion stamp (ABI v9), not the pull instant — see
|
||||
// async_loop: a pull stamp folds hand-off queue wait into "network".
|
||||
let received_ns = if frame.received_ns > 0 {
|
||||
frame.received_ns as i128
|
||||
} else {
|
||||
now_realtime_ns()
|
||||
};
|
||||
in_flight.push_back((frame.pts_ns / 1000, received_ns));
|
||||
if in_flight.len() > IN_FLIGHT_CAP {
|
||||
in_flight.pop_front(); // stale — codec never echoed it back
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
//! Shared-clipboard plane (text-only v1): Kotlin drives the Android `ClipboardManager`, these
|
||||
//! shims drive [`punktfunk_core::client::NativeClient`]'s clipboard surface.
|
||||
//!
|
||||
//! Model (mirrors the desktop clients): opt-in via `nativeClipControl(true)`; local copies are
|
||||
//! announced lazily as format-list offers (`nativeClipOfferText`) and the bytes cross only when
|
||||
//! the host pastes (a `fetch` event answered by `nativeClipServeText`); a host copy arrives as an
|
||||
//! `offer` event, which the Kotlin side fetches eagerly (Android's clipboard has no lazy provider
|
||||
//! path worth the complexity) and lands in the system clipboard on the `data` event.
|
||||
//!
|
||||
//! Events cross to Kotlin as compact strings from the blocking `nativeNextClip` poll (drained on
|
||||
//! a dedicated thread, same pattern as `nativeNextRumble`):
|
||||
//! `state:<0|1>` · `offer:<seq>:<has_text 0|1>` · `fetch:<req_id>` · `data:<xfer_id>:<text>` ·
|
||||
//! `cancel:<id>` · `error:<id>:<code>` · `closed` — null on a poll timeout. Non-text fetch
|
||||
//! requests are cancelled natively (only text is ever offered, so they shouldn't occur).
|
||||
|
||||
use std::time::Duration;
|
||||
|
||||
use jni::objects::{JObject, JString};
|
||||
use jni::sys::{jboolean, jint, jlong, jstring};
|
||||
use jni::JNIEnv;
|
||||
use punktfunk_core::clipboard::ClipEventCore;
|
||||
use punktfunk_core::error::PunktfunkError;
|
||||
use punktfunk_core::quic::{ClipKind, CLIP_FILE_INDEX_NONE, HOST_CAP_CLIPBOARD};
|
||||
|
||||
use super::SessionHandle;
|
||||
|
||||
/// The portable wire MIME both ends map to their platform text type.
|
||||
const TEXT_MIME: &str = "text/plain;charset=utf-8";
|
||||
|
||||
/// Deref the opaque handle (`0` → `None`).
|
||||
///
|
||||
/// SAFETY: live handle per the nativeConnect/nativeClose contract; every method used is `&self`
|
||||
/// on the `Sync` connector.
|
||||
fn client(handle: jlong) -> Option<&'static SessionHandle> {
|
||||
if handle == 0 {
|
||||
return None;
|
||||
}
|
||||
// SAFETY: see the function docs — the Kotlin side guarantees the handle outlives the call.
|
||||
Some(unsafe { &*(handle as *const SessionHandle) })
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipSupported(handle)` — the host advertised `HOST_CAP_CLIPBOARD`.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipSupported(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
) -> jboolean {
|
||||
client(handle).map_or(0, |h| {
|
||||
u8::from(h.client.host_caps() & HOST_CAP_CLIPBOARD != 0)
|
||||
})
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipControl(handle, enabled)` — session-level opt-in/out. Nothing
|
||||
/// clipboard-related happens on either side until an `enabled: true` crosses.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipControl(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
enabled: jboolean,
|
||||
) {
|
||||
if let Some(h) = client(handle) {
|
||||
let _ = h.client.clip_control(enabled != 0, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipOfferText(handle, seq)` — announce "the Android clipboard now holds
|
||||
/// text" (format list only; bytes cross when the host fetches). `seq` is Kotlin's monotonic
|
||||
/// counter, newest wins.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipOfferText(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
seq: jint,
|
||||
) {
|
||||
if let Some(h) = client(handle) {
|
||||
let _ = h.client.clip_offer(
|
||||
seq as u32,
|
||||
vec![ClipKind {
|
||||
mime: TEXT_MIME.into(),
|
||||
size_hint: 0,
|
||||
}],
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipFetchText(handle, seq)` — pull the text of the host's offer `seq`.
|
||||
/// Returns the transfer id echoed on the matching `data:`/`error:` event, or −1.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipFetchText(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
seq: jint,
|
||||
) -> jint {
|
||||
client(handle)
|
||||
.and_then(|h| {
|
||||
h.client
|
||||
.clip_fetch(seq as u32, TEXT_MIME.into(), CLIP_FILE_INDEX_NONE)
|
||||
.ok()
|
||||
})
|
||||
.map_or(-1, |xfer| xfer as jint)
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipServeText(handle, reqId, text)` — answer a `fetch:` event with the
|
||||
/// clipboard's current text (the host is pasting our offer).
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipServeText(
|
||||
mut env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
req_id: jint,
|
||||
text: JString,
|
||||
) {
|
||||
let Some(h) = client(handle) else { return };
|
||||
let Ok(s) = env.get_string(&text) else {
|
||||
let _ = h.client.clip_cancel(req_id as u32);
|
||||
return;
|
||||
};
|
||||
let _ = h
|
||||
.client
|
||||
.clip_serve(req_id as u32, String::from(s).into_bytes(), true);
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeClipCancel(handle, id)` — abort a transfer (either direction).
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeClipCancel(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
id: jint,
|
||||
) {
|
||||
if let Some(h) = client(handle) {
|
||||
let _ = h.client.clip_cancel(id as u32);
|
||||
}
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeNextClip(handle)` — block ≤250 ms for the next clipboard event, encoded
|
||||
/// as a compact string (module docs); null on timeout, `"closed"` once the session is gone.
|
||||
/// Call from a dedicated poll thread.
|
||||
///
|
||||
/// Text payloads ride `data:<xfer_id>:<text>` decoded lossily — safe because the phase-0
|
||||
/// clipboard task delivers a whole payload in ONE event (`last = true`), so a chunk boundary
|
||||
/// can never split a UTF-8 sequence.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeNextClip(
|
||||
env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
) -> jstring {
|
||||
let Some(h) = client(handle) else {
|
||||
return std::ptr::null_mut();
|
||||
};
|
||||
let msg = match h.client.next_clip(Duration::from_millis(250)) {
|
||||
Ok(ClipEventCore::State { enabled, .. }) => format!("state:{}", u8::from(enabled)),
|
||||
Ok(ClipEventCore::RemoteOffer { seq, kinds }) => {
|
||||
let has_text = kinds.iter().any(|k| k.mime.starts_with("text/plain"));
|
||||
format!("offer:{seq}:{}", u8::from(has_text))
|
||||
}
|
||||
Ok(ClipEventCore::FetchRequest { req_id, mime, .. }) => {
|
||||
if mime.starts_with("text/plain") {
|
||||
format!("fetch:{req_id}")
|
||||
} else {
|
||||
// We only ever offer text; cancel anything else rather than stall the host.
|
||||
let _ = h.client.clip_cancel(req_id);
|
||||
return std::ptr::null_mut();
|
||||
}
|
||||
}
|
||||
Ok(ClipEventCore::Data { xfer_id, bytes, .. }) => {
|
||||
format!("data:{xfer_id}:{}", String::from_utf8_lossy(&bytes))
|
||||
}
|
||||
Ok(ClipEventCore::Cancelled { id }) => format!("cancel:{id}"),
|
||||
Ok(ClipEventCore::Error { id, code }) => format!("error:{id}:{code}"),
|
||||
Err(PunktfunkError::NoFrame) => return std::ptr::null_mut(),
|
||||
Err(_) => "closed".into(),
|
||||
};
|
||||
env.new_string(msg)
|
||||
.map(|s| s.into_raw())
|
||||
.unwrap_or(std::ptr::null_mut())
|
||||
}
|
||||
@@ -6,11 +6,11 @@
|
||||
//! conventions: buttons 1=left/2=middle/3=right/4=X1/5=X2; scroll axis 0=vertical/1=horizontal,
|
||||
//! signed 120-unit delta, +=up/right; keys are Windows VK (mapped from KEYCODE_* on the Kotlin side).
|
||||
|
||||
use jni::objects::{JByteBuffer, JObject};
|
||||
use jni::objects::{JByteBuffer, JObject, JString};
|
||||
use jni::sys::{jboolean, jint, jlong};
|
||||
use jni::JNIEnv;
|
||||
use punktfunk_core::input::{InputEvent, InputKind};
|
||||
use punktfunk_core::quic::{RichInput, HID_REPORT_MAX};
|
||||
use punktfunk_core::quic::{RichInput, HID_REPORT_MAX, HOST_CAP_TEXT_INPUT};
|
||||
|
||||
use super::SessionHandle;
|
||||
|
||||
@@ -145,6 +145,45 @@ pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeSendKey(
|
||||
send_event(handle, kind, vk as u32, 0, 0, mods as u32);
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeTextInputSupported(handle)` — whether the host advertised
|
||||
/// `HOST_CAP_TEXT_INPUT` (its inject backend types committed text), so the Kotlin side can pick
|
||||
/// the real IME `InputConnection` over the TYPE_NULL raw-key fallback. `0` handle → false.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeTextInputSupported(
|
||||
_env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
) -> jboolean {
|
||||
if handle == 0 {
|
||||
return 0;
|
||||
}
|
||||
// SAFETY: live handle per the nativeConnect/nativeClose contract; host_caps is &self.
|
||||
let h = unsafe { &*(handle as *const SessionHandle) };
|
||||
u8::from(h.client.host_caps() & HOST_CAP_TEXT_INPUT != 0)
|
||||
}
|
||||
|
||||
/// `NativeBridge.nativeSendText(handle, text)` — committed IME text, one `TextInput` event per
|
||||
/// Unicode scalar (`code` = the scalar; multi-char commits are consecutive events in order).
|
||||
/// Control characters are skipped — Enter/Backspace/Tab ride the VK key path. Call only when
|
||||
/// [`Java_io_unom_punktfunk_kit_NativeBridge_nativeTextInputSupported`] returned true.
|
||||
#[no_mangle]
|
||||
pub extern "system" fn Java_io_unom_punktfunk_kit_NativeBridge_nativeSendText(
|
||||
mut env: JNIEnv,
|
||||
_this: JObject,
|
||||
handle: jlong,
|
||||
text: JString,
|
||||
) {
|
||||
if handle == 0 {
|
||||
return;
|
||||
}
|
||||
let Ok(s) = env.get_string(&text) else {
|
||||
return;
|
||||
};
|
||||
for ch in String::from(s).chars().filter(|c| !c.is_control()) {
|
||||
send_event(handle, InputKind::TextInput, ch as u32, 0, 0, 0);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Gamepad: Kotlin captures (KeyEvent/MotionEvent) → NativeClient::send_input ---------------
|
||||
// Multi-pad model: each physical controller is forwarded on its own wire pad index (0..15), carried
|
||||
// in the low byte of `flags` on every per-pad event — the Kotlin side (`GamepadRouter`) assigns a
|
||||
|
||||
@@ -17,6 +17,7 @@
|
||||
//! TODO(M4 Android stage 1): client→host DualSense rich input (`send_rich_input`), mode
|
||||
//! renegotiation. Port the remaining orchestration from `clients/linux`.
|
||||
|
||||
mod clipboard;
|
||||
mod connect;
|
||||
mod input;
|
||||
mod planes;
|
||||
|
||||
@@ -579,13 +579,21 @@ struct ContentView: View {
|
||||
model?.disconnect() // the captured-state ⌃⌥⇧D combo
|
||||
},
|
||||
onFrame: { [meter = model.meter, latency = model.latency,
|
||||
split = model.latencySplit, offset = conn.clockOffsetNs] au in
|
||||
split = model.latencySplit, queue = model.clientQueue,
|
||||
offset = conn.clockOffsetNs] au in
|
||||
meter.note(byteCount: au.data.count)
|
||||
latency.record(ptsNs: au.ptsNs, offsetNs: offset)
|
||||
// The same receipt, keyed by pts, awaiting its 0xCF host timing (the
|
||||
// host/network split — drained by the 1 s stats tick).
|
||||
// host/network split — drained by the 1 s stats tick). receivedNs is
|
||||
// the core's reassembly stamp (ABI v9), so the split's network term no
|
||||
// longer contains the client-queue wait...
|
||||
split.recordReceipt(
|
||||
ptsNs: au.ptsNs, receivedNs: au.receivedNs, offsetNs: offset)
|
||||
// ...which is measured as its own term instead (receipt→pull, both
|
||||
// client-local).
|
||||
queue.record(
|
||||
ptsNs: UInt64(bitPattern: au.receivedNs), atNs: au.pulledNs,
|
||||
offsetNs: 0)
|
||||
},
|
||||
onSessionEnd: { [weak model] in
|
||||
Task { @MainActor in model?.sessionEnded() }
|
||||
|
||||
@@ -102,6 +102,12 @@ final class SessionModel: ObservableObject {
|
||||
@Published var decodeValid = false
|
||||
@Published var displayP50Ms = 0.0
|
||||
@Published var displayValid = false
|
||||
/// Client-queue wait: core reassembly receipt → the pump's pull (`AccessUnit.pulledNs −
|
||||
/// receivedNs`, ABI v9 receipt split — the 2026-07 two-pair investigation). ~0 on a healthy
|
||||
/// stream; a persistent value is a client-side standing backlog that used to hide inside
|
||||
/// "network". Shown in the detailed tier only when it says something (≥ ~2 ms).
|
||||
@Published var clientQueueP50Ms = 0.0
|
||||
@Published var clientQueueValid = false
|
||||
/// The measured OS present floor (design/apple-presentation-rebuild.md): the deadline
|
||||
/// engine's vend→glass pipeline depth — an OS property no client can pace under (~2 refresh
|
||||
/// intervals composited; would read ~1 under direct-to-display). The HUD subtracts it from
|
||||
@@ -147,6 +153,9 @@ final class SessionModel: ObservableObject {
|
||||
let endToEnd = LatencyMeter()
|
||||
let decodeStage = LatencyMeter()
|
||||
let displayStage = LatencyMeter()
|
||||
/// Client-queue sampler (see `clientQueueP50Ms`) — fed per AU by the stream view's onFrame,
|
||||
/// drained by the same 1 s tick as the stage meters.
|
||||
let clientQueue = LatencyMeter()
|
||||
/// The OS present floor sampler (see `osFloorP50Ms`) — fed one sample per display-link
|
||||
/// update by the deadline engine, drained by the same 1 s tick as the stage meters.
|
||||
let presentFloor = LatencyMeter()
|
||||
@@ -489,6 +498,7 @@ final class SessionModel: ObservableObject {
|
||||
endToEndValid = false
|
||||
decodeValid = false
|
||||
displayValid = false
|
||||
clientQueueValid = false
|
||||
osFloorValid = false
|
||||
lostFrames = 0
|
||||
lostPct = 0
|
||||
@@ -679,6 +689,12 @@ final class SessionModel: ObservableObject {
|
||||
} else {
|
||||
self.osFloorValid = false
|
||||
}
|
||||
if let q = self.clientQueue.drain() {
|
||||
self.clientQueueP50Ms = q.p50Ms
|
||||
self.clientQueueValid = true
|
||||
} else {
|
||||
self.clientQueueValid = false
|
||||
}
|
||||
// Mirror the window to the unified log (see statsLog) — one line per second,
|
||||
// stages in ms, only while frames actually flowed. `fps` counts RECEIVED AUs;
|
||||
// `presents` counts frames that reached glass (the display meter's sample count)
|
||||
@@ -689,9 +705,12 @@ final class SessionModel: ObservableObject {
|
||||
// captured before the 2026-07 floor policy); the appended trio carries the
|
||||
// measured OS present floor and the floor-shaved values the HUD displays.
|
||||
let line = String(
|
||||
format: "fps=%d presents=%d e2e_p50=%.1f e2e_p95=%.1f hostnet_p50=%.1f "
|
||||
+ "decode_p50=%.1f display_p50=%.1f lost=%d "
|
||||
+ "floor_p50=%.1f display_adj=%.1f e2e_adj=%.1f",
|
||||
// Swift Int is 64-bit → %lld, NOT %d (which is a 32-bit C int); macOS 26's
|
||||
// strict String(format:) validator rejects the %d/Int mismatch and drops
|
||||
// the whole line (a cascade error that also mis-blames the float args).
|
||||
format: "fps=%lld presents=%lld e2e_p50=%.1f e2e_p95=%.1f hostnet_p50=%.1f "
|
||||
+ "decode_p50=%.1f display_p50=%.1f lost=%lld "
|
||||
+ "floor_p50=%.1f display_adj=%.1f e2e_adj=%.1f queue_p50=%.1f",
|
||||
frames,
|
||||
displayWindow?.count ?? 0,
|
||||
self.endToEndValid ? self.endToEndP50Ms : -1,
|
||||
@@ -702,7 +721,8 @@ final class SessionModel: ObservableObject {
|
||||
lost,
|
||||
self.osFloorValid ? self.osFloorP50Ms : -1,
|
||||
self.displayValid ? self.displayAdjP50Ms : -1,
|
||||
self.endToEndValid ? self.endToEndAdjP50Ms : -1)
|
||||
self.endToEndValid ? self.endToEndAdjP50Ms : -1,
|
||||
self.clientQueueValid ? self.clientQueueP50Ms : -1)
|
||||
statsLog.info("\(line, privacy: .public)")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -118,6 +118,16 @@ struct StreamHUDView: View {
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.tertiary)
|
||||
}
|
||||
// Client-queue wait (reassembly receipt → decode pull, ABI v9 split): ~0 on
|
||||
// a healthy stream and hidden as noise; shown from 2 ms — a persistent value
|
||||
// is a client-side standing backlog that pre-split builds displayed as
|
||||
// "network" (the 2026-07 two-pair plateau). The core's standing-latency
|
||||
// bleed logs alongside when it acts on the same state.
|
||||
if model.clientQueueValid && model.clientQueueP50Ms >= 2 {
|
||||
Text("client queue +\(model.clientQueueP50Ms, specifier: "%.1f") (receive backlog — standing if it persists)")
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.tertiary)
|
||||
}
|
||||
}
|
||||
} else if model.hostNetworkValid {
|
||||
// Stage-1 fallback presenter: the layer decodes + presents internally with no
|
||||
|
||||
@@ -43,6 +43,9 @@ struct GamepadSettingsView: View {
|
||||
@AppStorage(DefaultsKey.presentPriority) private var presentPriority =
|
||||
SettingsOptions.presentPriorityDefault
|
||||
@AppStorage(DefaultsKey.smoothBuffer) private var smoothBuffer = 0
|
||||
#if os(macOS)
|
||||
@AppStorage(DefaultsKey.windowedSafePresent) private var windowedSafePresent = true
|
||||
#endif
|
||||
#if os(iOS)
|
||||
@AppStorage(DefaultsKey.rumbleOnDevice) private var rumbleOnDevice = false
|
||||
#endif
|
||||
@@ -345,6 +348,22 @@ struct GamepadSettingsView: View {
|
||||
detail: "Turn off to use the touch interface even with a controller connected.",
|
||||
value: $gamepadUIEnabled),
|
||||
]
|
||||
#if os(macOS)
|
||||
// The windowed safe-present toggle slots in after "Smoothness buffer" (staying inside
|
||||
// the Video group) — macOS only, mirroring the touch SettingsView's Presentation row
|
||||
// (the DCP swapID-panic mitigation; see DefaultsKey.windowedSafePresent).
|
||||
if let at = list.firstIndex(where: { $0.id == "smoothBuffer" }) {
|
||||
list.insert(
|
||||
toggleRow(
|
||||
id: "windowedSafePresent", icon: "macwindow.badge.plus",
|
||||
label: "Safe windowed presentation",
|
||||
detail: "Windowed streams present in step with the compositor — avoids a "
|
||||
+ "macOS display-driver crash on high-refresh displays, at a small "
|
||||
+ "latency cost. Fullscreen always uses the fastest path.",
|
||||
value: $windowedSafePresent),
|
||||
at: at + 1)
|
||||
}
|
||||
#endif
|
||||
#if os(iOS)
|
||||
// The device-rumble mirror slots in after "Controller type" (staying inside the
|
||||
// Controller group — the next row carries the "Interface" header). iPhone only in
|
||||
|
||||
@@ -300,6 +300,18 @@ extension SettingsView {
|
||||
+ "of added latency. Off shows frames as soon as they're ready.") {
|
||||
Toggle("V-Sync", isOn: $vsync)
|
||||
}
|
||||
// The DCP swapID-panic mitigation's user handle (see DefaultsKey.windowedSafePresent
|
||||
// for the saga). Default ON: turning it off re-arms a WHOLE-MACHINE kernel panic on
|
||||
// affected setups, so the caption says so in plain words.
|
||||
described(windowedSafePresent
|
||||
? "Windowed streams present in step with the system compositor — avoids a macOS "
|
||||
+ "display-driver crash seen on high-refresh displays, at a small latency "
|
||||
+ "cost. Fullscreen always uses the fastest path."
|
||||
: "Windowed streams use the fastest present path. On some high-refresh setups "
|
||||
+ "this can crash macOS itself (kernel panic) — turn back on if your Mac "
|
||||
+ "restarts during windowed streaming.") {
|
||||
Toggle("Safe windowed presentation", isOn: $windowedSafePresent)
|
||||
}
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -40,6 +40,7 @@ struct SettingsView: View {
|
||||
@AppStorage(DefaultsKey.smoothBuffer) var smoothBuffer = 0
|
||||
#if os(macOS)
|
||||
@AppStorage(DefaultsKey.vsync) var vsync = false
|
||||
@AppStorage(DefaultsKey.windowedSafePresent) var windowedSafePresent = true
|
||||
#endif
|
||||
#if !os(tvOS)
|
||||
@AppStorage(DefaultsKey.allowVRR) var allowVRR = true
|
||||
|
||||
@@ -35,10 +35,31 @@ public struct AccessUnit: Sendable {
|
||||
public let ptsNs: UInt64
|
||||
public let frameIndex: UInt32
|
||||
public let flags: UInt32
|
||||
/// Client `CLOCK_REALTIME` instant the AU was handed over by the core (post-FEC, decrypted)
|
||||
/// — the **received** measurement point of design/stats-unification.md. The decode stage is
|
||||
/// `decodedNs - receivedNs`, both client-local (no skew offset applies).
|
||||
/// Client `CLOCK_REALTIME` instant the AU finished reassembly in the core (post-FEC,
|
||||
/// decrypted — `PunktfunkFrame.received_ns`, ABI v9) — the **received** measurement point of
|
||||
/// design/stats-unification.md. NOT the pull instant: stamping at the pull folded the
|
||||
/// pre-decode hand-off wait into the network term, which is how the 2026-07 two-pair
|
||||
/// standing-latency plateau hid as "network". The decode stage is `decodedNs - receivedNs`,
|
||||
/// both client-local (no skew offset applies).
|
||||
public let receivedNs: Int64
|
||||
/// Client `CLOCK_REALTIME` instant this pull returned. `pulledNs - receivedNs` is the
|
||||
/// client-queue wait (kernel hand-off + FrameChannel dwell) — the term the HUD splits out
|
||||
/// so a client-side standing backlog can never masquerade as network latency again.
|
||||
public let pulledNs: Int64
|
||||
|
||||
/// `pulledNs` defaults to `receivedNs` (zero queue wait) for callers with no pull instant —
|
||||
/// the synthetic probe AUs and decode tests, where the split is meaningless.
|
||||
public init(
|
||||
data: Data, ptsNs: UInt64, frameIndex: UInt32, flags: UInt32,
|
||||
receivedNs: Int64, pulledNs: Int64? = nil
|
||||
) {
|
||||
self.data = data
|
||||
self.ptsNs = ptsNs
|
||||
self.frameIndex = frameIndex
|
||||
self.flags = flags
|
||||
self.receivedNs = receivedNs
|
||||
self.pulledNs = pulledNs ?? receivedNs
|
||||
}
|
||||
}
|
||||
|
||||
/// One Opus audio packet (48 kHz stereo, 5 ms frames) — decode with AVAudioConverter
|
||||
@@ -662,11 +683,16 @@ public final class PunktfunkConnection {
|
||||
let data = Data(bytes: base, count: Int(frame.len)) // copy: ptr valid only until next call
|
||||
var ts = timespec()
|
||||
clock_gettime(CLOCK_REALTIME, &ts)
|
||||
let receivedNs = Int64(ts.tv_sec) * 1_000_000_000 + Int64(ts.tv_nsec)
|
||||
let pulledNs = Int64(ts.tv_sec) * 1_000_000_000 + Int64(ts.tv_nsec)
|
||||
// Receipt = the core's reassembly-completion stamp (ABI v9); the pull instant is
|
||||
// kept separately so the client-queue wait is its own measured term. 0 would mean a
|
||||
// pre-v9 core — impossible here (core and Kit ship in one binary), but fall back to
|
||||
// the pull instant rather than record a 1970 receipt.
|
||||
let receivedNs = frame.received_ns > 0 ? Int64(frame.received_ns) : pulledNs
|
||||
return AccessUnit(
|
||||
data: data, ptsNs: frame.pts_ns,
|
||||
frameIndex: frame.frame_index, flags: frame.flags,
|
||||
receivedNs: receivedNs)
|
||||
receivedNs: receivedNs, pulledNs: pulledNs)
|
||||
case statusNoFrame:
|
||||
return nil
|
||||
case statusClosed:
|
||||
|
||||
@@ -23,6 +23,34 @@ import os
|
||||
|
||||
private let presenterLog = Logger(subsystem: "io.unom.punktfunk", category: "presenter")
|
||||
|
||||
#if os(macOS)
|
||||
/// HOW a windowed (composited) macOS session pushes finished frames to glass — the DCP
|
||||
/// "mismatched swapID's" kernel-panic saga's mechanism picker. Fullscreen always presents
|
||||
/// `async` (direct-scanout promotion, lowest latency, no panic reports there); the windowed
|
||||
/// mechanism is resolved per session by SessionPresenter (user setting +
|
||||
/// PUNKTFUNK_WINDOWED_PRESENT env override) and routed here via `setWindowedPresent`.
|
||||
///
|
||||
/// - `async`: the CAMetalLayer image queue (`commandBuffer.present`) — the fastest composited
|
||||
/// path and the PANIC TRIGGER on high-refresh displays (the out-of-band swaps race
|
||||
/// WindowServer's compositor; it survived glass pacing and every codec).
|
||||
/// - `transaction`: `CAMetalLayer.presentsWithTransaction` — the swap commits WITH the layer
|
||||
/// tree, in lockstep with the compositor (Apple's documented remedy; validated no-panic on
|
||||
/// the 240 Hz repro machine). The present is committed from the RENDER thread inside an
|
||||
/// explicit CATransaction + flush — see `encodePresent` for why that beats the original
|
||||
/// main-thread hop.
|
||||
/// - `surface`: no image queue at all — render into a pooled IOSurface and swap it into a plain
|
||||
/// CALayer's `contents` (the f407f418 PyroWave mitigation, resurrected format-aware:
|
||||
/// rgba16Float + PQ tagging keeps HDR). WindowServer treats it as ordinary layer damage on
|
||||
/// its own composite cadence. PROTOTYPE: whether the compositor honors PQ/EDR for plain-layer
|
||||
/// IOSurface contents still needs an on-glass eyeball — the metal layer stays underneath with
|
||||
/// `wantsExtendedDynamicRangeContent` as the EDR anchor.
|
||||
enum WindowedPresentMode: String, Sendable {
|
||||
case async
|
||||
case transaction
|
||||
case surface
|
||||
}
|
||||
#endif
|
||||
|
||||
/// HDR reference white (BT.2408 "HDR Reference White"): the absolute luminance, in nits, that the
|
||||
/// PQ signal's diffuse white sits at. Passed to `CAEDRMetadata.hdr10(opticalOutputScale:)`, it anchors
|
||||
/// 203-nit diffuse white at EDR 1.0 (the display's SDR-white level) and lets the system tone-map the
|
||||
@@ -198,8 +226,8 @@ fragment float4 pf_frag_hdr(VOut in [[stage_in]],
|
||||
// in a genuine HDR10 output, PQ passthrough is the correct emission and the TV tone-maps.)
|
||||
// The shared PQ→display-referred-SDR tail (see pf_frag_hdr_tv's rationale above): ST 2084
|
||||
// EOTF → 203-nit-anchored scene light → BT.2020→709 primaries → extended-Reinhard rolloff →
|
||||
// BT.709 OETF. Used by the tvOS biplanar tone-map and the planar (PyroWave) tone-map — the
|
||||
// latter also on macOS windowed sessions, whose IOSurface present path is BGRA8-only.
|
||||
// BT.709 OETF. Used by the tvOS biplanar tone-map and the tvOS planar (PyroWave) tone-map (the
|
||||
// no-HDR-headroom fallback). macOS keeps real HDR windowed now — see `WindowedPresentMode`.
|
||||
static inline float3 pqToSdr(float3 pq) {
|
||||
const float m1 = 2610.0/16384.0;
|
||||
const float m2 = 78.84375;
|
||||
@@ -230,8 +258,7 @@ fragment float4 pf_frag_hdr_tv(VOut in [[stage_in]],
|
||||
|
||||
// PyroWave planar HDR tone-map: three separate R16 planes (P010-style studio codes; the rows
|
||||
// fold in depth-10 MSB packing) → PQ R′G′B′ → the shared SDR tail. Used when a PQ pyrowave
|
||||
// stream must land on an 8-bit surface: tvOS without HDR headroom, and macOS WINDOWED sessions
|
||||
// (the IOSurface present path — the DCP-panic mitigation — is BGRA8). The passthrough planar
|
||||
// stream must land on an 8-bit surface: tvOS without HDR headroom. The passthrough planar
|
||||
// HDR pipeline reuses pf_frag_planar itself on an rgba16Float drawable (identical math — the
|
||||
// layer's itur_2100_PQ colour space + EDR metadata do the interpretation).
|
||||
fragment float4 pf_frag_planar_tm(VOut in [[stage_in]],
|
||||
@@ -259,21 +286,51 @@ public final class MetalVideoPresenter {
|
||||
public let layer: CAMetalLayer
|
||||
|
||||
#if os(macOS)
|
||||
/// The WINDOWED-mode PyroWave present target: a plain CALayer sized like `layer` (installed
|
||||
/// as a sibling ABOVE it), fed IOSurfaces via `contents` inside ordinary CATransactions.
|
||||
/// WINDOWED-mode present coordination — the macOS DCP KERNEL PANIC mitigation.
|
||||
///
|
||||
/// Why this exists — the macOS DCP KERNEL PANIC ("mismatched swapID's" @UnifiedPipeline.cpp,
|
||||
/// WindowServer dies, machine reboots): out-of-band CAMetalLayer image-queue swaps into a
|
||||
/// COMPOSITED (windowed) session race WindowServer's own swap submissions on high-refresh
|
||||
/// displays, and the race survives glass pacing — a fully serialized one-in-flight present
|
||||
/// stream still panicked a 240 Hz Mac Studio (2026-07-18, twice). So in windowed mode we stop
|
||||
/// using the image queue entirely and present the way video players do: render the planar CSC
|
||||
/// into an IOSurface pool and swap `contents` on main — WindowServer treats it as ordinary
|
||||
/// damage on its own composite cadence, coalescing faster-than-refresh updates instead of
|
||||
/// latching queue swaps mid-cycle. Fullscreen keeps the CAMetalLayer path (direct-scanout
|
||||
/// promotion, no compositing, no panic reports). Contents updates are transparent to the
|
||||
/// layer below when nil, so flipping modes just covers/uncovers the metal layer.
|
||||
public let surfaceLayer: CALayer = {
|
||||
/// The panic ("mismatched swapID's" @UnifiedPipeline.cpp, WindowServer dies, machine reboots):
|
||||
/// the CAMetalLayer's ASYNCHRONOUS image queue (`commandBuffer.present(drawable)` — an
|
||||
/// out-of-band flip, mandatory with `displaySyncEnabled=false`) diverges from WindowServer's
|
||||
/// compositor on a high-refresh COMPOSITED (windowed) session — the compositor's notion of the
|
||||
/// current swap and the layer's queued swap disagree, and the DCP asserts. It survived glass
|
||||
/// pacing: a fully serialized one-in-flight present stream still panicked a 240 Hz Mac Studio
|
||||
/// (2026-07-18, PyroWave), and a windowed HEVC session panicked the same machine 2026-07-21 —
|
||||
/// so it is the async image queue itself, at any pacing or codec, not a present rate.
|
||||
///
|
||||
/// The fix keeps the full render path (rgba16Float / PQ / EDR — real HDR is preserved) and
|
||||
/// only changes HOW the drawable is presented: `CAMetalLayer.presentsWithTransaction`. With it
|
||||
/// set, we don't hand the drawable to the command buffer; we commit, wait until scheduled, then
|
||||
/// call `drawable.present()` INSIDE a CATransaction — the present is enrolled in Core
|
||||
/// Animation's transaction and committed together with the layer tree, so the swap stays in
|
||||
/// lockstep with the compositor instead of racing it (Apple's documented remedy for Metal
|
||||
/// presentation drifting out of sync with CA). Fullscreen keeps the async path (direct-scanout
|
||||
/// promotion, lowest latency, no compositor and no panic reports there).
|
||||
///
|
||||
/// 2026-07-21 latency rework: the mitigation MECHANISM is now a three-way pick
|
||||
/// (`WindowedPresentMode`) and the transactional present commits from the RENDER thread —
|
||||
/// see `encodePresent`. Staged under `stagingLock` (main pushes it via
|
||||
/// `setComposited`→`setWindowedPresent`); the render thread drains it and toggles the layer
|
||||
/// property + present style. `Active` is the render-thread copy so the layer property flips
|
||||
/// exactly once per mode change.
|
||||
private var windowedPresentStaged: WindowedPresentMode = .async
|
||||
private var windowedPresentActive: WindowedPresentMode = .async
|
||||
|
||||
/// PUNKTFUNK_TXN_PRESENT=main — the ORIGINAL transactional present (commit →
|
||||
/// waitUntilScheduled → hop to the MAIN thread and present inside its CATransaction), kept
|
||||
/// as a field A/B lever. The default is the render-thread commit: the present harness
|
||||
/// (2026-07-21, this saga) measured the main hop landing a runloop turn late on a busy main
|
||||
/// thread, and an ACTIVE implicit transaction there NESTS the explicit one — presents batch
|
||||
/// at runloop-iteration rate (the field's presents=55 @ fps=240, display_p50 18.6 ms).
|
||||
/// Off-main commits measured immune to main-thread churn (~10 ms glass p50 at 240 Hz
|
||||
/// full-size vs 14+ ms under a churned main hop).
|
||||
private let txnPresentOnMain =
|
||||
ProcessInfo.processInfo.environment["PUNKTFUNK_TXN_PRESENT"] == "main"
|
||||
|
||||
/// The WINDOWED-mode `surface` present target: a plain CALayer sized like `layer` (installed
|
||||
/// as a sibling ABOVE it by SessionPresenter), fed IOSurfaces via `contents` inside explicit
|
||||
/// CATransactions. Transparent (nil contents) whenever surface mode is off, so the metal
|
||||
/// layer below shows through. See `WindowedPresentMode.surface`.
|
||||
let surfaceLayer: CALayer = {
|
||||
let l = CALayer()
|
||||
l.contentsGravity = .resize // frame is already aspect-fit + pixel-snapped by layout
|
||||
l.isOpaque = true
|
||||
@@ -281,8 +338,8 @@ public final class MetalVideoPresenter {
|
||||
return l
|
||||
}()
|
||||
|
||||
/// One IOSurface-backed render target of the windowed present pool. All pool state is
|
||||
/// RENDER-THREAD confined; only the immutable surface refs cross to main (contents swap).
|
||||
/// One IOSurface-backed render target of the windowed surface-present pool. All pool state
|
||||
/// is RENDER-THREAD confined; only the immutable surface refs cross threads (contents swap).
|
||||
private struct SurfaceSlot {
|
||||
let surface: IOSurfaceRef
|
||||
let texture: MTLTexture
|
||||
@@ -292,15 +349,52 @@ public final class MetalVideoPresenter {
|
||||
|
||||
private var surfacePool: [SurfaceSlot] = []
|
||||
private var surfacePoolSize: CGSize = .zero
|
||||
private var surfacePoolHDR = false
|
||||
private var surfaceSeq: UInt64 = 0
|
||||
/// Index of the slot most recently handed to the layer — never rewritten next, even if its
|
||||
/// use count already dropped (the compositor may still be scanning out the previous frame).
|
||||
private var lastHandedOff: Int?
|
||||
/// Staged (under `stagingLock`, like every cross-thread input): the hosting view's windowed
|
||||
/// vs fullscreen state, pushed from main via `setSurfacePresents`. Drained in `renderPlanar`.
|
||||
private var surfacePresentsStaged = false
|
||||
/// Render-thread copy, so pool teardown happens exactly once on a mode flip.
|
||||
private var surfacePresentsActive = false
|
||||
|
||||
/// Once-per-second decomposition of the ACTIVE windowed present path (the field-diagnosis
|
||||
/// half of the DCP-latency work): scheduled/completed wait + commit/flush cost per present,
|
||||
/// and how many presents/swaps were issued. The pf-present line shows the GLASS side
|
||||
/// (latchMs / dropped); this shows the ISSUE side. Logged via `presenterLog` only while a
|
||||
/// windowed mechanism is active (zero cost fullscreen). Lock-guarded: transaction mode
|
||||
/// records from the render thread, surface mode from Metal completion threads.
|
||||
private final class WindowedPresentDiag: @unchecked Sendable {
|
||||
private let lock = NSLock()
|
||||
private var presents = 0
|
||||
private var schedMs: [Double] = []
|
||||
private var commitMs: [Double] = []
|
||||
private var last = CACurrentMediaTime()
|
||||
|
||||
func record(schedMs sched: Double, commitMs commit: Double, mode: WindowedPresentMode) {
|
||||
lock.lock()
|
||||
presents += 1
|
||||
schedMs.append(sched)
|
||||
commitMs.append(commit)
|
||||
let now = CACurrentMediaTime()
|
||||
guard now - last >= 1 else {
|
||||
lock.unlock()
|
||||
return
|
||||
}
|
||||
last = now
|
||||
let sSched = schedMs.sorted()
|
||||
let sCommit = commitMs.sorted()
|
||||
let line = String(
|
||||
format: "pf-windowed mode=%@ presents=%d schedMs p50=%.2f max=%.2f "
|
||||
+ "commitMs p50=%.2f max=%.2f",
|
||||
mode.rawValue, presents, sSched[sSched.count / 2], sSched.last ?? 0,
|
||||
sCommit[sCommit.count / 2], sCommit.last ?? 0)
|
||||
presents = 0
|
||||
schedMs.removeAll(keepingCapacity: true)
|
||||
commitMs.removeAll(keepingCapacity: true)
|
||||
lock.unlock()
|
||||
presenterLog.info("\(line, privacy: .public)")
|
||||
}
|
||||
}
|
||||
|
||||
private let windowedDiag = WindowedPresentDiag()
|
||||
#endif
|
||||
|
||||
private let device: MTLDevice
|
||||
@@ -316,7 +410,7 @@ public final class MetalVideoPresenter {
|
||||
private let pipelinePlanar: MTLRenderPipelineState
|
||||
/// PyroWave planar HDR passthrough (pf_frag_planar → rgba16Float; the layer's PQ colour
|
||||
/// space + EDR interpret the samples) and the planar PQ→SDR tone-map (pf_frag_planar_tm →
|
||||
/// bgra8; tvOS without headroom + macOS windowed IOSurface presents).
|
||||
/// bgra8; tvOS without HDR headroom).
|
||||
private let pipelinePlanarHDR: MTLRenderPipelineState
|
||||
private let pipelinePlanarToneMap: MTLRenderPipelineState
|
||||
private var textureCache: CVMetalTextureCache?
|
||||
@@ -591,13 +685,14 @@ public final class MetalVideoPresenter {
|
||||
}
|
||||
|
||||
#if os(macOS)
|
||||
/// Park the windowed-vs-fullscreen present routing (MAIN thread — the hosting view pushes its
|
||||
/// window state on every layout). true = PyroWave frames present via `surfaceLayer` contents
|
||||
/// (the DCP swapID-panic mitigation — see `surfaceLayer`); false = the CAMetalLayer path.
|
||||
/// Park the windowed present mechanism (MAIN thread — the hosting view pushes its window
|
||||
/// state on every layout; SessionPresenter resolves the mechanism per session). `.async` =
|
||||
/// FULLSCREEN (or the user opted out of the mitigation): the image queue. `.transaction` /
|
||||
/// `.surface` = COMPOSITED (windowed) mitigation mechanisms — see `WindowedPresentMode`.
|
||||
/// Applied by the render thread on the next frame, like every other staged value here.
|
||||
public func setSurfacePresents(_ on: Bool) {
|
||||
func setWindowedPresent(_ mode: WindowedPresentMode) {
|
||||
stagingLock.lock()
|
||||
surfacePresentsStaged = on
|
||||
windowedPresentStaged = mode
|
||||
stagingLock.unlock()
|
||||
}
|
||||
#endif
|
||||
@@ -734,36 +829,12 @@ public final class MetalVideoPresenter {
|
||||
) -> Bool {
|
||||
stagingLock.lock()
|
||||
let targetFromLayout = drawableTarget
|
||||
#if os(macOS)
|
||||
let surfaceMode = surfacePresentsStaged
|
||||
#endif
|
||||
stagingLock.unlock()
|
||||
// A PQ (HDR) pyrowave stream drives the same layer/EDR machinery as the biplanar path;
|
||||
// macOS WINDOWED sessions stay on the SDR layer (the IOSurface path tone-maps in-shader).
|
||||
#if os(macOS)
|
||||
configure(hdr: planes.pq && !surfaceMode)
|
||||
#else
|
||||
// A PQ (HDR) pyrowave stream drives the same layer/EDR machinery as the biplanar path —
|
||||
// including macOS windowed sessions, which keep real HDR (the DCP mitigation is the
|
||||
// transactional present in `encodePresent`, not a colour downgrade).
|
||||
configure(hdr: planes.pq)
|
||||
#endif
|
||||
var csc = planes.csc
|
||||
#if os(macOS)
|
||||
if surfaceMode != surfacePresentsActive {
|
||||
surfacePresentsActive = surfaceMode
|
||||
presenterLog.info(
|
||||
"stage2: windowed surface presents \(surfaceMode ? "ON" : "OFF", privacy: .public) (PyroWave DCP-panic mitigation)")
|
||||
if !surfaceMode {
|
||||
// Back to the metal path (fullscreen): drop the pool — at 5K it holds >100 MB,
|
||||
// and re-entering windowed mode rebuilds it in one frame.
|
||||
surfacePool.removeAll()
|
||||
surfacePoolSize = .zero
|
||||
lastHandedOff = nil
|
||||
}
|
||||
}
|
||||
if surfaceMode {
|
||||
return renderPlanarToSurface(
|
||||
planes, targetFromLayout: targetFromLayout, csc: &csc, onPresented: onPresented)
|
||||
}
|
||||
#endif
|
||||
// PQ passthrough needs the HDR drawable; a PQ frame while the drawable is (still)
|
||||
// 8-bit — tvOS without display headroom, or a not-yet-flipped layer — tone-maps
|
||||
// in-shader instead (the pipeline must match the drawable's pixel format).
|
||||
@@ -792,118 +863,6 @@ public final class MetalVideoPresenter {
|
||||
}
|
||||
}
|
||||
|
||||
#if os(macOS)
|
||||
/// The windowed-mode present tail (see `surfaceLayer` for why this path exists): render the
|
||||
/// planar CSC into a pooled IOSurface and hand it to `surfaceLayer.contents` on MAIN inside a
|
||||
/// plain CATransaction — an ordinary damaged-layer update on WindowServer's own composite
|
||||
/// cadence, no CAMetalLayer image-queue swap anywhere. `presentAtMediaTime` doesn't apply
|
||||
/// (the compositor paces); `onPresented` fires after the contents swap is committed, stamped
|
||||
/// with CLOCK_REALTIME then — the closest observable analogue of "reached glass" here (the
|
||||
/// composite follows within a refresh, so the meters' display stage reads slightly optimistic).
|
||||
private func renderPlanarToSurface(
|
||||
_ planes: WaveletPlanes, targetFromLayout: CGSize, csc: inout CscUniform,
|
||||
onPresented: ((Int64?) -> Void)?
|
||||
) -> Bool {
|
||||
let decodedSize = CGSize(width: planes.width, height: planes.height)
|
||||
let targetSize = (targetFromLayout.width > 0 && targetFromLayout.height > 0)
|
||||
? targetFromLayout : decodedSize
|
||||
ensureSurfacePool(size: targetSize)
|
||||
guard let slotIndex = takeSurfaceSlot(),
|
||||
let commandBuffer = queue.makeCommandBuffer()
|
||||
else { return false }
|
||||
let slot = surfacePool[slotIndex]
|
||||
|
||||
let pass = MTLRenderPassDescriptor()
|
||||
pass.colorAttachments[0].texture = slot.texture
|
||||
pass.colorAttachments[0].loadAction = .clear
|
||||
pass.colorAttachments[0].clearColor = MTLClearColor(red: 0, green: 0, blue: 0, alpha: 1)
|
||||
pass.colorAttachments[0].storeAction = .store
|
||||
guard let encoder = commandBuffer.makeRenderCommandEncoder(descriptor: pass) else {
|
||||
return false
|
||||
}
|
||||
encoder.setRenderPipelineState(planes.pq ? pipelinePlanarToneMap : pipelinePlanar)
|
||||
encoder.setFragmentTexture(planes.y, index: 0)
|
||||
encoder.setFragmentTexture(planes.cb, index: 1)
|
||||
encoder.setFragmentTexture(planes.cr, index: 2)
|
||||
encoder.setFragmentBytes(&csc, length: MemoryLayout<CscUniform>.stride, index: 0)
|
||||
encoder.drawPrimitives(type: .triangle, vertexStart: 0, vertexCount: 3)
|
||||
encoder.endEncoding()
|
||||
let surface = slot.surface
|
||||
let surfaceLayer = surfaceLayer // captured directly — the handler must not retain self
|
||||
let keepAlive: [Any] = [planes.y, planes.cb, planes.cr]
|
||||
commandBuffer.addCompletedHandler { _ in
|
||||
_ = keepAlive // ring textures pinned until the GPU finished sampling
|
||||
DispatchQueue.main.async {
|
||||
CATransaction.begin()
|
||||
CATransaction.setDisableActions(true)
|
||||
surfaceLayer.contents = surface
|
||||
CATransaction.commit()
|
||||
onPresented?(
|
||||
Stage2Pipeline.realtimeNs(forDisplayLinkTimestamp: CACurrentMediaTime()))
|
||||
}
|
||||
}
|
||||
commandBuffer.commit()
|
||||
lastHandedOff = slotIndex
|
||||
return true
|
||||
}
|
||||
|
||||
/// (Re)build the pool at `size` — 4 BGRA8 IOSurface render targets (one on glass, one queued
|
||||
/// in CA, one rendering, one spare). RENDER THREAD. A failed allocation leaves the pool empty;
|
||||
/// the caller returns false and the ring's putBack + display-link retry take over.
|
||||
private func ensureSurfacePool(size: CGSize) {
|
||||
guard size != surfacePoolSize else { return }
|
||||
surfacePool.removeAll()
|
||||
surfacePoolSize = size
|
||||
lastHandedOff = nil
|
||||
let w = Int(size.width)
|
||||
let h = Int(size.height)
|
||||
guard w > 0, h > 0 else { return }
|
||||
// 256-byte row alignment satisfies both IOSurface and Metal linear-texture rules.
|
||||
let bytesPerRow = ((w * 4) + 255) & ~255
|
||||
let props: [String: Any] = [
|
||||
kIOSurfaceWidth as String: w,
|
||||
kIOSurfaceHeight as String: h,
|
||||
kIOSurfaceBytesPerElement as String: 4,
|
||||
kIOSurfaceBytesPerRow as String: bytesPerRow,
|
||||
kIOSurfacePixelFormat as String: kCVPixelFormatType_32BGRA,
|
||||
]
|
||||
let desc = MTLTextureDescriptor.texture2DDescriptor(
|
||||
pixelFormat: .bgra8Unorm, width: w, height: h, mipmapped: false)
|
||||
desc.usage = [.renderTarget]
|
||||
desc.storageMode = .shared
|
||||
for _ in 0..<4 {
|
||||
guard let surface = IOSurfaceCreate(props as CFDictionary),
|
||||
let texture = device.makeTexture(descriptor: desc, iosurface: surface, plane: 0)
|
||||
else {
|
||||
surfacePool.removeAll()
|
||||
return
|
||||
}
|
||||
surfacePool.append(SurfaceSlot(surface: surface, texture: texture))
|
||||
}
|
||||
}
|
||||
|
||||
/// Pick the slot to render into: never the one just handed to the layer (the compositor may
|
||||
/// still scan it), prefer surfaces the window server isn't holding (`IOSurfaceIsInUse`), and
|
||||
/// among those the least recently rendered. Falls back to the LRU busy slot rather than
|
||||
/// stalling — a visible glitch at worst, never a queue-up. RENDER THREAD.
|
||||
private func takeSurfaceSlot() -> Int? {
|
||||
guard !surfacePool.isEmpty else { return nil }
|
||||
var free: Int?
|
||||
var busy: Int?
|
||||
for i in surfacePool.indices where i != lastHandedOff {
|
||||
if !IOSurfaceIsInUse(surfacePool[i].surface) {
|
||||
if free == nil || surfacePool[i].seq < surfacePool[free!].seq { free = i }
|
||||
} else {
|
||||
if busy == nil || surfacePool[i].seq < surfacePool[busy!].seq { busy = i }
|
||||
}
|
||||
}
|
||||
guard let pick = free ?? busy else { return nil }
|
||||
surfaceSeq += 1
|
||||
surfacePool[pick].seq = surfaceSeq
|
||||
return pick
|
||||
}
|
||||
#endif
|
||||
|
||||
/// The shared present tail of `render`/`renderPlanar`: size the drawable, encode one
|
||||
/// fullscreen triangle with `pipeline` (`bind` supplies the fragment resources), schedule
|
||||
/// the present and the on-glass callback.
|
||||
@@ -936,6 +895,36 @@ public final class MetalVideoPresenter {
|
||||
#if DEBUG
|
||||
logSizeIfChanged(decoded: decodedSize, drawable: targetSize)
|
||||
#endif
|
||||
#if os(macOS)
|
||||
// Windowed (composited) → the DCP swapID-panic mitigation mechanism (see
|
||||
// `WindowedPresentMode`). Toggle the layer property BEFORE vending a drawable so the
|
||||
// vend matches how it will be presented; drained here on the render thread, flipped
|
||||
// exactly once per mode change.
|
||||
stagingLock.lock()
|
||||
let windowedMode = windowedPresentStaged
|
||||
stagingLock.unlock()
|
||||
if windowedMode != windowedPresentActive {
|
||||
windowedPresentActive = windowedMode
|
||||
layer.presentsWithTransaction = windowedMode == .transaction
|
||||
if windowedMode != .surface, !surfacePool.isEmpty {
|
||||
// Leaving surface mode (fullscreen entry / mechanism A/B): drop the pool — at 5K
|
||||
// it holds >100 MB, and re-entering rebuilds it in one frame. SessionPresenter
|
||||
// clears the surface layer's contents on main.
|
||||
surfacePool.removeAll()
|
||||
surfacePoolSize = .zero
|
||||
lastHandedOff = nil
|
||||
}
|
||||
presenterLog.info(
|
||||
"stage2: windowed present mode \(windowedMode.rawValue, privacy: .public) (DCP swapID-panic mitigation)")
|
||||
}
|
||||
if windowedMode == .surface {
|
||||
// No image queue at all: render into a pooled IOSurface and swap it into the
|
||||
// sibling layer's contents. The drawable/queue tail below never runs.
|
||||
return encodeToSurface(
|
||||
targetSize: targetSize, pipeline: pipeline, onPresented: onPresented,
|
||||
keepAlive: keepAlive, bind: bind)
|
||||
}
|
||||
#endif
|
||||
if let providedDrawable,
|
||||
providedDrawable.texture.pixelFormat != layer.pixelFormat {
|
||||
return false // config outran the vend (HDR flip) — next vend has the new format
|
||||
@@ -974,6 +963,61 @@ public final class MetalVideoPresenter {
|
||||
}
|
||||
#endif
|
||||
}
|
||||
// Keep the bound sources alive until the GPU finishes sampling (see the callers).
|
||||
commandBuffer.addCompletedHandler { _ in _ = keepAlive }
|
||||
#if os(macOS)
|
||||
if windowedPresentActive == .transaction {
|
||||
// Windowed DCP mitigation: present the drawable THROUGH a Core Animation transaction
|
||||
// (`presentsWithTransaction`, set above) instead of the async image queue, so the swap
|
||||
// commits with the layer tree and stays in lockstep with the compositor (no out-of-band
|
||||
// flip to race WindowServer's swaps). Wait until the GPU work is scheduled (contents
|
||||
// will be ready — p50 ~0.1 ms), then present inside an EXPLICIT CATransaction ON THIS
|
||||
// RENDER THREAD and `flush()`. `presentAtMediaTime` does not apply — the transaction
|
||||
// paces.
|
||||
//
|
||||
// Threading history, because BOTH failure modes shipped or nearly shipped:
|
||||
// • A bare `present()` from this thread (no transaction) never flushes — nothing
|
||||
// commits a runloop-less thread's implicit transaction, so drawables are never
|
||||
// released; after maximumDrawableCount vends `nextDrawable()` blocks forever and
|
||||
// the stream FREEZES (the fullscreen→windowed switch did exactly this).
|
||||
// • The explicit begin/commit alone is NOT enough either: this thread has an ACTIVE
|
||||
// implicit transaction (the layer mutations above — drawableSize/colour — created
|
||||
// it), so the explicit transaction NESTS inside it and its commit defers to the
|
||||
// implicit one that never comes. The harness reproduced the exact freeze: every
|
||||
// present reported presentedTime=0, nothing reached glass. `CATransaction.flush()`
|
||||
// pushes the implicit transaction (present included) to the render server NOW.
|
||||
// • The original fix hopped to MAIN and presented there — correct, but slow in the
|
||||
// field (presents=55 @ fps=240, display_p50 18.6 ms on the 240 Hz Studio): each
|
||||
// present lands a runloop turn late, and main's own implicit transaction batches
|
||||
// enrolled presents at runloop-iteration rate. Kept as PUNKTFUNK_TXN_PRESENT=main.
|
||||
// The off-main commit measured immune to main-thread churn in the harness
|
||||
// (2026-07-21: glass p50 ~10 ms at 240 Hz full-size, cadence a clean 4.17 ms).
|
||||
commandBuffer.commit()
|
||||
let schedStart = CACurrentMediaTime()
|
||||
commandBuffer.waitUntilScheduled()
|
||||
let schedMs = (CACurrentMediaTime() - schedStart) * 1000
|
||||
let commitStart = CACurrentMediaTime()
|
||||
if txnPresentOnMain {
|
||||
let presentedDrawable = drawable
|
||||
DispatchQueue.main.async {
|
||||
CATransaction.begin()
|
||||
CATransaction.setDisableActions(true)
|
||||
presentedDrawable.present()
|
||||
CATransaction.commit()
|
||||
}
|
||||
} else {
|
||||
CATransaction.begin()
|
||||
CATransaction.setDisableActions(true)
|
||||
drawable.present()
|
||||
CATransaction.commit()
|
||||
CATransaction.flush()
|
||||
}
|
||||
windowedDiag.record(
|
||||
schedMs: schedMs, commitMs: (CACurrentMediaTime() - commitStart) * 1000,
|
||||
mode: .transaction)
|
||||
return true
|
||||
}
|
||||
#endif
|
||||
// Scheduled on the vsync when the pipeline gave us the link's target (see the doc comment);
|
||||
// immediate otherwise. A target already in the past presents immediately — same thing.
|
||||
if let presentAtMediaTime {
|
||||
@@ -981,12 +1025,146 @@ public final class MetalVideoPresenter {
|
||||
} else {
|
||||
commandBuffer.present(drawable)
|
||||
}
|
||||
// Keep the bound sources alive until the GPU finishes sampling (see the callers).
|
||||
commandBuffer.addCompletedHandler { _ in _ = keepAlive }
|
||||
commandBuffer.commit()
|
||||
return true
|
||||
}
|
||||
|
||||
#if os(macOS)
|
||||
/// The WINDOWED `surface` present tail (see `WindowedPresentMode.surface`): render with the
|
||||
/// same per-frame pipeline into a pooled IOSurface and hand it to `surfaceLayer.contents`
|
||||
/// from the command buffer's COMPLETION handler, inside an explicit CATransaction + flush
|
||||
/// (the same off-main commit discipline as the transactional present — an ordinary
|
||||
/// damaged-layer update on WindowServer's own composite cadence, no image queue anywhere).
|
||||
/// RENDER THREAD. `onPresented` is stamped right after the contents swap commits — the
|
||||
/// closest observable analogue of "reached glass" here (the composite follows within a
|
||||
/// refresh, so the display-stage meters read slightly OPTIMISTIC in this mode).
|
||||
///
|
||||
/// The pool tracks `hdrActive`: bgra8 for SDR, rgba16Float tagged BT.2100 PQ for HDR —
|
||||
/// `configure` already ran, so the caller's `pipeline` attachment format always matches.
|
||||
/// HDR OPEN RISK (why this whole mode is a prototype): whether the compositor honors the
|
||||
/// PQ tag + EDR for plain-CALayer IOSurface contents needs an on-glass eyeball; the metal
|
||||
/// layer underneath keeps `wantsExtendedDynamicRangeContent` as the EDR anchor (the harness
|
||||
/// measured the display's EDR headroom engaging with this arrangement).
|
||||
private func encodeToSurface(
|
||||
targetSize: CGSize, pipeline: MTLRenderPipelineState,
|
||||
onPresented: ((Int64?) -> Void)?,
|
||||
keepAlive: [Any], bind: (MTLRenderCommandEncoder) -> Void
|
||||
) -> Bool {
|
||||
ensureSurfacePool(size: targetSize, hdr: hdrActive)
|
||||
guard let slotIndex = takeSurfaceSlot(),
|
||||
let commandBuffer = queue.makeCommandBuffer()
|
||||
else { return false }
|
||||
let slot = surfacePool[slotIndex]
|
||||
|
||||
let pass = MTLRenderPassDescriptor()
|
||||
pass.colorAttachments[0].texture = slot.texture
|
||||
pass.colorAttachments[0].loadAction = .clear
|
||||
pass.colorAttachments[0].clearColor = MTLClearColor(red: 0, green: 0, blue: 0, alpha: 1)
|
||||
pass.colorAttachments[0].storeAction = .store
|
||||
guard let encoder = commandBuffer.makeRenderCommandEncoder(descriptor: pass) else {
|
||||
return false
|
||||
}
|
||||
encoder.setRenderPipelineState(pipeline)
|
||||
bind(encoder)
|
||||
encoder.drawPrimitives(type: .triangle, vertexStart: 0, vertexCount: 3)
|
||||
encoder.endEncoding()
|
||||
let surface = slot.surface
|
||||
let surfaceLayer = surfaceLayer // captured directly — the handler must not retain self
|
||||
let diag = windowedDiag
|
||||
let commitStamp = CACurrentMediaTime()
|
||||
commandBuffer.addCompletedHandler { _ in
|
||||
_ = keepAlive // sources pinned until the GPU finished sampling
|
||||
let completedAt = CACurrentMediaTime()
|
||||
// Swap on THIS Metal completion thread: explicit transaction + flush, so the commit
|
||||
// reaches the render server now, independent of main (completion handlers for one
|
||||
// queue fire in execution order, so swaps can't reorder).
|
||||
CATransaction.begin()
|
||||
CATransaction.setDisableActions(true)
|
||||
surfaceLayer.contents = surface
|
||||
CATransaction.commit()
|
||||
CATransaction.flush()
|
||||
diag.record(
|
||||
schedMs: (completedAt - commitStamp) * 1000,
|
||||
commitMs: (CACurrentMediaTime() - completedAt) * 1000, mode: .surface)
|
||||
onPresented?(Stage2Pipeline.realtimeNs(forDisplayLinkTimestamp: CACurrentMediaTime()))
|
||||
}
|
||||
commandBuffer.commit()
|
||||
lastHandedOff = slotIndex
|
||||
return true
|
||||
}
|
||||
|
||||
/// (Re)build the pool at `size`/`hdr` — 4 IOSurface render targets (one on glass, one
|
||||
/// committed in CA, one rendering, one spare). RENDER THREAD. A failed allocation leaves the
|
||||
/// pool empty; the caller returns false and the ring's putBack + display-link retry take
|
||||
/// over.
|
||||
private func ensureSurfacePool(size: CGSize, hdr: Bool) {
|
||||
guard size != surfacePoolSize || hdr != surfacePoolHDR else { return }
|
||||
surfacePool.removeAll()
|
||||
surfacePoolSize = size
|
||||
surfacePoolHDR = hdr
|
||||
lastHandedOff = nil
|
||||
let w = Int(size.width)
|
||||
let h = Int(size.height)
|
||||
guard w > 0, h > 0 else { return }
|
||||
// rgba16Float (8 B/px) carries the PQ-encoded HDR samples; bgra8 the SDR ones. 256-byte
|
||||
// row alignment satisfies both IOSurface and Metal linear-texture rules.
|
||||
let bytesPerElement = hdr ? 8 : 4
|
||||
let bytesPerRow = ((w * bytesPerElement) + 255) & ~255
|
||||
let props: [String: Any] = [
|
||||
kIOSurfaceWidth as String: w,
|
||||
kIOSurfaceHeight as String: h,
|
||||
kIOSurfaceBytesPerElement as String: bytesPerElement,
|
||||
kIOSurfaceBytesPerRow as String: bytesPerRow,
|
||||
kIOSurfacePixelFormat as String: hdr
|
||||
? kCVPixelFormatType_64RGBAHalf : kCVPixelFormatType_32BGRA,
|
||||
]
|
||||
let desc = MTLTextureDescriptor.texture2DDescriptor(
|
||||
pixelFormat: hdr ? .rgba16Float : .bgra8Unorm, width: w, height: h, mipmapped: false)
|
||||
desc.usage = [.renderTarget]
|
||||
desc.storageMode = .shared
|
||||
for _ in 0..<4 {
|
||||
guard let surface = IOSurfaceCreate(props as CFDictionary),
|
||||
let texture = device.makeTexture(descriptor: desc, iosurface: surface, plane: 0)
|
||||
else {
|
||||
surfacePool.removeAll()
|
||||
return
|
||||
}
|
||||
if hdr, let name = CGColorSpace(name: CGColorSpace.itur_2100_PQ)?.name {
|
||||
// Tag the surface BT.2100 PQ so the compositor interprets the half-float
|
||||
// samples as PQ-encoded HDR (the CALayer-contents analogue of the metal
|
||||
// layer's colorspace).
|
||||
IOSurfaceSetValue(surface, "IOSurfaceColorSpace" as CFString, name)
|
||||
}
|
||||
surfacePool.append(SurfaceSlot(surface: surface, texture: texture))
|
||||
}
|
||||
// The EDR request rides the SURFACE layer too (its contents are what composite); the
|
||||
// metal layer underneath keeps its own from configureColor as the anchor. Layer flags
|
||||
// are committed by the next swap's transaction flush.
|
||||
surfaceLayer.wantsExtendedDynamicRangeContent = hdr
|
||||
}
|
||||
|
||||
/// Pick the slot to render into: never the one just handed to the layer (the compositor may
|
||||
/// still scan it), prefer surfaces the window server isn't holding (`IOSurfaceIsInUse`), and
|
||||
/// among those the least recently rendered. Falls back to the LRU busy slot rather than
|
||||
/// stalling — a visible glitch at worst, never a queue-up. RENDER THREAD.
|
||||
private func takeSurfaceSlot() -> Int? {
|
||||
guard !surfacePool.isEmpty else { return nil }
|
||||
var free: Int?
|
||||
var busy: Int?
|
||||
for i in surfacePool.indices where i != lastHandedOff {
|
||||
if !IOSurfaceIsInUse(surfacePool[i].surface) {
|
||||
if free == nil || surfacePool[i].seq < surfacePool[free!].seq { free = i }
|
||||
} else {
|
||||
if busy == nil || surfacePool[i].seq < surfacePool[busy!].seq { busy = i }
|
||||
}
|
||||
}
|
||||
guard let pick = free ?? busy else { return nil }
|
||||
surfaceSeq += 1
|
||||
surfacePool[pick].seq = surfaceSeq
|
||||
return pick
|
||||
}
|
||||
#endif
|
||||
|
||||
/// Returns the CVMetalTexture (not just its MTLTexture) so the caller can keep it alive past the
|
||||
/// draw — the MTLTexture is only valid while its CVMetalTexture is retained.
|
||||
private func makeTexture(
|
||||
|
||||
@@ -142,20 +142,17 @@ enum PresentPriority: Equatable {
|
||||
|
||||
final class SessionPresenter {
|
||||
/// Present pacing for this session. Stage-3 always means glass gating; under the stage-2
|
||||
/// default, macOS PyroWave sessions ALSO get glass gating — a kernel-panic mitigation, not a
|
||||
/// latency tweak. macOS's DCP panics ("mismatched swapID's" @UnifiedPipeline.cpp, the whole
|
||||
/// machine dies) when WindowServer's swap submissions race, and the reliable trigger is
|
||||
/// out-of-band CAMetalLayer presents (displaySyncEnabled=false — mandatory for us, see
|
||||
/// MetalVideoPresenter's init) arriving faster than the compositor latches them in a
|
||||
/// COMPOSITED (windowed) session. Arrival pacing does exactly that with PyroWave: the wavelet
|
||||
/// decode is near-instant Metal compute, so a network clump of frames presents within the
|
||||
/// same millisecond, and PyroWave is the codec that sustains stream rates above the panel's
|
||||
/// refresh. The glass gate admits one presented-but-undisplayed swap at a time (serialized on
|
||||
/// the on-glass callback, 100 ms stale backstop), which removes the racing pattern outright;
|
||||
/// frames the panel couldn't have shown anyway coalesce in the newest-wins ring. An explicit
|
||||
/// stage-2 pick (setting/env) still forces arrival pacing — that A/B lever must stay honest.
|
||||
/// VideoToolbox codecs keep arrival pacing: decode latency spaces their presents, and years
|
||||
/// of stage-2 defaults there predate any panic report.
|
||||
/// default, macOS PyroWave sessions ALSO get glass gating — for SMOOTHNESS, not as the panic
|
||||
/// fix (that is the windowed transactional present — see `setComposited`). PyroWave's wavelet
|
||||
/// decode is near-instant Metal compute, so a network clump presents within the same
|
||||
/// millisecond, and it is the codec that sustains stream rates above the panel's refresh; the
|
||||
/// glass gate admits one presented-but-undisplayed swap at a time (serialized on the on-glass
|
||||
/// callback, 100 ms stale backstop) so those bursts coalesce in the newest-wins ring instead
|
||||
/// of flooding the queue. (Glass pacing was ALSO the original DCP-panic mitigation attempt —
|
||||
/// disproven: a fully serialized stream still panicked, which is why the real fix moved to the
|
||||
/// present mechanism.) An explicit stage-2 pick (setting/env) still forces arrival pacing —
|
||||
/// that A/B lever must stay honest. VideoToolbox codecs keep arrival pacing: decode latency
|
||||
/// spaces their presents.
|
||||
static func pacing(
|
||||
for choice: PresenterChoice, explicit: PresenterChoice?, codec: VideoCodec
|
||||
) -> PresentPacing {
|
||||
@@ -178,10 +175,25 @@ final class SessionPresenter {
|
||||
/// that doesn't exist after the first Wi-Fi clump. Sub-refresh display latency needs pacing
|
||||
/// that can't queue at all — that's stage-4 (`PresentPacing.deadline`), not a deeper gate.
|
||||
///
|
||||
#if os(macOS)
|
||||
/// Resolve the windowed (composited) present MECHANISM for this session — the DCP
|
||||
/// swapID-panic mitigation picker (see `WindowedPresentMode`). The
|
||||
/// `PUNKTFUNK_WINDOWED_PRESENT=async|transaction|surface` env lever wins (dev A/B);
|
||||
/// otherwise the user's safe-present setting: ON/unset → `.transaction` (the validated
|
||||
/// mitigation), OFF → `.async` (the fast pre-mitigation path — the panic returns on
|
||||
/// affected high-refresh setups; the Settings caption says so). `.surface` is currently
|
||||
/// env-only (prototype — HDR-composite verification owed). Fullscreen always presents
|
||||
/// async regardless (`setComposited`). Internal (not private) for unit tests.
|
||||
static func windowedPresentMode(setting: Bool?, env: String?) -> WindowedPresentMode {
|
||||
if let env, let mode = WindowedPresentMode(rawValue: env) { return mode }
|
||||
return (setting ?? true) ? .transaction : .async
|
||||
}
|
||||
#endif
|
||||
|
||||
/// `PUNKTFUNK_GATE_DEPTH` (1…3) still overrides on iOS/tvOS so the standing-queue ladder
|
||||
/// stays reproducible on-device; macOS is pinned to 1, env ignored — glass pacing exists
|
||||
/// there as the DCP swapID kernel-panic mitigation (see `pacing`), and STRICT present
|
||||
/// serialization is its point. Internal (not private) for unit tests.
|
||||
/// stays reproducible on-device; macOS is pinned to 1, env ignored — a deeper gate only builds
|
||||
/// a standing queue (see above), and macOS glass pacing exists for PyroWave smoothness
|
||||
/// (see `pacing`), where depth 1 is the point. Internal (not private) for unit tests.
|
||||
static func gateDepth(env: String?) -> Int {
|
||||
#if os(macOS)
|
||||
return 1
|
||||
@@ -196,10 +208,16 @@ final class SessionPresenter {
|
||||
private var stage2Link: CADisplayLink?
|
||||
private var metalLayer: CAMetalLayer?
|
||||
#if os(macOS)
|
||||
/// The windowed-mode PyroWave present target (sibling above `metalLayer`) and the last
|
||||
/// routing pushed to the pipeline — see `setComposited`. Main-thread only, like all of this.
|
||||
/// The windowed present MECHANISM this session runs while composited (resolved once per
|
||||
/// session in `start` — the user's safe-present setting + the PUNKTFUNK_WINDOWED_PRESENT
|
||||
/// dev override) and the routing last pushed to the pipeline — see `setComposited` (the DCP
|
||||
/// swapID-panic mitigation). Main-thread only, like all of this.
|
||||
private var windowedMode: WindowedPresentMode = .transaction
|
||||
private var windowedPresentApplied: WindowedPresentMode = .async
|
||||
/// The windowed `surface` present target (sibling above `metalLayer`, transparent while
|
||||
/// unused) — installed whenever stage-2 runs so a mechanism flip never has to mutate the
|
||||
/// layer tree mid-session.
|
||||
private var surfaceLayer: CALayer?
|
||||
private var surfacePresentsActive = false
|
||||
#endif
|
||||
private var connection: PunktfunkConnection?
|
||||
/// The decoded frame's REAL pixel dimensions (ground truth, pushed by the view from the pump's
|
||||
@@ -283,11 +301,17 @@ final class SessionPresenter {
|
||||
baseLayer.addSublayer(metal)
|
||||
metalLayer = metal
|
||||
#if os(macOS)
|
||||
// The windowed-PyroWave present target sits ABOVE the metal layer: transparent (nil
|
||||
// contents) while the metal path presents, covering it while surface presents run.
|
||||
windowedPresentApplied = .async
|
||||
// Resolve THIS session's windowed mechanism once (setting + dev env lever) —
|
||||
// `setComposited` routes between it and fullscreen-async from every layout.
|
||||
windowedMode = Self.windowedPresentMode(
|
||||
setting: UserDefaults.standard.object(
|
||||
forKey: DefaultsKey.windowedSafePresent) as? Bool,
|
||||
env: ProcessInfo.processInfo.environment["PUNKTFUNK_WINDOWED_PRESENT"])
|
||||
// The surface present target sits ABOVE the metal layer: transparent (nil contents)
|
||||
// unless the surface mechanism actually presents, covering it while it does.
|
||||
baseLayer.addSublayer(pipeline.surfaceLayer)
|
||||
surfaceLayer = pipeline.surfaceLayer
|
||||
surfacePresentsActive = false
|
||||
#endif
|
||||
stage2 = pipeline
|
||||
// The link is the vsync CLOCK + putBack-retry nudge, not the presentation trigger
|
||||
@@ -432,19 +456,23 @@ final class SessionPresenter {
|
||||
|
||||
#if os(macOS)
|
||||
/// Route presents for the window's composited state (MAIN thread — the view pushes it on
|
||||
/// every layout, which fullscreen transitions always trigger). PyroWave sessions in a
|
||||
/// COMPOSITED (windowed) session present via `surfaceLayer` contents instead of the
|
||||
/// CAMetalLayer image queue — the DCP "mismatched swapID's" kernel-panic mitigation (see
|
||||
/// `MetalVideoPresenter.surfaceLayer`; the metal-swap race survives glass pacing, so pacing
|
||||
/// alone was not enough). VT codecs keep the metal path: no panic reports there, and their
|
||||
/// HDR/EDR presentation has no surface-contents equivalent wired.
|
||||
/// every layout, which fullscreen transitions always trigger). A COMPOSITED (windowed)
|
||||
/// session presents through this session's resolved mitigation mechanism (`windowedMode` —
|
||||
/// transactional by default, see `windowedPresentMode`) instead of the async image queue —
|
||||
/// the DCP "mismatched swapID's" kernel-panic mitigation (see `MetalVideoPresenter`; the
|
||||
/// async-swap race survives glass pacing, so pacing alone was not enough). ALL codecs:
|
||||
/// PyroWave hit it 2026-07-18 and windowed HEVC hit the same 240 Hz Mac Studio 2026-07-21 —
|
||||
/// it is the async image queue itself, not any codec or present rate. Fullscreen keeps the
|
||||
/// async path (direct scanout, lowest latency, no panic there). The full HDR/EDR render
|
||||
/// path is preserved in every mechanism.
|
||||
func setComposited(_ composited: Bool) {
|
||||
guard let stage2, let connection else { return }
|
||||
let wantsSurface = composited && connection.videoCodec == .pyrowave
|
||||
guard wantsSurface != surfacePresentsActive else { return }
|
||||
surfacePresentsActive = wantsSurface
|
||||
stage2.setSurfacePresents(wantsSurface)
|
||||
if !wantsSurface {
|
||||
guard let stage2 else { return }
|
||||
let mode: WindowedPresentMode = composited ? windowedMode : .async
|
||||
guard mode != windowedPresentApplied else { return }
|
||||
let wasSurface = windowedPresentApplied == .surface
|
||||
windowedPresentApplied = mode
|
||||
stage2.setWindowedPresent(mode)
|
||||
if wasSurface {
|
||||
// Uncover the metal layer NOW (its last drawable is still attached, so fullscreen
|
||||
// entry shows the previous frame until the next present — no black flash).
|
||||
CATransaction.begin()
|
||||
@@ -471,7 +499,7 @@ final class SessionPresenter {
|
||||
#if os(macOS)
|
||||
surfaceLayer?.removeFromSuperlayer()
|
||||
surfaceLayer = nil
|
||||
surfacePresentsActive = false
|
||||
windowedPresentApplied = .async
|
||||
#endif
|
||||
connection = nil
|
||||
}
|
||||
|
||||
@@ -1114,15 +1114,15 @@ public final class Stage2Pipeline {
|
||||
}
|
||||
|
||||
#if os(macOS)
|
||||
/// The windowed-mode PyroWave present target (see `MetalVideoPresenter.surfaceLayer` — the
|
||||
/// DCP swapID-panic mitigation). The hosting view installs it as a sibling above `layer`.
|
||||
public var surfaceLayer: CALayer { presenter.surfaceLayer }
|
||||
|
||||
/// Forward the windowed-vs-fullscreen present routing (MAIN thread — see
|
||||
/// `MetalVideoPresenter.setSurfacePresents`).
|
||||
public func setSurfacePresents(_ on: Bool) {
|
||||
presenter.setSurfacePresents(on)
|
||||
/// Forward the windowed present mechanism (MAIN thread — see
|
||||
/// `MetalVideoPresenter.setWindowedPresent`, the DCP swapID-panic mitigation).
|
||||
func setWindowedPresent(_ mode: WindowedPresentMode) {
|
||||
presenter.setWindowedPresent(mode)
|
||||
}
|
||||
|
||||
/// The windowed `surface` present target the hosting SessionPresenter installs as a sibling
|
||||
/// ABOVE `layer` (transparent while unused — see `MetalVideoPresenter.surfaceLayer`).
|
||||
var surfaceLayer: CALayer { presenter.surfaceLayer }
|
||||
#endif
|
||||
|
||||
/// Forward the display's current EDR headroom to the presenter (MAIN thread — a `UIScreen`
|
||||
@@ -1213,7 +1213,9 @@ public final class Stage2Pipeline {
|
||||
let chunkAligned =
|
||||
au.flags & PunktfunkConnection.userFlagChunkAligned != 0
|
||||
let ptsNs = au.ptsNs
|
||||
let receivedNs = au.receivedNs
|
||||
// Decode stage starts at the PULL (matching the VT path's FrameContext —
|
||||
// receipt→pull is the HUD's separate client-queue term, ABI v9 split).
|
||||
let receivedNs = au.pulledNs
|
||||
let flags = au.flags
|
||||
let submitted = decoder.decode(
|
||||
au: au.data, chunkAligned: chunkAligned, windowSize: windowSize
|
||||
|
||||
@@ -32,9 +32,12 @@ public enum ReadyImage: @unchecked Sendable {
|
||||
public struct ReadyFrame: @unchecked Sendable {
|
||||
/// Host capture clock (the AU's pts), in nanoseconds.
|
||||
public let ptsNs: UInt64
|
||||
/// Client `CLOCK_REALTIME` instant the AU was received (`AccessUnit.receivedNs`, threaded
|
||||
/// through the decode via the frame refcon), in nanoseconds. 0 when unknown (a caller that
|
||||
/// didn't stamp receipt) — the decode-stage meter then drops the sample via its sanity guard.
|
||||
/// Client `CLOCK_REALTIME` instant the AU left `nextAU` (`AccessUnit.pulledNs`, threaded
|
||||
/// through the decode via the frame refcon), in nanoseconds — the decode stage's start
|
||||
/// point. (Named for its historical role; since the ABI v9 receipt split the true
|
||||
/// reassembly receipt lives on `AccessUnit.receivedNs`, and receipt→pull is the HUD's own
|
||||
/// client-queue term.) 0 when unknown (a caller that didn't stamp) — the decode-stage meter
|
||||
/// then drops the sample via its sanity guard.
|
||||
public let receivedNs: Int64
|
||||
/// Client `CLOCK_REALTIME` instant decode completed, in nanoseconds.
|
||||
public let decodedNs: Int64
|
||||
@@ -167,7 +170,11 @@ public final class VideoDecoder: @unchecked Sendable {
|
||||
var infoOut = VTDecodeInfoFlags()
|
||||
// The AU's receipt instant + wire flags ride through as a retained context; the output
|
||||
// callback reclaims it. Retain immediately before submit so no early return can leak it.
|
||||
let ctx = FrameContext(receivedNs: au.receivedNs, flags: au.flags)
|
||||
// The decode stage starts at the PULL (the AU leaving nextAU), not the reassembly
|
||||
// receipt: both consumers — the decode-stage meter and the ABR decode signal — are
|
||||
// specified from the pull, and the receipt→pull wait is the HUD's separate client-queue
|
||||
// term (see AccessUnit.pulledNs).
|
||||
let ctx = FrameContext(receivedNs: au.pulledNs, flags: au.flags)
|
||||
let refcon = Unmanaged.passRetained(ctx).toOpaque()
|
||||
let status = VTDecompressionSessionDecodeFrame(
|
||||
session,
|
||||
|
||||
@@ -700,9 +700,9 @@ public final class StreamLayerView: NSView {
|
||||
private func layoutPresenter() {
|
||||
presenter.layout(in: bounds, contentsScale: window?.backingScaleFactor ?? 1)
|
||||
// Present routing tracks the window's composited state (fullscreen transitions always
|
||||
// re-layout, so this stays current): windowed PyroWave presents via surface contents —
|
||||
// the DCP swapID kernel-panic mitigation (see SessionPresenter.setComposited). A view
|
||||
// not yet in a window counts as composited (the safe default).
|
||||
// re-layout, so this stays current): a windowed session presents through a Core Animation
|
||||
// transaction — the DCP swapID kernel-panic mitigation (see SessionPresenter.setComposited).
|
||||
// A view not yet in a window counts as composited (the safe default).
|
||||
presenter.setComposited(!(window?.styleMask.contains(.fullScreen) ?? false))
|
||||
// Feed the follower only once in a window (backing scale is real then) and with real
|
||||
// bounds — a pre-window layout would report point-sized dimensions.
|
||||
|
||||
@@ -70,6 +70,16 @@ public enum DefaultsKey {
|
||||
/// (lowest latency — the default, OFF). Resolved once per session;
|
||||
/// PUNKTFUNK_PRESENT_MODE=immediate|vsync overrides it for A/B. See Stage2Pipeline's header.
|
||||
public static let vsync = "punktfunk.vsync"
|
||||
/// macOS: present WINDOWED sessions in lockstep with the system compositor (the DCP
|
||||
/// "mismatched swapID's" kernel-panic mitigation — see SessionPresenter.windowedPresentMode
|
||||
/// and the MetalVideoPresenter saga notes). ON/unset (the default): windowed presents ride
|
||||
/// a Core Animation transaction — validated panic-free on the 240 Hz repro machine, at a
|
||||
/// small display-latency cost vs the raw path. OFF: windowed sessions keep the fast async
|
||||
/// image queue — ON AFFECTED SETUPS (high-refresh displays) THAT PATH KERNEL-PANICS THE
|
||||
/// WHOLE MAC, which is why the default is ON. Fullscreen always presents async (fast path)
|
||||
/// regardless. Resolved once per session; PUNKTFUNK_WINDOWED_PRESENT=async|transaction|
|
||||
/// surface overrides it for dev A/B.
|
||||
public static let windowedSafePresent = "punktfunk.windowedSafePresent"
|
||||
/// Allow variable refresh rate: hand the display link a wide frame-rate RANGE (low floor,
|
||||
/// preferred = stream rate) so a ProMotion / adaptive-sync display can vary its physical
|
||||
/// refresh to match the stream. On by default; a no-op on fixed-refresh displays. When off,
|
||||
|
||||
@@ -316,6 +316,41 @@ final class PresentPacingTests: XCTestCase {
|
||||
SessionPresenter.pacing(for: .stage4, explicit: .stage4, codec: .pyrowave), .deadline)
|
||||
}
|
||||
|
||||
// MARK: - Windowed present mechanism (the macOS DCP swapID-panic mitigation picker)
|
||||
|
||||
#if os(macOS)
|
||||
/// The safe-present setting: ON/unset → the validated transactional mitigation; an explicit
|
||||
/// OFF → the fast async path (the user accepted the affected-setup panic risk). The
|
||||
/// PUNKTFUNK_WINDOWED_PRESENT env lever overrides both ways, `surface` is env-only (the
|
||||
/// prototype mechanism), and garbage/empty env values are "unset", not an override.
|
||||
func testWindowedPresentModeResolution() {
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: nil, env: nil), .transaction,
|
||||
"unset defaults to the panic mitigation")
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: true, env: nil), .transaction)
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: false, env: nil), .async,
|
||||
"an explicit opt-out gets the fast async path")
|
||||
// The dev env lever wins over the setting, both directions.
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: true, env: "async"), .async)
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: false, env: "transaction"),
|
||||
.transaction)
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: true, env: "surface"), .surface,
|
||||
"the surface prototype is reachable via env only")
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: false, env: "surface"), .surface)
|
||||
// Garbage/empty env = unset.
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: nil, env: "garbage"), .transaction)
|
||||
XCTAssertEqual(
|
||||
SessionPresenter.windowedPresentMode(setting: false, env: ""), .async)
|
||||
}
|
||||
#endif
|
||||
|
||||
// MARK: - Glass-gate depth
|
||||
|
||||
/// The in-flight present budget is 1 EVERYWHERE: any deeper gate keeps a standing queue —
|
||||
|
||||
@@ -16,6 +16,8 @@
|
||||
# PF_LAUNCH library id to launch on connect (optional, e.g. steam:570 — pinned games)
|
||||
# PF_BROWSE non-empty = open the gamepad library (optional; --browse instead of --connect)
|
||||
# PF_MGMT management-API port for --browse (optional; client defaults to 47990)
|
||||
# PF_CONNECT_TIMEOUT connect budget in seconds (optional; the plugin stretches it after
|
||||
# firing Wake-on-LAN so the connect survives the host's resume)
|
||||
# PF_APPID flatpak app id (default io.unom.Punktfunk)
|
||||
# PF_FLATPAK override the flatpak binary path (default: `flatpak` on PATH)
|
||||
#
|
||||
@@ -61,10 +63,17 @@ if [ -z "${PF_HOST:-}" ]; then
|
||||
echo "punktfunkrun: PF_HOST is not set (the plugin sets it as a launch option)" >&2
|
||||
exit 2
|
||||
fi
|
||||
# Trailing args shared by both streaming execs. A stretched connect budget rides along when the
|
||||
# plugin set one (it just fired Wake-on-LAN, so the host may still be resuming); an older flatpak
|
||||
# without --connect-timeout ignores the flag harmlessly (hand-scanned argv).
|
||||
set -- --fullscreen
|
||||
if [ -n "${PF_CONNECT_TIMEOUT:-}" ]; then
|
||||
set -- --connect-timeout "$PF_CONNECT_TIMEOUT" "$@"
|
||||
fi
|
||||
if [ -n "${PF_LAUNCH:-}" ]; then
|
||||
# A pinned game: the id rides the session Hello and the host launches that title.
|
||||
echo "punktfunkrun: streaming $APPID --connect $PF_HOST --launch $PF_LAUNCH" >&2
|
||||
exec "$FLATPAK" run --arch=x86_64 "$APPID" --connect "$PF_HOST" --launch "$PF_LAUNCH" --fullscreen
|
||||
exec "$FLATPAK" run --arch=x86_64 "$APPID" --connect "$PF_HOST" --launch "$PF_LAUNCH" "$@"
|
||||
fi
|
||||
echo "punktfunkrun: streaming $APPID --connect $PF_HOST" >&2
|
||||
exec "$FLATPAK" run --arch=x86_64 "$APPID" --connect "$PF_HOST" --fullscreen
|
||||
exec "$FLATPAK" run --arch=x86_64 "$APPID" --connect "$PF_HOST" "$@"
|
||||
|
||||
@@ -70,7 +70,9 @@ function setShortcutHidden(appId: number, hidden: boolean): void {
|
||||
};
|
||||
|
||||
// Bump when the shipped artwork changes so existing shortcuts re-apply it once (per appId).
|
||||
const ART_VERSION = 2;
|
||||
// v3: CI zips through 0.17.1 shipped no assets/ at all, yet v2 was still recorded as applied
|
||||
// on those installs — the bump makes them re-apply once on the first build that has the files.
|
||||
const ART_VERSION = 3;
|
||||
function artKey(appId: number): string {
|
||||
return `punktfunk:shortcutArt:${appId}`;
|
||||
}
|
||||
@@ -79,7 +81,7 @@ function artKey(appId: number): string {
|
||||
* Apply the plugin's grid/hero/logo/icon to a shortcut (idempotent, once per ART_VERSION per
|
||||
* appId). Cosmetic and fully best-effort: any failure is swallowed and retried on the next call.
|
||||
*/
|
||||
async function applyArtwork(appId: number): Promise<void> {
|
||||
async function applyArtwork(appId: number, isRetry = false): Promise<void> {
|
||||
try {
|
||||
if (localStorage.getItem(artKey(appId)) === `${ART_VERSION}`) {
|
||||
return;
|
||||
@@ -91,16 +93,29 @@ async function applyArtwork(appId: number): Promise<void> {
|
||||
[art.logo, 2],
|
||||
[art.gridwide, 3],
|
||||
];
|
||||
let applied = false;
|
||||
for (const [data, assetType] of assets) {
|
||||
if (data) {
|
||||
await SteamClient.Apps.SetCustomArtworkForApp(appId, data, "png", assetType);
|
||||
applied = true;
|
||||
}
|
||||
}
|
||||
if (art.icon_path) {
|
||||
SteamClient.Apps.SetShortcutIcon(appId, art.icon_path);
|
||||
applied = true;
|
||||
}
|
||||
// Only record "done" when something actually landed — a plugin build whose assets/ is
|
||||
// missing/empty must keep retrying on later mounts instead of poisoning the marker.
|
||||
if (applied) {
|
||||
localStorage.setItem(artKey(appId), `${ART_VERSION}`);
|
||||
}
|
||||
} catch (e) {
|
||||
// A shortcut fresh out of AddShortcut may not be registered yet (the same race
|
||||
// setShortcutHidden defers around) — one deferred second attempt, then leave it to
|
||||
// the next mount.
|
||||
if (!isRetry) {
|
||||
setTimeout(() => void applyArtwork(appId, true), 2500);
|
||||
}
|
||||
console.warn("punktfunk: shortcut artwork not applied", e);
|
||||
}
|
||||
}
|
||||
@@ -157,7 +172,9 @@ async function ensureControllerConfig(): Promise<void> {
|
||||
return;
|
||||
}
|
||||
const r = await applyControllerConfig(SHORTCUT_NAME);
|
||||
if (r?.ok) {
|
||||
// `ok` alone isn't done: with zero account configset dirs (fresh Steam) the backend
|
||||
// succeeds without pointing any account at the template — keep retrying until one lands.
|
||||
if (r?.ok && (r.applied ?? []).some((a) => a.startsWith("configset:"))) {
|
||||
localStorage.setItem(CONFIG_KEY, `${CONFIG_VERSION}`);
|
||||
} else {
|
||||
console.warn("punktfunk: controller config not fully applied", r);
|
||||
@@ -283,13 +300,21 @@ export async function launchStream(
|
||||
opts: LaunchOpts = {},
|
||||
): Promise<void> {
|
||||
// Wake-on-LAN: if this host is asleep, nudge it awake before the stream connects. Kicked off now
|
||||
// so it races with the shortcut setup (near-zero added latency), and awaited just before RunGame.
|
||||
// so it races with the shortcut setup (near-zero added latency); its outcome is needed below
|
||||
// (the connect budget), and RunGame follows the await either way, so nothing is slower for it.
|
||||
// Best-effort — the flatpak client's --wake looks up the host's learned MAC (a no-op if none is
|
||||
// known), and the connect that follows has its own retry window, so a failure never blocks launch.
|
||||
const waking = wake(host, port).catch(() => ({ ok: false }));
|
||||
const { appId, runner } = await ensureStreamShortcut();
|
||||
const [{ appId, runner }, woke] = await Promise.all([ensureStreamShortcut(), waking]);
|
||||
const target = port && port !== 9777 ? `${host}:${port}` : host;
|
||||
const env = [`PF_HOST=${target}`];
|
||||
// A magic packet actually went out (a MAC was known), so the host may be mid-resume from
|
||||
// suspend — that takes far longer than the client's default 15 s connect budget. Stretch the
|
||||
// budget so the client's wake-tolerant dial keeps retrying across the resume; against an
|
||||
// already-awake host the connect still lands in under a second, so this costs nothing.
|
||||
if (woke.ok) {
|
||||
env.push("PF_CONNECT_TIMEOUT=75");
|
||||
}
|
||||
if (opts.browse) {
|
||||
env.push("PF_BROWSE=1");
|
||||
if (opts.mgmt) {
|
||||
@@ -303,9 +328,9 @@ export async function launchStream(
|
||||
env.push(`PF_LAUNCH=${opts.launchId}`);
|
||||
}
|
||||
// KEY=value ... %command% args — %command% expands to the shortcut exe (/bin/sh); the wrapper
|
||||
// script rides behind it as an argument and reads PF_* from the environment.
|
||||
// script rides behind it as an argument and reads PF_* from the environment. The wake was
|
||||
// awaited above, so the magic packet is out before the connect attempt.
|
||||
SteamClient.Apps.SetAppLaunchOptions(appId, `${env.join(" ")} %command% "${runner}"`);
|
||||
await waking; // ensure the magic packet is out before the connect attempt
|
||||
SteamClient.Apps.RunGame(gameIdFromAppId(appId), "", -1, 100);
|
||||
}
|
||||
|
||||
|
||||
@@ -856,7 +856,15 @@ pub fn show(
|
||||
s.render_scale =
|
||||
RENDER_SCALES[(scale_row.selected() as usize).min(RENDER_SCALES.len() - 1)];
|
||||
s.bitrate_kbps = (bitrate_row.value() * 1000.0) as u32;
|
||||
s.gamepad = GAMEPADS[(pad_row.selected() as usize).min(GAMEPADS.len() - 1)].to_string();
|
||||
// Keep a stored preference this table doesn't list (e.g. "switchpro" — valid to the
|
||||
// session, hand-edited or written by another client): it displays as "Automatic", and
|
||||
// writing that back would silently erase it just by opening + closing the dialog.
|
||||
// Persist the row only when the user picked a non-Auto entry or the stored value was
|
||||
// a listed one to begin with.
|
||||
let pad_sel = (pad_row.selected() as usize).min(GAMEPADS.len() - 1);
|
||||
if pad_sel != 0 || GAMEPADS.contains(&s.gamepad.as_str()) {
|
||||
s.gamepad = GAMEPADS[pad_sel].to_string();
|
||||
}
|
||||
s.touch_mode =
|
||||
TOUCH_MODES[(touch_row.selected() as usize).min(TOUCH_MODES.len() - 1)].to_string();
|
||||
s.forward_pad = chosen_pin.borrow().clone();
|
||||
|
||||
+26
-13
@@ -458,7 +458,11 @@ async fn session(args: Args) -> Result<()> {
|
||||
),
|
||||
(None, None) => tracing::info!(%remote, "punktfunk/1 connected"),
|
||||
}
|
||||
let (mut send, mut recv) = conn.open_bi().await.context("open control stream")?;
|
||||
let (mut send, recv) = conn.open_bi().await.context("open control stream")?;
|
||||
// Frame every read on the control stream through the resumable reader, exactly as the client
|
||||
// pump does: `clock_sync` bounds each read with a timeout, and a frame straddling two wakeups
|
||||
// would otherwise leave the stream permanently misaligned for the rest of the run.
|
||||
let mut recv = io::MsgReader::new(recv);
|
||||
|
||||
io::write_msg(
|
||||
&mut send,
|
||||
@@ -483,14 +487,24 @@ async fn session(args: Args) -> Result<()> {
|
||||
// host/network split is exactly what it exists to report. Old hosts ignore the bit.
|
||||
// PROBE_SEQ: the shared-core reassembler windows probe-space frames, so the probe
|
||||
// qualifies for `--speed-test` bursts; without the bit the host declines them.
|
||||
// STREAMED_AU: the same shared reassembler accepts sentinel-headed streamed
|
||||
// blocks, and the probe is exactly the tool that measures the overlap win.
|
||||
let mut caps = punktfunk_core::quic::VIDEO_CAP_HOST_TIMING
|
||||
| punktfunk_core::quic::VIDEO_CAP_PROBE_SEQ;
|
||||
| punktfunk_core::quic::VIDEO_CAP_PROBE_SEQ
|
||||
| punktfunk_core::quic::VIDEO_CAP_STREAMED_AU;
|
||||
if std::env::var_os("PUNKTFUNK_CLIENT_10BIT").is_some() {
|
||||
caps |= punktfunk_core::quic::VIDEO_CAP_10BIT;
|
||||
}
|
||||
if std::env::var_os("PUNKTFUNK_CLIENT_444").is_some() {
|
||||
caps |= punktfunk_core::quic::VIDEO_CAP_444;
|
||||
}
|
||||
// PUNKTFUNK_CLIENT_CHACHA20=1 advertises VIDEO_CAP_CHACHA20 — drives the
|
||||
// host's ChaCha20-Poly1305 session-cipher resolution (the soft-AES armv7
|
||||
// negotiation, design/chacha20-session-cipher.md §7) without a webOS build;
|
||||
// the negotiated cipher is reported in the welcome log line below.
|
||||
if std::env::var_os("PUNKTFUNK_CLIENT_CHACHA20").is_some() {
|
||||
caps |= punktfunk_core::quic::VIDEO_CAP_CHACHA20;
|
||||
}
|
||||
caps
|
||||
},
|
||||
// `--audio-channels` (default stereo); the probe multistream-decodes + validates the
|
||||
@@ -513,8 +527,8 @@ async fn session(args: Args) -> Result<()> {
|
||||
.encode(),
|
||||
)
|
||||
.await?;
|
||||
let welcome = Welcome::decode(&io::read_msg(&mut recv).await?)
|
||||
.map_err(|e| anyhow!("Welcome decode: {e:?}"))?;
|
||||
let welcome =
|
||||
Welcome::decode(&recv.read_msg().await?).map_err(|e| anyhow!("Welcome decode: {e:?}"))?;
|
||||
tracing::info!(
|
||||
mode = ?welcome.mode,
|
||||
fec = ?welcome.fec,
|
||||
@@ -528,6 +542,11 @@ async fn session(args: Args) -> Result<()> {
|
||||
chroma_444 = welcome.chroma_format == punktfunk_core::quic::CHROMA_IDC_444,
|
||||
chroma_format_idc = welcome.chroma_format,
|
||||
codec = codec_ext(welcome.codec),
|
||||
cipher = if welcome.cipher == punktfunk_core::quic::CIPHER_CHACHA20_POLY1305 {
|
||||
"chacha20-poly1305"
|
||||
} else {
|
||||
"aes-128-gcm"
|
||||
},
|
||||
"session offer"
|
||||
);
|
||||
|
||||
@@ -629,10 +648,7 @@ async fn session(args: Args) -> Result<()> {
|
||||
tracing::error!("Reconfigure write failed");
|
||||
return;
|
||||
}
|
||||
match io::read_msg(&mut rr)
|
||||
.await
|
||||
.map(|b| Reconfigured::decode(&b))
|
||||
{
|
||||
match rr.read_msg().await.map(|b| Reconfigured::decode(&b)) {
|
||||
Ok(Ok(ack)) if ack.accepted => {
|
||||
tracing::info!(mode = ?ack.mode, "mode switch ACCEPTED")
|
||||
}
|
||||
@@ -685,10 +701,7 @@ async fn session(args: Args) -> Result<()> {
|
||||
tracing::error!("SetBitrate write failed");
|
||||
return;
|
||||
}
|
||||
match io::read_msg(&mut rr)
|
||||
.await
|
||||
.map(|b| BitrateChanged::decode(&b))
|
||||
{
|
||||
match rr.read_msg().await.map(|b| BitrateChanged::decode(&b)) {
|
||||
Ok(Ok(ack)) => tracing::info!(
|
||||
applied_kbps = ack.bitrate_kbps,
|
||||
"BITRATE CHANGE acked by host"
|
||||
@@ -750,7 +763,7 @@ async fn session(args: Args) -> Result<()> {
|
||||
tracing::error!("ProbeRequest write failed");
|
||||
return;
|
||||
}
|
||||
let res = match io::read_msg(&mut sr).await.map(|b| ProbeResult::decode(&b)) {
|
||||
let res = match sr.read_msg().await.map(|b| ProbeResult::decode(&b)) {
|
||||
Ok(Ok(r)) => r,
|
||||
other => {
|
||||
tracing::error!(?other, "bad ProbeResult");
|
||||
|
||||
@@ -27,9 +27,13 @@ ui = ["dep:pf-console-ui", "dep:serde_json"]
|
||||
# Same Linux+Windows gating as the rest of the client stack; elsewhere this is a stub
|
||||
# binary.
|
||||
[target.'cfg(any(target_os = "linux", windows))'.dependencies]
|
||||
pf-presenter = { path = "../../crates/pf-presenter" }
|
||||
# `default-features = false` on both: THIS crate's `pyrowave` feature (above) is the single
|
||||
# switch that turns the wavelet codec on, and it enables it explicitly on each. Inheriting their
|
||||
# defaults instead would make `--no-default-features` a lie — the Windows ARM64 leg builds that
|
||||
# way precisely to skip the vendored PyroWave C++, which has no ARM64 SIMD path.
|
||||
pf-presenter = { path = "../../crates/pf-presenter", default-features = false }
|
||||
pf-console-ui = { path = "../../crates/pf-console-ui", optional = true }
|
||||
pf-client-core = { path = "../../crates/pf-client-core" }
|
||||
pf-client-core = { path = "../../crates/pf-client-core", default-features = false }
|
||||
punktfunk-core = { path = "../../crates/punktfunk-core", features = ["quic"] }
|
||||
# The fake-library dev hook (`PUNKTFUNK_FAKE_LIBRARY`, browse mode) parses GameEntry JSON.
|
||||
serde_json = { version = "1", optional = true }
|
||||
|
||||
@@ -29,7 +29,17 @@ punktfunk-core = { path = "../../crates/punktfunk-core", features = ["quic"] }
|
||||
# The shared client service layer: the trust/settings stores (ONE `Settings` struct for the
|
||||
# shell and the spawned session binary — src/trust.rs re-exports it) and the game-library
|
||||
# data model (fetch + art pipeline) behind the library page.
|
||||
pf-client-core = { path = "../../crates/pf-client-core" }
|
||||
#
|
||||
# `default-features = false` drops pf-client-core's default `pyrowave`, which would otherwise
|
||||
# build the vendored PyroWave C++ INTO THE SHELL — dead weight here (the shell never decodes;
|
||||
# it only offers "pyrowave" as a codec preference string the session binary acts on) and fatal
|
||||
# on ARM64, where Granite's math falls back to x86 SSE intrinsics and stops at
|
||||
# `simd.hpp: #error "Implement me."`. This does NOT drop PyroWave from the Windows client:
|
||||
# decode lives in the spawned punktfunk-session binary, whose own default enables the feature,
|
||||
# and cargo's feature unification turns it back on for the shared pf-client-core whenever that
|
||||
# binary is in the same build (x64). On the ARM64 leg both are built --no-default-features, so
|
||||
# nothing enables it and the C++ is never compiled.
|
||||
pf-client-core = { path = "../../crates/pf-client-core", default-features = false }
|
||||
|
||||
# WinUI 3 UI via windows-reactor (a declarative React-like framework backed by WinUI). Its
|
||||
# `build.rs` downloads the Windows App SDK NuGets and stages the bootstrap DLL + resources.pri
|
||||
|
||||
@@ -59,9 +59,14 @@
|
||||
</Application>
|
||||
<!--
|
||||
Second entry point: the couch/console UI, for an HTPC or a TV-attached box where the
|
||||
desktop shell is the wrong first screen. Same full-trust executable, launched with
|
||||
`--console`, which hands straight off to the session binary's controller-driven
|
||||
browse mode (host list, pairing, settings, library) fullscreen.
|
||||
desktop shell is the wrong first screen. Its own executable (punktfunk-console.exe)
|
||||
because an MSIX Application entry cannot pass arguments to a full-trust exe; it hands
|
||||
straight off to the session binary's controller-driven browse mode (host list,
|
||||
pairing, settings, library) fullscreen.
|
||||
|
||||
NOTE: never write a double hyphen in this file. XML forbids it inside a comment, and
|
||||
makepri rejects the whole manifest ("Appx manifest not found or is invalid") — which
|
||||
is exactly how the console flag spelled out here broke the v0.15.0 MSIX build.
|
||||
-->
|
||||
<Application Id="PunktfunkConsole" Executable="punktfunk-console.exe"
|
||||
EntryPoint="Windows.FullTrustApplication">
|
||||
|
||||
@@ -24,6 +24,16 @@ use pf_frame::DmabufFrame;
|
||||
pub trait Capturer: Send {
|
||||
fn next_frame(&mut self) -> Result<CapturedFrame>;
|
||||
|
||||
/// [`next_frame`](Self::next_frame) with a caller-chosen first-frame budget instead of the
|
||||
/// backend's default. The pipeline retry loop shortens its FIRST attempt's wait: a PipeWire
|
||||
/// stream connected while gamescope re-inits its headless takeover can negotiate a format,
|
||||
/// reach `Streaming`, and still never receive a buffer — a fresh connect then delivers within
|
||||
/// ~0.5 s, so waiting out the full default budget on a doomed stream just delays the retry
|
||||
/// that fixes it. Backends without an internal wait budget ignore it (the default delegates).
|
||||
fn next_frame_within(&mut self, _budget: std::time::Duration) -> Result<CapturedFrame> {
|
||||
self.next_frame()
|
||||
}
|
||||
|
||||
/// Non-blocking: the freshest frame available since the last call, or `None` if none has
|
||||
/// arrived (the caller reuses its last frame to hold a steady output rate). The default
|
||||
/// just produces a frame each call — fine for instant synthetic sources; the portal
|
||||
@@ -249,6 +259,12 @@ pub struct ZeroCopyPolicy {
|
||||
/// passthrough (like the VAAPI backend) instead of the EGL→CUDA import whose payloads only
|
||||
/// NVENC can consume. Per-session (the codec is negotiated), unlike `backend_is_vaapi`.
|
||||
pub pyrowave_session: bool,
|
||||
/// THIS session's encoder can ingest a producer-native NV12 capture (the Linux raw Vulkan
|
||||
/// Video backend on an H265/AV1 session — resolved by the host facade via
|
||||
/// `pf_encode::linux_native_nv12_ok`). Gates whether the negotiation PREFERS gamescope's
|
||||
/// producer-side NV12 pod: libav VAAPI (H264's backend) would misread the two-plane buffer,
|
||||
/// so H264/GameStream/PyroWave sessions must never see NV12 frames.
|
||||
pub native_nv12_session: bool,
|
||||
/// The PyroWave encoder's Vulkan-importable dmabuf modifiers for the capture's packed-RGB fourcc,
|
||||
/// resolved when the session encodes PyroWave (the passthrough advertises them so Mutter+NVIDIA,
|
||||
/// which allocates tiled-only, still negotiates zero-copy). Empty otherwise.
|
||||
|
||||
@@ -299,29 +299,11 @@ fn spawn_pipewire(
|
||||
|
||||
impl Capturer for PortalCapturer {
|
||||
fn next_frame(&mut self) -> Result<CapturedFrame> {
|
||||
// First frame can lag behind format negotiation; later frames arrive at ~fps. Wait in
|
||||
// short slices so a GPU-import poison (worker death) fails the capture within ~0.5 s
|
||||
// instead of sitting out the full first-frame budget.
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(10);
|
||||
loop {
|
||||
if self.broken.load(Ordering::Relaxed) {
|
||||
return Err(anyhow!(
|
||||
"zero-copy GPU import lost (node {}): the import worker died or tiled imports \
|
||||
failed repeatedly — rebuilding capture",
|
||||
self.node_id
|
||||
));
|
||||
}
|
||||
if let Some(f) = self.pending.take() {
|
||||
return Ok(f); // a wait_arrival stash outranks the channel (it's older)
|
||||
}
|
||||
let slice = Duration::from_millis(500)
|
||||
.min(deadline.saturating_duration_since(std::time::Instant::now()));
|
||||
match self.frames.recv_timeout(slice) {
|
||||
Ok(frame) => return Ok(frame),
|
||||
Err(RecvTimeoutError::Timeout) if std::time::Instant::now() < deadline => continue,
|
||||
Err(e) => return self.next_frame_timed_out(e),
|
||||
}
|
||||
self.frame_within(Duration::from_secs(10))
|
||||
}
|
||||
|
||||
fn next_frame_within(&mut self, budget: Duration) -> Result<CapturedFrame> {
|
||||
self.frame_within(budget)
|
||||
}
|
||||
|
||||
fn supports_arrival_wait(&self) -> bool {
|
||||
@@ -417,9 +399,41 @@ impl Capturer for PortalCapturer {
|
||||
}
|
||||
|
||||
impl PortalCapturer {
|
||||
/// The [`Capturer::next_frame`] budget expired (or the thread ended) — turn it into the
|
||||
/// diagnosis-bearing error. Split out of the slicing loop above; behavior unchanged.
|
||||
fn next_frame_timed_out(&self, err: RecvTimeoutError) -> Result<CapturedFrame> {
|
||||
/// The blocking first-frame wait behind [`Capturer::next_frame`] /
|
||||
/// [`Capturer::next_frame_within`]. First frame can lag behind format negotiation; later
|
||||
/// frames arrive at ~fps. Wait in short slices so a GPU-import poison (worker death) fails
|
||||
/// the capture within ~0.5 s instead of sitting out the full first-frame budget.
|
||||
fn frame_within(&mut self, budget: Duration) -> Result<CapturedFrame> {
|
||||
let deadline = std::time::Instant::now() + budget;
|
||||
loop {
|
||||
if self.broken.load(Ordering::Relaxed) {
|
||||
return Err(anyhow!(
|
||||
"zero-copy GPU import lost (node {}): the import worker died or tiled imports \
|
||||
failed repeatedly — rebuilding capture",
|
||||
self.node_id
|
||||
));
|
||||
}
|
||||
if let Some(f) = self.pending.take() {
|
||||
return Ok(f); // a wait_arrival stash outranks the channel (it's older)
|
||||
}
|
||||
let slice = Duration::from_millis(500)
|
||||
.min(deadline.saturating_duration_since(std::time::Instant::now()));
|
||||
match self.frames.recv_timeout(slice) {
|
||||
Ok(frame) => return Ok(frame),
|
||||
Err(RecvTimeoutError::Timeout) if std::time::Instant::now() < deadline => continue,
|
||||
Err(e) => return self.next_frame_timed_out(e, budget),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The [`frame_within`](Self::frame_within) budget expired (or the thread ended) — turn it
|
||||
/// into the diagnosis-bearing error. Split out of the slicing loop above; behavior unchanged.
|
||||
fn next_frame_timed_out(
|
||||
&self,
|
||||
err: RecvTimeoutError,
|
||||
budget: Duration,
|
||||
) -> Result<CapturedFrame> {
|
||||
let within = budget.as_secs_f32();
|
||||
match err {
|
||||
RecvTimeoutError::Timeout => {
|
||||
// Split the two black-screen root causes apart so the operator gets a cause, not
|
||||
@@ -427,9 +441,10 @@ impl PortalCapturer {
|
||||
// not (no acceptable format / node never emitted a param)?
|
||||
if self.negotiated.load(Ordering::Relaxed) {
|
||||
Err(anyhow!(
|
||||
"no PipeWire frame within 10s (node {}): format negotiated but no buffers \
|
||||
arrived — the compositor produced no frames (virtual output idle/unmapped, \
|
||||
or capture never started)",
|
||||
"no PipeWire frame within {within}s (node {}): format negotiated but no \
|
||||
buffers arrived — the compositor produced no frames (virtual output \
|
||||
idle/unmapped, capture never started, or a stream bound during a \
|
||||
compositor (re)start that will never deliver — a reconnect fixes that)",
|
||||
self.node_id
|
||||
))
|
||||
} else if self.hdr_offer {
|
||||
@@ -440,10 +455,10 @@ impl PortalCapturer {
|
||||
// auto-reconnects) negotiates SDR instead of re-running this same timeout.
|
||||
super::note_hdr_capture_failed();
|
||||
Err(anyhow!(
|
||||
"no PipeWire frame within 10s (node {}): the compositor never accepted \
|
||||
the HDR (10-bit PQ/BT.2020 dmabuf) offer — is the mirrored monitor in \
|
||||
HDR mode on GNOME 50+? Downgrading this host to SDR capture; reconnect \
|
||||
to stream SDR",
|
||||
"no PipeWire frame within {within}s (node {}): the compositor never \
|
||||
accepted the HDR (10-bit PQ/BT.2020 dmabuf) offer — is the mirrored \
|
||||
monitor in HDR mode on GNOME 50+? Downgrading this host to SDR capture; \
|
||||
reconnect to stream SDR",
|
||||
self.node_id
|
||||
))
|
||||
} else if self.vaapi_dmabuf && !pf_zerocopy::vaapi_dmabuf_forced() {
|
||||
@@ -452,14 +467,15 @@ impl PortalCapturer {
|
||||
// retries on the CPU offer instead of failing this same negotiation forever.
|
||||
pf_zerocopy::note_vaapi_dmabuf_failed();
|
||||
Err(anyhow!(
|
||||
"no PipeWire frame within 10s (node {}): the compositor never accepted \
|
||||
the LINEAR-dmabuf offer (VAAPI zero-copy) — downgrading this host to the \
|
||||
CPU capture path; the pipeline rebuild will renegotiate without dmabuf",
|
||||
"no PipeWire frame within {within}s (node {}): the compositor never \
|
||||
accepted the LINEAR-dmabuf offer (VAAPI zero-copy) — downgrading this \
|
||||
host to the CPU capture path; the pipeline rebuild will renegotiate \
|
||||
without dmabuf",
|
||||
self.node_id
|
||||
))
|
||||
} else {
|
||||
Err(anyhow!(
|
||||
"no PipeWire frame within 10s (node {}): format negotiation never \
|
||||
"no PipeWire frame within {within}s (node {}): format negotiation never \
|
||||
completed — the compositor offered no format this consumer accepts \
|
||||
(pixel-format/modifier mismatch) or the node never emitted a Format param",
|
||||
self.node_id
|
||||
@@ -824,6 +840,7 @@ mod pipewire {
|
||||
VideoFormat::RGBA => PixelFormat::Rgba,
|
||||
VideoFormat::RGB => PixelFormat::Rgb,
|
||||
VideoFormat::BGR => PixelFormat::Bgr,
|
||||
VideoFormat::NV12 => PixelFormat::Nv12,
|
||||
// The GNOME 50+ HDR screencast formats (packed 2:10:10:10; only ever negotiated by
|
||||
// the `want_hdr` offer, whose MANDATORY colorimetry props pin them to PQ/BT.2020).
|
||||
VideoFormat::xRGB_210LE => PixelFormat::X2Rgb10,
|
||||
@@ -989,10 +1006,10 @@ mod pipewire {
|
||||
.into_inner())
|
||||
}
|
||||
|
||||
/// Build a BGRx dmabuf `EnumFormat` pod advertising the EGL-importable `modifiers` as a
|
||||
/// mandatory enum Choice; the compositor fixates to one of them that it can allocate, which
|
||||
/// we read back in `param_changed`.
|
||||
/// Build a LINEAR/modifier DMA-BUF `EnumFormat` pod. Packed BGRx is the existing import path;
|
||||
/// NV12 is gamescope's producer-side RGB→YUV path (opt-in during bring-up).
|
||||
fn build_dmabuf_format(
|
||||
format: VideoFormat,
|
||||
modifiers: &[u64],
|
||||
preferred: Option<(u32, u32, u32)>,
|
||||
) -> Result<Vec<u8>> {
|
||||
@@ -1003,7 +1020,7 @@ mod pipewire {
|
||||
pw::spa::param::ParamType::EnumFormat,
|
||||
pw::spa::pod::property!(FormatProperties::MediaType, Id, MediaType::Video),
|
||||
pw::spa::pod::property!(FormatProperties::MediaSubtype, Id, MediaSubtype::Raw),
|
||||
pw::spa::pod::property!(FormatProperties::VideoFormat, Id, VideoFormat::BGRx),
|
||||
pw::spa::pod::property!(FormatProperties::VideoFormat, Id, format),
|
||||
pw::spa::pod::property!(
|
||||
FormatProperties::VideoSize,
|
||||
Choice,
|
||||
@@ -1032,6 +1049,22 @@ mod pipewire {
|
||||
pw::spa::utils::Fraction { num: 240, denom: 1 }
|
||||
),
|
||||
);
|
||||
if format == VideoFormat::NV12 {
|
||||
obj.properties.push(pw::spa::pod::Property {
|
||||
key: pw::spa::sys::SPA_FORMAT_VIDEO_colorMatrix,
|
||||
flags: pw::spa::pod::PropertyFlags::MANDATORY,
|
||||
value: pw::spa::pod::Value::Id(pw::spa::utils::Id(
|
||||
pw::spa::sys::SPA_VIDEO_COLOR_MATRIX_BT709,
|
||||
)),
|
||||
});
|
||||
obj.properties.push(pw::spa::pod::Property {
|
||||
key: pw::spa::sys::SPA_FORMAT_VIDEO_colorRange,
|
||||
flags: pw::spa::pod::PropertyFlags::MANDATORY,
|
||||
value: pw::spa::pod::Value::Id(pw::spa::utils::Id(
|
||||
pw::spa::sys::SPA_VIDEO_COLOR_RANGE_16_235,
|
||||
)),
|
||||
});
|
||||
}
|
||||
obj.properties.push(pw::spa::pod::Property {
|
||||
key: pw::spa::sys::SPA_FORMAT_VIDEO_modifier,
|
||||
flags: pw::spa::pod::PropertyFlags::MANDATORY,
|
||||
@@ -1310,21 +1343,25 @@ mod pipewire {
|
||||
/// (which Mutter delivers as metadata-only "corrupted" buffers) still refresh the position.
|
||||
fn update_cursor_meta(cursor: &mut CursorState, spa_buf: *mut spa::sys::spa_buffer) {
|
||||
// SAFETY: `spa_buf` is the live buffer we still hold (dequeued, not yet requeued).
|
||||
// `spa_buffer_find_meta_data` scans its metadata array for a `SPA_META_Cursor` of at least
|
||||
// `size_of::<spa_meta_cursor>()` bytes and returns a pointer into that buffer's metadata
|
||||
// (or null), valid until requeue. The size argument matches the struct the result is cast to.
|
||||
let cur = unsafe {
|
||||
spa::sys::spa_buffer_find_meta_data(
|
||||
spa_buf,
|
||||
spa::sys::SPA_META_Cursor,
|
||||
std::mem::size_of::<spa::sys::spa_meta_cursor>(),
|
||||
) as *const spa::sys::spa_meta_cursor
|
||||
};
|
||||
if cur.is_null() {
|
||||
// `spa_buffer_find_meta` returns the `spa_meta` (type + byte `size` + `data` pointer) for
|
||||
// `SPA_META_Cursor`, or null. We take `find_meta` rather than `find_meta_data` specifically
|
||||
// to obtain the region's real `size`: the bitmap offset, pixel offset and stride read below
|
||||
// are ALL producer-written, and without a bound against the actual region they drive
|
||||
// out-of-bounds pointer arithmetic and an oversized `slice::from_raw_parts` — an OOB read
|
||||
// that SIGSEGVs inside the PipeWire `.process` callback (a segfault `catch_unwind` cannot
|
||||
// catch). Every offset below is validated against `region_size` with checked arithmetic,
|
||||
// mirroring the fd-length guard the main frame path already applies to xdg-desktop-portal-wlr.
|
||||
let meta = unsafe { spa::sys::spa_buffer_find_meta(spa_buf, spa::sys::SPA_META_Cursor) };
|
||||
if meta.is_null() {
|
||||
return;
|
||||
}
|
||||
// SAFETY: `cur` is non-null and points to a `spa_meta_cursor` of at least its own size
|
||||
// inside the held buffer (guaranteed by the size arg above), so every field read is in bounds.
|
||||
// SAFETY: `meta` is non-null and points into the held buffer's metadata array.
|
||||
let (region_size, data) = unsafe { ((*meta).size as usize, (*meta).data as *const u8) };
|
||||
if data.is_null() || region_size < std::mem::size_of::<spa::sys::spa_meta_cursor>() {
|
||||
return;
|
||||
}
|
||||
let cur = data as *const spa::sys::spa_meta_cursor;
|
||||
// SAFETY: `region_size >= size_of::<spa_meta_cursor>()` checked above, so every field is in bounds.
|
||||
let (id, pos_x, pos_y, hot_x, hot_y, bmp_off) = unsafe {
|
||||
(
|
||||
(*cur).id,
|
||||
@@ -1347,13 +1384,18 @@ mod pipewire {
|
||||
// Position-only update — keep the cached bitmap.
|
||||
return;
|
||||
}
|
||||
// SAFETY: `bitmap_offset` is a byte offset from `cur` to a `spa_meta_bitmap`, which the
|
||||
// producer placed inside the same meta region it sized for this cursor (>= the size we
|
||||
// requested). The resulting pointer is in bounds and aligned for `spa_meta_bitmap`.
|
||||
let bmp =
|
||||
unsafe { (cur as *const u8).add(bmp_off as usize) as *const spa::sys::spa_meta_bitmap };
|
||||
// SAFETY: `bmp` is the in-bounds, aligned `spa_meta_bitmap` pointer computed just above; the
|
||||
// producer fully initialized this header, so reading its scalar fields is sound.
|
||||
let bmp_off = bmp_off as usize;
|
||||
// The `spa_meta_bitmap` header must fit entirely inside the region before we read it —
|
||||
// `bitmap_offset` is producer-controlled and otherwise reads past the metadata.
|
||||
match bmp_off.checked_add(std::mem::size_of::<spa::sys::spa_meta_bitmap>()) {
|
||||
Some(end) if end <= region_size => {}
|
||||
_ => return,
|
||||
}
|
||||
// SAFETY: `bmp_off + size_of::<spa_meta_bitmap>() <= region_size` (checked directly above),
|
||||
// so the header is fully in bounds; the producer places it aligned as before.
|
||||
let bmp = unsafe { data.add(bmp_off) as *const spa::sys::spa_meta_bitmap };
|
||||
// SAFETY: `bmp` is the in-bounds `spa_meta_bitmap` header validated just above; reading its
|
||||
// scalar fields is sound.
|
||||
let (vfmt, bw, bh, stride, pix_off) = unsafe {
|
||||
(
|
||||
(*bmp).format,
|
||||
@@ -1369,10 +1411,27 @@ mod pipewire {
|
||||
}
|
||||
let row = bw as usize * 4;
|
||||
let stride = if stride < row { row } else { stride };
|
||||
let span = stride * (bh as usize - 1) + row;
|
||||
// SAFETY: the bitmap pixels live at `bmp + pix_off` for `span` bytes, within the
|
||||
// producer-sized meta region. `span` is the exact extent the strided copy below reads.
|
||||
let src = unsafe { std::slice::from_raw_parts((bmp as *const u8).add(pix_off), span) };
|
||||
// `span` is the exact byte extent the strided loop reads: `stride·(bh-1) + row`. Compute it
|
||||
// with checked arithmetic (a producer stride near `i32::MAX` would otherwise overflow) and
|
||||
// require the whole pixel block `[bmp_off + pix_off, +span)` to lie inside the region before
|
||||
// fabricating the slice — this is the check whose absence made the read go out of bounds.
|
||||
let span = match stride
|
||||
.checked_mul(bh as usize - 1)
|
||||
.and_then(|v| v.checked_add(row))
|
||||
{
|
||||
Some(s) => s,
|
||||
None => return,
|
||||
};
|
||||
match bmp_off
|
||||
.checked_add(pix_off)
|
||||
.and_then(|v| v.checked_add(span))
|
||||
{
|
||||
Some(end) if end <= region_size => {}
|
||||
_ => return,
|
||||
}
|
||||
// SAFETY: `bmp_off + pix_off + span <= region_size` (checked directly above), so the slice
|
||||
// is fully within the producer's meta region; `span` is exactly the strided loop's extent.
|
||||
let src = unsafe { std::slice::from_raw_parts(data.add(bmp_off + pix_off), span) };
|
||||
let mut rgba = vec![0u8; bw as usize * bh as usize * 4];
|
||||
for y in 0..bh as usize {
|
||||
for x in 0..bw as usize {
|
||||
@@ -1578,8 +1637,8 @@ mod pipewire {
|
||||
}
|
||||
}
|
||||
|
||||
// VAAPI zero-copy passthrough: hand the raw dmabuf straight to the encoder, which imports
|
||||
// it into a VA surface and does RGB→NV12 on the GPU video engine. No CUDA importer here.
|
||||
// Raw DMA-BUF passthrough: packed RGB is imported for GPU CSC; producer-native NV12 can
|
||||
// be consumed by the Vulkan Video encoder without another color conversion.
|
||||
if ud.vaapi_passthrough {
|
||||
if let Some(fmt) = ud.format {
|
||||
if datas[0].type_() == pw::spa::buffer::DataType::DmaBuf {
|
||||
@@ -1587,9 +1646,41 @@ mod pipewire {
|
||||
let chunk = datas[0].chunk();
|
||||
let offset = chunk.offset();
|
||||
let stride = chunk.stride().max(0) as u32;
|
||||
// Native NV12 usually arrives as a two-plane SPA buffer over ONE buffer
|
||||
// object; plane 1's chunk carries the REAL UV offset/stride (compositors
|
||||
// may align the Y plane before UV). Pass it through instead of assuming
|
||||
// contiguity. Each spa_data holds its own (dup'd) fd, so BO identity is
|
||||
// by inode, not fd number; a genuinely two-BO frame cannot travel through
|
||||
// the single-fd import — drop it with a diagnosis instead of streaming
|
||||
// garbage chroma.
|
||||
let plane1 =
|
||||
if fmt == PixelFormat::Nv12 && datas.len() >= 2 && datas[1].fd() > 0 {
|
||||
// SAFETY: zeroed `libc::stat` is a valid POD initializer; both fds are
|
||||
// owned by the live PipeWire buffer for this callback, and `fstat`
|
||||
// only writes the out-param structs, whose fields are read only after
|
||||
// the `== 0` success checks.
|
||||
let same_bo = unsafe {
|
||||
let mut s0: libc::stat = std::mem::zeroed();
|
||||
let mut s1: libc::stat = std::mem::zeroed();
|
||||
libc::fstat(datas[0].fd() as i32, &mut s0) == 0
|
||||
&& libc::fstat(datas[1].fd() as i32, &mut s1) == 0
|
||||
&& (s0.st_dev, s0.st_ino) == (s1.st_dev, s1.st_ino)
|
||||
};
|
||||
if !same_bo {
|
||||
warn_once(
|
||||
"NV12 planes live in different buffer objects — frames \
|
||||
dropped (single-fd import only)",
|
||||
);
|
||||
return;
|
||||
}
|
||||
let c1 = datas[1].chunk();
|
||||
Some((c1.offset(), c1.stride().max(0) as u32))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
// dup the fd so it survives the SPA buffer recycle — the encode thread
|
||||
// imports it. (Content stability across the brief map+CSC window relies on
|
||||
// the compositor's buffer-pool depth, like any zero-copy capture.)
|
||||
// imports it. Content stability across the brief import/encode window relies
|
||||
// on the compositor's buffer-pool depth, like any zero-copy capture.
|
||||
// SAFETY: `datas[0].fd()` is the dmabuf fd owned by the live PipeWire buffer (valid
|
||||
// for this callback). `fcntl(fd, F_DUPFD_CLOEXEC, 0)` reads only the integer fd,
|
||||
// touches no Rust memory, and returns a fresh independent CLOEXEC duplicate (or -1).
|
||||
@@ -1616,9 +1707,10 @@ mod pipewire {
|
||||
modifier: ud.modifier,
|
||||
offset,
|
||||
stride,
|
||||
plane1,
|
||||
}),
|
||||
// Cursor-as-metadata: the encoder blends this into its owned VA
|
||||
// surface (raw dmabuf never touched).
|
||||
// Cursor-as-metadata is blended only by RGB→NV12 backends. Gamescope
|
||||
// embeds its pointer in the produced pixels, so native NV12 has none.
|
||||
cursor: ud.cursor.overlay(),
|
||||
});
|
||||
static ONCE: std::sync::atomic::AtomicBool =
|
||||
@@ -1629,7 +1721,12 @@ mod pipewire {
|
||||
h,
|
||||
modifier = ud.modifier,
|
||||
fourcc = format_args!("{:#010x}", fourcc),
|
||||
"zero-copy: handing the raw dmabuf to the encoder (GPU import + CSC)"
|
||||
source = if fmt == PixelFormat::Nv12 {
|
||||
"producer-native NV12"
|
||||
} else {
|
||||
"packed RGB (encoder GPU CSC)"
|
||||
},
|
||||
"zero-copy: handing the raw DMA-BUF to the encoder"
|
||||
);
|
||||
}
|
||||
return;
|
||||
@@ -1989,6 +2086,28 @@ mod pipewire {
|
||||
// PyroWave session (the wavelet encoder's own Vulkan device, any vendor) → hand the raw
|
||||
// dmabuf straight to the encoder.
|
||||
let vaapi_passthrough = zerocopy && !force_shm && importer.is_none() && raw_passthrough;
|
||||
// Producer-side NV12 (default-on; PUNKTFUNK_PIPEWIRE_NV12=0 escapes): gamescope offers a
|
||||
// one-fd LINEAR NV12 image when the consumer asks — its compositor pass does the RGB→YUV,
|
||||
// and the Vulkan Video encoder imports the buffer as its encode source directly (no host
|
||||
// CSC at all). `native_nv12_session` restricts this to sessions whose encoder can ingest
|
||||
// it (Linux vulkan-encode H265/AV1 — never H264/libav-VAAPI, GameStream-resolve, or
|
||||
// PyroWave, whose Vulkan compute CSC ingests packed RGB only). Raw passthrough is
|
||||
// required because the CUDA importer expects packed RGB, and 4:4:4/HDR must not be
|
||||
// silently subsampled/downconverted. Non-NV12 compositors (KWin/GNOME) simply match the
|
||||
// packed-RGB fallback pod.
|
||||
let prefer_native_nv12 = std::env::var("PUNKTFUNK_PIPEWIRE_NV12").as_deref() != Ok("0")
|
||||
&& policy.native_nv12_session
|
||||
&& backend_is_vaapi
|
||||
&& vaapi_passthrough
|
||||
&& !policy.pyrowave_session
|
||||
&& !want_444
|
||||
&& !want_hdr;
|
||||
if prefer_native_nv12 {
|
||||
tracing::info!(
|
||||
"zero-copy: preferring gamescope producer-side NV12 LINEAR DMA-BUF (no host \
|
||||
RGB CSC; PUNKTFUNK_PIPEWIRE_NV12=0 restores the packed-RGB negotiation)"
|
||||
);
|
||||
}
|
||||
// Modifiers our import stack handles for BGRx: the EGL-importable (tiled) set, plus LINEAR
|
||||
// (0) — NVIDIA's EGL won't list it, but LINEAR dmabufs (gamescope's only offer) import via
|
||||
// CUDA external memory instead. For the VAAPI passthrough path we advertise LINEAR only:
|
||||
@@ -2028,7 +2147,9 @@ mod pipewire {
|
||||
tracing::warn!("zero-copy: no importable dmabuf modifiers — using CPU path");
|
||||
} else if vaapi_passthrough && policy.pyrowave_modifiers.is_empty() {
|
||||
tracing::info!(
|
||||
"zero-copy: advertising LINEAR dmabuf for direct VAAPI import (GPU CSC)"
|
||||
native_nv12_preferred = prefer_native_nv12,
|
||||
"zero-copy: advertising LINEAR DMA-BUF for encoder import (native NV12 first \
|
||||
when enabled, packed RGB fallback)"
|
||||
);
|
||||
} else if want_dmabuf && !vaapi_passthrough {
|
||||
tracing::info!(
|
||||
@@ -2164,36 +2285,37 @@ mod pipewire {
|
||||
}
|
||||
})
|
||||
.process(|stream, ud| {
|
||||
// PipeWire dispatches this from a C trampoline with no catch_unwind; a
|
||||
// panic crossing that FFI boundary would abort the whole host. Contain it.
|
||||
let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
// Latest-frame-only (OBS pattern): Mutter delivers buffers in bursts and
|
||||
// recycles its pool; an older queued buffer carries a STALE frame. Drain all
|
||||
// queued buffers, requeue the older ones, keep only the newest.
|
||||
// SAFETY: `stream` is the live stream PipeWire passes into this `.process` callback on
|
||||
// the loop thread, where `pw_stream_dequeue_buffer` is the documented call. It returns
|
||||
// a `*mut pw_buffer` owned by the stream (or null when the queue is drained),
|
||||
// null-checked before any use. The loop is single-threaded, so no concurrent access.
|
||||
// Latest-frame-only (OBS pattern): Mutter delivers buffers in bursts and recycles its
|
||||
// pool; an older queued buffer carries a STALE frame. Drain all queued buffers, requeue
|
||||
// the older ones, keep only the newest. This dequeue/requeue runs OUTSIDE the
|
||||
// `catch_unwind` below — they are non-panicking C FFI pointer ops, and `newest` is
|
||||
// requeued exactly once AFTER the panic-containing region. Previously the whole thing was
|
||||
// inside the catch, so a caught panic (in `update_cursor_meta`/`consume_frame`) stranded
|
||||
// `newest` forever, permanently shrinking the stream's fixed pool until capture wedged.
|
||||
// SAFETY: `stream` is the live stream PipeWire passes into this `.process` callback on the
|
||||
// loop thread; `dequeue_raw_buffer` returns a stream-owned `*mut pw_buffer` or null
|
||||
// (null-checked), single-threaded so no concurrent access.
|
||||
let mut newest = unsafe { stream.dequeue_raw_buffer() };
|
||||
if newest.is_null() {
|
||||
return;
|
||||
}
|
||||
let mut drained = 1u32;
|
||||
loop {
|
||||
// SAFETY: same stream/loop-thread contract as the dequeue above; each call returns
|
||||
// the next stream-owned `*mut pw_buffer` or null (null-checked before use).
|
||||
// SAFETY: same stream/loop-thread contract; returns the next stream-owned buffer or null.
|
||||
let next = unsafe { stream.dequeue_raw_buffer() };
|
||||
if next.is_null() {
|
||||
break;
|
||||
}
|
||||
// SAFETY: `newest` is a non-null `*mut pw_buffer` previously dequeued from this same
|
||||
// stream and not yet requeued; `pw_stream_queue_buffer` hands ownership back to the
|
||||
// stream. We immediately overwrite `newest = next`, so the requeued pointer is never
|
||||
// touched again (no use-after-requeue). Loop thread, single-threaded.
|
||||
// SAFETY: `newest` was dequeued from this stream and not yet requeued; we immediately
|
||||
// overwrite it, so the requeued pointer is never touched again.
|
||||
unsafe { stream.queue_raw_buffer(newest) };
|
||||
newest = next;
|
||||
drained += 1;
|
||||
}
|
||||
// PipeWire dispatches from a C trampoline with no catch_unwind; a panic crossing that FFI
|
||||
// boundary would abort the whole host. Contain the inspect/consume work — the only Rust
|
||||
// code here that can panic — and requeue `newest` unconditionally after it.
|
||||
let outcome = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
// SAFETY: `newest` is the non-null buffer we still own (dequeued, not requeued);
|
||||
// `.buffer` is a `*mut spa_buffer` field libpipewire populated. This is a single field
|
||||
// load through a valid pointer — no mutation or aliasing.
|
||||
@@ -2272,19 +2394,18 @@ mod pipewire {
|
||||
"capture: skipped a stale CORRUPTED/cursor buffer (GNOME)"
|
||||
);
|
||||
}
|
||||
// SAFETY: `newest` is the non-null buffer we own (dequeued, never requeued on this
|
||||
// skip path); hand it back to the stream exactly once and return without touching it
|
||||
// again. Loop thread inside `.process`.
|
||||
unsafe { stream.queue_raw_buffer(newest) };
|
||||
// Skip this stale/cursor buffer — `newest` is requeued unconditionally below.
|
||||
return;
|
||||
}
|
||||
|
||||
consume_frame(ud, spa_buf);
|
||||
// SAFETY: `consume_frame` has finished reading `spa_buf` (and the `datas` borrows derived
|
||||
// from `newest`), so requeuing the owned `newest` exactly once here is sound — no
|
||||
// use-after-requeue. Loop thread inside `.process`.
|
||||
unsafe { stream.queue_raw_buffer(newest) };
|
||||
}));
|
||||
// Hand `newest` back to the stream exactly once, on EVERY path — normal, corrupted-skip,
|
||||
// or a caught panic in the closure above. This single requeue is what keeps the fixed
|
||||
// buffer pool from draining.
|
||||
// SAFETY: all reads of `spa_buf`/`newest` (update_cursor_meta, consume_frame) completed
|
||||
// inside the closure above; `newest` was dequeued from this stream and not yet requeued.
|
||||
unsafe { stream.queue_raw_buffer(newest) };
|
||||
if outcome.is_err() {
|
||||
// In the per-frame `.process` callback: a deterministic panic (e.g. a bad
|
||||
// format) would fire this every frame, so power-of-two throttle it — enough to
|
||||
@@ -2368,7 +2489,18 @@ mod pipewire {
|
||||
build_hdr_dmabuf_format(VideoFormat::xBGR_210LE, preferred)?,
|
||||
]
|
||||
} else if want_dmabuf {
|
||||
vec![build_dmabuf_format(&modifiers, preferred)?]
|
||||
let mut pods = Vec::with_capacity(if prefer_native_nv12 { 2 } else { 1 });
|
||||
if prefer_native_nv12 {
|
||||
// First compatible consumer pod wins. Gamescope advertises NV12 and BGRx; pinning
|
||||
// BT.709 limited here selects its RGB→NV12 shader with our bitstream colorimetry.
|
||||
pods.push(build_dmabuf_format(VideoFormat::NV12, &[0], preferred)?);
|
||||
}
|
||||
pods.push(build_dmabuf_format(
|
||||
VideoFormat::BGRx,
|
||||
&modifiers,
|
||||
preferred,
|
||||
)?);
|
||||
pods
|
||||
} else {
|
||||
vec![serialize_pod(obj)?]
|
||||
};
|
||||
|
||||
@@ -308,8 +308,9 @@ float2 main(float4 pos : SV_POSITION, float2 uv : TEXCOORD0) : SV_TARGET {
|
||||
/// Plane writes use per-plane render-target views of the single P010 texture: an `R16_UNORM` RTV
|
||||
/// selects plane 0 (luma, full WxH), an `R16G16_UNORM` RTV selects plane 1 (chroma, W/2 x H/2). This
|
||||
/// planar-RTV mechanism needs a D3D11.3+ runtime + driver support; [`HdrP010Converter::convert`]
|
||||
/// surfaces a clear error if `CreateRenderTargetView` rejects the plane format so the caller can fall
|
||||
/// back to the existing R10 path.
|
||||
/// surfaces a clear error if `CreateRenderTargetView` rejects the plane format. (There is no runtime
|
||||
/// fallback — the error propagates through `try_consume` and ends the session; the "R10 path" the
|
||||
/// original design referenced was never kept.)
|
||||
pub(crate) struct HdrP010Converter {
|
||||
vs: ID3D11VertexShader,
|
||||
ps_y: ID3D11PixelShader,
|
||||
@@ -737,14 +738,157 @@ fn p010_reference(r: f64, g: f64, b: f64) -> (f64, f64, f64) {
|
||||
/// Y ≤ 4 codes, U/V ≤ 5 codes (rounding + chroma averaging). Prints a per-colour table + PASS/FAIL.
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn hdr_p010_selftest() -> Result<()> {
|
||||
use windows::Win32::Graphics::Direct3D::D3D_DRIVER_TYPE_HARDWARE;
|
||||
use windows::Win32::Graphics::Dxgi::IDXGIAdapter;
|
||||
hdr_p010_selftest_at(64, 64, None)
|
||||
}
|
||||
|
||||
// 64x64, even dims. A 4x4 grid of 16x16 flat scRGB blocks (each 2x2 chroma footprint uniform →
|
||||
// exact chroma comparison) covering pure R/G/B/white/black/gray at plausible HDR nit levels, plus
|
||||
// a couple of bright (>1.0 scRGB) colours, then the rest is a gradient (compared on Y only).
|
||||
const W: u32 = 64;
|
||||
const H: u32 = 64;
|
||||
/// [`hdr_p010_selftest`] at an arbitrary even size and (optionally) on a specific GPU vendor
|
||||
/// (PCI vendor id, e.g. `0x8086` Intel / `0x10de` NVIDIA / `0x1002` AMD). The size matters on
|
||||
/// top of the 64×64 default because the field sessions run at capture resolutions whose height
|
||||
/// is NOT 16-aligned (1080 → the encoder's align16 pool seam) and a driver may treat the planar
|
||||
/// RTVs differently at real sizes; the vendor pin matters on dual-GPU boxes where the default
|
||||
/// adapter is not the one the session encodes on.
|
||||
/// Test support (used by pf-encode's live e2e): the 8 sRGB colour bars (white/yellow/cyan/green/
|
||||
/// magenta/red/blue/black, sRGB 1.0 = scRGB 1.0 = 80 nits) as a w×h FP16 scRGB texture on the
|
||||
/// adapter with `luid`, converted through the REAL [`HdrP010Converter`] into a P010 texture with
|
||||
/// **`BIND_RENDER_TARGET` only, `MiscFlags` 0 — the exact bind profile of the IDD out-ring** (the
|
||||
/// CPU-upload encoder tests can't use that profile, so only this path exercises "RTV-written P010
|
||||
/// → encoder ingest copy"). Returns `(device, p010)`; expected decoded codes per bar are the
|
||||
/// bars_pq2020 fixture's: (490,512,512) (478,423,518) (464,525,473) (450,432,476) (350,584,585)
|
||||
/// (325,448,598) (226,650,535) (64,512,512).
|
||||
#[cfg(target_os = "windows")]
|
||||
#[doc(hidden)]
|
||||
pub fn hdr_p010_convert_bars_on_luid(
|
||||
luid: [u8; 8],
|
||||
w: u32,
|
||||
h: u32,
|
||||
) -> Result<(ID3D11Device, ID3D11Texture2D)> {
|
||||
use windows::Win32::Graphics::Direct3D::D3D_DRIVER_TYPE_UNKNOWN;
|
||||
use windows::Win32::Graphics::Dxgi::{CreateDXGIFactory1, IDXGIAdapter1, IDXGIFactory4};
|
||||
|
||||
if w == 0 || h == 0 || w % 2 != 0 || h % 2 != 0 {
|
||||
bail!("bars pattern needs even non-zero dimensions, got {w}x{h}");
|
||||
}
|
||||
// sRGB primaries at full/zero channels: sRGB EOTF(1.0)=1.0, (0)=0 → the scRGB pattern is
|
||||
// pure 0/1 floats and the PQ/BT.2020 reference codes above are exact.
|
||||
const BARS: [(f32, f32, f32); 8] = [
|
||||
(1.0, 1.0, 1.0),
|
||||
(1.0, 1.0, 0.0),
|
||||
(0.0, 1.0, 1.0),
|
||||
(0.0, 1.0, 0.0),
|
||||
(1.0, 0.0, 1.0),
|
||||
(1.0, 0.0, 0.0),
|
||||
(0.0, 0.0, 1.0),
|
||||
(0.0, 0.0, 0.0),
|
||||
];
|
||||
let bar_w = (w / 8).max(1) as usize;
|
||||
let mut fp16 = vec![0u16; (w * h * 4) as usize];
|
||||
for y in 0..h as usize {
|
||||
for x in 0..w as usize {
|
||||
let (r, g, b) = BARS[(x / bar_w).min(7)];
|
||||
let i = (y * w as usize + x) * 4;
|
||||
fp16[i] = f32_to_f16(r);
|
||||
fp16[i + 1] = f32_to_f16(g);
|
||||
fp16[i + 2] = f32_to_f16(b);
|
||||
fp16[i + 3] = f32_to_f16(1.0);
|
||||
}
|
||||
}
|
||||
// SAFETY: same single-device/single-thread contract as `hdr_p010_selftest_at`; the FP16
|
||||
// initial-data Vec outlives the synchronous CreateTexture2D; the returned COM handles own
|
||||
// their references.
|
||||
unsafe {
|
||||
let luid = windows::Win32::Foundation::LUID {
|
||||
LowPart: u32::from_le_bytes(luid[..4].try_into().unwrap()),
|
||||
HighPart: i32::from_le_bytes(luid[4..].try_into().unwrap()),
|
||||
};
|
||||
let factory: IDXGIFactory4 = CreateDXGIFactory1().context("dxgi factory")?;
|
||||
let adapter: IDXGIAdapter1 = factory.EnumAdapterByLuid(luid).context("adapter by luid")?;
|
||||
let mut device: Option<ID3D11Device> = None;
|
||||
let mut context: Option<ID3D11DeviceContext> = None;
|
||||
D3D11CreateDevice(
|
||||
&adapter,
|
||||
D3D_DRIVER_TYPE_UNKNOWN,
|
||||
HMODULE::default(),
|
||||
D3D11_CREATE_DEVICE_BGRA_SUPPORT,
|
||||
Some(&[D3D_FEATURE_LEVEL_11_0]),
|
||||
D3D11_SDK_VERSION,
|
||||
Some(&mut device),
|
||||
None,
|
||||
Some(&mut context),
|
||||
)
|
||||
.context("D3D11CreateDevice(luid) for bars convert")?;
|
||||
let device = device.context("null device")?;
|
||||
let context = context.context("null context")?;
|
||||
|
||||
let src_desc = D3D11_TEXTURE2D_DESC {
|
||||
Width: w,
|
||||
Height: h,
|
||||
MipLevels: 1,
|
||||
ArraySize: 1,
|
||||
Format: DXGI_FORMAT_R16G16B16A16_FLOAT,
|
||||
SampleDesc: DXGI_SAMPLE_DESC {
|
||||
Count: 1,
|
||||
Quality: 0,
|
||||
},
|
||||
Usage: D3D11_USAGE_DEFAULT,
|
||||
BindFlags: D3D11_BIND_SHADER_RESOURCE.0 as u32,
|
||||
..Default::default()
|
||||
};
|
||||
let init = D3D11_SUBRESOURCE_DATA {
|
||||
pSysMem: fp16.as_ptr() as *const c_void,
|
||||
SysMemPitch: w * 8,
|
||||
SysMemSlicePitch: 0,
|
||||
};
|
||||
let mut src_tex: Option<ID3D11Texture2D> = None;
|
||||
device
|
||||
.CreateTexture2D(&src_desc, Some(&init), Some(&mut src_tex))
|
||||
.context("CreateTexture2D(fp16 bars)")?;
|
||||
let src_tex = src_tex.context("null src tex")?;
|
||||
let mut src_srv: Option<ID3D11ShaderResourceView> = None;
|
||||
device
|
||||
.CreateShaderResourceView(&src_tex, None, Some(&mut src_srv))
|
||||
.context("CreateShaderResourceView(fp16 bars)")?;
|
||||
let src_srv = src_srv.context("null src srv")?;
|
||||
|
||||
// The IDD out-ring's exact profile: P010, RENDER_TARGET only, MiscFlags 0.
|
||||
let p010_desc = D3D11_TEXTURE2D_DESC {
|
||||
Width: w,
|
||||
Height: h,
|
||||
MipLevels: 1,
|
||||
ArraySize: 1,
|
||||
Format: DXGI_FORMAT_P010,
|
||||
SampleDesc: DXGI_SAMPLE_DESC {
|
||||
Count: 1,
|
||||
Quality: 0,
|
||||
},
|
||||
Usage: D3D11_USAGE_DEFAULT,
|
||||
BindFlags: D3D11_BIND_RENDER_TARGET.0 as u32,
|
||||
..Default::default()
|
||||
};
|
||||
let mut p010: Option<ID3D11Texture2D> = None;
|
||||
device
|
||||
.CreateTexture2D(&p010_desc, None, Some(&mut p010))
|
||||
.context("CreateTexture2D(P010 bars dst)")?;
|
||||
let p010 = p010.context("null p010 tex")?;
|
||||
|
||||
let conv = HdrP010Converter::new(&device)?;
|
||||
conv.convert(&device, &context, &src_srv, &p010, w, h)?;
|
||||
Ok((device, p010))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn hdr_p010_selftest_at(w: u32, h: u32, vendor: Option<u32>) -> Result<()> {
|
||||
use windows::Win32::Graphics::Direct3D::{D3D_DRIVER_TYPE_HARDWARE, D3D_DRIVER_TYPE_UNKNOWN};
|
||||
use windows::Win32::Graphics::Dxgi::{CreateDXGIFactory1, IDXGIAdapter, IDXGIFactory1};
|
||||
|
||||
if w == 0 || h == 0 || w % 2 != 0 || h % 2 != 0 {
|
||||
bail!("hdr-p010-selftest needs even non-zero dimensions, got {w}x{h}");
|
||||
}
|
||||
// A grid of 16x16 flat scRGB blocks (each 2x2 chroma footprint uniform → exact chroma
|
||||
// comparison) covering pure R/G/B/white/black/gray at plausible HDR nit levels, plus a couple
|
||||
// of bright (>1.0 scRGB) colours, then the rest is a gradient (compared on Y only).
|
||||
#[allow(non_snake_case)]
|
||||
let (W, H) = (w, h);
|
||||
const BLK: u32 = 16;
|
||||
// (name, r, g, b) scRGB linear (1.0 = 80 nits). Mix of SDR-ish and HDR (>1.0) values.
|
||||
let named: [(&str, f32, f32, f32); 8] = [
|
||||
@@ -797,12 +941,36 @@ pub fn hdr_p010_selftest() -> Result<()> {
|
||||
// `fp16` outlives the synchronous `CreateTexture2D` that reads it. The mapped-pointer reads are
|
||||
// proven individually at the `read_u16` closure below.
|
||||
unsafe {
|
||||
// Hardware D3D11 device (no adapter pin — the default GPU is fine for the self-test).
|
||||
// Device on the requested vendor's adapter (dual-GPU boxes encode on a specific one), else
|
||||
// the default hardware GPU. Always says which adapter ran — a PASS is only meaningful for
|
||||
// the GPU it actually tested.
|
||||
let adapter: Option<IDXGIAdapter> = match vendor {
|
||||
None => None,
|
||||
Some(want) => {
|
||||
let factory: IDXGIFactory1 = CreateDXGIFactory1().context("dxgi factory")?;
|
||||
let mut found = None;
|
||||
for i in 0.. {
|
||||
let Ok(a) = factory.EnumAdapters(i) else {
|
||||
break;
|
||||
};
|
||||
let desc = a.GetDesc().context("adapter desc")?;
|
||||
if desc.VendorId == want {
|
||||
found = Some(a);
|
||||
break;
|
||||
}
|
||||
}
|
||||
Some(found.with_context(|| format!("no adapter with vendor id {want:#x}"))?)
|
||||
}
|
||||
};
|
||||
let mut device: Option<ID3D11Device> = None;
|
||||
let mut context: Option<ID3D11DeviceContext> = None;
|
||||
D3D11CreateDevice(
|
||||
None::<&IDXGIAdapter>,
|
||||
D3D_DRIVER_TYPE_HARDWARE,
|
||||
adapter.as_ref(),
|
||||
if adapter.is_some() {
|
||||
D3D_DRIVER_TYPE_UNKNOWN
|
||||
} else {
|
||||
D3D_DRIVER_TYPE_HARDWARE
|
||||
},
|
||||
HMODULE::default(),
|
||||
D3D11_CREATE_DEVICE_BGRA_SUPPORT,
|
||||
Some(&[D3D_FEATURE_LEVEL_11_0]),
|
||||
@@ -814,6 +982,22 @@ pub fn hdr_p010_selftest() -> Result<()> {
|
||||
.context("D3D11CreateDevice(hardware) for hdr-p010-selftest")?;
|
||||
let device = device.context("null device")?;
|
||||
let context = context.context("null context")?;
|
||||
{
|
||||
let dxgi: windows::Win32::Graphics::Dxgi::IDXGIDevice =
|
||||
device.cast().context("device -> IDXGIDevice")?;
|
||||
let desc = dxgi.GetAdapter().context("GetAdapter")?.GetDesc()?;
|
||||
let name = String::from_utf16_lossy(
|
||||
&desc.Description[..desc
|
||||
.Description
|
||||
.iter()
|
||||
.position(|&c| c == 0)
|
||||
.unwrap_or(desc.Description.len())],
|
||||
);
|
||||
println!(
|
||||
"adapter: {name} (vendor {:#06x}, luid {:08x}:{:08x})",
|
||||
desc.VendorId, desc.AdapterLuid.HighPart, desc.AdapterLuid.LowPart
|
||||
);
|
||||
}
|
||||
|
||||
// Source FP16 texture (initialized) + SRV.
|
||||
let src_desc = D3D11_TEXTURE2D_DESC {
|
||||
@@ -1175,3 +1359,16 @@ impl VideoConverter {
|
||||
blt.context("VideoProcessorBlt")
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod hdr_selftests {
|
||||
/// LIVE (needs the GPU): [`super::hdr_p010_selftest_at`] at the field capture size — 1080 is
|
||||
/// NOT 16-aligned, and the planar-RTV write path is driver-specific per vendor. Pinned to the
|
||||
/// Intel adapter (`0x8086`), so it runs on the Intel validation boxes and errors out cleanly
|
||||
/// ("no adapter") elsewhere. `cargo test -p pf-capture -- --ignored hdr_p010 --nocapture`.
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn hdr_p010_selftest_intel_1080_live() {
|
||||
super::hdr_p010_selftest_at(1920, 1080, Some(0x8086)).expect("hdr p010 selftest @1080");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1341,6 +1341,12 @@ impl IddPushCapturer {
|
||||
self.out_ring.clear(); // the output format changed → rebuild lazily at the new format
|
||||
self.video_conv = None; // converters are sized + HDR-specific → rebuild at the new mode
|
||||
self.hdr_p010_conv = None;
|
||||
// The PyroWave CSC is mode-baked too (BgraToYuvPlanes picks different SDR vs HDR shaders
|
||||
// and R8/R8G8 vs R16/R16G16 outputs). Without this, a display_hdr flip (Downgrade point D:
|
||||
// client_10bit=true but HDR couldn't enable at open) reused the stale SDR converter against
|
||||
// the freshly HDR-formatted pyro ring — every frame corrupted. `ensure_pyro_conv` only
|
||||
// builds when None, so it must be reset here like its siblings.
|
||||
self.pyro_conv = None;
|
||||
self.pyro_ring.clear(); // PyroWave two-plane ring is sized → rebuild at the new mode
|
||||
self.pyro_last = None;
|
||||
self.out_idx = 0;
|
||||
@@ -1861,6 +1867,7 @@ impl IddPushCapturer {
|
||||
cbcr,
|
||||
fence_handle,
|
||||
fence_value,
|
||||
ring_gen: self.generation,
|
||||
}),
|
||||
)
|
||||
} else {
|
||||
@@ -1919,6 +1926,7 @@ impl IddPushCapturer {
|
||||
cbcr: dst_cbcr,
|
||||
fence_handle,
|
||||
fence_value,
|
||||
ring_gen: self.generation,
|
||||
}),
|
||||
}),
|
||||
cursor: None,
|
||||
|
||||
@@ -876,10 +876,16 @@ impl Worker {
|
||||
);
|
||||
return;
|
||||
};
|
||||
let pref = self
|
||||
.pad_info(id)
|
||||
.map(|p| p.pref)
|
||||
.unwrap_or(GamepadPref::Xbox360);
|
||||
let pref = match self.pad_info(id) {
|
||||
// Steam Input's virtual pad standing in front of the Deck's built-in controls (the
|
||||
// only-pad-forwarded case, [`Self::forwarded_ids`]): declare the DECK kind, not the
|
||||
// wrapper's Xbox 360 identity. [`Self::auto_pref`] already resolves the SESSION
|
||||
// default this way, but a current host honors the per-pad arrival over the session
|
||||
// default — so without this the host builds an X-Box 360 pad on a real Deck.
|
||||
Some(p) if p.steam_virtual && is_steam_deck() => GamepadPref::SteamDeck,
|
||||
Some(p) => p.pref,
|
||||
None => GamepadPref::Xbox360,
|
||||
};
|
||||
match self.subsystem.open(sdl3::sys::joystick::SDL_JoystickID(id)) {
|
||||
Ok(pad) => {
|
||||
let mut slot = Slot::new(id, index, pref, pad);
|
||||
|
||||
@@ -447,8 +447,16 @@ fn pump(
|
||||
// every ~8–16 ms at 60–120 Hz anyway, so this rarely times out mid-stream).
|
||||
match connector.next_frame(Duration::from_millis(20)) {
|
||||
Ok(frame) => {
|
||||
// The `received` point: AU fully reassembled, in hand, before decode.
|
||||
let received_ns = now_ns();
|
||||
// The `received` point: reassembly COMPLETION, stamped by the core session as
|
||||
// the AU crossed poll_frame (ABI v9). Stamping here at the hand-off pull instead
|
||||
// would fold the pre-decode queue wait into `host+network` — a client-side
|
||||
// standing backlog masquerading as network latency (the 2026-07 two-pair
|
||||
// investigation). 0 = a core predating the stamp; fall back to the pull instant.
|
||||
let received_ns = if frame.received_ns > 0 {
|
||||
frame.received_ns
|
||||
} else {
|
||||
now_ns()
|
||||
};
|
||||
// fps / goodput count every received AU (spec), decoded or not.
|
||||
frames_n += 1;
|
||||
bytes_n += frame.data.len() as u64;
|
||||
|
||||
@@ -325,6 +325,11 @@ pub fn connect_reject_message(reason: punktfunk_core::reject::RejectReason) -> S
|
||||
"Client and host versions don't match — update both to the same release.".into()
|
||||
}
|
||||
R::Busy => "The host is busy with another session.".into(),
|
||||
R::SetupFailed => {
|
||||
"The host accepted the connection but couldn't start the stream — the host's log \
|
||||
(web console → Log) has the cause."
|
||||
.into()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -25,6 +25,11 @@ tracing = "0.1"
|
||||
# A test writer for the NVENC backend's unit tests (`with_test_writer().try_init()`).
|
||||
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
|
||||
|
||||
[target.'cfg(target_os = "windows")'.dev-dependencies]
|
||||
# The QSV live e2e drives the REAL HdrP010Converter output (an RTV-written, ring-profile P010
|
||||
# texture) into the encoder — the one seam the CPU-upload tests can't reach.
|
||||
pf-capture = { path = "../pf-capture" }
|
||||
|
||||
[target.'cfg(any(target_os = "linux", target_os = "windows"))'.dependencies]
|
||||
# Software H.264 (openh264, BSD-2) — the GPU-less encode path on both platforms.
|
||||
openh264 = "0.9"
|
||||
|
||||
@@ -32,6 +32,46 @@ pub struct EncodedFrame {
|
||||
pub chunk_aligned: bool,
|
||||
}
|
||||
|
||||
/// One slice-boundary chunk of an encoded AU, emitted by a chunked-poll backend
|
||||
/// ([`Encoder::poll_chunk`], latency plan §7 LN1): the encoder hands out completed slices while
|
||||
/// the rest of the frame is still encoding, so packetize/FEC/pacing can overlap the encode tail.
|
||||
/// The chunks of one AU concatenate to exactly the bytes [`Encoder::poll`] would have returned,
|
||||
/// and every cut lands on an Annex-B NAL boundary (slice starts). AU-level metadata
|
||||
/// (`pts_ns`/`keyframe`/`recovery_anchor`/`chunk_aligned`) is authoritative on the FIRST chunk
|
||||
/// (`first`) — the host opens the wire frame from it; `last` closes the AU. `keyframe` on a
|
||||
/// non-final chunk is the encoder's own prediction (exact under the P-only/infinite-GOP config —
|
||||
/// the driver only ever emits an IDR we asked for); the final chunk re-checks it against the
|
||||
/// driver's reported picture type.
|
||||
pub struct AuChunk {
|
||||
pub data: Vec<u8>,
|
||||
pub pts_ns: u64,
|
||||
pub keyframe: bool,
|
||||
/// See [`EncodedFrame::recovery_anchor`].
|
||||
pub recovery_anchor: bool,
|
||||
/// See [`EncodedFrame::chunk_aligned`].
|
||||
pub chunk_aligned: bool,
|
||||
/// Opens the AU (carries the authoritative AU metadata).
|
||||
pub first: bool,
|
||||
/// Closes the AU (the concatenation is complete; the encoder's in-flight slot is released).
|
||||
pub last: bool,
|
||||
}
|
||||
|
||||
impl AuChunk {
|
||||
/// A whole AU as a single self-closing chunk — what every non-chunked backend's
|
||||
/// [`Encoder::poll_chunk`] default emits, so a chunk consumer needs no per-backend fork.
|
||||
pub fn whole(f: EncodedFrame) -> Self {
|
||||
AuChunk {
|
||||
data: f.data,
|
||||
pts_ns: f.pts_ns,
|
||||
keyframe: f.keyframe,
|
||||
recovery_anchor: f.recovery_anchor,
|
||||
chunk_aligned: f.chunk_aligned,
|
||||
first: true,
|
||||
last: true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Codec selection negotiated with the client.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum Codec {
|
||||
@@ -219,6 +259,11 @@ pub struct EncoderCaps {
|
||||
|
||||
/// A hardware encoder. One per session; runs on the encode thread.
|
||||
pub trait Encoder: Send {
|
||||
/// Submit one captured frame for encoding. Lifetime contract: the caller must keep `frame`
|
||||
/// (and its GPU payload) alive until this frame's AU has been returned by
|
||||
/// [`poll`](Self::poll) — a stream-ordered backend (Linux direct-NVENC's IO-stream binding)
|
||||
/// may still be reading the payload asynchronously after `submit` returns. Both host encode
|
||||
/// loops already hold the frame across their poll drain; new callers must do the same.
|
||||
fn submit(&mut self, frame: &CapturedFrame) -> Result<()>;
|
||||
/// [`submit`](Self::submit) with the **wire frame index** this frame's AU will carry — the
|
||||
/// number the packetizer stamps on it and the client's loss reports/RFI requests name. The
|
||||
@@ -264,8 +309,37 @@ pub trait Encoder: Send {
|
||||
fn invalidate_ref_frames(&mut self, _first_frame: i64, _last_frame: i64) -> bool {
|
||||
false
|
||||
}
|
||||
/// Escalate into a pipelined (two-thread) retrieve mode under sustained GPU contention — the
|
||||
/// encoder analog of the capturer depth escalation: AUs ride ~one loop tick behind (`poll`
|
||||
/// may return `None` while an encode is in flight) in exchange for capture/submit no longer
|
||||
/// serializing on the encode wait. Returns whether pipelined retrieve is (now) active; the
|
||||
/// switch may be deferred to the next safe point internally. `false` from the default impl =
|
||||
/// unsupported — the session loop stops asking. De-escalation is a v2 item everywhere.
|
||||
fn set_pipelined(&mut self, _on: bool) -> bool {
|
||||
false
|
||||
}
|
||||
/// Pull the next encoded AU if one is ready.
|
||||
fn poll(&mut self) -> Result<Option<EncodedFrame>>;
|
||||
/// Whether [`poll_chunk`](Self::poll_chunk) currently emits sub-AU chunks — i.e. the LIVE
|
||||
/// session has slice-level readback armed (Linux direct-NVENC with the
|
||||
/// `PUNKTFUNK_NVENC_SLICES` and `PUNKTFUNK_NVENC_SUBFRAME` knobs on a sync depth-1
|
||||
/// retrieve). Dynamic, not static: a pipelined-retrieve escalation or a session rebuild can
|
||||
/// turn it off — re-query per AU, never cache across frames. `false` (the default) means
|
||||
/// `poll_chunk` degrades to one whole-AU chunk per frame.
|
||||
fn supports_chunked_poll(&self) -> bool {
|
||||
false
|
||||
}
|
||||
/// Pull the next slice-boundary chunk of the oldest in-flight AU (latency plan §7 LN1).
|
||||
/// Semantics when chunking is live: BLOCKS until the next chunk is readable, and the final
|
||||
/// (`last`) chunk blocks exactly like [`poll`](Self::poll) does — the depth-1 pump treats
|
||||
/// `None` as re-poll-next-tick, so a non-blocking tail would ride the AU one tick late (the
|
||||
/// `6dc195f9` Vulkan bug class). `Ok(None)` only when no AU is in flight. Each AU must be
|
||||
/// drained through ONE method: calling `poll` on a partially-chunked AU is a caller bug (the
|
||||
/// backend errors rather than double-emit bytes). Default: delegates to `poll`, wrapping the
|
||||
/// whole AU as a single `first && last` chunk.
|
||||
fn poll_chunk(&mut self) -> Result<Option<AuChunk>> {
|
||||
Ok(self.poll()?.map(AuChunk::whole))
|
||||
}
|
||||
/// Tear the underlying hardware encoder down and rebuild it in place, keeping the session's
|
||||
/// negotiated parameters — the encode-stall watchdog's recovery lever (a wedged AMF/QSV
|
||||
/// driver stops emitting AUs or accepting frames without ever returning an error). Returns
|
||||
@@ -293,6 +367,16 @@ pub trait Encoder: Send {
|
||||
/// flagged [`EncodedFrame::chunk_aligned`] and the session marks them on the wire.
|
||||
/// Default: no-op (the H.26x backends' bitstreams cannot be cut losslessly).
|
||||
fn set_wire_chunking(&mut self, _shard_payload: usize) {}
|
||||
/// How many frames the CAPTURER guarantees the encoder may hold in flight before it starts
|
||||
/// reusing an input texture (`Capturer::pipeline_depth`). Backends that encode the capturer's
|
||||
/// textures IN PLACE — no `CopyResource` — must not pipeline deeper than this: the capturer
|
||||
/// rotates its output ring per delivered frame with no regard for encode completion, so a
|
||||
/// deeper pipeline lets it overwrite a texture mid-encode. That is visual corruption (torn or
|
||||
/// mixed frames), not UB, so it fails silently and intermittently.
|
||||
///
|
||||
/// Called once by the session glue after the capturer is known; a backend that copies its
|
||||
/// input, or is synchronous, ignores it. Default: no-op.
|
||||
fn set_input_ring_depth(&mut self, _depth: usize) {}
|
||||
/// Signal end-of-stream. After this, drain the remaining AUs with [`poll`](Self::poll)
|
||||
/// until it returns `None` — NVENC buffers frames internally even at `delay=0`.
|
||||
fn flush(&mut self) -> Result<()>;
|
||||
@@ -412,6 +496,23 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// The whole-AU chunk (every non-chunked backend's `poll_chunk` shape) must carry the AU's
|
||||
/// metadata verbatim and be self-closing (`first && last`).
|
||||
#[test]
|
||||
fn whole_au_chunk_is_self_closing() {
|
||||
let c = AuChunk::whole(EncodedFrame {
|
||||
data: vec![0, 0, 0, 1, 0x40],
|
||||
pts_ns: 42,
|
||||
keyframe: true,
|
||||
recovery_anchor: true,
|
||||
chunk_aligned: false,
|
||||
});
|
||||
assert_eq!(c.data, vec![0, 0, 0, 1, 0x40]);
|
||||
assert_eq!(c.pts_ns, 42);
|
||||
assert!(c.keyframe && c.recovery_anchor && !c.chunk_aligned);
|
||||
assert!(c.first && c.last);
|
||||
}
|
||||
|
||||
/// Wire round-trip and the stats label stay in lockstep with the `quic::CODEC_*` bits.
|
||||
#[test]
|
||||
fn codec_wire_roundtrip_and_label() {
|
||||
|
||||
@@ -826,7 +826,7 @@ impl NvencEncoder {
|
||||
(*f).linesize[i] as usize,
|
||||
)
|
||||
});
|
||||
pf_zerocopy::cuda::copy_yuv444_to_device(buf, dsts)
|
||||
pf_zerocopy::cuda::copy_yuv444_to_device(buf, dsts, true)
|
||||
} else if self.want_444 {
|
||||
ffi::av_frame_free(&mut f);
|
||||
bail!(
|
||||
@@ -839,11 +839,11 @@ impl NvencEncoder {
|
||||
let y_pitch = (*f).linesize[0] as usize;
|
||||
let uv_ptr = (*f).data[1] as pf_zerocopy::cuda::CUdeviceptr;
|
||||
let uv_pitch = (*f).linesize[1] as usize;
|
||||
pf_zerocopy::cuda::copy_nv12_to_device(buf, y_ptr, y_pitch, uv_ptr, uv_pitch)
|
||||
pf_zerocopy::cuda::copy_nv12_to_device(buf, y_ptr, y_pitch, uv_ptr, uv_pitch, true)
|
||||
} else {
|
||||
let dst_ptr = (*f).data[0] as pf_zerocopy::cuda::CUdeviceptr;
|
||||
let dst_pitch = (*f).linesize[0] as usize;
|
||||
pf_zerocopy::cuda::copy_device_to_device(buf, dst_ptr, dst_pitch)
|
||||
pf_zerocopy::cuda::copy_device_to_device(buf, dst_ptr, dst_pitch, true)
|
||||
};
|
||||
if let Err(e) = copy_res {
|
||||
ffi::av_frame_free(&mut f);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -37,9 +37,11 @@ pub(crate) fn fourcc_to_vk(fourcc: u32) -> Option<vk::Format> {
|
||||
const AR24: u32 = 0x3432_5241; // ARGB8888
|
||||
const XB24: u32 = 0x3432_4258; // XBGR8888
|
||||
const AB24: u32 = 0x3432_4241; // ABGR8888
|
||||
const NV12: u32 = 0x3231_564e; // DRM_FORMAT_NV12
|
||||
match fourcc {
|
||||
XR24 | AR24 => Some(vk::Format::B8G8R8A8_UNORM),
|
||||
XB24 | AB24 => Some(vk::Format::R8G8B8A8_UNORM),
|
||||
NV12 => Some(vk::Format::G8_B8R8_2PLANE_420_UNORM),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -77,21 +79,62 @@ pub(crate) unsafe fn import_rgb_dmabuf(
|
||||
d: &pf_frame::DmabufFrame,
|
||||
cw: u32,
|
||||
ch: u32,
|
||||
) -> Result<(vk::Image, vk::DeviceMemory, vk::ImageView)> {
|
||||
import_rgb_dmabuf_as(
|
||||
device,
|
||||
ext_fd,
|
||||
mem_props,
|
||||
d,
|
||||
cw,
|
||||
ch,
|
||||
vk::ImageUsageFlags::SAMPLED,
|
||||
None,
|
||||
)
|
||||
}
|
||||
|
||||
/// [`import_rgb_dmabuf`] with the image usage explicit and an optional video-profile list.
|
||||
/// Despite the historical name, this also imports gamescope's one-fd LINEAR NV12: the UV
|
||||
/// subresource layout comes from the producer's plane-1 chunk when it reported one, falling
|
||||
/// back to the shared-stride contiguous-plane contract.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) unsafe fn import_rgb_dmabuf_as(
|
||||
device: &ash::Device,
|
||||
ext_fd: &ash::khr::external_memory_fd::Device,
|
||||
mem_props: &vk::PhysicalDeviceMemoryProperties,
|
||||
d: &pf_frame::DmabufFrame,
|
||||
cw: u32,
|
||||
ch: u32,
|
||||
usage: vk::ImageUsageFlags,
|
||||
profile_list: Option<&mut vk::VideoProfileListInfoKHR>,
|
||||
) -> Result<(vk::Image, vk::DeviceMemory, vk::ImageView)> {
|
||||
use anyhow::Context;
|
||||
use std::os::fd::IntoRawFd;
|
||||
let fmt = fourcc_to_vk(d.fourcc)
|
||||
.with_context(|| format!("unsupported dmabuf fourcc {:#x}", d.fourcc))?;
|
||||
let plane = [vk::SubresourceLayout::default()
|
||||
let planes: Vec<vk::SubresourceLayout> = if fmt == vk::Format::G8_B8R8_2PLANE_420_UNORM {
|
||||
let (uv_offset, uv_stride) = d.plane1.map(|(o, s)| (o as u64, s as u64)).unwrap_or((
|
||||
d.offset as u64 + d.stride as u64 * ch as u64,
|
||||
d.stride as u64,
|
||||
));
|
||||
vec![
|
||||
vk::SubresourceLayout::default()
|
||||
.offset(d.offset as u64)
|
||||
.row_pitch(d.stride as u64)];
|
||||
.row_pitch(d.stride as u64),
|
||||
vk::SubresourceLayout::default()
|
||||
.offset(uv_offset)
|
||||
.row_pitch(uv_stride),
|
||||
]
|
||||
} else {
|
||||
vec![vk::SubresourceLayout::default()
|
||||
.offset(d.offset as u64)
|
||||
.row_pitch(d.stride as u64)]
|
||||
};
|
||||
let mut drm = vk::ImageDrmFormatModifierExplicitCreateInfoEXT::default()
|
||||
.drm_format_modifier(d.modifier)
|
||||
.plane_layouts(&plane);
|
||||
.plane_layouts(&planes);
|
||||
let mut ext = vk::ExternalMemoryImageCreateInfo::default()
|
||||
.handle_types(vk::ExternalMemoryHandleTypeFlags::DMA_BUF_EXT);
|
||||
let img = device.create_image(
|
||||
&vk::ImageCreateInfo::default()
|
||||
let mut ci = vk::ImageCreateInfo::default()
|
||||
.image_type(vk::ImageType::TYPE_2D)
|
||||
.format(fmt)
|
||||
.extent(vk::Extent3D {
|
||||
@@ -103,13 +146,15 @@ pub(crate) unsafe fn import_rgb_dmabuf(
|
||||
.array_layers(1)
|
||||
.samples(vk::SampleCountFlags::TYPE_1)
|
||||
.tiling(vk::ImageTiling::DRM_FORMAT_MODIFIER_EXT)
|
||||
.usage(vk::ImageUsageFlags::SAMPLED)
|
||||
.usage(usage)
|
||||
.sharing_mode(vk::SharingMode::EXCLUSIVE)
|
||||
.initial_layout(vk::ImageLayout::UNDEFINED)
|
||||
.push_next(&mut ext)
|
||||
.push_next(&mut drm),
|
||||
None,
|
||||
)?;
|
||||
.push_next(&mut drm);
|
||||
if let Some(pl) = profile_list {
|
||||
ci = ci.push_next(pl);
|
||||
}
|
||||
let img = device.create_image(&ci, None)?;
|
||||
// dup the fd; Vulkan takes ownership of the dup on a successful import.
|
||||
let dup = d.fd.try_clone().context("dup dmabuf fd")?.into_raw_fd();
|
||||
let fd_props = {
|
||||
@@ -183,7 +228,8 @@ pub(crate) unsafe fn make_plain_image(
|
||||
None,
|
||||
)?;
|
||||
let req = device.get_image_memory_requirements(img);
|
||||
let mem = device.allocate_memory(
|
||||
// Unwind on failure: callers (the encoders' open paths) only ever see the completed triple.
|
||||
let mem = match device.allocate_memory(
|
||||
&vk::MemoryAllocateInfo::default()
|
||||
.allocation_size(req.size)
|
||||
.memory_type_index(find_mem(
|
||||
@@ -192,8 +238,24 @@ pub(crate) unsafe fn make_plain_image(
|
||||
vk::MemoryPropertyFlags::DEVICE_LOCAL,
|
||||
)),
|
||||
None,
|
||||
)?;
|
||||
device.bind_image_memory(img, mem, 0)?;
|
||||
let view = make_view(device, img, fmt, 0)?;
|
||||
Ok((img, mem, view))
|
||||
) {
|
||||
Ok(m) => m,
|
||||
Err(e) => {
|
||||
device.destroy_image(img, None);
|
||||
return Err(e.into());
|
||||
}
|
||||
};
|
||||
if let Err(e) = device.bind_image_memory(img, mem, 0) {
|
||||
device.destroy_image(img, None);
|
||||
device.free_memory(mem, None);
|
||||
return Err(e.into());
|
||||
}
|
||||
match make_view(device, img, fmt, 0) {
|
||||
Ok(view) => Ok((img, mem, view)),
|
||||
Err(e) => {
|
||||
device.destroy_image(img, None);
|
||||
device.free_memory(mem, None);
|
||||
Err(e)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
//! Vendored `VK_VALVE_video_encode_rgb_conversion` bindings — the RGB→YCbCr encode-source
|
||||
//! extension (Vulkan 1.4.327; RADV since Mesa 26.0, hardware-gated on the VCN EFC front-end
|
||||
//! conversion block). Our pinned `ash 0.38.0+1.3.281` predates it entirely; same vendoring
|
||||
//! rationale as [`vk_av1_encode`](super::vk_av1_encode) — definitions copied from the registry
|
||||
//! so the layouts are correct-by-construction, chained via raw `p_next`. Consumed by
|
||||
//! `vulkan_video.rs`: B0 probes + logs availability (design/vulkan-rgb-direct-encode.md);
|
||||
//! B1 makes the captured BGRx dmabuf the direct encode source with EFC doing the 709-narrow CSC.
|
||||
#![allow(dead_code)]
|
||||
|
||||
use ash::vk;
|
||||
use std::ffi::{c_void, CStr};
|
||||
|
||||
pub const EXTENSION_NAME: &CStr = c"VK_VALVE_video_encode_rgb_conversion";
|
||||
|
||||
// ---------- struct-type (VkStructureType) values — construct via `stype` ----------
|
||||
pub const ST_PHYSICAL_DEVICE_FEATURES: i32 = 1_000_390_000;
|
||||
pub const ST_CAPABILITIES: i32 = 1_000_390_001;
|
||||
pub const ST_PROFILE_INFO: i32 = 1_000_390_002;
|
||||
pub const ST_SESSION_CREATE_INFO: i32 = 1_000_390_003;
|
||||
|
||||
// `VkVideoEncodeRgbModelConversionFlagBitsVALVE`
|
||||
pub const MODEL_RGB_IDENTITY: u32 = 0x01;
|
||||
pub const MODEL_YCBCR_IDENTITY: u32 = 0x02;
|
||||
pub const MODEL_YCBCR_709: u32 = 0x04;
|
||||
pub const MODEL_YCBCR_601: u32 = 0x08;
|
||||
pub const MODEL_YCBCR_2020: u32 = 0x10;
|
||||
// `VkVideoEncodeRgbRangeCompressionFlagBitsVALVE`
|
||||
pub const RANGE_FULL: u32 = 0x01;
|
||||
pub const RANGE_NARROW: u32 = 0x02;
|
||||
// `VkVideoEncodeRgbChromaOffsetFlagBitsVALVE`
|
||||
pub const CHROMA_OFFSET_COSITED_EVEN: u32 = 0x01;
|
||||
pub const CHROMA_OFFSET_MIDPOINT: u32 = 0x02;
|
||||
|
||||
/// `VkPhysicalDeviceVideoEncodeRgbConversionFeaturesVALVE` — chain into
|
||||
/// `VkPhysicalDeviceFeatures2` (query) / `VkDeviceCreateInfo` (enable).
|
||||
#[repr(C)]
|
||||
pub struct PhysicalDeviceVideoEncodeRgbConversionFeaturesVALVE {
|
||||
pub s_type: vk::StructureType,
|
||||
pub p_next: *mut c_void,
|
||||
pub video_encode_rgb_conversion: vk::Bool32,
|
||||
}
|
||||
|
||||
/// `VkVideoEncodeRgbConversionCapabilitiesVALVE` — chain into the
|
||||
/// `vkGetPhysicalDeviceVideoCapabilitiesKHR` output when the queried profile carries
|
||||
/// [`VideoEncodeProfileRgbConversionInfoVALVE`]; reports which conversions the HW does.
|
||||
#[repr(C)]
|
||||
pub struct VideoEncodeRgbConversionCapabilitiesVALVE {
|
||||
pub s_type: vk::StructureType,
|
||||
pub p_next: *mut c_void,
|
||||
pub rgb_models: u32,
|
||||
pub rgb_ranges: u32,
|
||||
pub x_chroma_offsets: u32,
|
||||
pub y_chroma_offsets: u32,
|
||||
}
|
||||
|
||||
/// `VkVideoEncodeProfileRgbConversionInfoVALVE` — part of the video-profile *identity*: every
|
||||
/// consumer of the profile (caps query, format query, session, image profile lists) must carry
|
||||
/// the same chain.
|
||||
#[repr(C)]
|
||||
pub struct VideoEncodeProfileRgbConversionInfoVALVE {
|
||||
pub s_type: vk::StructureType,
|
||||
pub p_next: *const c_void,
|
||||
pub perform_encode_rgb_conversion: vk::Bool32,
|
||||
}
|
||||
|
||||
/// `VkVideoEncodeSessionRgbConversionCreateInfoVALVE` — chain into
|
||||
/// `VkVideoSessionCreateInfoKHR`; single-bit selections of the conversion actually performed.
|
||||
#[repr(C)]
|
||||
pub struct VideoEncodeSessionRgbConversionCreateInfoVALVE {
|
||||
pub s_type: vk::StructureType,
|
||||
pub p_next: *const c_void,
|
||||
pub rgb_model: u32,
|
||||
pub rgb_range: u32,
|
||||
pub x_chroma_offset: u32,
|
||||
pub y_chroma_offset: u32,
|
||||
}
|
||||
|
||||
/// `vk::StructureType` for a raw `ST_*` constant above.
|
||||
#[inline]
|
||||
pub fn stype(raw: i32) -> vk::StructureType {
|
||||
vk::StructureType::from_raw(raw)
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -35,6 +35,38 @@ pub(super) fn codec_guid(codec: Codec) -> nv::GUID {
|
||||
}
|
||||
}
|
||||
|
||||
/// Resolved per-frame slice count for a session (latency plan §7 LN1, Phase 3): the
|
||||
/// `PUNKTFUNK_NVENC_SLICES` env override wins (1..=32; **1 = the explicit single-slice
|
||||
/// escape**, needed now that a backend can default higher), else the backend's
|
||||
/// `default_slices` — 4 on Linux direct-NVENC since the Phase-3 default-on, 1 everywhere else
|
||||
/// (the Windows async path is deliberately untouched). H.264/HEVC only (AV1 partitions via
|
||||
/// tiles). ONE parse shared by the config author ([`apply_low_latency_config`] via
|
||||
/// [`LowLatencyConfig::slices`]) and the Linux backend's chunked-poll arming, so the two can
|
||||
/// never disagree about whether a session is multi-slice.
|
||||
pub(super) fn resolve_slices(codec: Codec, default_slices: u32) -> u32 {
|
||||
if !matches!(codec, Codec::H264 | Codec::H265) {
|
||||
return 1;
|
||||
}
|
||||
std::env::var("PUNKTFUNK_NVENC_SLICES")
|
||||
.ok()
|
||||
.and_then(|s| s.parse::<u32>().ok())
|
||||
.filter(|n| (1..=32).contains(n))
|
||||
.unwrap_or(default_slices)
|
||||
}
|
||||
|
||||
/// Resolved sub-frame readback (`enableSubFrameWrite` + `reportSliceOffsets`; sync sessions
|
||||
/// only, see [`build_init_params`]): `PUNKTFUNK_NVENC_SUBFRAME` tri-state — `0` = never (the
|
||||
/// default-on escape), `1` = force (even where the caps probe says unsupported — an operator
|
||||
/// explicitly testing), unset = the backend's `default_on` (Linux direct-NVENC passes its
|
||||
/// SUBFRAME_READBACK caps-probe result since Phase 3; Windows passes `false`).
|
||||
pub(super) fn resolve_subframe(default_on: bool) -> bool {
|
||||
match std::env::var("PUNKTFUNK_NVENC_SUBFRAME").as_deref() {
|
||||
Ok("0") => false,
|
||||
Ok("1") => true,
|
||||
_ => default_on,
|
||||
}
|
||||
}
|
||||
|
||||
/// Reference-frame DPB depth when RFI is supported (Apollo uses 5). A deeper DPB lets an invalidated
|
||||
/// reference fall back to an older still-valid frame instead of a full IDR; `numRefL0 = 1` keeps each
|
||||
/// P-frame single-reference for low latency. Also the window the backends' `invalidate_ref_frames`
|
||||
@@ -66,6 +98,9 @@ pub(super) struct LowLatencyConfig {
|
||||
pub hdr: bool,
|
||||
/// This GPU supports reference-frame invalidation (a deeper DPB for graceful loss recovery).
|
||||
pub rfi_supported: bool,
|
||||
/// Resolved per-frame slice count ([`resolve_slices`] — env override, else the backend
|
||||
/// default). ≤ 1 leaves the preset's single slice untouched.
|
||||
pub slices: u32,
|
||||
}
|
||||
|
||||
/// Author the shared `NV_ENC_INITIALIZE_PARAMS` (P1/ULL preset, PTD, the session dimensions/rate)
|
||||
@@ -82,6 +117,7 @@ pub(super) fn build_init_params(
|
||||
cfg: &mut nv::NV_ENC_CONFIG,
|
||||
split_mode: u32,
|
||||
enable_async: bool,
|
||||
subframe: bool,
|
||||
) -> nv::NV_ENC_INITIALIZE_PARAMS {
|
||||
let mut init = nv::NV_ENC_INITIALIZE_PARAMS {
|
||||
version: nv::NV_ENC_INITIALIZE_PARAMS_VER,
|
||||
@@ -101,6 +137,16 @@ pub(super) fn build_init_params(
|
||||
};
|
||||
// splitEncodeMode is a C bitfield — set via the generated accessor, not a struct field.
|
||||
init.set_splitEncodeMode(split_mode);
|
||||
// Sub-frame readback (latency plan §7 LN1; default-on for Linux direct-NVENC since Phase 3 —
|
||||
// the caller resolves `subframe` via [`resolve_subframe`] + its caps probe): the driver
|
||||
// writes each slice into the output buffer as it completes and reports per-slice offsets, so
|
||||
// a sync-mode consumer can read slices out while the frame is still encoding. Pair with
|
||||
// multi-slice (a single-slice frame yields nothing to read early). `reportSliceOffsets`
|
||||
// requires `enableEncodeAsync = 0`, so async (Windows) sessions never arm.
|
||||
if !enable_async && subframe {
|
||||
init.set_enableSubFrameWrite(1);
|
||||
init.set_reportSliceOffsets(1);
|
||||
}
|
||||
init
|
||||
}
|
||||
|
||||
@@ -118,6 +164,9 @@ pub(super) unsafe fn apply_low_latency_config(cfg: &mut nv::NV_ENC_CONFIG, c: Lo
|
||||
cfg.gopLength = nv::NVENC_INFINITE_GOPLENGTH;
|
||||
cfg.frameIntervalP = 1;
|
||||
cfg.rcParams.rateControlMode = nv::NV_ENC_PARAMS_RC_MODE::NV_ENC_PARAMS_RC_CBR;
|
||||
// Explicit zero reorder delay: with P-only + no lookahead there is no reordering to buffer,
|
||||
// but pin the bit so no preset/driver default can ever slip a frame of reorder delay in.
|
||||
cfg.rcParams.set_zeroReorderDelay(1);
|
||||
let bps = c.bitrate.min(u32::MAX as u64) as u32;
|
||||
cfg.rcParams.averageBitRate = bps;
|
||||
cfg.rcParams.maxBitRate = bps;
|
||||
@@ -146,6 +195,26 @@ pub(super) unsafe fn apply_low_latency_config(cfg: &mut nv::NV_ENC_CONFIG, c: Lo
|
||||
Codec::PyroWave => unreachable!("PyroWave never opens the direct-NVENC backend"),
|
||||
}
|
||||
|
||||
// Multi-slice frames (latency plan §7 LN1): `c.slices` splits every frame into N slices
|
||||
// (sliceMode 3 = "N slices per frame"), the unit sub-frame readback ships early and loss
|
||||
// concealment can discard independently. Costs ~1-2 % bitrate in slice headers. H.264/HEVC
|
||||
// only — AV1 partitions via tiles, not slices (the resolver already returns 1 there).
|
||||
// Default 4 on Linux direct-NVENC (Phase 3), env-only elsewhere; ≤ 1 keeps the preset's
|
||||
// single slice.
|
||||
if let Some(n) = Some(c.slices).filter(|n| *n >= 2) {
|
||||
match c.codec {
|
||||
Codec::H264 => {
|
||||
cfg.encodeCodecConfig.h264Config.sliceMode = 3;
|
||||
cfg.encodeCodecConfig.h264Config.sliceModeData = n;
|
||||
}
|
||||
Codec::H265 => {
|
||||
cfg.encodeCodecConfig.hevcConfig.sliceMode = 3;
|
||||
cfg.encodeCodecConfig.hevcConfig.sliceModeData = n;
|
||||
}
|
||||
Codec::Av1 | Codec::PyroWave => {}
|
||||
}
|
||||
}
|
||||
|
||||
// Chroma + bit depth. Full-chroma 4:4:4 (HEVC Range Extensions, chromaFormatIDC=3 under the FREXT
|
||||
// profile) takes precedence and composes with 10-bit (Main 4:4:4 10); it needs a full-chroma-
|
||||
// capable input. Otherwise 10-bit selects Main10 (HEVC) or the AV1 output depth — stamping the
|
||||
|
||||
@@ -37,7 +37,8 @@
|
||||
#![deny(clippy::undocumented_unsafe_blocks)]
|
||||
|
||||
use super::nvenc_core::{
|
||||
apply_low_latency_config, build_init_params, codec_guid, LowLatencyConfig, NvStatusExt, RFI_DPB,
|
||||
apply_low_latency_config, build_init_params, codec_guid, resolve_slices, resolve_subframe,
|
||||
LowLatencyConfig, NvStatusExt, RFI_DPB,
|
||||
};
|
||||
use super::nvenc_status;
|
||||
use super::{ChromaFormat, Codec, EncodedFrame, Encoder, EncoderCaps};
|
||||
@@ -409,6 +410,12 @@ pub struct NvencD3d11Encoder {
|
||||
events: Vec<usize>,
|
||||
/// Async mode: the retrieve thread + its channels (`None` = classic same-thread sync retrieve).
|
||||
async_rt: Option<AsyncRetrieve>,
|
||||
/// The capturer's `pipeline_depth` (`set_input_ring_depth`). This backend encodes the
|
||||
/// capturer's textures IN PLACE, so it is a HARD ceiling on async in-flight depth: the
|
||||
/// capturer rotates its ring per delivered frame regardless of encode completion, so
|
||||
/// pipelining deeper lets it overwrite a texture mid-encode (torn frames). `None` until the
|
||||
/// session glue reports it — treated as "unknown, don't pipeline past the env cap".
|
||||
input_ring_depth: Option<usize>,
|
||||
/// `NV_ENC_CAPS_ASYNC_ENCODE_SUPPORT` from the caps probe — gates the async retrieve mode.
|
||||
async_supported: bool,
|
||||
/// (bitstream, mapped input resource to unmap after retrieval, pts_ns, recovery-anchor) per
|
||||
@@ -505,6 +512,7 @@ impl NvencD3d11Encoder {
|
||||
bitstreams: Vec::new(),
|
||||
events: Vec::new(),
|
||||
async_rt: None,
|
||||
input_ring_depth: None,
|
||||
async_supported: false,
|
||||
pending: VecDeque::new(),
|
||||
frame_idx: 0,
|
||||
@@ -737,6 +745,10 @@ impl NvencD3d11Encoder {
|
||||
av1_input_depth_minus8: if ten_bit_in { 2 } else { 0 },
|
||||
hdr: self.hdr,
|
||||
rfi_supported: self.rfi_supported,
|
||||
// Env-only on Windows (default single slice): the Phase-3 default-on is a
|
||||
// Linux-direct-NVENC decision — the Windows async path stays untouched, and a
|
||||
// Windows operator opting in must choose slices+sync over async retrieve.
|
||||
slices: resolve_slices(self.codec, 1),
|
||||
},
|
||||
);
|
||||
Ok(cfg)
|
||||
@@ -784,6 +796,9 @@ impl NvencD3d11Encoder {
|
||||
&mut cfg,
|
||||
split_mode,
|
||||
enable_async,
|
||||
// Windows: env opt-in only ("1"), never a default — and build_init_params
|
||||
// additionally refuses it on an async session.
|
||||
resolve_subframe(false),
|
||||
);
|
||||
|
||||
match (api().initialize_encoder)(enc, &mut init).nv_ok() {
|
||||
@@ -1156,11 +1171,21 @@ impl Encoder for NvencD3d11Encoder {
|
||||
// index, which is non-zero on a mid-session encoder rebuild's first frame.
|
||||
let opening = self.next == 0;
|
||||
// Async backpressure: never hand NVENC an output bitstream that is still in flight, and
|
||||
// keep in-flight depth within the capturer's texture ring (see `async_inflight_cap`). At
|
||||
// the cap, block on the OLDEST completion (the retrieve thread is already waiting on its
|
||||
// event) before submitting more — bounding depth exactly like the sync path's per-tick
|
||||
// blocking poll, just `cap` deep instead of 1.
|
||||
while self.async_rt.is_some() && self.pending.len() >= async_inflight_cap() {
|
||||
// keep in-flight depth within the capturer's texture ring. At the cap, block on the OLDEST
|
||||
// completion (the retrieve thread is already waiting on its event) before submitting more —
|
||||
// bounding depth exactly like the sync path's per-tick blocking poll, just `cap` deep
|
||||
// instead of 1.
|
||||
//
|
||||
// The ring term is the one that matters for correctness: `async_inflight_cap()` is only the
|
||||
// output-bitstream-pool ceiling plus an env knob, and consults NOTHING about the capturer,
|
||||
// despite this comment previously claiming otherwise. Since this backend encodes the
|
||||
// capturer's textures in place, exceeding the capturer's declared `pipeline_depth` lets it
|
||||
// rotate a texture out from under a live encode — torn frames, silently.
|
||||
let cap = match self.input_ring_depth {
|
||||
Some(d) => async_inflight_cap().min(d.max(1)),
|
||||
None => async_inflight_cap(),
|
||||
};
|
||||
while self.async_rt.is_some() && self.pending.len() >= cap {
|
||||
let done = {
|
||||
let rt = self.async_rt.as_mut().expect("checked in loop condition");
|
||||
rt.done_rx
|
||||
@@ -1336,6 +1361,17 @@ impl Encoder for NvencD3d11Encoder {
|
||||
self.submit(frame)
|
||||
}
|
||||
|
||||
fn set_input_ring_depth(&mut self, depth: usize) {
|
||||
// This backend registers and encodes the capturer's textures in place (no CopyResource),
|
||||
// so the capturer's ring depth is a hard ceiling on how deep async may pipeline.
|
||||
self.input_ring_depth = Some(depth);
|
||||
tracing::debug!(
|
||||
depth,
|
||||
env_cap = async_inflight_cap(),
|
||||
"NVENC: capturer input-ring depth reported — async in-flight bounded by the smaller"
|
||||
);
|
||||
}
|
||||
|
||||
fn request_keyframe(&mut self) {
|
||||
self.force_kf = true;
|
||||
}
|
||||
@@ -1540,6 +1576,7 @@ impl Encoder for NvencD3d11Encoder {
|
||||
&mut cfg,
|
||||
self.split_mode,
|
||||
self.session_async,
|
||||
resolve_subframe(false),
|
||||
),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
@@ -139,6 +139,10 @@ pub struct PyroWaveEncoder {
|
||||
// Imported plane textures, cached by the out-ring texture's raw pointer (stable per ring slot):
|
||||
// the full-res R8 Y plane and the half-res R8G8 CbCr plane, imported SEPARATELY (a single planar
|
||||
// NV12 import is unreliable on NVIDIA at arbitrary sizes).
|
||||
/// The capturer ring generation the cached plane imports below belong to. A recreate bumps it,
|
||||
/// and every cached import is destroyed — the COM addresses they are keyed on can be recycled
|
||||
/// by the allocator after a recreate, so identity cannot rest on the pointer alone.
|
||||
ring_gen: Option<u32>,
|
||||
y_images: Vec<(PlaneKey, pw::pyrowave_image)>,
|
||||
cbcr_images: Vec<(PlaneKey, pw::pyrowave_image)>,
|
||||
|
||||
@@ -271,6 +275,7 @@ impl PyroWaveEncoder {
|
||||
pw_dev,
|
||||
pw_enc,
|
||||
sync: std::ptr::null_mut(),
|
||||
ring_gen: None,
|
||||
y_images: Vec::new(),
|
||||
cbcr_images: Vec::new(),
|
||||
width,
|
||||
@@ -455,6 +460,25 @@ impl PyroWaveEncoder {
|
||||
in pyrowave mode (session_plan::output_format must set OutputFormat::pyrowave)",
|
||||
)?;
|
||||
|
||||
// Ring recreate ⇒ every cached plane import belongs to textures that no longer exist. Their
|
||||
// COM addresses can be handed back out by the allocator, so a pointer-keyed hit could return
|
||||
// an image bound to freed memory. Flush on the generation change rather than relying on the
|
||||
// address (or the FIFO cap) to notice.
|
||||
if self.ring_gen != Some(share.ring_gen) {
|
||||
if self.ring_gen.is_some() {
|
||||
tracing::info!(
|
||||
from = ?self.ring_gen,
|
||||
to = share.ring_gen,
|
||||
cached = self.y_images.len() + self.cbcr_images.len(),
|
||||
"pyrowave: capturer recreated its ring — flushing stale plane imports"
|
||||
);
|
||||
}
|
||||
for (_, img) in self.y_images.drain(..).chain(self.cbcr_images.drain(..)) {
|
||||
pw::pyrowave_image_destroy(img);
|
||||
}
|
||||
self.ring_gen = Some(share.ring_gen);
|
||||
}
|
||||
|
||||
// Import the fence whenever this encoder has no timeline yet — the first frame, OR a fresh
|
||||
// encoder after a client mode-switch rebuild (the capturer passes the persistent handle on
|
||||
// every frame precisely so a rebuilt encoder can re-import it).
|
||||
@@ -1000,6 +1024,9 @@ mod tests {
|
||||
cbcr: cbcr_tex,
|
||||
fence_handle: Some(fence_handle.0 as isize),
|
||||
fence_value: 1,
|
||||
// One synthetic ring for the whole case: a constant generation exercises the
|
||||
// steady-state cache-hit path (a changing one would flush every frame).
|
||||
ring_gen: 1,
|
||||
}),
|
||||
}),
|
||||
cursor: None,
|
||||
|
||||
@@ -478,10 +478,14 @@ fn build_params(cfg: &EncodeConfig) -> ParamSet {
|
||||
b
|
||||
});
|
||||
|
||||
// HDR signalling (10-bit sessions are the HDR path on Windows — same coupling as NVENC):
|
||||
// BT.2020/PQ colour description + the source's mastering/CLL grade at every IDR.
|
||||
// Colour signalling, written UNCONDITIONALLY (mirrors nvenc_core.rs): the input is already
|
||||
// CSC'd to a specific matrix — BT.709 limited for SDR (the capture-side VideoConverter),
|
||||
// BT.2020 PQ for HDR (HdrP010Converter) — so the stream must say so. An SDR stream without a
|
||||
// colour description leaves the choice to the decoder's "unspecified" default, and
|
||||
// Moonlight/third-party/Android-vendor decoders default to 601 at sub-HD → mis-rendered
|
||||
// colours. (10-bit sessions are the HDR path on Windows — same coupling as NVENC.)
|
||||
let hdr = cfg.ten_bit && cfg.codec != Codec::H264;
|
||||
let vsi = hdr.then(|| {
|
||||
let vsi = {
|
||||
// SAFETY: all-zero is valid; header stamped below.
|
||||
let mut b: Box<vpl::mfxExtVideoSignalInfo> = Box::new(unsafe { std::mem::zeroed() });
|
||||
b.Header.BufferId = vpl::MFX_EXTBUFF_VIDEO_SIGNAL_INFO as u32;
|
||||
@@ -489,11 +493,17 @@ fn build_params(cfg: &EncodeConfig) -> ParamSet {
|
||||
b.VideoFormat = 5; // unspecified
|
||||
b.VideoFullRange = 0;
|
||||
b.ColourDescriptionPresent = 1;
|
||||
if hdr {
|
||||
b.ColourPrimaries = 9; // BT.2020
|
||||
b.TransferCharacteristics = 16; // SMPTE ST 2084 (PQ)
|
||||
b.MatrixCoefficients = 9; // BT.2020 non-constant
|
||||
b
|
||||
});
|
||||
} else {
|
||||
b.ColourPrimaries = 1; // BT.709
|
||||
b.TransferCharacteristics = 1; // BT.709
|
||||
b.MatrixCoefficients = 1; // BT.709
|
||||
}
|
||||
Some(b)
|
||||
};
|
||||
let mastering = cfg.hdr_meta.filter(|_| hdr).map(|m| {
|
||||
// SAFETY: all-zero is valid; header stamped below.
|
||||
let mut b: Box<vpl::mfxExtMasteringDisplayColourVolume> =
|
||||
@@ -1994,4 +2004,246 @@ mod tests {
|
||||
"the bitrate retarget emitted a keyframe (StartNewSequence leak)"
|
||||
);
|
||||
}
|
||||
|
||||
/// FULL-CHAIN colour check at the field capture size: a known P010 colour-bar source at
|
||||
/// 1920x1080 — whose height is NOT 16-aligned, so the ingest `CopySubresourceRegion` copies
|
||||
/// into a 1920x1088 runtime pool surface whose chroma plane sits at a DIFFERENT row offset
|
||||
/// than the source's (the seam no 640x480 test exercises) — encoded to Main10 HEVC and
|
||||
/// dumped to `%TEMP%\pf_qsv_1080_bars.h265` for off-box decode verification against the
|
||||
/// same codes. On-box this asserts stream shape; the pixel verdict needs a decoder.
|
||||
#[test]
|
||||
fn qsv_live_p010_1080_colorbars_dump() {
|
||||
use windows::Win32::Graphics::Direct3D::D3D_DRIVER_TYPE_UNKNOWN;
|
||||
use windows::Win32::Graphics::Direct3D11::{
|
||||
D3D11CreateDevice, D3D11_BIND_RENDER_TARGET, D3D11_BIND_SHADER_RESOURCE,
|
||||
D3D11_SDK_VERSION, D3D11_SUBRESOURCE_DATA, D3D11_USAGE_DEFAULT,
|
||||
};
|
||||
use windows::Win32::Graphics::Dxgi::Common::{DXGI_FORMAT_P010, DXGI_SAMPLE_DESC};
|
||||
use windows::Win32::Graphics::Dxgi::{CreateDXGIFactory1, IDXGIAdapter1, IDXGIFactory4};
|
||||
|
||||
// (Y, Cb, Cr) 10-bit limited codes for the 8 sRGB bars white/yellow/cyan/green/magenta/
|
||||
// red/blue/black at 80-nit SDR white under PQ/BT.2020 — the same math as pf-capture's
|
||||
// `p010_reference` (and the bars_pq2020 client fixture). Stored MSB-aligned (`<<6`).
|
||||
const BARS: [(u16, u16, u16); 8] = [
|
||||
(490, 512, 512),
|
||||
(478, 423, 518),
|
||||
(464, 525, 473),
|
||||
(450, 432, 476),
|
||||
(350, 584, 585),
|
||||
(325, 448, 598),
|
||||
(226, 650, 535),
|
||||
(64, 512, 512),
|
||||
];
|
||||
const W: u32 = 1920;
|
||||
const H: u32 = 1080;
|
||||
|
||||
init_tracing();
|
||||
let Ok((_l, impls)) = intel_loader() else {
|
||||
eprintln!("skipping: no VPL loader");
|
||||
return;
|
||||
};
|
||||
let Some(imp) = impls.iter().find(|i| i.luid_valid) else {
|
||||
eprintln!("skipping: no Intel VPL implementation on this box");
|
||||
return;
|
||||
};
|
||||
if !probe_can_encode_10bit(Codec::H265) {
|
||||
eprintln!("skipping: this GPU declines 10-bit HEVC");
|
||||
return;
|
||||
}
|
||||
|
||||
// P010 initial data: plane 0 = H rows of W u16 luma; plane 1 = H/2 rows of W u16
|
||||
// (interleaved Cb,Cr pairs), same pitch. Bars are vertical: bar index = x / (W/8).
|
||||
let bar_w = (W / 8) as usize;
|
||||
let mut init = vec![0u16; (W as usize) * (H as usize + H as usize / 2)];
|
||||
for y in 0..H as usize {
|
||||
for x in 0..W as usize {
|
||||
init[y * W as usize + x] = BARS[(x / bar_w).min(7)].0 << 6;
|
||||
}
|
||||
}
|
||||
let chroma_base = (W as usize) * (H as usize);
|
||||
for cy in 0..(H as usize / 2) {
|
||||
for cx in 0..(W as usize / 2) {
|
||||
let (_, cb, cr) = BARS[((cx * 2) / bar_w).min(7)];
|
||||
init[chroma_base + cy * W as usize + cx * 2] = cb << 6;
|
||||
init[chroma_base + cy * W as usize + cx * 2 + 1] = cr << 6;
|
||||
}
|
||||
}
|
||||
|
||||
// SAFETY: self-contained harness on one thread/device (same contract as `drive_live`);
|
||||
// the initial-data pointer outlives the synchronous CreateTexture2D that reads it.
|
||||
let (device, tex) = unsafe {
|
||||
let luid = windows::Win32::Foundation::LUID {
|
||||
LowPart: u32::from_le_bytes(imp.luid[..4].try_into().unwrap()),
|
||||
HighPart: i32::from_le_bytes(imp.luid[4..].try_into().unwrap()),
|
||||
};
|
||||
let factory: IDXGIFactory4 = CreateDXGIFactory1().expect("dxgi factory");
|
||||
let adapter: IDXGIAdapter1 = factory.EnumAdapterByLuid(luid).expect("intel adapter");
|
||||
let mut device = None;
|
||||
D3D11CreateDevice(
|
||||
&adapter,
|
||||
D3D_DRIVER_TYPE_UNKNOWN,
|
||||
windows::Win32::Foundation::HMODULE::default(),
|
||||
Default::default(),
|
||||
None,
|
||||
D3D11_SDK_VERSION,
|
||||
Some(&mut device),
|
||||
None,
|
||||
None,
|
||||
)
|
||||
.expect("d3d11 device on intel adapter");
|
||||
let device: ID3D11Device = device.expect("device");
|
||||
let desc = D3D11_TEXTURE2D_DESC {
|
||||
Width: W,
|
||||
Height: H,
|
||||
MipLevels: 1,
|
||||
ArraySize: 1,
|
||||
Format: DXGI_FORMAT_P010,
|
||||
SampleDesc: DXGI_SAMPLE_DESC {
|
||||
Count: 1,
|
||||
Quality: 0,
|
||||
},
|
||||
Usage: D3D11_USAGE_DEFAULT,
|
||||
BindFlags: (D3D11_BIND_RENDER_TARGET.0 | D3D11_BIND_SHADER_RESOURCE.0) as u32,
|
||||
CPUAccessFlags: 0,
|
||||
MiscFlags: 0,
|
||||
};
|
||||
let data = D3D11_SUBRESOURCE_DATA {
|
||||
pSysMem: init.as_ptr() as *const std::ffi::c_void,
|
||||
SysMemPitch: W * 2,
|
||||
SysMemSlicePitch: 0,
|
||||
};
|
||||
let mut t: Option<ID3D11Texture2D> = None;
|
||||
device
|
||||
.CreateTexture2D(&desc, Some(&data), Some(&mut t))
|
||||
.expect("bar texture");
|
||||
(device.clone(), t.expect("texture"))
|
||||
};
|
||||
|
||||
let mut enc = QsvEncoder::open(
|
||||
Codec::H265,
|
||||
PixelFormat::P010,
|
||||
W,
|
||||
H,
|
||||
30,
|
||||
10_000_000,
|
||||
10,
|
||||
ChromaFormat::Yuv420,
|
||||
)
|
||||
.expect("open");
|
||||
enc.set_hdr_meta(Some(test_hdr_meta()));
|
||||
let mut stream = Vec::new();
|
||||
let mut aus = 0usize;
|
||||
let mut keyframes = 0usize;
|
||||
for i in 0..12u32 {
|
||||
let frame = CapturedFrame {
|
||||
width: W,
|
||||
height: H,
|
||||
pts_ns: i as u64 * 33_333_333,
|
||||
format: PixelFormat::P010,
|
||||
payload: FramePayload::D3d11(pf_frame::dxgi::D3d11Frame {
|
||||
texture: tex.clone(),
|
||||
device: device.clone(),
|
||||
pyro: None,
|
||||
}),
|
||||
cursor: None,
|
||||
};
|
||||
enc.submit_indexed(&frame, i).expect("submit");
|
||||
if let Some(au) = enc.poll().expect("poll") {
|
||||
aus += 1;
|
||||
keyframes += au.keyframe as usize;
|
||||
stream.extend_from_slice(&au.data);
|
||||
}
|
||||
}
|
||||
enc.flush().expect("flush");
|
||||
while let Some(au) = enc.poll().expect("drain") {
|
||||
aus += 1;
|
||||
keyframes += au.keyframe as usize;
|
||||
stream.extend_from_slice(&au.data);
|
||||
}
|
||||
assert!(aus >= 10, "expected ≥10 AUs, got {aus}");
|
||||
assert!(keyframes >= 1, "expected an IDR in the dump");
|
||||
let path = std::env::temp_dir().join("pf_qsv_1080_bars.h265");
|
||||
std::fs::write(&path, &stream).expect("write dump");
|
||||
println!(
|
||||
"wrote {} AUs ({} bytes, {keyframes} keyframes) to {}",
|
||||
aus,
|
||||
stream.len(),
|
||||
path.display()
|
||||
);
|
||||
}
|
||||
|
||||
/// The PRODUCTION host chain minus the IDD ring: the REAL `HdrP010Converter` renders the 8
|
||||
/// sRGB bars into a ring-profile P010 texture (`BIND_RENDER_TARGET` only — RTV-written, not
|
||||
/// CPU-uploaded) on the VPL implementation's own adapter, and THAT texture goes through the
|
||||
/// unaligned-height ingest copy into a Main10 encode. Dumped to
|
||||
/// `%TEMP%\pf_qsv_conv_1080_bars.h265`; expected decode codes = the bars_pq2020 fixture set
|
||||
/// (see `hdr_p010_convert_bars_on_luid`).
|
||||
#[test]
|
||||
fn qsv_live_hdr_converter_e2e_1080_dump() {
|
||||
const W: u32 = 1920;
|
||||
const H: u32 = 1080;
|
||||
|
||||
init_tracing();
|
||||
let Ok((_l, impls)) = intel_loader() else {
|
||||
eprintln!("skipping: no VPL loader");
|
||||
return;
|
||||
};
|
||||
let Some(imp) = impls.iter().find(|i| i.luid_valid) else {
|
||||
eprintln!("skipping: no Intel VPL implementation on this box");
|
||||
return;
|
||||
};
|
||||
if !probe_can_encode_10bit(Codec::H265) {
|
||||
eprintln!("skipping: this GPU declines 10-bit HEVC");
|
||||
return;
|
||||
}
|
||||
let (device, tex) = pf_capture::dxgi::hdr_p010_convert_bars_on_luid(imp.luid, W, H)
|
||||
.expect("converter bars");
|
||||
|
||||
let mut enc = QsvEncoder::open(
|
||||
Codec::H265,
|
||||
PixelFormat::P010,
|
||||
W,
|
||||
H,
|
||||
30,
|
||||
10_000_000,
|
||||
10,
|
||||
ChromaFormat::Yuv420,
|
||||
)
|
||||
.expect("open");
|
||||
enc.set_hdr_meta(Some(test_hdr_meta()));
|
||||
let mut stream = Vec::new();
|
||||
let mut aus = 0usize;
|
||||
for i in 0..12u32 {
|
||||
let frame = CapturedFrame {
|
||||
width: W,
|
||||
height: H,
|
||||
pts_ns: i as u64 * 33_333_333,
|
||||
format: PixelFormat::P010,
|
||||
payload: FramePayload::D3d11(pf_frame::dxgi::D3d11Frame {
|
||||
texture: tex.clone(),
|
||||
device: device.clone(),
|
||||
pyro: None,
|
||||
}),
|
||||
cursor: None,
|
||||
};
|
||||
enc.submit_indexed(&frame, i).expect("submit");
|
||||
if let Some(au) = enc.poll().expect("poll") {
|
||||
aus += 1;
|
||||
stream.extend_from_slice(&au.data);
|
||||
}
|
||||
}
|
||||
enc.flush().expect("flush");
|
||||
while let Some(au) = enc.poll().expect("drain") {
|
||||
aus += 1;
|
||||
stream.extend_from_slice(&au.data);
|
||||
}
|
||||
assert!(aus >= 10, "expected ≥10 AUs, got {aus}");
|
||||
let path = std::env::temp_dir().join("pf_qsv_conv_1080_bars.h265");
|
||||
std::fs::write(&path, &stream).expect("write dump");
|
||||
println!(
|
||||
"wrote {aus} AUs ({} bytes) to {}",
|
||||
stream.len(),
|
||||
path.display()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -129,6 +129,9 @@ pub fn open_video(
|
||||
cuda: bool,
|
||||
bit_depth: u8,
|
||||
chroma: ChromaFormat,
|
||||
// The session may hand this encoder cursor bitmaps to composite (cursor-as-metadata
|
||||
// captures). Backends whose fast path can't blend (Vulkan EFC RGB-direct) key off it.
|
||||
cursor_blend: bool,
|
||||
) -> Result<Box<dyn Encoder>> {
|
||||
let (inner, backend) = open_video_backend(
|
||||
codec,
|
||||
@@ -140,6 +143,7 @@ pub fn open_video(
|
||||
cuda,
|
||||
bit_depth,
|
||||
chroma,
|
||||
cursor_blend,
|
||||
)?;
|
||||
// Record what this session encodes on (the mgmt API's "currently used GPU"): the backend label
|
||||
// is reported by `open_video_backend` from the branch that ACTUALLY opened — not re-derived by
|
||||
@@ -203,15 +207,34 @@ impl Encoder for TrackedEncoder {
|
||||
fn invalidate_ref_frames(&mut self, first_frame: i64, last_frame: i64) -> bool {
|
||||
self.inner.invalidate_ref_frames(first_frame, last_frame)
|
||||
}
|
||||
// Forwarded for the same reason as `set_wire_chunking` below — the unforwarded default
|
||||
// (`false` = "backend can't pipeline, stop asking") silently killed the §7 LN3 contention
|
||||
// escalation for every session, since the host loop only ever holds the wrapped box.
|
||||
fn set_pipelined(&mut self, on: bool) -> bool {
|
||||
self.inner.set_pipelined(on)
|
||||
}
|
||||
// The classic TrackedEncoder trap: a defaulted trait method that isn't forwarded
|
||||
// silently no-ops through the wrapper (bit the direct-NVENC work, then THIS — the
|
||||
// §4.4 chunking probe run hit the default while the plan said Some(1408)).
|
||||
fn set_wire_chunking(&mut self, shard_payload: usize) {
|
||||
self.inner.set_wire_chunking(shard_payload)
|
||||
}
|
||||
// Forwarded for the same reason as `set_wire_chunking` above — an unforwarded default here
|
||||
// would silently leave the in-place backends pipelining past the capturer's ring.
|
||||
fn set_input_ring_depth(&mut self, depth: usize) {
|
||||
self.inner.set_input_ring_depth(depth)
|
||||
}
|
||||
fn poll(&mut self) -> Result<Option<EncodedFrame>> {
|
||||
self.inner.poll()
|
||||
}
|
||||
// Both chunked-poll methods forwarded (the same trap class): the defaults would report
|
||||
// "not chunked" and wrap whole AUs, silently discarding the sub-frame overlap.
|
||||
fn supports_chunked_poll(&self) -> bool {
|
||||
self.inner.supports_chunked_poll()
|
||||
}
|
||||
fn poll_chunk(&mut self) -> Result<Option<AuChunk>> {
|
||||
self.inner.poll_chunk()
|
||||
}
|
||||
fn reset(&mut self) -> bool {
|
||||
self.inner.reset()
|
||||
}
|
||||
@@ -245,7 +268,9 @@ fn open_video_backend(
|
||||
cuda: bool,
|
||||
bit_depth: u8,
|
||||
chroma: ChromaFormat,
|
||||
cursor_blend: bool,
|
||||
) -> Result<(Box<dyn Encoder>, &'static str)> {
|
||||
let _ = cursor_blend; // consumed only by the Linux vulkan-encode arm below
|
||||
validate_dimensions(codec, width, height)?;
|
||||
// Refresh/fps must be positive and sane: fps feeds the encoder time_base (`Rational(1, fps)`)
|
||||
// and the pts→ns conversion (`pts * 1e9 / fps`), so 0 builds a 1/0 rational / divides by zero.
|
||||
@@ -303,8 +328,15 @@ fn open_video_backend(
|
||||
&& vulkan_encode_enabled()
|
||||
&& !(bit_depth == 10 && format.is_hdr_rgb10())
|
||||
{
|
||||
match vulkan_video::VulkanVideoEncoder::open(codec, width, height, fps, bitrate_bps)
|
||||
{
|
||||
match vulkan_video::VulkanVideoEncoder::open(
|
||||
codec,
|
||||
format,
|
||||
width,
|
||||
height,
|
||||
fps,
|
||||
bitrate_bps,
|
||||
cursor_blend,
|
||||
) {
|
||||
Ok(e) => {
|
||||
tracing::info!(
|
||||
codec = ?codec,
|
||||
@@ -313,12 +345,32 @@ fn open_video_backend(
|
||||
);
|
||||
return Ok((Box::new(e) as Box<dyn Encoder>, "vulkan"));
|
||||
}
|
||||
// Native NV12 (PUNKTFUNK_PIPEWIRE_NV12 capture) has no VAAPI fallback:
|
||||
// libav's dmabuf lane would import the two-plane buffer as packed RGB
|
||||
// (silent garbage) and its CPU lane bails per frame — die crisply instead.
|
||||
Err(e) if format == PixelFormat::Nv12 => {
|
||||
return Err(e.context(
|
||||
"Vulkan Video open failed on a native-NV12 capture \
|
||||
— no VAAPI fallback exists; set PUNKTFUNK_PIPEWIRE_NV12=0 to \
|
||||
restore the packed-RGB negotiation",
|
||||
));
|
||||
}
|
||||
Err(e) => tracing::warn!(
|
||||
error = %format!("{e:#}"),
|
||||
"Vulkan Video encode open failed — falling back to libav VAAPI"
|
||||
),
|
||||
}
|
||||
}
|
||||
// Same rule when the Vulkan backend was never eligible (H264 session,
|
||||
// PUNKTFUNK_VULKAN_ENCODE=0, or a build without the feature).
|
||||
if format == PixelFormat::Nv12 {
|
||||
anyhow::bail!(
|
||||
"native NV12 capture requires the Vulkan Video encoder (HEVC/AV1 \
|
||||
session, --features vulkan-encode, PUNKTFUNK_VULKAN_ENCODE not 0) — this \
|
||||
session resolved to libav VAAPI; set PUNKTFUNK_PIPEWIRE_NV12=0 to restore \
|
||||
the packed-RGB negotiation"
|
||||
);
|
||||
}
|
||||
vaapi::VaapiEncoder::open(
|
||||
codec,
|
||||
format,
|
||||
@@ -357,7 +409,15 @@ fn open_video_backend(
|
||||
"the Vulkan Video encoder supports HEVC + AV1; the session negotiated {codec:?}"
|
||||
);
|
||||
}
|
||||
vulkan_video::VulkanVideoEncoder::open(codec, width, height, fps, bitrate_bps)
|
||||
vulkan_video::VulkanVideoEncoder::open(
|
||||
codec,
|
||||
format,
|
||||
width,
|
||||
height,
|
||||
fps,
|
||||
bitrate_bps,
|
||||
cursor_blend,
|
||||
)
|
||||
.map(|e| (Box::new(e) as Box<dyn Encoder>, "vulkan"))
|
||||
}
|
||||
#[cfg(not(feature = "vulkan-encode"))]
|
||||
@@ -781,6 +841,33 @@ fn vulkan_encode_enabled() -> bool {
|
||||
.unwrap_or(true)
|
||||
}
|
||||
|
||||
/// Whether THIS session's encoder can ingest a producer-native NV12 capture: only the raw
|
||||
/// Vulkan Video backend does (libav VAAPI would misread the two-plane buffer as packed RGB —
|
||||
/// [`open_video`] refuses the combination), so the session's codec must be one it encodes and
|
||||
/// the backend must be eligible to open. The host facade threads the verdict into the capture
|
||||
/// negotiation (`OutputFormat::nv12_native` → `ZeroCopyPolicy::native_nv12_session`), which
|
||||
/// then PREFERS gamescope's producer-side NV12 pod (default-on; `PUNKTFUNK_PIPEWIRE_NV12=0`
|
||||
/// escapes at the capture gate).
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn linux_native_nv12_ok(codec: Codec) -> bool {
|
||||
#[cfg(feature = "vulkan-encode")]
|
||||
{
|
||||
matches!(codec, Codec::H265 | Codec::Av1)
|
||||
&& vulkan_encode_enabled()
|
||||
// NVENC/PyroWave prefs never open the Vulkan Video backend; every other pref
|
||||
// (auto/vaapi/amd/intel/vulkan) tries it first on AMD/Intel — see [`open_video`].
|
||||
&& !matches!(
|
||||
pf_host_config::config().encoder_pref.as_str(),
|
||||
"nvenc" | "nvidia" | "cuda" | "pyrowave"
|
||||
)
|
||||
}
|
||||
#[cfg(not(feature = "vulkan-encode"))]
|
||||
{
|
||||
let _ = codec;
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
/// Cheap, side-effect-free NVIDIA-presence probe for the `auto` backend selector: the NVIDIA
|
||||
/// kernel driver exposes these device nodes, AMD/Intel boxes have neither. Deliberately does NOT
|
||||
/// create a CUDA context (that would allocate GPU state on every host that merely *might* be
|
||||
@@ -1338,6 +1425,12 @@ mod vulkan_video;
|
||||
#[cfg(all(target_os = "linux", feature = "vulkan-encode"))]
|
||||
#[path = "enc/linux/vk_av1_encode.rs"]
|
||||
mod vk_av1_encode;
|
||||
// Vendored `VK_VALVE_video_encode_rgb_conversion` bindings (host-only) — RGB encode source with
|
||||
// the VCN EFC front-end doing the CSC (design/vulkan-rgb-direct-encode.md). ash 0.38 predates
|
||||
// the extension; same vendoring rationale as `vk_av1_encode`.
|
||||
#[cfg(all(target_os = "linux", feature = "vulkan-encode"))]
|
||||
#[path = "enc/linux/vk_valve_rgb.rs"]
|
||||
mod vk_valve_rgb;
|
||||
// Small ash leaf helpers shared by the Linux Vulkan encode backends (dmabuf import, image/memory
|
||||
// utilities) — extracted from `vulkan_video.rs` when the PyroWave backend arrived.
|
||||
#[cfg(all(
|
||||
|
||||
@@ -52,6 +52,12 @@ pub struct PyroFrameShare {
|
||||
/// The fence value the capturer signalled after THIS frame's convert. The encoder's Vulkan
|
||||
/// acquire waits on it, so the wavelet read is ordered after the D3D11 CSC.
|
||||
pub fence_value: u64,
|
||||
/// The capturer's ring generation, bumped every time it recreates its texture ring. The
|
||||
/// PyroWave encoder caches its plane imports keyed on the texture's COM address, which carries
|
||||
/// no reference — after a recreate those addresses can be recycled by the allocator, so a
|
||||
/// cached import may describe a texture that no longer exists. The encoder flushes its import
|
||||
/// cache whenever this changes, making cache identity independent of allocator behaviour.
|
||||
pub ring_gen: u32,
|
||||
}
|
||||
|
||||
/// A GPU-resident captured texture (the Windows zero-copy path: NVENC/AMF/QSV encode it in place;
|
||||
|
||||
+31
-10
@@ -105,13 +105,15 @@ pub fn drm_fourcc(format: PixelFormat) -> Option<u32> {
|
||||
Bgra => drm_fourcc_code(b"AR24"), // DRM_FORMAT_ARGB8888
|
||||
Rgbx => drm_fourcc_code(b"XB24"), // DRM_FORMAT_XBGR8888
|
||||
Rgba => drm_fourcc_code(b"AB24"), // DRM_FORMAT_ABGR8888
|
||||
// Linux native NV12 capture (gamescope PipeWire): one LINEAR dmabuf with contiguous Y then
|
||||
// interleaved UV, exposed under DRM_FORMAT_NV12.
|
||||
Nv12 => drm_fourcc_code(b"NV12"),
|
||||
// The GNOME 50+ HDR screencast formats (packed 2:10:10:10, PQ/BT.2020).
|
||||
X2Rgb10 => drm_fourcc_code(b"XR30"), // DRM_FORMAT_XRGB2101010
|
||||
X2Bgr10 => drm_fourcc_code(b"XB30"), // DRM_FORMAT_XBGR2101010
|
||||
// 24-bit packed RGB/BGR have no straightforward dmabuf import here; use the CPU path.
|
||||
// Rgb10a2/Nv12/P010 are the Windows HDR / video-processor formats — never produced on
|
||||
// Linux; Yuv444 is OUR convert's OUTPUT, never a capture source format.
|
||||
Rgb | Bgr | Rgb10a2 | Nv12 | P010 | Yuv444 => return None,
|
||||
// Rgb10a2/P010 are Windows formats; Yuv444 is OUR convert output, never a capture source.
|
||||
Rgb | Bgr | Rgb10a2 | P010 | Yuv444 => return None,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -144,6 +146,12 @@ pub struct OutputFormat {
|
||||
/// (never BGRA-passthrough / P010). `false` on every non-PyroWave session and on Linux (the
|
||||
/// wavelet encoder ingests dmabufs / CPU RGB there, not a D3D11 texture).
|
||||
pub pyrowave: bool,
|
||||
/// THIS session's encoder can ingest a producer-native NV12 capture (Linux raw Vulkan Video
|
||||
/// backend on an H265/AV1 session — see `pf_encode::linux_native_nv12_ok`). The Linux capture
|
||||
/// negotiation only offers gamescope the NV12 pod when this is set: libav VAAPI (the H264
|
||||
/// codec's backend, and the fallback family) would misread the two-plane buffer as packed
|
||||
/// RGB. Always `false` on Windows (the IDD-push capturer owns its own formats).
|
||||
pub nv12_native: bool,
|
||||
}
|
||||
|
||||
impl OutputFormat {
|
||||
@@ -161,6 +169,11 @@ impl OutputFormat {
|
||||
chroma_444: false,
|
||||
// GameStream never negotiates PyroWave (native punktfunk/1 only).
|
||||
pyrowave: false,
|
||||
// Conservative: the GameStream + spike paths don't resolve the codec here, and a
|
||||
// Moonlight client may negotiate H264 (whose VAAPI backend can't ingest NV12) — so
|
||||
// they never prefer the producer-native NV12 pod. The punktfunk/1 plane opts in via
|
||||
// `SessionPlan::output_format()`, which knows the codec.
|
||||
nv12_native: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -201,18 +214,26 @@ pub struct CapturedFrame {
|
||||
pub cursor: Option<CursorOverlay>,
|
||||
}
|
||||
|
||||
/// A captured frame still living in a single-plane packed-RGB dmabuf (the VAAPI zero-copy path).
|
||||
/// A captured frame still living in a DMA-BUF. Packed RGB uses one plane. Native Linux NV12
|
||||
/// (gamescope PipeWire) travels in ONE fd: Y starts at `offset`, and the interleaved UV plane
|
||||
/// lives at `plane1`'s offset/stride when the producer reported them — else at the contiguous
|
||||
/// fallback `offset + stride * frame_height` with the shared `stride`.
|
||||
///
|
||||
/// Owns a *dup* of the PipeWire buffer's fd, so the frame can travel to the encode thread and be
|
||||
/// imported into a VA surface there without the compositor's buffer being closed underneath it.
|
||||
/// (Content stability across the brief import window relies on the compositor's buffer pool depth,
|
||||
/// same as any zero-copy capture — the VAAPI importer copies into its own NV12 surface promptly.)
|
||||
/// imported there without the compositor's buffer being closed underneath it. Content stability
|
||||
/// across the brief import window relies on the compositor's buffer pool depth, like any zero-copy
|
||||
/// capture.
|
||||
#[cfg(target_os = "linux")]
|
||||
pub struct DmabufFrame {
|
||||
pub fd: std::os::fd::OwnedFd,
|
||||
/// DRM FourCC of the packed-RGB plane (e.g. `XR24` for BGRx).
|
||||
/// DRM FourCC (`XR24` for BGRx, `NV12` for native 4:2:0).
|
||||
pub fourcc: u32,
|
||||
/// DRM format modifier the compositor allocated (0 = LINEAR).
|
||||
pub modifier: u64,
|
||||
/// Second-plane `(offset, stride)` within the SAME fd, when the producer reported one (the
|
||||
/// PipeWire buffer's plane-1 chunk — NV12's interleaved UV). `None` falls back to the
|
||||
/// contiguous-plane contract above. Always `None` for single-plane packed RGB.
|
||||
pub plane1: Option<(u32, u32)>,
|
||||
pub offset: u32,
|
||||
pub stride: u32,
|
||||
}
|
||||
@@ -225,8 +246,8 @@ pub enum FramePayload {
|
||||
/// The dmabuf has already been imported + copied into this owned device buffer.
|
||||
#[cfg(target_os = "linux")]
|
||||
Cuda(pf_zerocopy::DeviceBuffer),
|
||||
/// A raw packed-RGB dmabuf — the AMD/Intel (VAAPI) zero-copy path. The encoder imports it into
|
||||
/// a VA surface and does RGB→NV12 on the GPU video engine (no host CSC, no upload).
|
||||
/// A raw DMA-BUF: packed RGB for the existing GPU CSC paths, or native NV12 from a producer
|
||||
/// such as gamescope. The encoder imports it without a host copy.
|
||||
#[cfg(target_os = "linux")]
|
||||
Dmabuf(DmabufFrame),
|
||||
/// A GPU-resident D3D11 texture (Windows zero-copy path for NVENC). Owns the copied frame.
|
||||
|
||||
@@ -63,6 +63,13 @@ pub struct HostConfig {
|
||||
/// deliver full chroma, and the GPU/driver passed the encode probe — otherwise 4:2:0.
|
||||
/// `PUNKTFUNK_444=0`/`false`/`off`/`no` disables. Independent of `ten_bit` (chroma vs depth).
|
||||
pub four_four_four: bool,
|
||||
/// `PUNKTFUNK_CHACHA20` — host policy gate for the negotiated ChaCha20-Poly1305 session
|
||||
/// cipher (design/chacha20-session-cipher.md). **Default ON** (pure rollout safety — perf-only,
|
||||
/// both AEADs are full-strength): the host merely *allows* it — a session only seals with
|
||||
/// ChaCha when the client advertised `VIDEO_CAP_CHACHA20` (set by soft-AES armv7 clients,
|
||||
/// e.g. webOS TVs, whose GCM decrypt caps at ~100 Mbps); everyone else stays AES-128-GCM.
|
||||
/// `PUNKTFUNK_CHACHA20=0`/`false`/`off`/`no` disables.
|
||||
pub chacha20: bool,
|
||||
/// `PUNKTFUNK_PERF` — per-stage timing instrumentation.
|
||||
pub perf: bool,
|
||||
/// `PUNKTFUNK_VIDEO_SOURCE` — GameStream video source select (`virtual` / `portal` / unset → synthetic).
|
||||
@@ -147,6 +154,16 @@ impl HostConfig {
|
||||
)
|
||||
})
|
||||
.unwrap_or(true),
|
||||
// Default ON, explicit-off grammar (the client's VIDEO_CAP_CHACHA20 bit is the real
|
||||
// per-session switch; see the field doc).
|
||||
chacha20: val("PUNKTFUNK_CHACHA20")
|
||||
.map(|s| {
|
||||
!matches!(
|
||||
s.trim().to_ascii_lowercase().as_str(),
|
||||
"0" | "false" | "off" | "no"
|
||||
)
|
||||
})
|
||||
.unwrap_or(true),
|
||||
perf: flag("PUNKTFUNK_PERF"),
|
||||
video_source: val("PUNKTFUNK_VIDEO_SOURCE"),
|
||||
compositor: val("PUNKTFUNK_COMPOSITOR"),
|
||||
|
||||
@@ -36,6 +36,12 @@ pub fn vk_to_evdev(vk: u8) -> Option<u16> {
|
||||
0x2D => Some(110), // VK_INSERT -> KEY_INSERT
|
||||
0x2E => Some(111), // VK_DELETE -> KEY_DELETE
|
||||
|
||||
// --- Consumer/media keys (Android TV remotes, keyboard media rows) ---
|
||||
0xB0 => Some(163), // VK_MEDIA_NEXT_TRACK -> KEY_NEXTSONG
|
||||
0xB1 => Some(165), // VK_MEDIA_PREV_TRACK -> KEY_PREVIOUSSONG
|
||||
0xB2 => Some(166), // VK_MEDIA_STOP -> KEY_STOPCD
|
||||
0xB3 => Some(164), // VK_MEDIA_PLAY_PAUSE -> KEY_PLAYPAUSE
|
||||
|
||||
// --- Generic modifiers ---
|
||||
0x10 => Some(42), // VK_SHIFT -> KEY_LEFTSHIFT
|
||||
0x11 => Some(29), // VK_CONTROL -> KEY_LEFTCTRL
|
||||
|
||||
@@ -424,6 +424,9 @@ impl InputInjector for KwinFakeInjector {
|
||||
self.fake.touch_up(event.code);
|
||||
self.fake.touch_frame();
|
||||
}
|
||||
// fake_input can only press host-layout keycodes — no committed-text path (the
|
||||
// HOST_CAP_TEXT_INPUT cap is not advertised on this backend).
|
||||
InputKind::TextInput => {}
|
||||
// Gamepads are injected through uinput, not the compositor.
|
||||
InputKind::GamepadState
|
||||
| InputKind::GamepadButton
|
||||
|
||||
@@ -109,7 +109,7 @@ async fn session_main(mut rx: UnboundedReceiver<InputEvent>, source: EiSource) {
|
||||
// Keep `_rd`/`_session` bound for the whole loop — dropping the portal session closes the
|
||||
// EIS connection. Bound the setup so a headless approval dialog (un-bypassed grant) can't
|
||||
// hang the worker forever.
|
||||
let (_keepalive, context, mut events) = match tokio::time::timeout(
|
||||
let (_keepalive, context, mut events, output_hint) = match tokio::time::timeout(
|
||||
Duration::from_secs(30),
|
||||
connect(source),
|
||||
)
|
||||
@@ -130,6 +130,7 @@ async fn session_main(mut rx: UnboundedReceiver<InputEvent>, source: EiSource) {
|
||||
tracing::info!("libei: EIS connected — awaiting devices");
|
||||
|
||||
let mut state = EiState::new();
|
||||
state.output_hint = output_hint;
|
||||
// Watchdog: a healthy EIS server adds + resumes an input device within a beat of the handshake.
|
||||
// If none has resumed by this deadline, the connection is dead-on-arrival (stale/half-ready
|
||||
// gamescope socket the handshake passed but no real server is behind) — exit so the next
|
||||
@@ -177,20 +178,28 @@ type Connected = (
|
||||
Box<dyn Send>,
|
||||
ei::Context,
|
||||
reis::tokio::EiConvertEventStream,
|
||||
// The compositor's output size ("WxH" relay-file hint) — the scale target for absolute
|
||||
// coordinates when the EIS advertises only a degenerate region (gamescope). `None` for
|
||||
// the portal/Mutter paths, whose regions are real.
|
||||
Option<(u32, u32)>,
|
||||
);
|
||||
|
||||
/// Reach an EIS server per `source` and run the EI sender handshake.
|
||||
async fn connect(source: EiSource) -> Result<Connected> {
|
||||
let (keepalive, stream): (Box<dyn Send>, UnixStream) = match source {
|
||||
let (keepalive, stream, output_hint): (Box<dyn Send>, UnixStream, Option<(u32, u32)>) =
|
||||
match source {
|
||||
EiSource::Portal => {
|
||||
let (rd, session, fd) = connect_portal().await?;
|
||||
(Box::new((rd, session)), UnixStream::from(fd))
|
||||
(Box::new((rd, session)), UnixStream::from(fd), None)
|
||||
}
|
||||
EiSource::MutterEis => {
|
||||
let (keepalive, fd) = connect_mutter().await?;
|
||||
(keepalive, UnixStream::from(fd))
|
||||
(keepalive, UnixStream::from(fd), None)
|
||||
}
|
||||
EiSource::SocketPathFile(file) => {
|
||||
let (stream, hint) = connect_socket_file(&file).await?;
|
||||
(Box::new(()), stream, hint)
|
||||
}
|
||||
EiSource::SocketPathFile(file) => (Box::new(()), connect_socket_file(&file).await?),
|
||||
};
|
||||
let context = ei::Context::new(stream).map_err(|e| anyhow!("reis EI context: {e}"))?;
|
||||
// Bound the handshake. `UnixStream::connect` to a socket *file* succeeds the moment the path
|
||||
@@ -206,7 +215,7 @@ async fn connect(source: EiSource) -> Result<Connected> {
|
||||
anyhow!("EI handshake timed out (EIS server not responding — stale/half-ready socket?)")
|
||||
})?
|
||||
.map_err(|e| anyhow!("EI handshake: {e}"))?;
|
||||
Ok((keepalive, context, events))
|
||||
Ok((keepalive, context, events, output_hint))
|
||||
}
|
||||
|
||||
/// Open a RemoteDesktop portal session (pointer + keyboard) and obtain the EIS socket fd.
|
||||
@@ -294,8 +303,10 @@ async fn connect_mutter() -> Result<(Box<dyn Send>, std::os::fd::OwnedFd)> {
|
||||
|
||||
/// Poll `file` for the EIS socket path (the gamescope backend relays `LIBEI_SOCKET` there once
|
||||
/// the nested app launches), then connect. A bare name is resolved against `XDG_RUNTIME_DIR`,
|
||||
/// mirroring libei's own `LIBEI_SOCKET` semantics.
|
||||
async fn connect_socket_file(file: &std::path::Path) -> Result<UnixStream> {
|
||||
/// mirroring libei's own `LIBEI_SOCKET` semantics. Line 2 of the relay file, when present,
|
||||
/// carries the compositor's output size as `WxH` — returned as the absolute-coordinate scale
|
||||
/// hint (gamescope's EIS region is degenerate, so the geometry can't come from the protocol).
|
||||
async fn connect_socket_file(file: &std::path::Path) -> Result<(UnixStream, Option<(u32, u32)>)> {
|
||||
// The relay file is rewritten each session with the CURRENT gamescope's `LIBEI_SOCKET`, and the
|
||||
// socket may not be `listen()`ing the instant its name appears — or the file may briefly still
|
||||
// hold a prior, now-dead session's name (the host-lifetime injector reconnecting between
|
||||
@@ -319,7 +330,12 @@ async fn connect_socket_file(file: &std::path::Path) -> Result<UnixStream> {
|
||||
));
|
||||
}
|
||||
if let Ok(s) = std::fs::read_to_string(file) {
|
||||
let name = s.trim();
|
||||
let mut file_lines = s.lines();
|
||||
let name = file_lines.next().unwrap_or("").trim();
|
||||
let hint = file_lines.next().and_then(|l| {
|
||||
let (w, h) = l.trim().split_once('x')?;
|
||||
Some((w.parse::<u32>().ok()?, h.parse::<u32>().ok()?))
|
||||
});
|
||||
if !name.is_empty() {
|
||||
let full = if name.starts_with('/') {
|
||||
std::path::PathBuf::from(name)
|
||||
@@ -334,7 +350,7 @@ async fn connect_socket_file(file: &std::path::Path) -> Result<UnixStream> {
|
||||
logged = name.to_string();
|
||||
}
|
||||
match UnixStream::connect(&full) {
|
||||
Ok(stream) => return Ok(stream),
|
||||
Ok(stream) => return Ok((stream, hint)),
|
||||
// Refused = socket file exists but no listener yet (or a dead session);
|
||||
// NotFound = path not created yet. Both heal once the live gamescope's EIS is
|
||||
// up — retry. Anything else (e.g. permission) is a real failure.
|
||||
@@ -386,6 +402,22 @@ struct EiState {
|
||||
held_keys: Vec<u32>,
|
||||
held_buttons: Vec<u32>,
|
||||
held_touches: Vec<u32>,
|
||||
/// The touch id currently degraded to the absolute pointer ([`EiState::degrade_touch`]) —
|
||||
/// the "primary finger" while the EIS has no touchscreen device. `None` between touches.
|
||||
degraded_touch: Option<u32>,
|
||||
/// The compositor's output size (relay-file "WxH" hint) — scale target for absolute
|
||||
/// coordinates when the device's region is degenerate/absent (gamescope). Without it the
|
||||
/// fallback is raw client pixels, correct only when the stream runs at the output's size.
|
||||
output_hint: Option<(u32, u32)>,
|
||||
}
|
||||
|
||||
/// Is this EIS region a plausible OUTPUT geometry — something to map normalized coordinates
|
||||
/// into? gamescope advertises a degenerate `(0,0,INT32_MAX,INT32_MAX)` "everything" region on
|
||||
/// its virtual input device, meaning "absolute coordinates are raw"; normalizing into it
|
||||
/// explodes a center tap to x≈1e9, which the compositor clamps to the far corner. 16384
|
||||
/// comfortably covers real multi-monitor layouts while rejecting the sentinel.
|
||||
fn sane_region(r: &reis::event::Region) -> bool {
|
||||
r.width > 0 && r.height > 0 && r.width <= 16_384 && r.height <= 16_384
|
||||
}
|
||||
|
||||
/// Stable small index per [`InputKind`] for the `seen_kinds` bitmask.
|
||||
@@ -406,6 +438,7 @@ fn kind_bit(kind: InputKind) -> u32 {
|
||||
InputKind::GamepadState => 12,
|
||||
InputKind::GamepadRemove => 13,
|
||||
InputKind::GamepadArrival => 14,
|
||||
InputKind::TextInput => 15,
|
||||
};
|
||||
1 << i
|
||||
}
|
||||
@@ -422,6 +455,8 @@ impl EiState {
|
||||
held_keys: Vec::new(),
|
||||
held_buttons: Vec::new(),
|
||||
held_touches: Vec::new(),
|
||||
degraded_touch: None,
|
||||
output_hint: None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -430,6 +465,10 @@ impl EiState {
|
||||
/// normal [`EiState::inject`] path so the compositor sees proper key-up / button-up /
|
||||
/// touch-up frames before the devices disappear.
|
||||
fn release_all(&mut self, ctx: &ei::Context) {
|
||||
// A degraded touch is held as a synthesized left button (in `held_buttons`), so the
|
||||
// loop below releases it — but the primary-finger latch must clear too, or the NEXT
|
||||
// session's first TouchDown reads as a second finger and is ignored.
|
||||
self.degraded_touch = None;
|
||||
let (keys, buttons, touches) = (
|
||||
std::mem::take(&mut self.held_keys),
|
||||
std::mem::take(&mut self.held_buttons),
|
||||
@@ -506,6 +545,8 @@ impl EiState {
|
||||
keyboard = dev.has_capability(DeviceCapability::Keyboard),
|
||||
button = dev.has_capability(DeviceCapability::Button),
|
||||
scroll = dev.has_capability(DeviceCapability::Scroll),
|
||||
regions = dev.regions().len(),
|
||||
region0 = ?dev.regions().first().map(|r| (r.x, r.y, r.width, r.height)),
|
||||
"libei: device RESUMED (now emittable)"
|
||||
);
|
||||
}
|
||||
@@ -537,8 +578,85 @@ impl EiState {
|
||||
}
|
||||
}
|
||||
|
||||
/// Degrade touch to a single-finger ABSOLUTE POINTER on compositors whose EIS never
|
||||
/// creates a touchscreen device (gamescope's "Gamescope Virtual Input" advertises
|
||||
/// pointer/pointer_abs/button but no touch — observed live; headless KWin the same).
|
||||
/// Down = abs-move + left press, move = abs-move, up = left release — synthesized through
|
||||
/// the normal [`EiState::inject`] Mouse* paths so region mapping, held-state tracking and
|
||||
/// [`EiState::release_all`] all apply. Only the FIRST finger drives the pointer; later
|
||||
/// fingers are ignored (a pointer has no second contact — a pinch degrades to a drag).
|
||||
fn degrade_touch(&mut self, ev: &InputEvent, ctx: &ei::Context) {
|
||||
const GS_BUTTON_LEFT: u32 = 1;
|
||||
match ev.kind {
|
||||
InputKind::TouchDown => {
|
||||
if self.degraded_touch.is_some() {
|
||||
return; // secondary finger — single-pointer degradation
|
||||
}
|
||||
self.degraded_touch = Some(ev.code);
|
||||
static NOTED: std::sync::atomic::AtomicBool =
|
||||
std::sync::atomic::AtomicBool::new(false);
|
||||
if !NOTED.swap(true, std::sync::atomic::Ordering::Relaxed) {
|
||||
tracing::info!(
|
||||
"compositor's EIS has no touchscreen device — degrading touch to a \
|
||||
single-finger absolute pointer (tap = left click; multi-touch \
|
||||
gestures unavailable)"
|
||||
);
|
||||
}
|
||||
self.inject(
|
||||
&InputEvent {
|
||||
kind: InputKind::MouseMoveAbs,
|
||||
..*ev
|
||||
},
|
||||
ctx,
|
||||
);
|
||||
self.inject(
|
||||
&InputEvent {
|
||||
kind: InputKind::MouseButtonDown,
|
||||
code: GS_BUTTON_LEFT,
|
||||
..*ev
|
||||
},
|
||||
ctx,
|
||||
);
|
||||
}
|
||||
InputKind::TouchMove if self.degraded_touch == Some(ev.code) => {
|
||||
self.inject(
|
||||
&InputEvent {
|
||||
kind: InputKind::MouseMoveAbs,
|
||||
..*ev
|
||||
},
|
||||
ctx,
|
||||
);
|
||||
}
|
||||
InputKind::TouchUp if self.degraded_touch == Some(ev.code) => {
|
||||
self.degraded_touch = None;
|
||||
self.inject(
|
||||
&InputEvent {
|
||||
kind: InputKind::MouseButtonUp,
|
||||
code: GS_BUTTON_LEFT,
|
||||
..*ev
|
||||
},
|
||||
ctx,
|
||||
);
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
/// Translate and emit one client input event, committing it as a single `frame`.
|
||||
fn inject(&mut self, ev: &InputEvent, ctx: &ei::Context) {
|
||||
// No ei_touchscreen device but an absolute pointer exists → degrade rather than drop
|
||||
// (the game-mode "touch simply does not work" trap). Checked per event, not latched:
|
||||
// a touchscreen device appearing later (compositor restart, capability change) takes
|
||||
// over seamlessly on the next touch.
|
||||
if matches!(
|
||||
ev.kind,
|
||||
InputKind::TouchDown | InputKind::TouchMove | InputKind::TouchUp
|
||||
) && self.device_for(DeviceCapability::Touch).is_none()
|
||||
&& self.device_for(DeviceCapability::PointerAbsolute).is_some()
|
||||
{
|
||||
self.degrade_touch(ev, ctx);
|
||||
return;
|
||||
}
|
||||
let cap = match ev.kind {
|
||||
InputKind::MouseMove => DeviceCapability::Pointer,
|
||||
InputKind::MouseMoveAbs => DeviceCapability::PointerAbsolute,
|
||||
@@ -553,6 +671,9 @@ impl EiState {
|
||||
| InputKind::GamepadAxis
|
||||
| InputKind::GamepadRemove
|
||||
| InputKind::GamepadArrival => return, // uinput path (later)
|
||||
// libei presses keycodes against the server's negotiated keymap — no committed-text
|
||||
// path (the HOST_CAP_TEXT_INPUT cap is not advertised on this backend).
|
||||
InputKind::TextInput => return,
|
||||
};
|
||||
self.injected += 1;
|
||||
let n = self.injected;
|
||||
@@ -605,16 +726,31 @@ impl EiState {
|
||||
InputKind::MouseMoveAbs => {
|
||||
let w = ((ev.flags >> 16) & 0xffff) as f32;
|
||||
let h = (ev.flags & 0xffff) as f32;
|
||||
match (
|
||||
slot.interface::<ei::PointerAbsolute>(),
|
||||
slot.regions().first(),
|
||||
) {
|
||||
(Some(p), Some(region)) if w > 0.0 && h > 0.0 => {
|
||||
// Map the normalized client position into the device's first region.
|
||||
match slot.interface::<ei::PointerAbsolute>() {
|
||||
Some(p) if w > 0.0 && h > 0.0 => {
|
||||
// Map the normalized client position into the device's first region —
|
||||
// but only when the region looks like a real output geometry.
|
||||
// gamescope's "Gamescope Virtual Input" advertises a degenerate
|
||||
// (0,0,INT32_MAX,INT32_MAX) region meaning "coordinates are raw":
|
||||
// normalizing into it explodes a center tap to x≈1e9, which gamescope
|
||||
// clamps to the far corner (the observed cursor-parked-at-1279,799).
|
||||
// There the managed session runs at the client's mode, so client
|
||||
// pixels ARE output pixels: emit them raw.
|
||||
let nx = (ev.x as f32 / w).clamp(0.0, 1.0);
|
||||
let ny = (ev.y as f32 / h).clamp(0.0, 1.0);
|
||||
let x = region.x as f32 + nx * region.width as f32;
|
||||
let y = region.y as f32 + ny * region.height as f32;
|
||||
let (x, y) = match slot.regions().first().filter(|r| sane_region(r)) {
|
||||
Some(region) => (
|
||||
region.x as f32 + nx * region.width as f32,
|
||||
region.y as f32 + ny * region.height as f32,
|
||||
),
|
||||
// Degenerate/absent region: scale into the relay-file output hint
|
||||
// (correct even when the client streams at a different resolution
|
||||
// than the session runs); raw client pixels as the last resort.
|
||||
None => match self.output_hint {
|
||||
Some((ow, oh)) => (nx * ow as f32, ny * oh as f32),
|
||||
None => (ev.x as f32, ev.y as f32),
|
||||
},
|
||||
};
|
||||
p.motion_absolute(x, y);
|
||||
}
|
||||
_ => emitted = false,
|
||||
@@ -680,12 +816,21 @@ impl EiState {
|
||||
InputKind::TouchDown | InputKind::TouchMove => {
|
||||
let w = ((ev.flags >> 16) & 0xffff) as f32;
|
||||
let h = (ev.flags & 0xffff) as f32;
|
||||
match (slot.interface::<ei::Touchscreen>(), slot.regions().first()) {
|
||||
(Some(t), Some(region)) if w > 0.0 && h > 0.0 => {
|
||||
match slot.interface::<ei::Touchscreen>() {
|
||||
Some(t) if w > 0.0 && h > 0.0 => {
|
||||
let nx = (ev.x as f32 / w).clamp(0.0, 1.0);
|
||||
let ny = (ev.y as f32 / h).clamp(0.0, 1.0);
|
||||
let x = region.x as f32 + nx * region.width as f32;
|
||||
let y = region.y as f32 + ny * region.height as f32;
|
||||
// Same degenerate-region fallback ladder as MouseMoveAbs.
|
||||
let (x, y) = match slot.regions().first().filter(|r| sane_region(r)) {
|
||||
Some(region) => (
|
||||
region.x as f32 + nx * region.width as f32,
|
||||
region.y as f32 + ny * region.height as f32,
|
||||
),
|
||||
None => match self.output_hint {
|
||||
Some((ow, oh)) => (nx * ow as f32, ny * oh as f32),
|
||||
None => (ev.x as f32, ev.y as f32),
|
||||
},
|
||||
};
|
||||
if ev.kind == InputKind::TouchDown {
|
||||
t.down(ev.code, x, y);
|
||||
} else {
|
||||
@@ -703,7 +848,8 @@ impl EiState {
|
||||
| InputKind::GamepadButton
|
||||
| InputKind::GamepadAxis
|
||||
| InputKind::GamepadRemove
|
||||
| InputKind::GamepadArrival => emitted = false,
|
||||
| InputKind::GamepadArrival
|
||||
| InputKind::TextInput => emitted = false,
|
||||
}
|
||||
|
||||
if emitted {
|
||||
|
||||
@@ -17,7 +17,7 @@
|
||||
use anyhow::{bail, Context, Result};
|
||||
use std::mem::size_of;
|
||||
use std::os::fd::RawFd;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::thread::JoinHandle;
|
||||
|
||||
@@ -196,6 +196,45 @@ impl Drop for GadgetFd {
|
||||
}
|
||||
}
|
||||
|
||||
/// The signal used to break a worker thread out of a blocking raw_gadget ioctl at teardown.
|
||||
/// `EVENT_FETCH`/`EP_WRITE` are `wait_event_interruptible` in the kernel with no timeout and no
|
||||
/// `O_NONBLOCK` honouring, and closing the fd cannot wake a thread already inside the ioctl (the
|
||||
/// in-flight syscall holds a reference to the struct file). A signal is the only reliable lever:
|
||||
/// delivered with a no-op, non-`SA_RESTART` handler it forces the ioctl to return `EINTR`, after
|
||||
/// which the loop's top-of-iteration `running` check exits. `SIGUSR1` is unused elsewhere in this
|
||||
/// process; the handler is a no-op, so a stray `SIGUSR1` becomes harmless rather than fatal.
|
||||
const WAKE_SIGNAL: libc::c_int = libc::SIGUSR1;
|
||||
|
||||
/// Install the no-op `WAKE_SIGNAL` handler exactly once. Crucially `sa_flags = 0` (no `SA_RESTART`)
|
||||
/// so a delivered signal makes the interruptible ioctl return `EINTR` instead of auto-restarting.
|
||||
fn install_wake_handler() {
|
||||
static ONCE: std::sync::Once = std::sync::Once::new();
|
||||
ONCE.call_once(|| {
|
||||
extern "C" fn noop(_: libc::c_int) {}
|
||||
// SAFETY: installing a well-formed `sigaction` with an empty mask and a valid no-op handler
|
||||
// for a single signal; touches only this process's disposition for `WAKE_SIGNAL`.
|
||||
unsafe {
|
||||
let mut sa: libc::sigaction = std::mem::zeroed();
|
||||
// Via `*const ()`: casting a function item straight to an integer is what
|
||||
// `clippy::function_casts_as_integer` rejects, and the pointer hop is the documented
|
||||
// way to spell it. `sa_sigaction` is a `usize`-typed handler slot, so the value is
|
||||
// unchanged.
|
||||
sa.sa_sigaction = noop as *const () as usize;
|
||||
libc::sigemptyset(&mut sa.sa_mask);
|
||||
sa.sa_flags = 0;
|
||||
libc::sigaction(WAKE_SIGNAL, &sa, std::ptr::null_mut());
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// Lets `Drop` wake a specific worker thread parked in a blocking ioctl. `tid` is the thread's
|
||||
/// `pthread_self()` (0 until it starts); `done` is set right before the thread returns, so `Drop`
|
||||
/// stops signalling a thread that has already exited.
|
||||
struct Waker {
|
||||
tid: Arc<AtomicU64>,
|
||||
done: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
/// A virtual Steam Deck presented over the USB gadget subsystem. Dropping it stops the threads and
|
||||
/// closes the gadget (the kernel tears down the device).
|
||||
pub struct SteamDeckGadget {
|
||||
@@ -203,6 +242,7 @@ pub struct SteamDeckGadget {
|
||||
feedback: Arc<Mutex<super::steam_proto::SteamFeedback>>,
|
||||
running: Arc<AtomicBool>,
|
||||
threads: Vec<JoinHandle<()>>,
|
||||
wakers: Vec<Waker>,
|
||||
_fd: Arc<GadgetFd>,
|
||||
seq: u32,
|
||||
}
|
||||
@@ -243,6 +283,18 @@ impl SteamDeckGadget {
|
||||
let ctrl_ep = Arc::new(std::sync::atomic::AtomicI32::new(-1));
|
||||
let configured = Arc::new(AtomicBool::new(false));
|
||||
|
||||
// The teardown wake path (see `WAKE_SIGNAL`) needs the handler installed before any thread
|
||||
// can park in a blocking ioctl.
|
||||
install_wake_handler();
|
||||
let ctrl_waker = Waker {
|
||||
tid: Arc::new(AtomicU64::new(0)),
|
||||
done: Arc::new(AtomicBool::new(false)),
|
||||
};
|
||||
let stream_waker = Waker {
|
||||
tid: Arc::new(AtomicU64::new(0)),
|
||||
done: Arc::new(AtomicBool::new(false)),
|
||||
};
|
||||
|
||||
// Control thread: enumerate + answer every control transfer.
|
||||
let control = {
|
||||
let fd = fd.clone();
|
||||
@@ -250,10 +302,15 @@ impl SteamDeckGadget {
|
||||
let ctrl_ep = ctrl_ep.clone();
|
||||
let configured = configured.clone();
|
||||
let feedback = feedback.clone();
|
||||
let tid = ctrl_waker.tid.clone();
|
||||
let done = ctrl_waker.done.clone();
|
||||
std::thread::Builder::new()
|
||||
.name("pf-deck-gadget-ctrl".into())
|
||||
.spawn(move || {
|
||||
control_loop(fd, running, ctrl_ep, configured, feedback, serial, unit_id)
|
||||
// SAFETY: `pthread_self` is always valid on the calling thread.
|
||||
tid.store(unsafe { libc::pthread_self() } as u64, Ordering::SeqCst);
|
||||
control_loop(fd, running, ctrl_ep, configured, feedback, serial, unit_id);
|
||||
done.store(true, Ordering::SeqCst);
|
||||
})
|
||||
.context("spawn gadget control thread")?
|
||||
};
|
||||
@@ -264,9 +321,16 @@ impl SteamDeckGadget {
|
||||
let ctrl_ep = ctrl_ep.clone();
|
||||
let configured = configured.clone();
|
||||
let report = report.clone();
|
||||
let tid = stream_waker.tid.clone();
|
||||
let done = stream_waker.done.clone();
|
||||
std::thread::Builder::new()
|
||||
.name("pf-deck-gadget-stream".into())
|
||||
.spawn(move || stream_loop(fd, running, ctrl_ep, configured, report))
|
||||
.spawn(move || {
|
||||
// SAFETY: `pthread_self` is always valid on the calling thread.
|
||||
tid.store(unsafe { libc::pthread_self() } as u64, Ordering::SeqCst);
|
||||
stream_loop(fd, running, ctrl_ep, configured, report);
|
||||
done.store(true, Ordering::SeqCst);
|
||||
})
|
||||
.context("spawn gadget stream thread")?
|
||||
};
|
||||
|
||||
@@ -275,6 +339,7 @@ impl SteamDeckGadget {
|
||||
feedback,
|
||||
running,
|
||||
threads: vec![control, stream],
|
||||
wakers: vec![ctrl_waker, stream_waker],
|
||||
_fd: fd,
|
||||
seq: 0,
|
||||
})
|
||||
@@ -302,6 +367,32 @@ impl SteamDeckGadget {
|
||||
impl Drop for SteamDeckGadget {
|
||||
fn drop(&mut self) {
|
||||
self.running.store(false, Ordering::SeqCst);
|
||||
// The control thread spends steady state parked in a blocking `EVENT_FETCH` ioctl that only
|
||||
// tests `running` at the top of its loop, so clearing the flag is not enough — it must be
|
||||
// signalled out of the syscall (see `WAKE_SIGNAL`). Without this the join below can hang the
|
||||
// caller (the session input thread, via `PadSlots::sweep`) indefinitely. Retry until each
|
||||
// thread reports done, to cover the race where the signal lands just before the thread
|
||||
// re-enters the ioctl; bounded (~1 s) so a genuinely stuck thread can't wedge teardown either.
|
||||
for _ in 0..200 {
|
||||
let mut all_done = true;
|
||||
for w in &self.wakers {
|
||||
if w.done.load(Ordering::SeqCst) {
|
||||
continue;
|
||||
}
|
||||
all_done = false;
|
||||
let tid = w.tid.load(Ordering::SeqCst);
|
||||
if tid != 0 {
|
||||
// SAFETY: the thread is joinable and not yet joined (join runs after this loop),
|
||||
// so `tid` names a live pthread; `pthread_kill` on a finished-but-unjoined thread
|
||||
// is defined (returns ESRCH), never UB.
|
||||
unsafe { libc::pthread_kill(tid as libc::pthread_t, WAKE_SIGNAL) };
|
||||
}
|
||||
}
|
||||
if all_done {
|
||||
break;
|
||||
}
|
||||
std::thread::sleep(std::time::Duration::from_millis(5));
|
||||
}
|
||||
for t in self.threads.drain(..) {
|
||||
let _ = t.join();
|
||||
}
|
||||
|
||||
@@ -98,9 +98,26 @@ pub struct WlrootsInjector {
|
||||
keyboard: ZwpVirtualKeyboardV1,
|
||||
xkb_state: xkb::State,
|
||||
_keymap_file: std::fs::File, // keep the memfd alive for the compositor's mmap
|
||||
/// Dedicated committed-text device ([`InputKind::TextInput`]), created on first use.
|
||||
text: Option<TextKeyboard>,
|
||||
start: Instant,
|
||||
}
|
||||
|
||||
/// Cap on distinct characters the dynamic text keymap holds before it restarts from scratch
|
||||
/// (keycodes grow upward from 9; xkb tops out at 255, so stay well under).
|
||||
const TEXT_KEYMAP_MAX: usize = 200;
|
||||
|
||||
/// The dedicated **text** virtual keyboard: types committed IME text (`InputKind::TextInput`,
|
||||
/// one Unicode scalar per event) by growing a keymap of Unicode keysyms on demand and pressing
|
||||
/// the character's keycode — the `wtype` model. A separate `zwp_virtual_keyboard` so keymap
|
||||
/// re-uploads never disturb the main device's layout/modifier state that VK key events ride on.
|
||||
struct TextKeyboard {
|
||||
keyboard: ZwpVirtualKeyboardV1,
|
||||
/// Characters in keycode order: `chars[i]` types on wire keycode `i + 1` (xkb `i + 9`).
|
||||
chars: Vec<char>,
|
||||
_keymap_file: Option<std::fs::File>, // keep the memfd alive for the compositor's mmap
|
||||
}
|
||||
|
||||
impl WlrootsInjector {
|
||||
pub fn open() -> Result<Self> {
|
||||
let conn = Connection::connect_to_env()
|
||||
@@ -171,6 +188,7 @@ impl WlrootsInjector {
|
||||
keyboard,
|
||||
xkb_state,
|
||||
_keymap_file: file,
|
||||
text: None,
|
||||
start: Instant::now(),
|
||||
})
|
||||
}
|
||||
@@ -179,6 +197,54 @@ impl WlrootsInjector {
|
||||
self.start.elapsed().as_millis() as u32
|
||||
}
|
||||
|
||||
/// Type one committed-text Unicode scalar on the dedicated text device (created lazily),
|
||||
/// growing its keymap when the character is new. Control characters are dropped — Enter,
|
||||
/// Backspace and Tab ride the VK key-event path.
|
||||
fn type_text(&mut self, cp: u32) -> Result<()> {
|
||||
let Some(ch) = char::from_u32(cp) else {
|
||||
return Ok(()); // lone surrogate / out of range
|
||||
};
|
||||
if ch.is_control() {
|
||||
return Ok(());
|
||||
}
|
||||
if self.text.is_none() {
|
||||
let (Some(mgr), Some(seat)) =
|
||||
(self.globals.keyboard_mgr.clone(), self.globals.seat.clone())
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
let kb = mgr.create_virtual_keyboard(&seat, &self.queue.handle(), ());
|
||||
self.text = Some(TextKeyboard {
|
||||
keyboard: kb,
|
||||
chars: Vec::new(),
|
||||
_keymap_file: None,
|
||||
});
|
||||
}
|
||||
let t = self.now_ms();
|
||||
let text = self.text.as_mut().expect("created above");
|
||||
let code = match text.chars.iter().position(|&c| c == ch) {
|
||||
Some(i) => (i + 1) as u32,
|
||||
None => {
|
||||
if text.chars.len() >= TEXT_KEYMAP_MAX {
|
||||
text.chars.clear(); // restart the map; old codes are re-assigned lazily
|
||||
}
|
||||
text.chars.push(ch);
|
||||
let keymap_str = text_keymap(&text.chars);
|
||||
let file = memfd_with(&keymap_str)?;
|
||||
text.keyboard.keymap(
|
||||
1, /* XKB_V1 */
|
||||
file.as_fd(),
|
||||
keymap_str.len() as u32 + 1,
|
||||
);
|
||||
text._keymap_file = Some(file);
|
||||
text.chars.len() as u32
|
||||
}
|
||||
};
|
||||
text.keyboard.key(t, code, 1);
|
||||
text.keyboard.key(t, code, 0);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Update xkb state for a key and tell the compositor the resulting modifier mask.
|
||||
fn send_modifiers(&mut self, evdev: u16, down: bool) {
|
||||
let kc = xkb::Keycode::new(evdev as u32 + 8); // evdev -> xkb keycode
|
||||
@@ -254,6 +320,9 @@ impl InputInjector for WlrootsInjector {
|
||||
tracing::debug!(vk = event.code, "unmapped VK keycode — dropped");
|
||||
}
|
||||
}
|
||||
InputKind::TextInput => {
|
||||
self.type_text(event.code)?;
|
||||
}
|
||||
InputKind::GamepadState
|
||||
| InputKind::GamepadButton
|
||||
| InputKind::GamepadAxis
|
||||
@@ -271,6 +340,33 @@ impl InputInjector for WlrootsInjector {
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a minimal xkb keymap whose keycode `i + 9` (wire code `i + 1`) types `chars[i]`, using
|
||||
/// Unicode keysym names (`U<hex>` — xkbcommon resolves them for any scalar, emoji included).
|
||||
/// Types/compat `include "complete"` mirrors `wtype`'s generated keymap — proven on wlroots
|
||||
/// compositors, and the system XKB data is present (the main keymap compiled from it in `open`).
|
||||
fn text_keymap(chars: &[char]) -> String {
|
||||
use std::fmt::Write as _;
|
||||
let mut keycodes = String::new();
|
||||
let mut symbols = String::new();
|
||||
for (i, ch) in chars.iter().enumerate() {
|
||||
let _ = writeln!(keycodes, " <T{i}> = {};", i + 9);
|
||||
let _ = writeln!(symbols, " key <T{i}> {{ [ U{:04X} ] }};", *ch as u32);
|
||||
}
|
||||
format!(
|
||||
"xkb_keymap {{\n\
|
||||
xkb_keycodes \"punktfunk-text\" {{\n\
|
||||
minimum = 8;\n\
|
||||
maximum = {};\n\
|
||||
{keycodes}\
|
||||
}};\n\
|
||||
xkb_types \"punktfunk-text\" {{ include \"complete\" }};\n\
|
||||
xkb_compatibility \"punktfunk-text\" {{ include \"complete\" }};\n\
|
||||
xkb_symbols \"punktfunk-text\" {{\n{symbols} }};\n\
|
||||
}};\n",
|
||||
chars.len() + 9,
|
||||
)
|
||||
}
|
||||
|
||||
/// Create an anonymous in-memory file holding `s` + a trailing NUL (for the keymap fd).
|
||||
fn memfd_with(s: &str) -> Result<std::fs::File> {
|
||||
let name = b"punktfunk-keymap\0";
|
||||
|
||||
@@ -299,13 +299,20 @@ pub(super) fn create_swdevice(p: &SwDeviceProfile) -> Result<(HSWDEVICE, Option<
|
||||
let event = unsafe { CreateEventW(None, true, false, PCWSTR::null())? };
|
||||
// `result` starts as E_FAIL, NOT S_OK: if the wait below times out, a zero-initialised HRESULT
|
||||
// would read as success and mask the failure (found by the 2026-07 driver-health audit).
|
||||
let mut ctx = SwCreateCtx {
|
||||
// HEAP-allocated, deliberately: `sw_create_cb` writes `result` + up to 127 u16 of instance id
|
||||
// through this pointer and then `SetEvent`s. The wait below is bounded (10 s), so on a wedged-PnP
|
||||
// timeout the callback may still be PENDING — a stack context would be popped and a late callback
|
||||
// would corrupt whatever the input thread put there next, and SetEvent a closed/recycled handle.
|
||||
// On the timeout path we therefore LEAK the box and leave the event open (a one-off ~264 B + one
|
||||
// HANDLE, only on that rare path) so a late callback always writes to live memory.
|
||||
let ctx = Box::into_raw(Box::new(SwCreateCtx {
|
||||
event,
|
||||
result: E_FAIL,
|
||||
instance_id: [0; 128],
|
||||
};
|
||||
// SAFETY: info + the buffers + ctx outlive the call (we wait on the event before returning);
|
||||
// windows-rs returns the HSWDEVICE (the C out-param) as the Result value.
|
||||
}));
|
||||
// SAFETY: info + the buffers outlive the call; `ctx` is a live heap allocation that outlives every
|
||||
// path below (reclaimed only where the callback provably ran). windows-rs returns the HSWDEVICE
|
||||
// (the C out-param) as the Result value.
|
||||
let hsw = match unsafe {
|
||||
SwDeviceCreate(
|
||||
w!("punktfunk"),
|
||||
@@ -313,13 +320,15 @@ pub(super) fn create_swdevice(p: &SwDeviceProfile) -> Result<(HSWDEVICE, Option<
|
||||
&info,
|
||||
None,
|
||||
Some(sw_create_cb),
|
||||
Some(&mut ctx as *mut SwCreateCtx as *const c_void),
|
||||
Some(ctx as *const c_void),
|
||||
)
|
||||
} {
|
||||
Ok(h) => h,
|
||||
Err(e) => {
|
||||
// SAFETY: event is valid.
|
||||
// SAFETY: the call failed, so no callback was registered and `ctx` is ours to reclaim;
|
||||
// `event` is valid and unreferenced.
|
||||
unsafe {
|
||||
drop(Box::from_raw(ctx));
|
||||
let _ = CloseHandle(event);
|
||||
}
|
||||
return Err(anyhow!("SwDeviceCreate failed: {e}"));
|
||||
@@ -328,17 +337,22 @@ pub(super) fn create_swdevice(p: &SwDeviceProfile) -> Result<(HSWDEVICE, Option<
|
||||
// Block until PnP finishes enumerating (the callback signals), then check its result.
|
||||
// SAFETY: event is valid.
|
||||
let wait = unsafe { WaitForSingleObject(event, 10_000) };
|
||||
// SAFETY: event is valid.
|
||||
unsafe {
|
||||
let _ = CloseHandle(event);
|
||||
}
|
||||
if wait != WAIT_OBJECT_0 {
|
||||
// Timed out: the callback may still fire. Intentionally leak `ctx` AND leave `event` open so
|
||||
// its eventual write + SetEvent target live memory/handle rather than freed ones.
|
||||
// SAFETY: hsw is the handle SwDeviceCreate returned.
|
||||
unsafe { SwDeviceClose(hsw) };
|
||||
return Err(anyhow!(
|
||||
"SwDeviceCreate enumeration callback never fired (10s) — PnP may be wedged"
|
||||
));
|
||||
}
|
||||
// The callback ran (it is what signalled the event), so nothing else will touch `ctx`/`event`.
|
||||
// SAFETY: `ctx` came from `Box::into_raw` above and is reclaimed exactly once here; `event` is
|
||||
// valid and no longer referenced by a pending callback.
|
||||
let ctx = unsafe {
|
||||
let _ = CloseHandle(event);
|
||||
Box::from_raw(ctx)
|
||||
};
|
||||
if ctx.result.is_err() {
|
||||
// SAFETY: hsw is the handle SwDeviceCreate returned.
|
||||
unsafe { SwDeviceClose(hsw) };
|
||||
|
||||
@@ -62,7 +62,7 @@ impl Ds4WinPad {
|
||||
std::ptr::write_unaligned(base as *mut u32, SHM_MAGIC);
|
||||
}
|
||||
let inst = format!("pf_ds4_{index}");
|
||||
let (hsw, instance_id) = match create_swdevice(&SwDeviceProfile {
|
||||
let (hsw, instance_id) = create_swdevice(&SwDeviceProfile {
|
||||
instance: &inst,
|
||||
container_tag: 0x5046_4453, // "PFDS"
|
||||
container_index: index,
|
||||
@@ -70,13 +70,13 @@ impl Ds4WinPad {
|
||||
usb_vid_pid: "VID_054C&PID_09CC",
|
||||
usb_mi: None,
|
||||
description: "punktfunk Virtual DualShock 4",
|
||||
}) {
|
||||
Ok((h, id)) => (Some(h), id),
|
||||
Err(e) => {
|
||||
tracing::warn!(error = %format!("{e:#}"), "SwDeviceCreate failed; DualShock 4 devnode unavailable");
|
||||
(None, None)
|
||||
}
|
||||
};
|
||||
})?; // Propagate, do NOT swallow — see below.
|
||||
let (hsw, instance_id) = (Some(hsw), instance_id);
|
||||
// Swallowing a create failure here (the previous behaviour) latched the pad slot to
|
||||
// `Some(pad)` with no live devnode: `PadSlots::ensure` short-circuits on `is_some()` and
|
||||
// `gate.on_success()` cleared the backoff, so the create-gate that exists precisely to
|
||||
// self-heal a transient PnP failure never retried. The game saw no controller for the whole
|
||||
// session unless the client unplugged the pad. Matches the XUSB sibling, which propagates.
|
||||
let _sw = hsw.map(super::gamepad_raii::SwDevice::new);
|
||||
// Bounded eager delivery — for the DS4 this is what closes the identity race: the driver
|
||||
// must read `device_type = 1` from the delivered DATA section before hidclass asks it for
|
||||
|
||||
@@ -82,12 +82,16 @@ fn create_swdevice(index: u8) -> Result<(HSWDEVICE, Option<String>)> {
|
||||
let event = unsafe { CreateEventW(None, true, false, PCWSTR::null())? };
|
||||
// `result` starts as E_FAIL, NOT S_OK: if the wait below times out, a zero-initialised HRESULT
|
||||
// would read as success and mask the failure (found by the 2026-07 driver-health audit).
|
||||
let mut ctx = SwCreateCtx {
|
||||
// HEAP-allocated for the same reason as the DualSense sibling: the callback writes through this
|
||||
// pointer and SetEvents, and the wait below is bounded — a stack context would be popped while a
|
||||
// late callback still holds it. On the timeout path the box is deliberately leaked and the event
|
||||
// left open so a late write/SetEvent always targets live memory/handle.
|
||||
let ctx = Box::into_raw(Box::new(SwCreateCtx {
|
||||
event,
|
||||
result: E_FAIL,
|
||||
instance_id: [0; 128],
|
||||
};
|
||||
// SAFETY: info + buffers + ctx outlive the call (we wait on the event before returning).
|
||||
}));
|
||||
// SAFETY: info + buffers outlive the call; `ctx` is a live heap allocation outliving every path.
|
||||
let hsw = match unsafe {
|
||||
SwDeviceCreate(
|
||||
w!("punktfunk"),
|
||||
@@ -95,13 +99,14 @@ fn create_swdevice(index: u8) -> Result<(HSWDEVICE, Option<String>)> {
|
||||
&info,
|
||||
None,
|
||||
Some(sw_create_cb),
|
||||
Some(&mut ctx as *mut SwCreateCtx as *const c_void),
|
||||
Some(ctx as *const c_void),
|
||||
)
|
||||
} {
|
||||
Ok(h) => h,
|
||||
Err(e) => {
|
||||
// SAFETY: event is valid.
|
||||
// SAFETY: the call failed, so no callback is pending and `ctx` is ours to reclaim.
|
||||
unsafe {
|
||||
drop(Box::from_raw(ctx));
|
||||
let _ = CloseHandle(event);
|
||||
}
|
||||
return Err(anyhow!("SwDeviceCreate(pf_xusb) failed: {e}"));
|
||||
@@ -109,17 +114,20 @@ fn create_swdevice(index: u8) -> Result<(HSWDEVICE, Option<String>)> {
|
||||
};
|
||||
// SAFETY: event valid; block until PnP finishes enumerating, then check the callback result.
|
||||
let wait = unsafe { WaitForSingleObject(event, 10_000) };
|
||||
// SAFETY: event is valid.
|
||||
unsafe {
|
||||
let _ = CloseHandle(event);
|
||||
}
|
||||
if wait != WAIT_OBJECT_0 {
|
||||
// Timed out — intentionally leak `ctx` and leave `event` open (see above).
|
||||
// SAFETY: hsw is the handle SwDeviceCreate returned.
|
||||
unsafe { SwDeviceClose(hsw) };
|
||||
return Err(anyhow!(
|
||||
"SwDeviceCreate(pf_xusb) enumeration callback never fired (10s) — PnP may be wedged"
|
||||
));
|
||||
}
|
||||
// The callback ran (it signalled the event), so nothing else will touch `ctx`/`event`.
|
||||
// SAFETY: `ctx` came from `Box::into_raw` and is reclaimed exactly once here.
|
||||
let ctx = unsafe {
|
||||
let _ = CloseHandle(event);
|
||||
Box::from_raw(ctx)
|
||||
};
|
||||
if ctx.result.is_err() {
|
||||
// SAFETY: hsw is the handle SwDeviceCreate returned.
|
||||
unsafe { SwDeviceClose(hsw) };
|
||||
|
||||
@@ -27,10 +27,10 @@ use windows::Win32::System::StationsAndDesktops::{
|
||||
use windows::Win32::UI::Input::KeyboardAndMouse::{
|
||||
GetKeyboardLayout, MapVirtualKeyExW, SendInput, HKL, INPUT, INPUT_0, INPUT_KEYBOARD,
|
||||
INPUT_MOUSE, KEYBDINPUT, KEYEVENTF_EXTENDEDKEY, KEYEVENTF_KEYUP, KEYEVENTF_SCANCODE,
|
||||
MAPVK_VK_TO_VSC_EX, MOUSEEVENTF_ABSOLUTE, MOUSEEVENTF_HWHEEL, MOUSEEVENTF_LEFTDOWN,
|
||||
MOUSEEVENTF_LEFTUP, MOUSEEVENTF_MIDDLEDOWN, MOUSEEVENTF_MIDDLEUP, MOUSEEVENTF_MOVE,
|
||||
MOUSEEVENTF_RIGHTDOWN, MOUSEEVENTF_RIGHTUP, MOUSEEVENTF_VIRTUALDESK, MOUSEEVENTF_WHEEL,
|
||||
MOUSEEVENTF_XDOWN, MOUSEEVENTF_XUP, MOUSEINPUT, VIRTUAL_KEY,
|
||||
KEYEVENTF_UNICODE, MAPVK_VK_TO_VSC_EX, MOUSEEVENTF_ABSOLUTE, MOUSEEVENTF_HWHEEL,
|
||||
MOUSEEVENTF_LEFTDOWN, MOUSEEVENTF_LEFTUP, MOUSEEVENTF_MIDDLEDOWN, MOUSEEVENTF_MIDDLEUP,
|
||||
MOUSEEVENTF_MOVE, MOUSEEVENTF_RIGHTDOWN, MOUSEEVENTF_RIGHTUP, MOUSEEVENTF_VIRTUALDESK,
|
||||
MOUSEEVENTF_WHEEL, MOUSEEVENTF_XDOWN, MOUSEEVENTF_XUP, MOUSEINPUT, VIRTUAL_KEY,
|
||||
};
|
||||
use windows::Win32::UI::WindowsAndMessaging::{
|
||||
GetForegroundWindow, GetSystemMetrics, GetWindowThreadProcessId, SM_CXVIRTUALSCREEN,
|
||||
@@ -297,6 +297,33 @@ impl InputInjector for SendInputInjector {
|
||||
};
|
||||
self.send(&[key(ki)])
|
||||
}
|
||||
InputKind::TextInput => {
|
||||
// Committed IME text: one Unicode scalar per event, injected as
|
||||
// `KEYEVENTF_UNICODE` packets (wScan = UTF-16 unit, no scancode/layout involved
|
||||
// — the receiving app gets the character verbatim via WM_CHAR). An astral-plane
|
||||
// scalar (emoji) is its surrogate pair, each unit down+up in order — exactly how
|
||||
// Windows expects supplementary characters from unicode injection.
|
||||
let Some(ch) = char::from_u32(event.code) else {
|
||||
return Ok(()); // lone surrogate / out of range — drop
|
||||
};
|
||||
if ch.is_control() {
|
||||
return Ok(()); // control chars ride the VK path (Enter/Backspace/Tab)
|
||||
}
|
||||
let mut units = [0u16; 2];
|
||||
let mut inputs: Vec<INPUT> = Vec::with_capacity(4);
|
||||
for &unit in ch.encode_utf16(&mut units).iter() {
|
||||
for flags in [KEYEVENTF_UNICODE, KEYEVENTF_UNICODE | KEYEVENTF_KEYUP] {
|
||||
inputs.push(key(KEYBDINPUT {
|
||||
wVk: VIRTUAL_KEY(0),
|
||||
wScan: unit,
|
||||
dwFlags: flags,
|
||||
time: 0,
|
||||
dwExtraInfo: 0,
|
||||
}));
|
||||
}
|
||||
}
|
||||
self.send(&inputs)
|
||||
}
|
||||
// Gamepad goes through the XUSB backend. Touch: no SendInput equivalent -> no-op.
|
||||
InputKind::GamepadButton
|
||||
| InputKind::GamepadAxis
|
||||
|
||||
@@ -66,7 +66,7 @@ impl DeckWinPad {
|
||||
std::ptr::write_unaligned(base as *mut u32, SHM_MAGIC);
|
||||
}
|
||||
let inst = format!("pf_deck_{index}");
|
||||
let (hsw, instance_id) = match create_swdevice(&SwDeviceProfile {
|
||||
let (hsw, instance_id) = create_swdevice(&SwDeviceProfile {
|
||||
instance: &inst,
|
||||
container_tag: 0x5046_4453, // "PFDS"
|
||||
container_index: index,
|
||||
@@ -77,13 +77,8 @@ impl DeckWinPad {
|
||||
// spike's run-1 failure).
|
||||
usb_mi: Some(2),
|
||||
description: "punktfunk Virtual Steam Deck",
|
||||
}) {
|
||||
Ok((h, i)) => (Some(h), i),
|
||||
Err(e) => {
|
||||
tracing::warn!(error = %format!("{e:#}"), "SwDeviceCreate failed; Steam Deck devnode unavailable");
|
||||
(None, None)
|
||||
}
|
||||
};
|
||||
})?; // Propagate — swallowing latched the slot to a pad with no devnode (see the DS4 twin).
|
||||
let (hsw, instance_id) = (Some(hsw), instance_id);
|
||||
let _sw = hsw.map(super::gamepad_raii::SwDevice::new);
|
||||
// Bounded eager delivery — the driver must read `device_type = 3` before hidclass asks
|
||||
// it for descriptors, or the pad would enumerate with the default DualSense identity.
|
||||
|
||||
@@ -174,6 +174,29 @@ pub fn default_backend() -> Backend {
|
||||
Backend::Unsupported
|
||||
}
|
||||
|
||||
/// Whether the session's inject backend can type **committed text**
|
||||
/// ([`InputKind::TextInput`] — see `HOST_CAP_TEXT_INPUT`): Windows always (`KEYEVENTF_UNICODE`);
|
||||
/// Linux only on the wlroots backend (a dedicated virtual keyboard with a dynamically-grown
|
||||
/// Unicode keymap) — KWin fake-input/libei/gamescope can only press keycodes of the host layout.
|
||||
/// Consulted at Welcome time to advertise the cap; a mid-session backend switch away from a
|
||||
/// capable one just degrades to dropped text events (input is lossy by design).
|
||||
#[cfg(target_os = "windows")]
|
||||
pub fn text_input_supported() -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
/// See the Windows variant: Linux types text only through the wlroots virtual-keyboard backend.
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn text_input_supported() -> bool {
|
||||
matches!(default_backend(), Backend::WlrVirtual)
|
||||
}
|
||||
|
||||
/// No injector ⇒ no text.
|
||||
#[cfg(not(any(target_os = "linux", target_os = "windows")))]
|
||||
pub fn text_input_supported() -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
#[path = "inject/service.rs"]
|
||||
mod service;
|
||||
pub use service::InjectorService;
|
||||
|
||||
@@ -11,7 +11,11 @@ repository.workspace = true
|
||||
# Same Linux+Windows gating as the rest of the client stack (dmabuf import is the one
|
||||
# Linux-only module — see lib.rs).
|
||||
[target.'cfg(any(target_os = "linux", windows))'.dependencies]
|
||||
pf-client-core = { path = "../pf-client-core" }
|
||||
# `default-features = false`: the PyroWave decode backend is turned on through THIS crate's own
|
||||
# `pyrowave` feature (which re-exports it below), never by inheriting the dependency's default.
|
||||
# Otherwise a consumer that deliberately builds us without `pyrowave` still drags the vendored
|
||||
# C++ in — fatal on Windows ARM64, where Granite has no SIMD path.
|
||||
pf-client-core = { path = "../pf-client-core", default-features = false }
|
||||
# AVVkFrame access (Vulkan Video frames: live sync state under the frames lock).
|
||||
pf-ffvk = { path = "../pf-ffvk" }
|
||||
punktfunk-core = { path = "../punktfunk-core", features = ["quic"] }
|
||||
|
||||
@@ -333,6 +333,11 @@ fn run_inner(mut opts: SessionOpts, mut mode: ModeCtl) -> Result<Option<Outcome>
|
||||
// bottom-right corner (the reported bug). The menu/library is keyboard+gamepad-driven
|
||||
// and consumes no mouse, so nothing wanted these synthetic events anyway.
|
||||
sdl3::hint::set("SDL_TOUCH_MOUSE_EVENTS", "0");
|
||||
// The Wayland `app_id` (and X11 WM_CLASS) — compositors match it against
|
||||
// io.unom.Punktfunk.desktop for the window/taskbar icon. Without it SDL uses a generic
|
||||
// identity and the session window gets the default-Wayland icon (the Linux analog of
|
||||
// the AppUserModelID adoption above).
|
||||
sdl3::hint::set("SDL_APP_ID", "io.unom.Punktfunk");
|
||||
let sdl = sdl3::init().context("SDL init")?;
|
||||
let video = sdl.video().context("SDL video")?;
|
||||
let events = sdl.event().context("SDL events")?;
|
||||
|
||||
@@ -30,9 +30,17 @@ impl Presenter {
|
||||
// switch modes before anything touches this frame. Only where the surface
|
||||
// offers HDR10 — otherwise PQ stays on the SDR swapchain and the CSC shader
|
||||
// tonemaps (mode 1).
|
||||
//
|
||||
// CPU frames NEVER take the HDR10 surface: software decode uploads swscale RGBA with
|
||||
// no CSC/tonemap pass, so on a mode-0 swapchain that sRGB-encoded content would be
|
||||
// composed as PQ — the field-reported psychedelic cyan/magenta picture (reproduced
|
||||
// 2026-07-21: Fedora-class client, no hw HEVC decode, GNOME/Mesa offering HDR10 even
|
||||
// on an SDR desktop). On the SDR swapchain the same frames are merely untonemapped
|
||||
// (washed out) — wrong in the known, benign way until the CPU lane grows a real
|
||||
// PQ→sRGB pass.
|
||||
let frame_pq = match &input {
|
||||
FrameInput::Redraw => None,
|
||||
FrameInput::Cpu(f) => Some(f.color.is_pq()),
|
||||
FrameInput::Cpu(_) => Some(false),
|
||||
#[cfg(target_os = "linux")]
|
||||
FrameInput::Dmabuf(d) => Some(d.color.is_pq()),
|
||||
FrameInput::VkFrame(v) => Some(v.color.is_pq()),
|
||||
|
||||
@@ -617,6 +617,15 @@ pub(super) fn pick_formats(
|
||||
surface: vk::SurfaceKHR,
|
||||
colorspace_ext: bool,
|
||||
) -> Result<(vk::SurfaceFormatKHR, Option<vk::SurfaceFormatKHR>)> {
|
||||
// `PUNKTFUNK_HDR10=0` (explicit-off grammar) refuses the HDR10/ST.2084 swapchain outright,
|
||||
// pinning PQ streams to the shader tonemap on an SDR surface. Two reasons this exists:
|
||||
// desktop compositors newly offer HDR10 even on SDR desktops (GNOME 48 / Plasma 6 with
|
||||
// Mesa ≥ 25.1 — a lane that otherwise engages silently), and it is the A/B lever that
|
||||
// splits "HDR10 passthrough composes wrong" from "the decoded planes are wrong" in the
|
||||
// field without rebuilding anything.
|
||||
let colorspace_ext = colorspace_ext
|
||||
&& !std::env::var("PUNKTFUNK_HDR10")
|
||||
.is_ok_and(|v| matches!(v.as_str(), "0" | "false" | "off" | "no"));
|
||||
let formats = unsafe { surface_i.get_physical_device_surface_formats(pdev, surface) }?;
|
||||
let mut sdr = None;
|
||||
for want in [vk::Format::B8G8R8A8_UNORM, vk::Format::R8G8B8A8_UNORM] {
|
||||
|
||||
@@ -72,8 +72,8 @@ pub use session::{session_epoch, try_recover_session};
|
||||
#[path = "vdisplay/routing.rs"]
|
||||
pub(crate) mod routing;
|
||||
pub use routing::{
|
||||
apply_input_env, restore_managed_session, restore_takeover_on_startup, start_restore_worker,
|
||||
wants_dedicated_game_session,
|
||||
apply_input_env, managed_session_available, restore_managed_session,
|
||||
restore_takeover_on_startup, start_restore_worker, wants_dedicated_game_session,
|
||||
};
|
||||
#[cfg(target_os = "linux")]
|
||||
pub use routing::{
|
||||
@@ -254,6 +254,34 @@ pub fn detect() -> Result<Compositor> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Attach-only probes: while any scope is held, backend `create` paths must not stop, relaunch,
|
||||
/// or take over box sessions — they may only attach to an already-live output, and fail fast
|
||||
/// otherwise. The capture-loss rebuild holds one for its first seconds: right after a capture
|
||||
/// loss the active-session detection can be STALE (a Game→Desktop switch observed live: the
|
||||
/// probe's gamescope re-acquire restarted `gamescope-session.target` and yanked the user out of
|
||||
/// the KDE session they had just switched to). A counter, so overlapping scopes compose.
|
||||
static REBUILD_PROBES: std::sync::atomic::AtomicU32 = std::sync::atomic::AtomicU32::new(0);
|
||||
|
||||
/// RAII scope marking pipeline builds as attach-only probes (see [`rebuild_probe_active`]).
|
||||
pub struct RebuildProbeScope(());
|
||||
|
||||
pub fn rebuild_probe_scope() -> RebuildProbeScope {
|
||||
REBUILD_PROBES.fetch_add(1, std::sync::atomic::Ordering::SeqCst);
|
||||
RebuildProbeScope(())
|
||||
}
|
||||
|
||||
impl Drop for RebuildProbeScope {
|
||||
fn drop(&mut self) {
|
||||
REBUILD_PROBES.fetch_sub(1, std::sync::atomic::Ordering::SeqCst);
|
||||
}
|
||||
}
|
||||
|
||||
/// Is any [`rebuild_probe_scope`] active? Destructive session operations (stop/relaunch/
|
||||
/// takeover-restart) must be skipped while true.
|
||||
pub fn rebuild_probe_active() -> bool {
|
||||
REBUILD_PROBES.load(std::sync::atomic::Ordering::SeqCst) > 0
|
||||
}
|
||||
|
||||
/// Open the virtual-display driver for `compositor`.
|
||||
pub fn open(compositor: Compositor) -> Result<Box<dyn VirtualDisplay>> {
|
||||
#[cfg(target_os = "linux")]
|
||||
|
||||
@@ -325,6 +325,29 @@ fn create_managed_session(client: &str, mode: Mode) -> Result<VirtualOutput> {
|
||||
if steamos_session_present() {
|
||||
return create_managed_session_steamos(mode);
|
||||
}
|
||||
// Attach-only rebuild probe: reuse a live same-mode session, but NEVER stop/relaunch box
|
||||
// sessions — right after a capture loss the caller's session detection can be stale, and a
|
||||
// destructive rebuild here would fight the session the user just switched to.
|
||||
if crate::rebuild_probe_active() {
|
||||
let guard = MANAGED_SESSION.lock().unwrap_or_else(|e| e.into_inner());
|
||||
let same_mode = guard.as_ref().is_some_and(|s| {
|
||||
s.width == mode.width && s.height == mode.height && s.refresh_hz == mode.refresh_hz
|
||||
});
|
||||
if same_mode {
|
||||
if let Some(node_id) = find_gamescope_node() {
|
||||
point_injector_at_eis();
|
||||
tracing::info!(
|
||||
node_id,
|
||||
"gamescope session: attach-only probe reusing live node"
|
||||
);
|
||||
return Ok(managed_output(node_id, mode));
|
||||
}
|
||||
}
|
||||
return Err(anyhow!(
|
||||
"gamescope session has no attachable live node — attach-only rebuild probe refuses \
|
||||
to stop/relaunch box sessions (re-detection follows the live session)"
|
||||
));
|
||||
}
|
||||
// Steam is single-instance: if the box autologged into gaming mode on a physical display (the
|
||||
// Bazzite default — `gamescope-session-plus@ogui-steam` on the TV), that session holds Steam and
|
||||
// renders to the TV's native mode, which we'd capture instead of the client's. Free Steam by
|
||||
@@ -607,12 +630,17 @@ fn write_steamos_dropin(shim_dir: &std::path::Path, mode: Mode) -> Result<()> {
|
||||
if let Some(parent) = path.parent() {
|
||||
std::fs::create_dir_all(parent).with_context(|| format!("mkdir {}", parent.display()))?;
|
||||
}
|
||||
// UnsetEnvironment: the same headless-must-not-attach armor `launch_session` gives its
|
||||
// transient unit — the manager env can carry a stale desktop DISPLAY/WAYLAND_DISPLAY (from a
|
||||
// portal settle), and gamescope would abort trying to attach to it instead of becoming the
|
||||
// display server. Unit-scoped belt-and-suspenders on top of the observe_session_instance scrub.
|
||||
let body = format!(
|
||||
"[Service]\n\
|
||||
Environment=PATH={shim}:/usr/bin:/bin:/usr/local/bin\n\
|
||||
Environment=PF_W={w}\n\
|
||||
Environment=PF_H={h}\n\
|
||||
Environment=PF_HZ={hz}\n",
|
||||
Environment=PF_HZ={hz}\n\
|
||||
UnsetEnvironment=DISPLAY WAYLAND_DISPLAY\n",
|
||||
shim = shim_dir.display(),
|
||||
w = mode.width,
|
||||
h = mode.height,
|
||||
@@ -650,6 +678,16 @@ fn create_managed_session_steamos(mode: Mode) -> Result<VirtualOutput> {
|
||||
}
|
||||
*guard = None; // tracked session lost its node — fall through to a clean restart
|
||||
}
|
||||
// Attach-only rebuild probe: the reuse path above may attach, but a restart of the session
|
||||
// target is out of bounds — observed live on a Deck: a stale post-capture-loss detection made
|
||||
// this restart steal the seat back from the KDE session the user had just switched to.
|
||||
if crate::rebuild_probe_active() {
|
||||
return Err(anyhow!(
|
||||
"gamescope has no live node and this is an attach-only rebuild probe — refusing to \
|
||||
restart {STEAMOS_SESSION_TARGET} (the box may be mid-switch to another session; \
|
||||
re-detection follows it)"
|
||||
));
|
||||
}
|
||||
let shim_dir = write_headless_shim()?;
|
||||
write_steamos_dropin(&shim_dir, mode)?;
|
||||
systemctl_user(&["daemon-reload"]);
|
||||
@@ -1078,6 +1116,26 @@ pub fn schedule_restore_tv_session() {
|
||||
}
|
||||
}
|
||||
|
||||
/// Does any DRM connector report a physically `connected` display? Scans
|
||||
/// `/sys/class/drm/*/status` — only connector nodes (`card0-eDP-1`, `card0-HDMI-A-1`, …) have a
|
||||
/// `status` file, so the bare `cardN` device dirs and `renderD*` nodes filter themselves out. A
|
||||
/// headless box (VM, panel-less mini PC) has none — in which case a "restore to the physical
|
||||
/// panel" can only fail, gamescope having no output to drive. Errors (no DRM at all, sysfs
|
||||
/// unreadable) read as headless: the safe direction is keeping the working session.
|
||||
fn physical_display_connected() -> bool {
|
||||
connected_connector_under(std::path::Path::new("/sys/class/drm"))
|
||||
}
|
||||
|
||||
/// [`physical_display_connected`] against an arbitrary sysfs root (the unit-testable core).
|
||||
fn connected_connector_under(base: &std::path::Path) -> bool {
|
||||
let Ok(entries) = std::fs::read_dir(base) else {
|
||||
return false;
|
||||
};
|
||||
entries.flatten().any(|e| {
|
||||
std::fs::read_to_string(e.path().join("status")).is_ok_and(|s| s.trim() == "connected")
|
||||
})
|
||||
}
|
||||
|
||||
/// Tear down our host-managed session (freeing Steam) and restart the autologin gaming session(s)
|
||||
/// we stopped on connect — so the TV returns to gaming mode when no one is streaming. Invoked by
|
||||
/// [`start_restore_worker`] once the debounce deadline passes; takes the stopped-unit list so a
|
||||
@@ -1089,6 +1147,19 @@ fn do_restore_tv_session() {
|
||||
{
|
||||
let mut took = STEAMOS_TOOK_OVER.lock().unwrap_or_else(|e| e.into_inner());
|
||||
if *took {
|
||||
// A box with no physically connected display (a VM, a panel-less mini PC) has no
|
||||
// "physical gaming session" to restore TO: removing the drop-in and restarting the
|
||||
// target just crash-loops gamescope (no output to drive) and strands every later
|
||||
// connect on "no usable compositor". Keep the headless session — and the takeover
|
||||
// state, so a same-mode reconnect reuses it warm — instead. Checked at restore time
|
||||
// (not connect time) so plugging a panel in later restores normally.
|
||||
if !physical_display_connected() {
|
||||
tracing::info!(
|
||||
"gamescope (SteamOS): no physical display connected — keeping the headless \
|
||||
session (nothing to restore to)"
|
||||
);
|
||||
return;
|
||||
}
|
||||
*took = false;
|
||||
clear_takeover(); // A3: takeover undone — drop the persisted crash-restore marker
|
||||
*MANAGED_SESSION.lock().unwrap_or_else(|e| e.into_inner()) = None;
|
||||
@@ -1188,15 +1259,31 @@ pub fn start_restore_worker() -> std::sync::Arc<()> {
|
||||
/// session). Shared by the attach and host-managed-session paths.
|
||||
fn point_injector_at_eis() {
|
||||
match find_gamescope_eis_socket() {
|
||||
Some(sock) => match std::fs::write(ei_socket_file(), &sock) {
|
||||
Some(sock) => {
|
||||
// Relay format: line 1 = socket, optional line 2 = the session's CURRENT output
|
||||
// size as "WxH". gamescope's EIS advertises only a degenerate INT32_MAX region, so
|
||||
// the injector can't learn the output geometry from the protocol — the hint lets
|
||||
// it scale normalized client positions correctly even when the client streams at
|
||||
// a different resolution than the session runs (foreign attach, supersample).
|
||||
let size = current_gamescope_output_size();
|
||||
let body = match size {
|
||||
Some((w, h)) => format!("{sock}\n{w}x{h}"),
|
||||
None => sock.clone(),
|
||||
};
|
||||
match std::fs::write(ei_socket_file(), body) {
|
||||
Ok(()) => {
|
||||
tracing::info!(socket = %sock, "gamescope: pointed injector at the session's EIS socket")
|
||||
tracing::info!(
|
||||
socket = %sock,
|
||||
output = ?size,
|
||||
"gamescope: pointed injector at the session's EIS socket"
|
||||
)
|
||||
}
|
||||
Err(e) => tracing::warn!(
|
||||
error = %e,
|
||||
"gamescope: could not write the EIS relay file — input may not reach the session"
|
||||
),
|
||||
},
|
||||
}
|
||||
}
|
||||
None => tracing::warn!(
|
||||
"gamescope: no connectable gamescope EIS socket found — input won't reach the session"
|
||||
),
|
||||
@@ -1486,7 +1573,35 @@ impl Drop for GamescopeProc {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{cgroup_is_punktfunk_owned, is_steam_launch, shape_dedicated_command};
|
||||
use super::{
|
||||
cgroup_is_punktfunk_owned, connected_connector_under, is_steam_launch,
|
||||
shape_dedicated_command,
|
||||
};
|
||||
|
||||
#[test]
|
||||
fn connector_status_scan() {
|
||||
let base = std::env::temp_dir().join(format!("pf-drm-scan-{}", std::process::id()));
|
||||
let mk = |name: &str, status: Option<&str>| {
|
||||
let dir = base.join(name);
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
if let Some(s) = status {
|
||||
std::fs::write(dir.join("status"), s).unwrap();
|
||||
}
|
||||
};
|
||||
// Headless layout: device + render nodes only (no status files) → not connected.
|
||||
mk("card0", None);
|
||||
mk("renderD128", None);
|
||||
assert!(!connected_connector_under(&base));
|
||||
// Connectors present but nothing plugged in → still not connected.
|
||||
mk("card0-HDMI-A-1", Some("disconnected\n"));
|
||||
assert!(!connected_connector_under(&base));
|
||||
// A live panel → connected.
|
||||
mk("card0-eDP-1", Some("connected\n"));
|
||||
assert!(connected_connector_under(&base));
|
||||
// A missing base dir (no DRM at all) reads as headless.
|
||||
assert!(!connected_connector_under(&base.join("nope")));
|
||||
std::fs::remove_dir_all(&base).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn steam_launch_detection() {
|
||||
|
||||
@@ -207,6 +207,20 @@ pub fn cancel_pending_tv_restore() {
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
pub fn cancel_pending_tv_restore() {}
|
||||
|
||||
/// Can the MANAGED gamescope path stand a session up from nothing on this box (SteamOS's
|
||||
/// `gamescope-session` launcher or Bazzite's `gamescope-session-plus` present)? Lets the connect
|
||||
/// path route a "no live graphical session" box to the gamescope takeover — which rebuilds the
|
||||
/// session at the client's mode — instead of failing the connect. Always `false` off Linux.
|
||||
#[cfg(target_os = "linux")]
|
||||
pub fn managed_session_available() -> bool {
|
||||
gamescope::managed_session_available()
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
pub fn managed_session_available() -> bool {
|
||||
false
|
||||
}
|
||||
|
||||
/// Call when a client session ends: if the host-managed gamescope path took over a box's autologin
|
||||
/// gaming session (stopped its single-instance Steam to stream at the client's mode), **schedule** a
|
||||
/// debounced restore so the TV returns to gaming mode — unless a client reconnects within the window
|
||||
|
||||
@@ -57,6 +57,13 @@ pub fn observe_session_instance(active: &ActiveSession) {
|
||||
if let Some(old) = compositor_for_kind(prev.0) {
|
||||
registry::invalidate_backend(old.id());
|
||||
}
|
||||
// The dead desktop's socket vars may still sit in the systemd --user manager env
|
||||
// ([`settle_desktop_portal`]'s import-environment) — scrub them NOW, or the next
|
||||
// `gamescope-session.target` start inherits a stale WAYLAND_DISPLAY and gamescope
|
||||
// runs NESTED against the dead desktop socket instead of becoming the display
|
||||
// server ("Failed to connect to wayland socket: wayland-0" — kept a Deck's Game
|
||||
// Mode from starting at all, observed live 2026-07-21).
|
||||
scrub_desktop_manager_env();
|
||||
}
|
||||
let epoch = bump_session_epoch();
|
||||
tracing::info!(
|
||||
@@ -70,6 +77,23 @@ pub fn observe_session_instance(active: &ActiveSession) {
|
||||
*last = Some(cur);
|
||||
}
|
||||
|
||||
/// Counterpart to [`settle_desktop_portal`]'s `import-environment`: drop the desktop session's
|
||||
/// socket vars from the systemd `--user` manager env once that desktop instance is GONE. They
|
||||
/// persist in the manager otherwise, and every later user unit inherits them — including
|
||||
/// `gamescope-session.target`, whose gamescope then aborts trying to attach to the dead desktop
|
||||
/// socket. Best-effort; the D-Bus activation env has no unset op, but gamescope-session is
|
||||
/// systemd-started, so the manager scrub is the one that matters. (A desktop restart re-imports
|
||||
/// via the next [`settle_desktop_portal`], so scrubbing on a bounce is harmless.)
|
||||
#[cfg(target_os = "linux")]
|
||||
fn scrub_desktop_manager_env() {
|
||||
let _ = std::process::Command::new("systemctl")
|
||||
.args(["--user", "unset-environment", "WAYLAND_DISPLAY", "DISPLAY"])
|
||||
.status();
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
fn scrub_desktop_manager_env() {}
|
||||
|
||||
/// Is `kind` a **desktop** compositor (KWin / Mutter / wlroots) — one whose kept PipeWire outputs die
|
||||
/// with the compositor instance, so the session epoch tracks it? `Gaming` (gamescope) and `None` are
|
||||
/// not (gamescope spawns are independent nested sessions — see [`observe_session_instance`]).
|
||||
|
||||
@@ -861,7 +861,18 @@ pub unsafe fn isolate_displays_ccd(keep_target_ids: &[u32]) -> Option<SavedConfi
|
||||
continue;
|
||||
}
|
||||
if p.flags & DISPLAYCONFIG_PATH_ACTIVE != 0 {
|
||||
p.flags &= !DISPLAYCONFIG_PATH_ACTIVE; // mark this path inactive
|
||||
// Mark the path inactive AND unpin its modes: per the SetDisplayConfig
|
||||
// contract a path being turned OFF needs BOTH mode indexes marked invalid,
|
||||
// and leaving them referencing the queried mode entries gets the whole
|
||||
// supplied config rejected with 0x57 ERROR_INVALID_PARAMETER on some
|
||||
// driver/topology combinations (field-reported: exclusive mode left the
|
||||
// physical panel lit, every retry failing 0x57). Writing the all-ones
|
||||
// sentinel to the whole union is also correct under the virtual-mode-aware
|
||||
// interpretation (cloneGroupId/sourceModeInfoIdx both become their 0xffff
|
||||
// INVALID values).
|
||||
p.flags &= !DISPLAYCONFIG_PATH_ACTIVE;
|
||||
p.sourceInfo.Anonymous.modeInfoIdx = DISPLAYCONFIG_PATH_MODE_IDX_INVALID;
|
||||
p.targetInfo.Anonymous.modeInfoIdx = DISPLAYCONFIG_PATH_MODE_IDX_INVALID;
|
||||
others += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@ use std::path::{Path, PathBuf};
|
||||
use std::process::{Child, Command};
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
use std::time::Duration;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// Handshake budget: EGL + CUDA bring-up is ~200 ms; a cold driver load can take seconds.
|
||||
const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(20);
|
||||
@@ -64,11 +64,27 @@ impl Drop for Shared {
|
||||
/// Children whose worker hasn't exited yet at `RemoteImporter` drop time (it exits on socket
|
||||
/// EOF, i.e. after the last in-flight frame drops). Swept on every spawn and every drop so
|
||||
/// workers don't linger as zombies for more than one capture generation.
|
||||
static REAPER: Mutex<Vec<Child>> = Mutex::new(Vec::new());
|
||||
static REAPER: Mutex<Vec<(Child, Instant)>> = Mutex::new(Vec::new());
|
||||
|
||||
/// How long past `REPLY_TIMEOUT` a parked worker may linger before it is force-killed. A worker
|
||||
/// wedged INSIDE a driver call never observes socket EOF, so `try_wait` alone would keep it (and
|
||||
/// its CUcontext + BufferPool — order hundreds of MB of VRAM) forever.
|
||||
const REAPER_KILL_DEADLINE: Duration = Duration::from_secs(20);
|
||||
|
||||
fn sweep_reaper() {
|
||||
let mut list = REAPER.lock().unwrap();
|
||||
list.retain_mut(|c| !matches!(c.try_wait(), Ok(Some(_))));
|
||||
let now = Instant::now();
|
||||
list.retain_mut(|(c, parked)| {
|
||||
if matches!(c.try_wait(), Ok(Some(_))) {
|
||||
return false; // exited on its own → reaped
|
||||
}
|
||||
if now.duration_since(*parked) > REAPER_KILL_DEADLINE {
|
||||
let _ = c.kill();
|
||||
let _ = c.wait();
|
||||
return false; // wedged past the deadline → force-killed + reaped
|
||||
}
|
||||
true
|
||||
});
|
||||
}
|
||||
|
||||
/// Fd pinned to this process's own executable image, opened (once, lazily) via the
|
||||
@@ -455,7 +471,7 @@ impl Drop for RemoteImporter {
|
||||
// gone; park the rest for the next sweep.
|
||||
if let Some(mut child) = self.child.take() {
|
||||
if !matches!(child.try_wait(), Ok(Some(_))) {
|
||||
REAPER.lock().unwrap().push(child);
|
||||
REAPER.lock().unwrap().push((child, Instant::now()));
|
||||
}
|
||||
}
|
||||
sweep_reaper();
|
||||
|
||||
@@ -251,6 +251,32 @@ unsafe fn copy_blocking(copy: &CUDA_MEMCPY2D, what: &str) -> Result<()> {
|
||||
ck(cuStreamSynchronize(stream), "cuStreamSynchronize")
|
||||
}
|
||||
|
||||
/// Issue `copy` on this thread's priority stream WITHOUT waiting — for stream-ordered consumers
|
||||
/// only (the direct-NVENC submit path with `NvEncSetIOCudaStreams` bound to this stream): the
|
||||
/// stream, not the CPU, orders completion, so the SOURCE must stay valid until the downstream
|
||||
/// stream work (the encode) has finished.
|
||||
unsafe fn copy_async(copy: &CUDA_MEMCPY2D, what: &str) -> Result<()> {
|
||||
ck(cuMemcpy2DAsync_v2(copy, copy_stream()), what)
|
||||
}
|
||||
|
||||
/// `copy_blocking` when `sync`, else `copy_async` — the shared tail of the public `copy_*_to_device`
|
||||
/// helpers, whose `sync: false` mode carries `copy_async`'s source-lifetime contract.
|
||||
unsafe fn copy_issue(copy: &CUDA_MEMCPY2D, what: &str, sync: bool) -> Result<()> {
|
||||
if sync {
|
||||
copy_blocking(copy, what)
|
||||
} else {
|
||||
copy_async(copy, what)
|
||||
}
|
||||
}
|
||||
|
||||
/// The calling thread's copy/launch stream as a raw handle, for binding external stream-ordering
|
||||
/// (the direct-NVENC `NvEncSetIOCudaStreams` hookup). Null = the NULL stream (priority-stream
|
||||
/// creation failed) — callers should treat null as "stream-ordering unavailable" and keep their
|
||||
/// blocking copies. The shared context must be current on this thread.
|
||||
pub fn copy_stream_handle() -> *mut c_void {
|
||||
copy_stream() // CUstream IS *mut c_void (opaque CUstream_st*)
|
||||
}
|
||||
|
||||
/// Max cursor-overlay bitmap edge (px) uploaded to the device blend buffer — matches the Vulkan path.
|
||||
pub const CURSOR_MAX: u32 = 256;
|
||||
|
||||
@@ -354,6 +380,7 @@ impl CursorBlend {
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_surf, mut a_cur) = (surf, self.cur_buf);
|
||||
let (mut a_pitch, mut a_w, mut a_h) = (pitch as i32, w as i32, h as i32);
|
||||
@@ -370,7 +397,7 @@ impl CursorBlend {
|
||||
&mut a_ox as *mut _ as *mut c_void,
|
||||
&mut a_oy as *mut _ as *mut c_void,
|
||||
];
|
||||
self.launch(self.f_argb, a_cw as u32, a_ch as u32, &mut args)
|
||||
self.launch(self.f_argb, a_cw as u32, a_ch as u32, &mut args, sync)
|
||||
}
|
||||
|
||||
/// Blend into an owned planar YUV444 surface (3 stacked full-res planes) at `(ox,oy)`.
|
||||
@@ -385,6 +412,7 @@ impl CursorBlend {
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_base, mut a_cur) = (base, self.cur_buf);
|
||||
let (mut a_pitch, mut a_w, mut a_h) = (pitch as i32, w as i32, h as i32);
|
||||
@@ -401,7 +429,7 @@ impl CursorBlend {
|
||||
&mut a_ox as *mut _ as *mut c_void,
|
||||
&mut a_oy as *mut _ as *mut c_void,
|
||||
];
|
||||
self.launch(self.f_yuv444, a_cw as u32, a_ch as u32, &mut args)
|
||||
self.launch(self.f_yuv444, a_cw as u32, a_ch as u32, &mut args, sync)
|
||||
}
|
||||
|
||||
/// Blend into an owned NV12 surface (Y plane at `base`, interleaved UV at `base + pitch*h`).
|
||||
@@ -416,6 +444,7 @@ impl CursorBlend {
|
||||
ch: u32,
|
||||
ox: i32,
|
||||
oy: i32,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (mut a_yb, mut a_uvb, mut a_cur) = (base, base + pitch as u64 * h as u64, self.cur_buf);
|
||||
let (mut a_yp, mut a_uvp) = (pitch as i32, pitch as i32);
|
||||
@@ -441,16 +470,20 @@ impl CursorBlend {
|
||||
(a_cw as u32).div_ceil(2),
|
||||
(a_ch as u32).div_ceil(2),
|
||||
&mut args,
|
||||
sync,
|
||||
)
|
||||
}
|
||||
|
||||
/// Launch `f` over a `work_w × work_h` grid (16×16 blocks) on the copy stream, then synchronize.
|
||||
/// Launch `f` over a `work_w × work_h` grid (16×16 blocks) on the copy stream; `sync` waits
|
||||
/// for it, `!sync` leaves completion to the stream (stream-ordered consumers only — the
|
||||
/// kernel PARAMETERS are copied at launch time, so the arg locals need not outlive the call).
|
||||
fn launch(
|
||||
&self,
|
||||
f: CUfunction,
|
||||
work_w: u32,
|
||||
work_h: u32,
|
||||
args: &mut [*mut c_void],
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
if work_w == 0 || work_h == 0 {
|
||||
return Ok(());
|
||||
@@ -458,9 +491,11 @@ impl CursorBlend {
|
||||
const B: u32 = 16;
|
||||
let stream = copy_stream();
|
||||
// SAFETY: `f` is a resolved kernel from our loaded module; `args` holds pointers to live
|
||||
// locals whose types match the kernel's C parameters (per the call site above); grid/block
|
||||
// dims are non-zero. Launched on the copy stream (ordered after the input-surface copy that
|
||||
// `copy_into_slot` already synchronized) then synchronized. Requires the context current.
|
||||
// locals whose types match the kernel's C parameters (per the call site above) — CUDA
|
||||
// copies the parameter values during `cuLaunchKernel` itself, so they need not outlive
|
||||
// the call. Grid/block dims are non-zero. Launched on the copy stream (ordered after the
|
||||
// input-surface copy issued on the same stream); `sync` waits, `!sync` leaves ordering to
|
||||
// the stream (the NVENC IO-stream binding). Requires the context current.
|
||||
unsafe {
|
||||
ck(
|
||||
cuLaunchKernel(
|
||||
@@ -478,7 +513,10 @@ impl CursorBlend {
|
||||
),
|
||||
"cuLaunchKernel(cursor)",
|
||||
)?;
|
||||
ck(cuStreamSynchronize(stream), "cuStreamSynchronize(cursor)")
|
||||
if sync {
|
||||
ck(cuStreamSynchronize(stream), "cuStreamSynchronize(cursor)")?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1003,7 +1041,13 @@ impl RegisteredTexture {
|
||||
// SAFETY: `self.resource` is the valid `CUgraphicsResource` from a successful `register_gl`
|
||||
// (its only constructor), so the wrappers forward to the live table; the caller holds the
|
||||
// GL+CUDA contexts current (the registration's contract). `cuGraphicsMapResources` maps
|
||||
// `count == 1` resource via `&mut self.resource` (a live field) on the default stream;
|
||||
// `count == 1` resource via `&mut self.resource` (a live field). It is issued on
|
||||
// `copy_stream()` — NOT the NULL stream — because map's only ordering guarantee is that
|
||||
// prior GL work completes before subsequent CUDA work issued IN THE STREAM PASSED TO IT;
|
||||
// the copy below runs on `copy_stream()` (a `CU_STREAM_NON_BLOCKING` stream, exempt from
|
||||
// implicit NULL-stream ordering), so mapping on NULL left the copy free to race the GL
|
||||
// de-tile/CSC that produced this texture (glFlush only, no fence) — intermittent torn or
|
||||
// stale frames under GPU load. Map, copy, and unmap now all share `copy_stream()`.
|
||||
// `cuGraphicsSubResourceGetMappedArray` writes the mapped `CUarray` into the live local
|
||||
// `array` (index 0, mip 0). On failure we unmap and bail (balanced). `©` is a live
|
||||
// local `CUDA_MEMCPY2D` outliving the synchronous `copy_blocking`: `srcArray` is valid
|
||||
@@ -1012,12 +1056,12 @@ impl RegisteredTexture {
|
||||
// we always unmap afterward (even on error), keeping the map/unmap pair balanced.
|
||||
unsafe {
|
||||
ck(
|
||||
cuGraphicsMapResources(1, &mut self.resource, std::ptr::null_mut()),
|
||||
cuGraphicsMapResources(1, &mut self.resource, copy_stream()),
|
||||
"cuGraphicsMapResources",
|
||||
)?;
|
||||
let mut array: CUarray = std::ptr::null_mut();
|
||||
if cuGraphicsSubResourceGetMappedArray(&mut array, self.resource, 0, 0) != 0 {
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, std::ptr::null_mut());
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, copy_stream());
|
||||
bail!("cuGraphicsSubResourceGetMappedArray failed");
|
||||
}
|
||||
let copy = CUDA_MEMCPY2D {
|
||||
@@ -1031,7 +1075,7 @@ impl RegisteredTexture {
|
||||
..Default::default()
|
||||
};
|
||||
let res = copy_blocking(©, "cuMemcpy2DAsync_v2");
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, std::ptr::null_mut());
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, copy_stream());
|
||||
res
|
||||
}
|
||||
}
|
||||
@@ -1058,12 +1102,12 @@ impl RegisteredTexture {
|
||||
// so the map/unmap pair stays balanced and the array outlives the copy.
|
||||
unsafe {
|
||||
ck(
|
||||
cuGraphicsMapResources(1, &mut self.resource, std::ptr::null_mut()),
|
||||
cuGraphicsMapResources(1, &mut self.resource, copy_stream()),
|
||||
"cuGraphicsMapResources",
|
||||
)?;
|
||||
let mut array: CUarray = std::ptr::null_mut();
|
||||
if cuGraphicsSubResourceGetMappedArray(&mut array, self.resource, 0, 0) != 0 {
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, std::ptr::null_mut());
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, copy_stream());
|
||||
bail!("cuGraphicsSubResourceGetMappedArray failed");
|
||||
}
|
||||
let copy = CUDA_MEMCPY2D {
|
||||
@@ -1077,7 +1121,7 @@ impl RegisteredTexture {
|
||||
..Default::default()
|
||||
};
|
||||
let res = copy_blocking(©, "cuMemcpy2DAsync_v2(plane)");
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, std::ptr::null_mut());
|
||||
let _ = cuGraphicsUnmapResources(1, &mut self.resource, copy_stream());
|
||||
res
|
||||
}
|
||||
}
|
||||
@@ -1122,10 +1166,13 @@ pub fn copy_mapped_yuv444(
|
||||
/// Copy a pitched device buffer into another device region (device→device), e.g. our imported
|
||||
/// [`DeviceBuffer`] into a pooled CUDA surface NVENC owns. Both are 4-byte (BGRx) pixels.
|
||||
/// The caller must have the shared context current on this thread (see [`make_current`]).
|
||||
/// `sync: false` enqueues without a CPU wait (stream-ordered consumers only — `src` must stay
|
||||
/// valid until the downstream stream work completes; see [`copy_stream_handle`]).
|
||||
pub fn copy_device_to_device(
|
||||
src: &DeviceBuffer,
|
||||
dst_ptr: CUdeviceptr,
|
||||
dst_pitch: usize,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let copy = CUDA_MEMCPY2D {
|
||||
srcMemoryType: CU_MEMORYTYPE_DEVICE,
|
||||
@@ -1138,23 +1185,27 @@ pub fn copy_device_to_device(
|
||||
Height: src.height as usize,
|
||||
..Default::default()
|
||||
};
|
||||
// SAFETY: `copy_blocking` is unsafe (issues a CUDA copy); the caller must have the shared
|
||||
// SAFETY: `copy_issue` is unsafe (issues a CUDA copy); the caller must have the shared
|
||||
// context current (documented). `©` is a live local device→device `CUDA_MEMCPY2D` outliving
|
||||
// the synchronous call: `srcDevice`/`srcPitch` are `src`'s live allocation, `dstDevice`/
|
||||
// `dstPitch` the caller's live region, `width*4`×`height` within both. Wrapper → live table.
|
||||
unsafe { copy_blocking(©, "cuMemcpy2DAsync_v2(dev->dev)") }
|
||||
// the enqueue: `srcDevice`/`srcPitch` are `src`'s live allocation, `dstDevice`/`dstPitch` the
|
||||
// caller's live region, `width*4`×`height` within both; `sync: false` shifts the source-
|
||||
// lifetime obligation to the caller (documented above). Wrapper → live table.
|
||||
unsafe { copy_issue(©, "cuMemcpy2DAsync_v2(dev->dev)", sync) }
|
||||
}
|
||||
|
||||
/// Copy our imported NV12 [`DeviceBuffer`] (Y + UV planes) into NVENC's two-plane CUDA surface
|
||||
/// `(y_dst, y_pitch)` / `(uv_dst, uv_pitch)` (`av_hwframe_get_buffer`'s `data[0]`/`data[1]` +
|
||||
/// `linesize[0]`/`linesize[1]`). The Y plane is `width`×`height` bytes; the chroma plane is
|
||||
/// `(width/2)·2` bytes × `height/2` rows. The caller must have the shared context current.
|
||||
/// `sync: false` enqueues without a CPU wait (stream-ordered consumers only — `src` must stay
|
||||
/// valid until the downstream stream work completes; see [`copy_stream_handle`]).
|
||||
pub fn copy_nv12_to_device(
|
||||
src: &DeviceBuffer,
|
||||
y_dst: CUdeviceptr,
|
||||
y_pitch: usize,
|
||||
uv_dst: CUdeviceptr,
|
||||
uv_pitch: usize,
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
let (src_uv_ptr, src_uv_pitch) = src
|
||||
.uv
|
||||
@@ -1183,15 +1234,16 @@ pub fn copy_nv12_to_device(
|
||||
Height: h / 2,
|
||||
..Default::default()
|
||||
};
|
||||
// SAFETY: two unsafe `copy_blocking` device→device copies; the caller must have the shared
|
||||
// SAFETY: two unsafe `copy_issue` device→device copies; the caller must have the shared
|
||||
// context current (documented). `&y`/`&uv` are live local `CUDA_MEMCPY2D`s outliving each
|
||||
// synchronous call. All four device pointers are valid: `src.ptr`/`src_uv_ptr` come from a live
|
||||
// enqueue. All four device pointers are valid: `src.ptr`/`src_uv_ptr` come from a live
|
||||
// NV12 `DeviceBuffer` (its `.uv` presence was checked via `ok_or_else`), `y_dst`/`uv_dst` are
|
||||
// the caller's live NVENC surface planes; the luma copy is `w`×`h`, the chroma copy
|
||||
// `(w/2)*2`×`h/2`, each within its planes. Wrappers → live table.
|
||||
// `(w/2)*2`×`h/2`, each within its planes; `sync: false` shifts the source-lifetime obligation
|
||||
// to the caller (documented above). Wrappers → live table.
|
||||
unsafe {
|
||||
copy_blocking(&y, "cuMemcpy2DAsync_v2(nv12 Y dev->dev)")?;
|
||||
copy_blocking(&uv, "cuMemcpy2DAsync_v2(nv12 UV dev->dev)")
|
||||
copy_issue(&y, "cuMemcpy2DAsync_v2(nv12 Y dev->dev)", sync)?;
|
||||
copy_issue(&uv, "cuMemcpy2DAsync_v2(nv12 UV dev->dev)", sync)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1199,7 +1251,13 @@ pub fn copy_nv12_to_device(
|
||||
/// (`av_hwframe_get_buffer`'s `data[0..3]` + `linesize[0..3]` for a `yuv444p` frames context).
|
||||
/// Each plane is `width`×`height` bytes; the source planes sit at row offsets `0/H/2H` of the
|
||||
/// single allocation. The caller must have the shared context current.
|
||||
pub fn copy_yuv444_to_device(src: &DeviceBuffer, dsts: [(CUdeviceptr, usize); 3]) -> Result<()> {
|
||||
/// `sync: false` enqueues without a CPU wait (stream-ordered consumers only — `src` must stay
|
||||
/// valid until the downstream stream work completes; see [`copy_stream_handle`]).
|
||||
pub fn copy_yuv444_to_device(
|
||||
src: &DeviceBuffer,
|
||||
dsts: [(CUdeviceptr, usize); 3],
|
||||
sync: bool,
|
||||
) -> Result<()> {
|
||||
anyhow::ensure!(src.yuv444, "copy_yuv444_to_device on a non-YUV444 buffer");
|
||||
let w = src.width as usize;
|
||||
let h = src.height as usize;
|
||||
@@ -1215,12 +1273,13 @@ pub fn copy_yuv444_to_device(src: &DeviceBuffer, dsts: [(CUdeviceptr, usize); 3]
|
||||
Height: h,
|
||||
..Default::default()
|
||||
};
|
||||
// SAFETY: unsafe `copy_blocking` device→device copy; the caller must have the shared
|
||||
// context current (documented). `©` is a live local outliving the synchronous call;
|
||||
// SAFETY: unsafe `copy_issue` device→device copy; the caller must have the shared
|
||||
// context current (documented). `©` is a live local outliving the enqueue;
|
||||
// `src.ptr + pitch·h·i` stays within the live 3·H-row stacked allocation (`yuv444`
|
||||
// checked above), `dst_ptr`/`dst_pitch` is the caller's live NVENC plane; `w`×`h` fits
|
||||
// both. Wrapper → live table.
|
||||
unsafe { copy_blocking(©, "cuMemcpy2DAsync_v2(yuv444 plane dev->dev)")? };
|
||||
// both; `sync: false` shifts the source-lifetime obligation to the caller (documented
|
||||
// above). Wrapper → live table.
|
||||
unsafe { copy_issue(©, "cuMemcpy2DAsync_v2(yuv444 plane dev->dev)", sync)? };
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -691,6 +691,15 @@ impl EglImporter {
|
||||
width: u32,
|
||||
height: u32,
|
||||
) -> Result<DeviceBuffer> {
|
||||
// Even dimensions only: the UV copy walks `height.div_ceil(2)` chroma rows (the correct NV12
|
||||
// count), but the pooled UV plane is sized at `height/2` rows — for an odd height those
|
||||
// disagree by one row and the copy writes a full `uv_pitch` past the allocation (OOB device
|
||||
// write / CUDA_ERROR_ILLEGAL_ADDRESS that poisons the shared context). Reject here, matching
|
||||
// the guards `Nv12Blit::new`/`Yuv444Blit::new` already carry.
|
||||
anyhow::ensure!(
|
||||
width % 2 == 0 && height % 2 == 0,
|
||||
"LINEAR NV12 needs even dimensions (got {width}x{height})"
|
||||
);
|
||||
cuda::make_current()?;
|
||||
if self
|
||||
.linear_nv12_pool
|
||||
|
||||
@@ -86,9 +86,11 @@ impl VkBridge {
|
||||
// SAFETY: standard ash bring-up — every call is `unsafe` only because ash cannot statically
|
||||
// verify Vulkan handle/CreateInfo validity. `ash::Entry::load` dlopens a real system
|
||||
// libvulkan. Each `*CreateInfo`/`AllocateInfo` is built by ash's builders from locals (`app`,
|
||||
// `exts`, `prio`, `qci`, and the inline infos) that all live for the duration of the
|
||||
// synchronous `create_*`/`enumerate_*` call that reads them — in particular the
|
||||
// `enabled_extension_names(&exts)` and `queue_priorities(&prio)` borrows outlive their calls.
|
||||
// `exts`, `prio`, `qci`, `gp_info`, and the inline infos) that all live for the duration of
|
||||
// the synchronous `create_*`/`enumerate_*` call that reads them — the ladder loop rebuilds
|
||||
// `prio`/`gp_info`/`qci`/`exts` fresh per attempt, so every `enabled_extension_names(&exts)`
|
||||
// / `queue_priorities(&prio)` / `push_next(&mut gp_info)` borrow outlives its own
|
||||
// `create_device` call.
|
||||
// Every handle passed (`instance`, `phys`, `device`, `qf`, `cmd_pool`) was just created and
|
||||
// checked via `?`/`ok_or_else` in this same function, so no invalid handle is ever used. This
|
||||
// constructor shares nothing across threads.
|
||||
@@ -122,23 +124,93 @@ impl VkBridge {
|
||||
.ok_or_else(|| anyhow!("no compute-capable queue family"))?
|
||||
as u32;
|
||||
|
||||
let exts = [
|
||||
// Global-priority queue (latency plan §7 LN4, PyroWave's `ac0e7332` lever for the
|
||||
// VkBridge): the LINEAR/gamescope CSC dispatch shares the SM/compute cores with the
|
||||
// game, so ask for an elevated global priority to get scheduled ahead of it.
|
||||
// `PUNKTFUNK_VK_QUEUE_PRIORITY` = off | high | realtime (default realtime); the
|
||||
// create loop downgrades REALTIME→HIGH→none on NOT_PERMITTED (and retries a plain
|
||||
// create on INITIALIZATION_FAILED) so a refused class never fails the bridge.
|
||||
let gp_ext = std::env::var("PUNKTFUNK_VK_QUEUE_PRIORITY")
|
||||
.ok()
|
||||
.as_deref()
|
||||
.map_or(Some(vk::QueueGlobalPriorityKHR::REALTIME), |v| match v {
|
||||
"off" | "0" => None,
|
||||
"high" => Some(vk::QueueGlobalPriorityKHR::HIGH),
|
||||
_ => Some(vk::QueueGlobalPriorityKHR::REALTIME),
|
||||
})
|
||||
.and_then(|want| {
|
||||
// Enable whichever alias the driver advertises (KHR = the promoted name).
|
||||
let props = instance.enumerate_device_extension_properties(phys).ok()?;
|
||||
let has = |name: &std::ffi::CStr| {
|
||||
props
|
||||
.iter()
|
||||
.any(|p| p.extension_name_as_c_str() == Ok(name))
|
||||
};
|
||||
if has(vk::KHR_GLOBAL_PRIORITY_NAME) {
|
||||
Some((vk::KHR_GLOBAL_PRIORITY_NAME, want))
|
||||
} else if has(vk::EXT_GLOBAL_PRIORITY_NAME) {
|
||||
Some((vk::EXT_GLOBAL_PRIORITY_NAME, want))
|
||||
} else {
|
||||
None
|
||||
}
|
||||
});
|
||||
let base_exts = [
|
||||
ash::khr::external_memory_fd::NAME.as_ptr(),
|
||||
ash::ext::external_memory_dma_buf::NAME.as_ptr(),
|
||||
];
|
||||
let mut try_priority = gp_ext.map(|(_, want)| want);
|
||||
let device = loop {
|
||||
let prio = [1.0f32];
|
||||
let qci = [vk::DeviceQueueCreateInfo::default()
|
||||
let mut gp_info = vk::DeviceQueueGlobalPriorityCreateInfoKHR::default()
|
||||
.global_priority(try_priority.unwrap_or(vk::QueueGlobalPriorityKHR::MEDIUM));
|
||||
let mut qci0 = vk::DeviceQueueCreateInfo::default()
|
||||
.queue_family_index(qf)
|
||||
.queue_priorities(&prio)];
|
||||
let device = instance
|
||||
.create_device(
|
||||
.queue_priorities(&prio);
|
||||
let mut exts: Vec<*const std::ffi::c_char> = base_exts.to_vec();
|
||||
if try_priority.is_some() {
|
||||
qci0 = qci0.push_next(&mut gp_info);
|
||||
exts.push(gp_ext.expect("try_priority implies gp_ext").0.as_ptr());
|
||||
}
|
||||
let qci = [qci0];
|
||||
match instance.create_device(
|
||||
phys,
|
||||
&vk::DeviceCreateInfo::default()
|
||||
.queue_create_infos(&qci)
|
||||
.enabled_extension_names(&exts),
|
||||
None,
|
||||
)
|
||||
.context("vkCreateDevice (external-memory extensions supported?)")?;
|
||||
) {
|
||||
Ok(d) => {
|
||||
if let Some(p) = try_priority {
|
||||
tracing::info!(
|
||||
priority = ?p,
|
||||
"VkBridge queue at elevated global priority (CSC schedules \
|
||||
ahead of a GPU-bound game where the driver honors it)"
|
||||
);
|
||||
}
|
||||
break d;
|
||||
}
|
||||
// A refused class must never fail the bridge — walk the ladder down.
|
||||
Err(
|
||||
vk::Result::ERROR_NOT_PERMITTED_KHR
|
||||
| vk::Result::ERROR_INITIALIZATION_FAILED,
|
||||
) if try_priority == Some(vk::QueueGlobalPriorityKHR::REALTIME) => {
|
||||
try_priority = Some(vk::QueueGlobalPriorityKHR::HIGH);
|
||||
}
|
||||
Err(
|
||||
vk::Result::ERROR_NOT_PERMITTED_KHR
|
||||
| vk::Result::ERROR_INITIALIZATION_FAILED,
|
||||
) if try_priority.is_some() => {
|
||||
tracing::debug!(
|
||||
"global-priority queue not permitted — VkBridge at default priority"
|
||||
);
|
||||
try_priority = None;
|
||||
}
|
||||
Err(e) => {
|
||||
return Err(e)
|
||||
.context("vkCreateDevice (external-memory extensions supported?)")
|
||||
}
|
||||
}
|
||||
};
|
||||
let ext_fd = ash::khr::external_memory_fd::Device::new(&instance, &device);
|
||||
let queue = device.get_device_queue(qf, 0);
|
||||
|
||||
@@ -193,10 +265,17 @@ impl VkBridge {
|
||||
|
||||
/// Import `fd` (dup'd internally; Vulkan owns the dup) as a transfer-src buffer of `size`.
|
||||
unsafe fn import_src(&mut self, fd: i32, size: u64) -> Result<()> {
|
||||
use std::os::fd::{AsRawFd, FromRawFd, IntoRawFd, OwnedFd};
|
||||
let dup = libc::dup(fd);
|
||||
if dup < 0 {
|
||||
bail!("dup(dmabuf fd)");
|
||||
}
|
||||
// Own the dup so every early return BEFORE Vulkan consumes it (at `allocate_memory` success)
|
||||
// closes it. `SrcBuf` holds raw handles with no Drop and is only populated on the success
|
||||
// path, so each fallible step below must also destroy the buffer it created — otherwise a
|
||||
// failed import (which the worker survives and the caller retries every frame) leaks a
|
||||
// VkBuffer + VkDeviceMemory + fd per frame for the worker's whole lifetime.
|
||||
let dup = OwnedFd::from_raw_fd(dup);
|
||||
let mut ext_info = vk::ExternalMemoryBufferCreateInfo::default()
|
||||
.handle_types(vk::ExternalMemoryHandleTypeFlags::DMA_BUF_EXT);
|
||||
let buffer = self
|
||||
@@ -212,41 +291,55 @@ impl VkBridge {
|
||||
.push_next(&mut ext_info),
|
||||
None,
|
||||
)
|
||||
.context("create import buffer")?;
|
||||
.context("create import buffer")?; // `dup` drops → closes on failure
|
||||
let mut fd_props = vk::MemoryFdPropertiesKHR::default();
|
||||
self.ext_fd
|
||||
.get_memory_fd_properties(
|
||||
if let Err(e) = self.ext_fd.get_memory_fd_properties(
|
||||
vk::ExternalMemoryHandleTypeFlags::DMA_BUF_EXT,
|
||||
dup,
|
||||
dup.as_raw_fd(),
|
||||
&mut fd_props,
|
||||
)
|
||||
.context("vkGetMemoryFdPropertiesKHR")?;
|
||||
) {
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("vkGetMemoryFdPropertiesKHR");
|
||||
}
|
||||
let reqs = self.device.get_buffer_memory_requirements(buffer);
|
||||
let mem_type = self.memory_type(
|
||||
let mem_type = match self.memory_type(
|
||||
reqs.memory_type_bits & fd_props.memory_type_bits,
|
||||
vk::MemoryPropertyFlags::empty(),
|
||||
)?;
|
||||
) {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
// Vulkan takes ownership of the fd on a SUCCESSFUL import: hand over the raw fd now, and on
|
||||
// failure close it ourselves (matching the original contract) plus destroy the buffer.
|
||||
let raw = dup.into_raw_fd();
|
||||
let mut import = vk::ImportMemoryFdInfoKHR::default()
|
||||
.handle_type(vk::ExternalMemoryHandleTypeFlags::DMA_BUF_EXT)
|
||||
.fd(dup); // Vulkan takes ownership of `dup` on success
|
||||
.fd(raw);
|
||||
let mut dedicated = vk::MemoryDedicatedAllocateInfo::default().buffer(buffer);
|
||||
let memory = self
|
||||
.device
|
||||
.allocate_memory(
|
||||
let memory = match self.device.allocate_memory(
|
||||
&vk::MemoryAllocateInfo::default()
|
||||
.allocation_size(reqs.size.max(size))
|
||||
.memory_type_index(mem_type)
|
||||
.push_next(&mut import)
|
||||
.push_next(&mut dedicated),
|
||||
None,
|
||||
)
|
||||
.map_err(|e| {
|
||||
libc::close(dup); // failed import does not consume the fd
|
||||
anyhow!("import dmabuf memory: {e}")
|
||||
})?;
|
||||
self.device
|
||||
.bind_buffer_memory(buffer, memory, 0)
|
||||
.context("bind import memory")?;
|
||||
) {
|
||||
Ok(m) => m,
|
||||
Err(e) => {
|
||||
libc::close(raw); // failed import does not consume the fd
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(anyhow!("import dmabuf memory: {e}"));
|
||||
}
|
||||
};
|
||||
if let Err(e) = self.device.bind_buffer_memory(buffer, memory, 0) {
|
||||
// `memory` owns the imported fd — freeing it releases the fd too.
|
||||
self.device.free_memory(memory, None);
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("bind import memory");
|
||||
}
|
||||
self.src_cache.insert(
|
||||
fd,
|
||||
SrcBuf {
|
||||
@@ -263,11 +356,11 @@ impl VkBridge {
|
||||
if self.dst.as_ref().is_some_and(|d| d.size >= size) {
|
||||
return Ok(());
|
||||
}
|
||||
if let Some(old) = self.dst.take() {
|
||||
self.device.destroy_buffer(old.buffer, None);
|
||||
self.device.free_memory(old.memory, None);
|
||||
// old.cuda drops its mapping with it
|
||||
}
|
||||
// Build the replacement FULLY before retiring the old one. Previously the old dst was
|
||||
// destroyed and `self.dst` nulled up front, so a failed rebuild both dropped the working
|
||||
// buffer AND leaked every object the partial rebuild created (`buffer`/`memory` are raw ash
|
||||
// handles with no Drop, and `VkBridge::drop` only frees the live `self.dst`). Now every
|
||||
// fallible step unwinds locally, and the swap happens only on full success.
|
||||
let mut ext_info = vk::ExternalMemoryBufferCreateInfo::default()
|
||||
.handle_types(vk::ExternalMemoryHandleTypeFlags::OPAQUE_FD);
|
||||
let buffer = self
|
||||
@@ -285,35 +378,63 @@ impl VkBridge {
|
||||
.context("create export buffer")?;
|
||||
let reqs = self.device.get_buffer_memory_requirements(buffer);
|
||||
let mem_type =
|
||||
self.memory_type(reqs.memory_type_bits, vk::MemoryPropertyFlags::DEVICE_LOCAL)?;
|
||||
match self.memory_type(reqs.memory_type_bits, vk::MemoryPropertyFlags::DEVICE_LOCAL) {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e);
|
||||
}
|
||||
};
|
||||
let mut export = vk::ExportMemoryAllocateInfo::default()
|
||||
.handle_types(vk::ExternalMemoryHandleTypeFlags::OPAQUE_FD);
|
||||
let mut dedicated = vk::MemoryDedicatedAllocateInfo::default().buffer(buffer);
|
||||
let memory = self
|
||||
.device
|
||||
.allocate_memory(
|
||||
let memory = match self.device.allocate_memory(
|
||||
&vk::MemoryAllocateInfo::default()
|
||||
.allocation_size(reqs.size)
|
||||
.memory_type_index(mem_type)
|
||||
.push_next(&mut export)
|
||||
.push_next(&mut dedicated),
|
||||
None,
|
||||
)
|
||||
.context("allocate exportable memory")?;
|
||||
self.device
|
||||
.bind_buffer_memory(buffer, memory, 0)
|
||||
.context("bind export memory")?;
|
||||
let opaque_fd = self
|
||||
.ext_fd
|
||||
.get_memory_fd(
|
||||
) {
|
||||
Ok(m) => m,
|
||||
Err(e) => {
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("allocate exportable memory");
|
||||
}
|
||||
};
|
||||
if let Err(e) = self.device.bind_buffer_memory(buffer, memory, 0) {
|
||||
self.device.free_memory(memory, None);
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("bind export memory");
|
||||
}
|
||||
let opaque_fd = match self.ext_fd.get_memory_fd(
|
||||
&vk::MemoryGetFdInfoKHR::default()
|
||||
.memory(memory)
|
||||
.handle_type(vk::ExternalMemoryHandleTypeFlags::OPAQUE_FD),
|
||||
)
|
||||
.context("vkGetMemoryFdKHR")?;
|
||||
) {
|
||||
Ok(f) => f,
|
||||
Err(e) => {
|
||||
self.device.free_memory(memory, None);
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("vkGetMemoryFdKHR");
|
||||
}
|
||||
};
|
||||
// CUDA imports (and on success owns) the exported fd. Size must match the allocation.
|
||||
let cuda = cuda::ExternalDmabuf::import_owned_fd(opaque_fd, reqs.size)
|
||||
.context("cuImportExternalMemory(OPAQUE_FD from Vulkan)")?;
|
||||
// `import_owned_fd` closes `opaque_fd` on its own failure, so only the Vulkan objects unwind.
|
||||
let cuda = match cuda::ExternalDmabuf::import_owned_fd(opaque_fd, reqs.size) {
|
||||
Ok(c) => c,
|
||||
Err(e) => {
|
||||
self.device.free_memory(memory, None);
|
||||
self.device.destroy_buffer(buffer, None);
|
||||
return Err(e).context("cuImportExternalMemory(OPAQUE_FD from Vulkan)");
|
||||
}
|
||||
};
|
||||
// Full success: retire the previous buffer now, then publish the new one.
|
||||
if let Some(old) = self.dst.take() {
|
||||
self.device.destroy_buffer(old.buffer, None);
|
||||
self.device.free_memory(old.memory, None);
|
||||
// old.cuda drops its mapping with it
|
||||
}
|
||||
tracing::info!(size, "Vulkan→CUDA exportable staging buffer ready");
|
||||
self.dst = Some(DstBuf {
|
||||
buffer,
|
||||
@@ -544,9 +665,19 @@ impl VkBridge {
|
||||
self.device
|
||||
.queue_submit(self.queue, &[submit], self.fence)
|
||||
.context("queue submit")?;
|
||||
self.device
|
||||
// Exception-safe wait: a TIMEOUT/DEVICE_LOST must not `?` out with the submission still
|
||||
// executing — `self.cmd` and `self.fence` are reused every frame, and the caller retries
|
||||
// on the SAME bridge (and `ensure_dst` later destroys `dst.buffer` assuming no in-flight
|
||||
// work references it). Drain the GPU and reset the fence before propagating so the shared
|
||||
// cmd/fence return clean.
|
||||
if let Err(e) = self
|
||||
.device
|
||||
.wait_for_fences(&[self.fence], true, 1_000_000_000)
|
||||
.context("fence wait")?;
|
||||
{
|
||||
let _ = self.device.device_wait_idle();
|
||||
let _ = self.device.reset_fences(&[self.fence]);
|
||||
return Err(e).context("fence wait");
|
||||
}
|
||||
self.device
|
||||
.reset_fences(&[self.fence])
|
||||
.context("reset fence")?;
|
||||
@@ -639,9 +770,19 @@ impl VkBridge {
|
||||
self.device
|
||||
.queue_submit(self.queue, &[submit], self.fence)
|
||||
.context("queue submit")?;
|
||||
self.device
|
||||
// Exception-safe wait: a TIMEOUT/DEVICE_LOST must not `?` out with the submission still
|
||||
// executing — `self.cmd` and `self.fence` are reused every frame, and the caller retries
|
||||
// on the SAME bridge (and `ensure_dst` later destroys `dst.buffer` assuming no in-flight
|
||||
// work references it). Drain the GPU and reset the fence before propagating so the shared
|
||||
// cmd/fence return clean.
|
||||
if let Err(e) = self
|
||||
.device
|
||||
.wait_for_fences(&[self.fence], true, 1_000_000_000)
|
||||
.context("fence wait")?;
|
||||
{
|
||||
let _ = self.device.device_wait_idle();
|
||||
let _ = self.device.reset_fences(&[self.fence]);
|
||||
return Err(e).context("fence wait");
|
||||
}
|
||||
self.device
|
||||
.reset_fences(&[self.fence])
|
||||
.context("reset fence")?;
|
||||
|
||||
@@ -33,6 +33,11 @@ reed-solomon-simd = "3.1" # GF(2^16) Leopard-RS, SIMD, O(n log n) — the w
|
||||
# NOT interoperable.) See vendor/fec-rs/LICENSE (BSD-2-Clause).
|
||||
fec-rs = { path = "vendor/fec-rs" }
|
||||
aes-gcm = "0.10" # AES-128-GCM session crypto, matches GameStream
|
||||
# ChaCha20-Poly1305 session crypto, negotiated by clients without hardware AES (the soft-AES
|
||||
# armv7 targets — webOS TVs — where GCM caps decrypt at ~100 Mbps; ARX runs 4-7x faster there).
|
||||
# Same RustCrypto `aead 0.5` generation as aes-gcm: identical trait/nonce/tag shapes, pure Rust,
|
||||
# cross-compiles like aes-gcm (no cmake). See design/chacha20-session-cipher.md.
|
||||
chacha20poly1305 = "0.10"
|
||||
zerocopy = { version = "0.8", features = ["derive"] }
|
||||
bytes = "1"
|
||||
socket2 = { version = "0.6", features = [
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
//! Tier-1 microbenchmarks for the punktfunk/1 hot path — GPU-free, so they run in normal CI.
|
||||
//!
|
||||
//! Two layers:
|
||||
//! - `crypto/*` — the isolated AES-128-GCM primitives on one ~MTU shard.
|
||||
//! - `crypto/*` — the isolated AEAD primitives (AES-128-GCM + the negotiated
|
||||
//! ChaCha20-Poly1305) on one ~MTU shard.
|
||||
//! - `pipeline/*`— a whole frame through the real per-frame path end to end over the in-process
|
||||
//! loopback transport: FEC encode → AES-GCM seal → packetize → (loopback) → reassemble →
|
||||
//! FEC decode → open. This is what a throughput/latency regression in the core would show up in.
|
||||
@@ -11,11 +12,11 @@
|
||||
|
||||
use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput};
|
||||
use punktfunk_core::config::{Config, FecConfig, FecScheme, ProtocolPhase, Role};
|
||||
use punktfunk_core::crypto::SessionCrypto;
|
||||
use punktfunk_core::crypto::{SessionCrypto, SessionKey};
|
||||
use punktfunk_core::session::Session;
|
||||
use punktfunk_core::transport::loopback_pair;
|
||||
|
||||
const TAG_LEN: usize = 16; // AES-GCM authentication tag
|
||||
const TAG_LEN: usize = 16; // AEAD authentication tag (GCM and Poly1305 share the size)
|
||||
const SHARD: usize = punktfunk_core::config::mtu1500_shard_payload(); // one MTU-safe data shard
|
||||
|
||||
fn cfg(role: Role, scheme: FecScheme) -> Config {
|
||||
@@ -38,21 +39,29 @@ fn cfg(role: Role, scheme: FecScheme) -> Config {
|
||||
shard_payload: SHARD,
|
||||
max_frame_bytes: 8 * 1024 * 1024,
|
||||
encrypt: true, // bench the real path — crypto is always on for punktfunk/1
|
||||
key: [7u8; 16],
|
||||
key: SessionKey::Aes128Gcm([7u8; 16]),
|
||||
salt: [1, 2, 3, 4],
|
||||
loopback_drop_period: 0, // throughput run: no induced loss (loss-harness covers recovery)
|
||||
}
|
||||
}
|
||||
|
||||
fn bench_crypto(c: &mut Criterion) {
|
||||
let host = SessionCrypto::new(&[7u8; 16], [1, 2, 3, 4], Role::Host);
|
||||
let client = SessionCrypto::new(&[7u8; 16], [1, 2, 3, 4], Role::Client);
|
||||
let mut g = c.benchmark_group("crypto");
|
||||
g.throughput(Throughput::Bytes(SHARD as u64));
|
||||
// Both negotiated session AEADs. On the x86 / Apple Silicon this runs on, both must be
|
||||
// line-rate-trivial — the chacha20 series is the host-side sealing-cost check for the
|
||||
// negotiated soft-AES-armv7 path (design/chacha20-session-cipher.md §7). The AES series
|
||||
// keeps its unsuffixed names so the CI regression compare retains its history.
|
||||
for (suffix, key) in [
|
||||
("", SessionKey::Aes128Gcm([7u8; 16])),
|
||||
("_chacha20", SessionKey::ChaCha20Poly1305([7u8; 32])),
|
||||
] {
|
||||
let host = SessionCrypto::new(&key, [1, 2, 3, 4], Role::Host);
|
||||
let client = SessionCrypto::new(&key, [1, 2, 3, 4], Role::Client);
|
||||
let payload = vec![0xABu8; SHARD];
|
||||
let sealed = host.seal(0, &payload).unwrap();
|
||||
|
||||
let mut g = c.benchmark_group("crypto");
|
||||
g.throughput(Throughput::Bytes(SHARD as u64));
|
||||
g.bench_function("seal", |b| {
|
||||
g.bench_function(format!("seal{suffix}"), |b| {
|
||||
let mut seq = 0u64;
|
||||
b.iter(|| {
|
||||
let ct = host.seal(seq, black_box(&payload)).unwrap();
|
||||
@@ -60,7 +69,7 @@ fn bench_crypto(c: &mut Criterion) {
|
||||
black_box(ct)
|
||||
})
|
||||
});
|
||||
g.bench_function("seal_in_place", |b| {
|
||||
g.bench_function(format!("seal_in_place{suffix}"), |b| {
|
||||
let mut seq = 0u64;
|
||||
let mut buf = vec![0xABu8; SHARD + TAG_LEN];
|
||||
b.iter(|| {
|
||||
@@ -68,10 +77,10 @@ fn bench_crypto(c: &mut Criterion) {
|
||||
seq += 1;
|
||||
})
|
||||
});
|
||||
g.bench_function("open", |b| {
|
||||
g.bench_function(format!("open{suffix}"), |b| {
|
||||
b.iter(|| black_box(client.open(0, black_box(&sealed)).unwrap()))
|
||||
});
|
||||
g.bench_function("open_in_place", |b| {
|
||||
g.bench_function(format!("open_in_place{suffix}"), |b| {
|
||||
// In-place open consumes the buffer, so each iteration restores the ciphertext first —
|
||||
// one memcpy, mirroring what the recv ring does when the next datagram lands in the slot.
|
||||
let mut buf = sealed.clone();
|
||||
@@ -80,6 +89,7 @@ fn bench_crypto(c: &mut Criterion) {
|
||||
black_box(client.open_in_place(0, &mut buf).unwrap());
|
||||
})
|
||||
});
|
||||
}
|
||||
g.finish();
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,6 @@ fn main() {
|
||||
println!("cargo:rerun-if-changed=src/abi.rs");
|
||||
println!("cargo:rerun-if-changed=src/config.rs");
|
||||
println!("cargo:rerun-if-changed=src/input.rs");
|
||||
println!("cargo:rerun-if-changed=src/client.rs");
|
||||
println!("cargo:rerun-if-changed=src/error.rs");
|
||||
println!("cargo:rerun-if-changed=cbindgen.toml");
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
//! - Panics never cross the boundary: every entry point is wrapped in `catch_unwind`.
|
||||
|
||||
use crate::config::{Config, FecConfig, FecScheme, ProtocolPhase, Role};
|
||||
use crate::crypto::SessionKey;
|
||||
use crate::error::PunktfunkStatus;
|
||||
use crate::input::InputEvent;
|
||||
use crate::reanchor::{GateVerdict, ReanchorGate};
|
||||
@@ -78,6 +79,11 @@ impl PunktfunkConfig {
|
||||
u8::try_from(self.fec_percent).map_err(|_| PunktfunkStatus::InvalidArg)?;
|
||||
let max_data_per_block =
|
||||
u16::try_from(self.max_data_per_block).map_err(|_| PunktfunkStatus::InvalidArg)?;
|
||||
// The one narrowing here that differs by target width: on 32-bit (armeabi-v7a) an
|
||||
// `as usize` silently truncates a >4 GiB value to a plausible-looking residue that
|
||||
// passes validate() — reject it instead, like every narrowing above.
|
||||
let max_frame_bytes =
|
||||
usize::try_from(self.max_frame_bytes).map_err(|_| PunktfunkStatus::InvalidArg)?;
|
||||
let cfg = Config {
|
||||
role,
|
||||
phase,
|
||||
@@ -87,9 +93,12 @@ impl PunktfunkConfig {
|
||||
max_data_per_block,
|
||||
},
|
||||
shard_payload: self.shard_payload as usize,
|
||||
max_frame_bytes: self.max_frame_bytes as usize,
|
||||
max_frame_bytes,
|
||||
encrypt: self.encrypt != 0,
|
||||
key: self.key,
|
||||
// The C ABI keeps its fixed 16-byte key and always selects AES-128-GCM — no
|
||||
// ABI_VERSION bump. Raw-`Config` C embedders can't negotiate ChaCha; the Swift/
|
||||
// Kotlin clients are aarch64 with AES CE and never want it.
|
||||
key: SessionKey::Aes128Gcm(self.key),
|
||||
salt: self.salt,
|
||||
loopback_drop_period: self.loopback_drop_period,
|
||||
};
|
||||
@@ -125,6 +134,12 @@ pub struct PunktfunkFrame {
|
||||
pub frame_index: u32,
|
||||
pub pts_ns: u64,
|
||||
pub flags: u32,
|
||||
/// Wall-clock reassembly-completion instant (ns since the Unix epoch, CLOCK_REALTIME — the
|
||||
/// clock `pts_ns` and the skew handshake use). THIS is the receipt stamp for latency math:
|
||||
/// a stamp the embedder takes itself at the poll return additionally contains the
|
||||
/// pre-decode hand-off queue wait, so a client-side standing backlog would masquerade as
|
||||
/// network latency (ABI v9 — the 2026-07 two-pair standing-latency investigation).
|
||||
pub received_ns: u64,
|
||||
}
|
||||
|
||||
/// Snapshot of session counters.
|
||||
@@ -391,6 +406,7 @@ pub unsafe extern "C" fn punktfunk_client_poll_frame(
|
||||
frame_index: f.frame_index,
|
||||
pts_ns: f.pts_ns,
|
||||
flags: f.flags,
|
||||
received_ns: f.received_ns,
|
||||
};
|
||||
}
|
||||
PunktfunkStatus::Ok
|
||||
@@ -456,24 +472,32 @@ pub unsafe extern "C" fn punktfunk_set_input_callback(
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn punktfunk_host_poll_input(s: *mut PunktfunkSession) -> i32 {
|
||||
let r = std::panic::catch_unwind(AssertUnwindSafe(|| {
|
||||
let mut count = 0i32;
|
||||
loop {
|
||||
// Narrow scope: re-derive the handle and pull ONE event, then drop the borrow
|
||||
// before dispatching. The callback may legally re-enter `punktfunk_*` on this
|
||||
// handle (get_stats, send_input, clearing the callback) — with a `&mut` held
|
||||
// across the call that re-entry aliased it (UB under noalias). Re-reading
|
||||
// `input_cb` per iteration also makes a mid-drain
|
||||
// `punktfunk_set_input_callback(s, NULL, NULL)` take effect immediately instead
|
||||
// of firing the cleared callback for the queued remainder. (Freeing the session
|
||||
// from inside the callback remains forbidden, as on every entry point.)
|
||||
let (ev, cb) = {
|
||||
let s = match unsafe { s.as_mut() } {
|
||||
Some(s) => s,
|
||||
None => return PunktfunkStatus::NullPointer as i32,
|
||||
};
|
||||
let cb = s.input_cb;
|
||||
let mut count = 0i32;
|
||||
loop {
|
||||
match s.inner.poll_input() {
|
||||
Ok(Some(ev)) => {
|
||||
Ok(Some(ev)) => (ev, s.input_cb),
|
||||
Ok(None) => break,
|
||||
Err(e) => return e.status() as i32,
|
||||
}
|
||||
};
|
||||
if let Some((cb, user)) = cb {
|
||||
cb(&ev as *const InputEvent, user);
|
||||
}
|
||||
count += 1;
|
||||
}
|
||||
Ok(None) => break,
|
||||
Err(e) => return e.status() as i32,
|
||||
}
|
||||
}
|
||||
count
|
||||
}));
|
||||
r.unwrap_or(PunktfunkStatus::Panic as i32)
|
||||
@@ -878,8 +902,8 @@ pub const PUNKTFUNK_GAMEPAD_AUTO: u32 = 0;
|
||||
/// uinput X-Box 360 pad (the universal default — every game speaks XInput).
|
||||
pub const PUNKTFUNK_GAMEPAD_XBOX360: u32 = 1;
|
||||
/// UHID DualSense (kernel `hid-playstation`): adaptive triggers, lightbar, touchpad, motion —
|
||||
/// feedback arrives on the HID-output plane ([`punktfunk_connection_next_hidout`]). Honored
|
||||
/// only where available (Linux hosts); otherwise the host falls back to X-Box 360.
|
||||
/// feedback arrives on the HID-output plane ([`punktfunk_connection_next_hidout`]). Honored on
|
||||
/// Linux (UHID) and Windows (UMDF minidriver) hosts; otherwise the host falls back to X-Box 360.
|
||||
pub const PUNKTFUNK_GAMEPAD_DUALSENSE: u32 = 2;
|
||||
/// uinput X-Box One / Series pad — the X-Box 360 backend with the One/Series USB identity, so
|
||||
/// games show One/Series glyphs. XInput-identical to `XBOX360` otherwise (no game-visible gain;
|
||||
@@ -888,8 +912,8 @@ pub const PUNKTFUNK_GAMEPAD_DUALSENSE: u32 = 2;
|
||||
pub const PUNKTFUNK_GAMEPAD_XBOXONE: u32 = 3;
|
||||
/// UHID DualShock 4 (kernel `hid-playstation` ≥ 6.2): lightbar, touchpad, motion, rumble — the
|
||||
/// touchpad/motion arrive over the rich-input plane and lightbar over the HID-output plane, like
|
||||
/// DualSense (minus adaptive triggers / player LEDs / mute). Honored only where available (Linux
|
||||
/// hosts); otherwise the host falls back to X-Box 360.
|
||||
/// DualSense (minus adaptive triggers / player LEDs / mute). Honored on Linux (UHID) and Windows
|
||||
/// (UMDF minidriver) hosts; otherwise the host falls back to X-Box 360.
|
||||
pub const PUNKTFUNK_GAMEPAD_DUALSHOCK4: u32 = 4;
|
||||
/// UHID classic Steam Controller (Valve `28DE:1102`, kernel `hid-steam`): one stick + dual
|
||||
/// trackpads + two grip paddles. Honored only where available (Linux hosts); else Xbox 360.
|
||||
@@ -899,10 +923,12 @@ pub const PUNKTFUNK_GAMEPAD_STEAMCONTROLLER: u32 = 5;
|
||||
/// host. Honored on Linux AND Windows hosts; else folds to X-Box 360.
|
||||
pub const PUNKTFUNK_GAMEPAD_STEAMDECK: u32 = 6;
|
||||
/// DualSense Edge (Sony `054C:0DF2`): the DualSense plus two back buttons + two Fn buttons, so a
|
||||
/// client's back paddles land on native slots. Folds to `DUALSENSE` until its backend lands.
|
||||
/// client's back paddles land on native slots. Honored on Linux (UHID `hid-playstation`) and
|
||||
/// Windows (UMDF) hosts; otherwise the host falls back to X-Box 360.
|
||||
pub const PUNKTFUNK_GAMEPAD_DUALSENSEEDGE: u32 = 7;
|
||||
/// Nintendo Switch Pro Controller (Nintendo `057E:2009`, kernel `hid-nintendo`): Nintendo glyphs +
|
||||
/// positional layout, gyro/accel, HD rumble. Folds to `XBOX360` until its backend lands.
|
||||
/// positional layout, gyro/accel, HD rumble. Honored only where available (Linux hosts, UHID
|
||||
/// `hid-nintendo`); otherwise the host falls back to X-Box 360.
|
||||
pub const PUNKTFUNK_GAMEPAD_SWITCHPRO: u32 = 8;
|
||||
/// New Steam Controller (2026, Valve `28DE:1302`) passed through AS-IS: the host mirrors the
|
||||
/// client's raw Triton input reports out of a virtual SC2 with the real identity, and Steam's
|
||||
@@ -1744,6 +1770,7 @@ pub unsafe extern "C" fn punktfunk_connection_next_au(
|
||||
frame_index: f.frame_index,
|
||||
pts_ns: f.pts_ns,
|
||||
flags: f.flags,
|
||||
received_ns: f.received_ns,
|
||||
};
|
||||
}
|
||||
PunktfunkStatus::Ok
|
||||
@@ -1906,6 +1933,13 @@ pub unsafe extern "C" fn punktfunk_connection_next_audio_pcm(
|
||||
}
|
||||
let AudioPcmState { decoder, pcm } = &mut *state;
|
||||
let dec = decoder.as_mut().unwrap();
|
||||
// A header-only datagram (DTX silence — a legal wire form) must be SKIPPED, not
|
||||
// decoded: `decode_float` treats an empty payload as a loss and synthesizes a full
|
||||
// 120 ms of concealment for a ~5 ms slot, growing the playout ring without bound.
|
||||
// Mirrors the host mic pump's guard; the sink underruns to silence on its own.
|
||||
if pkt.data.is_empty() {
|
||||
return PunktfunkStatus::NoFrame;
|
||||
}
|
||||
// `decode_float` divides the output buffer length by the channel count to get the
|
||||
// per-channel capacity; an empty payload requests packet-loss concealment.
|
||||
match dec.decode_float(&pkt.data, pcm, false) {
|
||||
@@ -2956,7 +2990,14 @@ pub unsafe extern "C" fn punktfunk_connection_next_clipboard(
|
||||
unsafe { *out = out_ev };
|
||||
PunktfunkStatus::Ok
|
||||
}
|
||||
Err(e) => e.status(),
|
||||
Err(e) => {
|
||||
// Release the parked payload once the embedder polls past it: clipboard
|
||||
// traffic is sporadic, so without this a one-off 50 MiB paste stays resident
|
||||
// for the rest of the session (there is no other release entry point). The
|
||||
// borrow contract already says `out` data is valid only until the next call.
|
||||
*c.last_clip.lock().unwrap() = None;
|
||||
e.status()
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -3046,6 +3087,35 @@ pub unsafe extern "C" fn punktfunk_connection_clock_offset_ns(
|
||||
})
|
||||
}
|
||||
|
||||
/// The **live** host↔client wall-clock offset (nanoseconds, host minus client): the
|
||||
/// connect-time estimate of [`punktfunk_connection_clock_offset_ns`], updated by every applied
|
||||
/// mid-stream clock re-sync. Ongoing latency math (per-frame `received − pts` splits, the
|
||||
/// glass-to-glass meter) must use this one — after a wall-clock step/slew the frozen
|
||||
/// connect-time value reads tens of milliseconds wrong for the rest of the session, while the
|
||||
/// core itself has already re-synced. Same clock contract as the connect-time getter.
|
||||
///
|
||||
/// # Safety
|
||||
/// `c` is a valid connection handle; `offset_ns` is writable (NULL is skipped).
|
||||
#[cfg(feature = "quic")]
|
||||
#[no_mangle]
|
||||
pub unsafe extern "C" fn punktfunk_connection_clock_offset_now_ns(
|
||||
c: *const PunktfunkConnection,
|
||||
offset_ns: *mut i64,
|
||||
) -> PunktfunkStatus {
|
||||
guard(|| {
|
||||
let c = match unsafe { c.as_ref() } {
|
||||
Some(c) => c,
|
||||
None => return PunktfunkStatus::NullPointer,
|
||||
};
|
||||
unsafe {
|
||||
if !offset_ns.is_null() {
|
||||
*offset_ns = c.inner.clock_offset_now_ns();
|
||||
}
|
||||
}
|
||||
PunktfunkStatus::Ok
|
||||
})
|
||||
}
|
||||
|
||||
/// Ask the host to switch the live session to `width`x`height`@`refresh_hz` without
|
||||
/// reconnecting (window resized, refresh changed). Non-blocking enqueue: on acceptance the
|
||||
/// stream continues at the new mode — the first new-mode access unit is an IDR with
|
||||
@@ -3185,6 +3255,11 @@ pub unsafe extern "C" fn punktfunk_connection_frames_dropped(
|
||||
out: *mut u64,
|
||||
) -> PunktfunkStatus {
|
||||
guard(|| {
|
||||
// The header promises "writes 0 on a NULL connection" — honor it BEFORE the handle
|
||||
// check, so an embedder that skips the status never reads an uninitialized slot.
|
||||
if !out.is_null() {
|
||||
unsafe { *out = 0 };
|
||||
}
|
||||
let c = match unsafe { c.as_ref() } {
|
||||
Some(c) => c,
|
||||
None => return PunktfunkStatus::NullPointer,
|
||||
@@ -3242,6 +3317,11 @@ pub unsafe extern "C" fn punktfunk_connection_wants_decode_latency(
|
||||
out: *mut bool,
|
||||
) -> PunktfunkStatus {
|
||||
guard(|| {
|
||||
// The header promises "writes 0 on a NULL connection" — honor it BEFORE the handle
|
||||
// check: an uninitialized byte is not even a valid C++/Swift bool to read.
|
||||
if !out.is_null() {
|
||||
unsafe { *out = false };
|
||||
}
|
||||
let c = match unsafe { c.as_ref() } {
|
||||
Some(c) => c,
|
||||
None => return PunktfunkStatus::NullPointer,
|
||||
|
||||
@@ -73,6 +73,142 @@ pub(crate) const NOOP_CLOCK_FLUSHES_TO_DISARM: u32 = 2;
|
||||
/// FIRST no-op clock flush — the moment a step is actually suspected.
|
||||
pub(crate) const CLOCK_RESYNC_INTERVAL: Duration = Duration::from_secs(60);
|
||||
|
||||
/// Standing-latency bleed (the 2026-07 two-pair investigation): how far above the session's own
|
||||
/// one-way-delay floor a report window's MINIMUM must sit to count as a standing elevation. The
|
||||
/// jump-to-live detectors above deliberately ignore anything below ~6 frames / 400 ms, so a
|
||||
/// small standing state — a sub-frame kernel/reassembly backlog, or a stale clock offset after a
|
||||
/// wall-clock step — is carried forever and reads as permanent extra "network" latency. 10 ms
|
||||
/// sits above skew-handshake error + normal LAN jitter, and below a single 60 fps frame period,
|
||||
/// so the observed one-frame plateau (~17 ms) trips it while a healthy stream cannot.
|
||||
pub(crate) const STANDING_LAT_THRESH_NS: i128 = 10_000_000;
|
||||
|
||||
/// Consecutive elevated report windows (~750 ms each) before the bleed escalates — ~4.5 s of a
|
||||
/// continuously standing, loss-free elevation. Windows with any loss reset the run: loss means
|
||||
/// genuine congestion, which the FEC/ABR machinery owns, not this detector.
|
||||
pub(crate) const STANDING_LAT_WINDOWS: u32 = 6;
|
||||
|
||||
/// Per-session cap on flush+keyframe bleeds. A standing state that survives a clock re-sync AND
|
||||
/// this many local flushes is not local and not clock — the path latency itself changed; the
|
||||
/// detector disarms with a warning instead of paying a recovery keyframe every few seconds.
|
||||
pub(crate) const STANDING_LAT_MAX_BLEEDS: u32 = 3;
|
||||
|
||||
/// What the standing-latency detector asks the pump to do this window (see [`StandingLatency`]).
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub(crate) enum StandingLatAction {
|
||||
None,
|
||||
/// First escalation: ask for a mid-stream clock re-sync — free, and a stale offset from a
|
||||
/// stepped/slewed wall clock produces exactly this signature (an applied re-sync re-bases
|
||||
/// the floor via the pump's `clock_gen` watch, clearing the elevation if that was the cause).
|
||||
Resync {
|
||||
above_ms: i64,
|
||||
},
|
||||
/// The elevation survived a re-sync attempt: flush the local receive backlog + request a
|
||||
/// keyframe (the jump-to-live action), draining a real sub-threshold standing queue. The
|
||||
/// pump reports execution back via [`StandingLatency::bled`]; an unexecuted action simply
|
||||
/// re-arms next window.
|
||||
Bleed {
|
||||
above_ms: i64,
|
||||
},
|
||||
/// Bleed cap reached and the elevation is back: give up and say so.
|
||||
Disarm {
|
||||
above_ms: i64,
|
||||
},
|
||||
}
|
||||
|
||||
/// Detector for a small, constant, loss-free one-way-delay elevation — the standing state the
|
||||
/// jump-to-live thresholds deliberately tolerate. Tracks the session's OWD floor (minimum of
|
||||
/// report-window minimums since start / last re-base) and escalates when windows sit
|
||||
/// persistently above it: re-sync first, then a bounded number of flush+keyframe bleeds, then
|
||||
/// disarm. Pure state machine (no clocks, no I/O) so the escalation ladder is unit-testable.
|
||||
pub(crate) struct StandingLatency {
|
||||
/// Lowest window-minimum OWD seen since session start / last [`rebase`](Self::rebase).
|
||||
floor_ns: Option<i128>,
|
||||
/// Minimum per-frame OWD this report window; `None` = no frames yet.
|
||||
window_min_ns: Option<i128>,
|
||||
/// Consecutive elevated windows.
|
||||
run: u32,
|
||||
/// The current elevation already got its re-sync request — next escalation is a bleed.
|
||||
resync_tried: bool,
|
||||
bleeds: u32,
|
||||
disarmed: bool,
|
||||
}
|
||||
|
||||
impl StandingLatency {
|
||||
pub(crate) fn new() -> Self {
|
||||
StandingLatency {
|
||||
floor_ns: None,
|
||||
window_min_ns: None,
|
||||
run: 0,
|
||||
resync_tried: false,
|
||||
bleeds: 0,
|
||||
disarmed: false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Feed one frame's skew-corrected OWD (capture→reassembly-complete, ns). Caller gates on a
|
||||
/// live clock offset and plausibility (0 < owd < 10 s), like the ABR OWD signal.
|
||||
pub(crate) fn note_frame(&mut self, owd_ns: i128) {
|
||||
self.window_min_ns = Some(match self.window_min_ns {
|
||||
Some(m) => m.min(owd_ns),
|
||||
None => owd_ns,
|
||||
});
|
||||
}
|
||||
|
||||
/// Close a report window. `loss_free` = the window carried zero loss (loss resets the run —
|
||||
/// congestion is the FEC/ABR machinery's problem, and queues under loss are not "standing").
|
||||
pub(crate) fn on_window(&mut self, loss_free: bool) -> StandingLatAction {
|
||||
let Some(wmin) = self.window_min_ns.take() else {
|
||||
return StandingLatAction::None; // no frames this window — no evidence either way
|
||||
};
|
||||
let floor = *self.floor_ns.get_or_insert(wmin);
|
||||
self.floor_ns = Some(floor.min(wmin));
|
||||
let above_ns = wmin - floor;
|
||||
if self.disarmed {
|
||||
return StandingLatAction::None;
|
||||
}
|
||||
if !loss_free || above_ns < STANDING_LAT_THRESH_NS {
|
||||
self.run = 0;
|
||||
if above_ns < STANDING_LAT_THRESH_NS {
|
||||
self.resync_tried = false; // elevation cleared — a future one re-syncs first again
|
||||
}
|
||||
return StandingLatAction::None;
|
||||
}
|
||||
self.run += 1;
|
||||
if self.run < STANDING_LAT_WINDOWS {
|
||||
return StandingLatAction::None;
|
||||
}
|
||||
self.run = 0; // each escalation gets a fresh observation run
|
||||
let above_ms = (above_ns / 1_000_000) as i64;
|
||||
if !self.resync_tried {
|
||||
self.resync_tried = true;
|
||||
StandingLatAction::Resync { above_ms }
|
||||
} else if self.bleeds < STANDING_LAT_MAX_BLEEDS {
|
||||
StandingLatAction::Bleed { above_ms }
|
||||
} else {
|
||||
self.disarmed = true;
|
||||
StandingLatAction::Disarm { above_ms }
|
||||
}
|
||||
}
|
||||
|
||||
/// The pump executed a [`StandingLatAction::Bleed`] (flush + keyframe). The floor is KEPT: a
|
||||
/// successful bleed brings OWD back down to it (elevation clears naturally); an unsuccessful
|
||||
/// one leaves the elevation visible so the ladder continues toward the cap.
|
||||
pub(crate) fn bled(&mut self) {
|
||||
self.bleeds += 1;
|
||||
self.window_min_ns = None;
|
||||
}
|
||||
|
||||
/// A mid-stream clock re-sync was APPLIED (the pump's `clock_gen` watch): every OWD reading
|
||||
/// shifted, so the floor and any elevation measured under the old offset are meaningless —
|
||||
/// re-learn from scratch. The bleed budget survives (it caps keyframes per session).
|
||||
pub(crate) fn rebase(&mut self) {
|
||||
self.floor_ns = None;
|
||||
self.window_min_ns = None;
|
||||
self.run = 0;
|
||||
self.resync_tried = false;
|
||||
}
|
||||
}
|
||||
|
||||
/// Client decode-stage latency accumulator for the adaptive-bitrate controller's decode signal.
|
||||
/// The embedder adds one sample per decoded frame ([`NativeClient::report_decode_us`], µs from the
|
||||
/// AU leaving [`NativeClient::next_frame`] to its decoded output) and the data-plane pump drains a
|
||||
@@ -191,6 +327,7 @@ mod frame_channel_tests {
|
||||
pts_ns: i as u64,
|
||||
flags: 0,
|
||||
complete: true,
|
||||
received_ns: 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -258,3 +395,143 @@ mod frame_channel_tests {
|
||||
assert_eq!(popped(&ch), Some(total - FRAME_QUEUE_HARD_CAP as u32));
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod standing_latency_tests {
|
||||
use super::{
|
||||
StandingLatAction, StandingLatency, STANDING_LAT_MAX_BLEEDS, STANDING_LAT_THRESH_NS,
|
||||
STANDING_LAT_WINDOWS,
|
||||
};
|
||||
|
||||
const FLOOR: i128 = 2_000_000; // a healthy 2 ms LAN OWD
|
||||
const ELEVATED: i128 = FLOOR + STANDING_LAT_THRESH_NS + 7_000_000; // ~one 60fps frame above
|
||||
|
||||
/// Run `n` windows at `owd`, asserting every window but the last returns None; returns the
|
||||
/// last window's action.
|
||||
fn run_windows(d: &mut StandingLatency, owd: i128, n: u32) -> StandingLatAction {
|
||||
for i in 0..n {
|
||||
d.note_frame(owd);
|
||||
let a = d.on_window(true);
|
||||
if i + 1 < n {
|
||||
assert_eq!(a, StandingLatAction::None, "window {i} escalated early");
|
||||
} else {
|
||||
return a;
|
||||
}
|
||||
}
|
||||
unreachable!("n > 0 by construction");
|
||||
}
|
||||
|
||||
/// Learn a clean floor: one window at the healthy OWD.
|
||||
fn learned(d: &mut StandingLatency) {
|
||||
d.note_frame(FLOOR);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn healthy_stream_never_escalates() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
// Jitter riding above the floor but under the threshold: never a run.
|
||||
for _ in 0..(STANDING_LAT_WINDOWS * 4) {
|
||||
d.note_frame(FLOOR + STANDING_LAT_THRESH_NS - 1);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn escalation_ladder_resync_then_bleeds_then_disarm() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
// First full elevated run asks for the free fix: a clock re-sync.
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
// Re-sync didn't help (no rebase came) — each further run is a bleed, up to the cap...
|
||||
for _ in 0..STANDING_LAT_MAX_BLEEDS {
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Bleed { .. }
|
||||
));
|
||||
d.bled();
|
||||
}
|
||||
// ...then the detector gives up loudly, once, and stays quiet.
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Disarm { .. }
|
||||
));
|
||||
d.note_frame(ELEVATED);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn loss_windows_reset_the_run() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
for _ in 0..(STANDING_LAT_WINDOWS - 1) {
|
||||
d.note_frame(ELEVATED);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
// A lossy window means congestion, not a standing state: run resets...
|
||||
d.note_frame(ELEVATED);
|
||||
assert_eq!(d.on_window(false), StandingLatAction::None);
|
||||
// ...so the ladder needs the full run again before acting.
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_resets_the_ladder_to_resync_first() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
// The elevation clears on its own (e.g. the successful bleed case, or transient): the
|
||||
// next episode starts back at the free escalation, not at a bleed.
|
||||
d.note_frame(FLOOR);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn applied_resync_rebases_and_clears_a_stale_offset_elevation() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
assert!(matches!(
|
||||
run_windows(&mut d, ELEVATED, STANDING_LAT_WINDOWS),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
// The re-sync APPLIES (pump sees clock_gen move) → rebase. The corrected offset brings
|
||||
// OWD readings back to truth; the floor re-learns and nothing ever escalates to a bleed.
|
||||
d.rebase();
|
||||
for _ in 0..(STANDING_LAT_WINDOWS * 2) {
|
||||
d.note_frame(FLOOR);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_windows_are_no_evidence() {
|
||||
let mut d = StandingLatency::new();
|
||||
learned(&mut d);
|
||||
for _ in 0..(STANDING_LAT_WINDOWS - 1) {
|
||||
d.note_frame(ELEVATED);
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
}
|
||||
// A frameless window (paused stream) neither advances nor resets the run...
|
||||
assert_eq!(d.on_window(true), StandingLatAction::None);
|
||||
// ...so one more elevated window completes it.
|
||||
d.note_frame(ELEVATED);
|
||||
assert!(matches!(
|
||||
d.on_window(true),
|
||||
StandingLatAction::Resync { .. }
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -393,6 +393,7 @@ impl NativeClient {
|
||||
launch,
|
||||
pin,
|
||||
identity,
|
||||
connect_timeout: timeout,
|
||||
frames: frame_chan_w,
|
||||
audio_tx,
|
||||
rumble_tx,
|
||||
@@ -425,6 +426,12 @@ impl NativeClient {
|
||||
Ok(Ok(t)) => t,
|
||||
Ok(Err(e)) => return Err(e),
|
||||
Err(_) => {
|
||||
// A connect we already reported as failed must not leave a lingering host
|
||||
// session if the handshake lands late: mark it a deliberate QUIT (not a plain
|
||||
// drop / close code 0) so the worker's close tells the host to tear down now
|
||||
// instead of holding the session (and its virtual display) for a reconnect
|
||||
// that will never come.
|
||||
quit.store(true, Ordering::SeqCst);
|
||||
shutdown.store(true, Ordering::SeqCst);
|
||||
return Err(PunktfunkError::Timeout);
|
||||
}
|
||||
@@ -456,9 +463,14 @@ impl NativeClient {
|
||||
hot_tids,
|
||||
clock_offset,
|
||||
decode_lat,
|
||||
// The controller arms exactly when the pump does (see `abr::BitrateController::new`
|
||||
// below): Automatic (the user asked for bitrate 0) and not a rate-pinned PyroWave stream.
|
||||
wants_decode: bitrate_kbps == 0 && negotiated.codec != crate::quic::CODEC_PYROWAVE,
|
||||
// The controller arms exactly when the pump does — all three terms, not two: Automatic
|
||||
// (the user asked for bitrate 0), not a rate-pinned PyroWave stream, AND the host
|
||||
// echoed the rate it actually configured. Dropping the last term made this
|
||||
// over-advertise against an old host that reports no rate, so an embedder fed decode
|
||||
// latency to a controller that never runs.
|
||||
wants_decode: bitrate_kbps == 0
|
||||
&& negotiated.codec != crate::quic::CODEC_PYROWAVE
|
||||
&& negotiated.bitrate_kbps > 0,
|
||||
mode: mode_slot,
|
||||
host_fingerprint: negotiated.host_fingerprint,
|
||||
resolved_compositor: negotiated.compositor,
|
||||
@@ -703,14 +715,23 @@ impl NativeClient {
|
||||
// Reset the accumulator so a fresh run doesn't blend into the previous one.
|
||||
*self.probe.lock().unwrap() = ProbeState {
|
||||
active: true,
|
||||
duration_ms,
|
||||
..Default::default()
|
||||
};
|
||||
self.ctrl_tx
|
||||
let sent = self
|
||||
.ctrl_tx
|
||||
.try_send(CtrlRequest::Probe(ProbeRequest {
|
||||
target_kbps,
|
||||
duration_ms,
|
||||
}))
|
||||
.map_err(|_| PunktfunkError::Closed)
|
||||
.map_err(|_| PunktfunkError::Closed);
|
||||
if sent.is_err() {
|
||||
// Nothing was asked of the host, so nothing will ever answer. Leaving `active` latched
|
||||
// would suppress the pump's entire report tick for the rest of the session (the pump
|
||||
// mirrors the startup path's rollback at the same point).
|
||||
self.probe.lock().unwrap().active = false;
|
||||
}
|
||||
sent
|
||||
}
|
||||
|
||||
/// Read the current speed-test measurement (partial until `done`, final once the host's
|
||||
@@ -745,7 +766,9 @@ impl NativeClient {
|
||||
0.0
|
||||
} as f32;
|
||||
// Host-side drop: what the send buffer couldn't even accept (the host-side ceiling).
|
||||
let offered_wire = p.host_wire_packets + p.host_send_dropped;
|
||||
// Saturating: both counters arrive verbatim off the wire (same discipline as the
|
||||
// saturating_sub/mul above — a hostile sum must not overflow-panic a debug build).
|
||||
let offered_wire = p.host_wire_packets.saturating_add(p.host_send_dropped);
|
||||
let host_drop_pct = if offered_wire > 0 {
|
||||
p.host_send_dropped as f64 / offered_wire as f64 * 100.0
|
||||
} else {
|
||||
|
||||
@@ -86,10 +86,22 @@ impl NativeClient {
|
||||
// A typed application close from the host (pairing not armed / armed for a
|
||||
// different device / rate-limited / version mismatch) beats the generic
|
||||
// transport error the aborted exchange produced — it is the actual answer.
|
||||
Err(e) => Err(match reject_from_close(&conn) {
|
||||
// Same close-vs-stream-error race as the connect handshake: give the
|
||||
// host's CONNECTION_CLOSE a short grace to be processed before deciding
|
||||
// the error was plain transport trouble.
|
||||
Err(e) => {
|
||||
if conn.close_reason().is_none() {
|
||||
let _ = tokio::time::timeout(
|
||||
std::time::Duration::from_millis(300),
|
||||
conn.closed(),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
Err(match reject_from_close(&conn) {
|
||||
Some(r) => PunktfunkError::Rejected(r),
|
||||
None => e,
|
||||
}),
|
||||
})
|
||||
}
|
||||
ok => ok,
|
||||
};
|
||||
// Always tell the host we're done so it never blocks at its read — code 0 on
|
||||
|
||||
@@ -32,6 +32,11 @@ pub(crate) struct ProbeState {
|
||||
pub(crate) host_duration_ms: u32,
|
||||
/// The host's `ProbeResult` arrived → the measurement is final.
|
||||
pub(crate) done: bool,
|
||||
/// The requested burst length, so the pump can arm a watchdog for a host that never answers.
|
||||
/// Without one, an ignored `ProbeRequest` latches `active` forever and the pump's whole report
|
||||
/// tick — loss reports, the ABR window feed, the standing-latency ladder and pending clock
|
||||
/// re-syncs — stays suppressed for the rest of the session.
|
||||
pub(crate) duration_ms: u32,
|
||||
}
|
||||
|
||||
/// A finished/partial speed-test measurement, returned by [`NativeClient::probe_result`].
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,181 @@
|
||||
//! Control task: the handshake stream stays open for mid-stream renegotiation + speed tests.
|
||||
//! Outbound requests (mode switch, probe) and inbound replies (Reconfigured, ProbeResult) are
|
||||
//! multiplexed with `select!`; a single outbound channel (`ctrl_rx`) keeps one writer so the
|
||||
//! two `&mut ctrl_send` borrows don't collide across branches.
|
||||
|
||||
use super::super::*;
|
||||
use super::*;
|
||||
|
||||
pub(super) struct ControlTask {
|
||||
pub(super) ctrl_rx: tokio::sync::mpsc::Receiver<CtrlRequest>,
|
||||
pub(super) ctrl_send: quinn::SendStream,
|
||||
pub(super) ctrl_recv: io::MsgReader,
|
||||
/// `None` = no connect-time skew handshake (old host) — clock re-sync stays off.
|
||||
pub(super) clock_rtt_ns: Option<u64>,
|
||||
pub(super) mode_slot: Arc<Mutex<Mode>>,
|
||||
pub(super) probe: Arc<Mutex<ProbeState>>,
|
||||
/// The latest host `BitrateChanged` ack, drained by the pump's ABR on its report tick.
|
||||
pub(super) bitrate_ack: Arc<Mutex<Option<u32>>>,
|
||||
pub(super) clock_offset: Arc<std::sync::atomic::AtomicI64>,
|
||||
pub(super) clock_gen: Arc<AtomicU32>,
|
||||
/// Clipboard metadata events (ClipState/ClipOffer) feed the same event plane the
|
||||
/// clipboard task uses for fetch data.
|
||||
pub(super) clip_event_tx: std::sync::mpsc::SyncSender<ClipEventCore>,
|
||||
}
|
||||
|
||||
impl ControlTask {
|
||||
pub(super) async fn run(self) {
|
||||
let ControlTask {
|
||||
mut ctrl_rx,
|
||||
mut ctrl_send,
|
||||
mut ctrl_recv,
|
||||
clock_rtt_ns,
|
||||
mode_slot,
|
||||
probe,
|
||||
bitrate_ack,
|
||||
clock_offset,
|
||||
clock_gen,
|
||||
clip_event_tx,
|
||||
} = self;
|
||||
// Mid-stream clock re-sync (see [`ClockResync`]): a batch runs every
|
||||
// CLOCK_RESYNC_INTERVAL and whenever the pump asks (CtrlRequest::ClockResync after
|
||||
// its first no-op clock flush). Echoes interleave with the other control replies in
|
||||
// the read arm below; only when the host answered the connect-time handshake — an
|
||||
// old host would just eat the probes.
|
||||
let mut resync = ClockResync::new();
|
||||
let mut resync_tick = tokio::time::interval_at(
|
||||
tokio::time::Instant::now() + CLOCK_RESYNC_INTERVAL,
|
||||
CLOCK_RESYNC_INTERVAL,
|
||||
);
|
||||
resync_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||
loop {
|
||||
tokio::select! {
|
||||
req = ctrl_rx.recv() => {
|
||||
let Some(req) = req else { break }; // client dropped
|
||||
let bytes = match req {
|
||||
CtrlRequest::Mode(m) => Reconfigure { mode: m }.encode(),
|
||||
CtrlRequest::Probe(p) => p.encode(),
|
||||
CtrlRequest::Keyframe => RequestKeyframe.encode(),
|
||||
CtrlRequest::Rfi(r) => r.encode(),
|
||||
CtrlRequest::Loss(r) => r.encode(),
|
||||
CtrlRequest::SetBitrate(k) => SetBitrate { bitrate_kbps: k }.encode(),
|
||||
CtrlRequest::ClockResync => {
|
||||
if clock_rtt_ns.is_none() {
|
||||
continue; // no connect-time handshake — host can't answer
|
||||
}
|
||||
resync.begin(wall_clock_ns()).encode()
|
||||
}
|
||||
CtrlRequest::ClipControl(c) => c.encode(),
|
||||
CtrlRequest::ClipOffer(o) => o.encode(),
|
||||
};
|
||||
if io::write_msg(&mut ctrl_send, &bytes).await.is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
_ = resync_tick.tick(), if clock_rtt_ns.is_some() => {
|
||||
let probe = resync.begin(wall_clock_ns());
|
||||
if io::write_msg(&mut ctrl_send, &probe.encode()).await.is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
msg = ctrl_recv.read_msg() => {
|
||||
let Ok(msg) = msg else { break }; // stream closed
|
||||
if let Ok(ack) = Reconfigured::decode(&msg) {
|
||||
if ack.accepted {
|
||||
*mode_slot.lock().unwrap() = ack.mode;
|
||||
tracing::info!(mode = ?ack.mode, "host accepted mode switch");
|
||||
} else {
|
||||
tracing::warn!(active = ?ack.mode, "host rejected mode switch");
|
||||
}
|
||||
} else if let Ok(result) = ProbeResult::decode(&msg) {
|
||||
let mut p = probe.lock().unwrap();
|
||||
// Freeze the delivered figures now (the burst is done), before resumed
|
||||
// video can inflate the packet counters.
|
||||
let base_p = p.base_packets.unwrap_or(p.rx_packets_now);
|
||||
let base_b = p.base_bytes.unwrap_or(p.rx_bytes_now);
|
||||
p.delivered_packets = p.rx_packets_now.saturating_sub(base_p);
|
||||
p.delivered_bytes = p.rx_bytes_now.saturating_sub(base_b);
|
||||
p.host_goodput_bytes = result.bytes_sent;
|
||||
p.host_au = result.packets_sent;
|
||||
p.host_wire_packets = result.wire_packets_sent;
|
||||
p.host_send_dropped = result.send_dropped;
|
||||
p.host_duration_ms = result.duration_ms;
|
||||
p.done = true;
|
||||
p.active = false; // burst over — the pump stops mirroring counters
|
||||
tracing::info!(
|
||||
host_goodput_bytes = result.bytes_sent,
|
||||
wire_packets_sent = result.wire_packets_sent,
|
||||
send_dropped = result.send_dropped,
|
||||
duration_ms = result.duration_ms,
|
||||
delivered_packets = p.delivered_packets,
|
||||
"speed-test probe result"
|
||||
);
|
||||
} else if let Ok(ack) = BitrateChanged::decode(&msg) {
|
||||
// Adaptive bitrate: the host's clamp is authoritative — park it for
|
||||
// the pump's controller (which also reads any ack as "this host
|
||||
// renegotiates", arming further steps).
|
||||
tracing::info!(
|
||||
kbps = ack.bitrate_kbps,
|
||||
"host re-targeted encoder bitrate"
|
||||
);
|
||||
*bitrate_ack.lock().unwrap() = Some(ack.bitrate_kbps);
|
||||
} else if let Ok(echo) = ClockEcho::decode(&msg) {
|
||||
match resync.on_echo(&echo, wall_clock_ns()) {
|
||||
ResyncStep::Probe(p) => {
|
||||
if io::write_msg(&mut ctrl_send, &p.encode()).await.is_err() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
ResyncStep::Done { offset_ns, rtt_ns } => {
|
||||
// Never let a congested window bias the offset (frames read
|
||||
// late exactly then) — keep the old estimate and let the next
|
||||
// periodic batch try again.
|
||||
if accept_resync(rtt_ns, clock_rtt_ns.unwrap_or(0)) {
|
||||
// info, not debug: ≤1/min, and it is THE forensic
|
||||
// trail for a stale-offset (stepped/slewed wall clock)
|
||||
// latency plateau — the 2026-07 two-pair investigation
|
||||
// had to reconstruct this blind.
|
||||
tracing::info!(
|
||||
offset_ns,
|
||||
rtt_us = rtt_ns / 1000,
|
||||
"mid-stream clock re-sync applied"
|
||||
);
|
||||
clock_offset.store(offset_ns, Ordering::Relaxed);
|
||||
clock_gen.fetch_add(1, Ordering::Relaxed);
|
||||
} else {
|
||||
tracing::info!(
|
||||
rtt_us = rtt_ns / 1000,
|
||||
"clock re-sync batch discarded — RTT above the \
|
||||
connect-time baseline (congested window)"
|
||||
);
|
||||
}
|
||||
}
|
||||
ResyncStep::Idle => {}
|
||||
}
|
||||
} else if let Ok(state) = ClipState::decode(&msg) {
|
||||
// Host ack / policy / backend update for the toggle UI (try_send: a
|
||||
// lagging embedder drops the newest — a stale toggle heals on the next).
|
||||
let _ = clip_event_tx.try_send(ClipEventCore::State {
|
||||
enabled: state.enabled,
|
||||
policy: state.policy,
|
||||
reason: state.reason,
|
||||
});
|
||||
} else if let Ok(offer) = ClipOffer::decode(&msg) {
|
||||
// The host copied something: surface the lazy format list; the embedder
|
||||
// fetches only if a local app pastes.
|
||||
let _ = clip_event_tx.try_send(ClipEventCore::RemoteOffer {
|
||||
seq: offer.seq,
|
||||
kinds: offer.kinds,
|
||||
});
|
||||
} else {
|
||||
tracing::warn!(
|
||||
tag = ?msg.first(),
|
||||
len = msg.len(),
|
||||
"unknown control message — ignoring"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,568 @@
|
||||
//! The blocking data-plane pump: poll the session for access units, run the adaptive-FEC
|
||||
//! loss reports, the ABR controller + startup capacity probe, the jump-to-live detectors,
|
||||
//! and the standing-latency bleed, and hand frames to the embedder.
|
||||
|
||||
use super::super::*;
|
||||
use super::*;
|
||||
|
||||
/// Data-plane pump on a blocking thread: poll the session, hand frames to the embedder.
|
||||
/// try_send drops the newest frame when the embedder lags (freshness over completeness).
|
||||
/// Speed-test filler ([`FLAG_PROBE`]) is folded into the probe accumulator instead of the
|
||||
/// decoder queue — it isn't video.
|
||||
pub(super) struct DataPump {
|
||||
pub(super) session: Session,
|
||||
pub(super) frames: Arc<FrameChannel>,
|
||||
pub(super) ctrl_tx: tokio::sync::mpsc::Sender<CtrlRequest>,
|
||||
pub(super) shutdown: Arc<std::sync::atomic::AtomicBool>,
|
||||
pub(super) probe: Arc<Mutex<ProbeState>>,
|
||||
pub(super) hot_tids: Arc<Mutex<Vec<i32>>>,
|
||||
pub(super) clock_offset: Arc<std::sync::atomic::AtomicI64>,
|
||||
pub(super) clock_gen: Arc<AtomicU32>,
|
||||
pub(super) decode_lat: Arc<Mutex<DecodeLatAcc>>,
|
||||
pub(super) frames_dropped: Arc<std::sync::atomic::AtomicU64>,
|
||||
pub(super) fec_recovered: Arc<std::sync::atomic::AtomicU64>,
|
||||
pub(super) bitrate_ack: Arc<Mutex<Option<u32>>>,
|
||||
/// The embedder's REQUESTED rate (0 = Automatic — the only case the ABR arms).
|
||||
pub(super) bitrate_kbps: u32,
|
||||
/// The rate the host actually configured (echoed in Welcome).
|
||||
pub(super) resolved_bitrate_kbps: u32,
|
||||
pub(super) negotiated_codec: u8,
|
||||
}
|
||||
|
||||
impl DataPump {
|
||||
pub(super) fn run(self) {
|
||||
let DataPump {
|
||||
mut session,
|
||||
frames,
|
||||
ctrl_tx,
|
||||
shutdown: pump_shutdown,
|
||||
probe: pump_probe,
|
||||
hot_tids: pump_hot_tids,
|
||||
clock_offset: pump_clock_offset,
|
||||
clock_gen: pump_clock_gen,
|
||||
decode_lat: pump_decode_lat,
|
||||
frames_dropped,
|
||||
fec_recovered,
|
||||
bitrate_ack,
|
||||
bitrate_kbps,
|
||||
resolved_bitrate_kbps,
|
||||
negotiated_codec,
|
||||
} = self;
|
||||
pin_thread_user_interactive(); // feeds the frame channel → the user-interactive video pump
|
||||
register_hot_tid(&pump_hot_tids); // this thread does UDP receive + FEC reassembly — hint it
|
||||
// Adaptive-FEC loss reporting: every ADAPT_REPORT_INTERVAL, report the loss observed over the
|
||||
// window (shards FEC recovered, plus a bump if any frame went unrecoverable) so the host can
|
||||
// size FEC to the link. Suppressed during a speed test (its FLAG_PROBE filler would skew it).
|
||||
const ADAPT_REPORT_INTERVAL: Duration = Duration::from_millis(750);
|
||||
let mut last_report = Instant::now();
|
||||
let (
|
||||
mut last_recovered,
|
||||
mut last_late,
|
||||
mut last_received,
|
||||
mut last_dropped,
|
||||
mut last_bytes,
|
||||
) = (0u64, 0u64, 0u64, 0u64, 0u64);
|
||||
// PUNKTFUNK_PERF: per-window pump observability — the Session's receive stage split
|
||||
// (recv / decrypt / reassemble+FEC, see `Session::take_pump_perf`) and completed-AU
|
||||
// inter-arrival jitter. Smoothness has no metric otherwise: jump-to-live counters only
|
||||
// fire after the stream is already seconds behind.
|
||||
let pump_perf_on = std::env::var("PUNKTFUNK_PERF").is_ok_and(|v| v != "0");
|
||||
let mut arrivals_us: Vec<u32> = Vec::new();
|
||||
let mut last_arrival: Option<Instant> = None;
|
||||
// Adaptive bitrate (see `crate::abr`): armed only when the embedder asked for Automatic
|
||||
// (`bitrate_kbps == 0`) and the host echoed the rate it actually configured (an old host
|
||||
// echoes 0 → controller stays permanently off). Fed once per report window with the same
|
||||
// deltas the LossReport uses, plus the window's mean skew-corrected one-way delay, the
|
||||
// actual delivered throughput (climb gate + proven-throughput mark), and whether a
|
||||
// jump-to-live flush fired.
|
||||
// PyroWave sessions PIN their rate (§4.6): AIMD descent turns wavelets to mush well
|
||||
// above its floor, and the climb probe's VBV reasoning doesn't apply to hard
|
||||
// per-frame CBR — controller and capacity probe stay off (0 = permanently off).
|
||||
let rate_pinned = negotiated_codec == crate::quic::CODEC_PYROWAVE;
|
||||
let mut abr = BitrateController::new(if bitrate_kbps == 0 && !rate_pinned {
|
||||
resolved_bitrate_kbps
|
||||
} else {
|
||||
0
|
||||
});
|
||||
// Startup link-capacity probe (Automatic sessions): the controller's ceiling is the
|
||||
// negotiated start rate — the conservative 20 Mbps default, historically a box Automatic
|
||||
// could NEVER climb out of. One speed-test burst shortly after the stream settles
|
||||
// measures what the link actually delivers; ×0.7 (headroom for FEC overhead + variance)
|
||||
// becomes the climb ceiling and slow start does the rest. Old hosts decline (all-zero
|
||||
// reply) or never answer (timeout clears the state so LossReports resume) — either way
|
||||
// the ceiling stays negotiated, exactly the old behavior. PUNKTFUNK_ABR_PROBE=0 opts out.
|
||||
const CAPACITY_PROBE_KBPS: u32 = 2_000_000;
|
||||
const CAPACITY_PROBE_MS: u32 = 800;
|
||||
const CAPACITY_PROBE_DELAY: Duration = Duration::from_secs(2);
|
||||
const CAPACITY_PROBE_TIMEOUT: Duration = Duration::from_secs(6);
|
||||
let mut capacity_probe_at: Option<Instant> = (bitrate_kbps == 0
|
||||
&& !rate_pinned
|
||||
&& resolved_bitrate_kbps > 0
|
||||
&& std::env::var("PUNKTFUNK_ABR_PROBE").map_or(true, |v| v != "0"))
|
||||
.then(|| Instant::now() + CAPACITY_PROBE_DELAY);
|
||||
let mut capacity_probe_deadline: Option<Instant> = None;
|
||||
// Edge detector + watchdog for a probe of EITHER origin (the startup capacity probe or an
|
||||
// embedder speed test via `NativeClient::request_probe`). The startup path had both built
|
||||
// in; the embedder path had neither, so an unanswered request wedged the report tick and a
|
||||
// finished one left the ABR window anchored before the burst.
|
||||
let mut was_probing = false;
|
||||
let mut probe_watchdog: Option<Instant> = None;
|
||||
let (mut owd_sum_ns, mut owd_frames) = (0i128, 0u32);
|
||||
let mut flush_in_window = false;
|
||||
// Jump-to-live state (see the guard in the loop below): when the clock-based over-bound
|
||||
// run began (`stale_since`, armed only when the skew handshake succeeded so the clocks
|
||||
// are comparable), when the clock-free non-draining-queue run began (`standing_since`),
|
||||
// and the last-jump instant for the shared cooldown. Wall-clock runs (T1.4), not frame
|
||||
// counts — the detectors' sensitivity must not scale with fps or repeat cadence.
|
||||
let mut stale_since: Option<Instant> = None;
|
||||
let mut standing_since: Option<Instant> = None;
|
||||
let mut last_flush: Option<Instant> = None;
|
||||
// Clock-detector health: consecutive clock-triggered flushes that found no local backlog
|
||||
// (see NOOP_FLUSH_DATAGRAMS). Reaching NOOP_CLOCK_FLUSHES_TO_DISARM turns the clock-based
|
||||
// detector off (a clock step / upstream queue it can't fix) — until a mid-stream clock
|
||||
// re-sync lands and re-arms it (`pump_clock_gen` below). The FIRST no-op flush also asks
|
||||
// the control task for an immediate re-sync (via the report tick): the flush finding no
|
||||
// local backlog IS the "the wall clock stepped under me" signal.
|
||||
let mut noop_clock_flushes: u32 = 0;
|
||||
let mut clock_detector_armed = true;
|
||||
let mut resync_wanted = false;
|
||||
let mut seen_clock_gen = pump_clock_gen.load(Ordering::Relaxed);
|
||||
// Standing-latency bleed (see StandingLatency): the third detector, for the small,
|
||||
// constant, loss-free OWD elevation the two jump-to-live detectors deliberately
|
||||
// tolerate (< QUEUE_HIGH frames, < FLUSH_LATENCY behind) — a sub-frame standing
|
||||
// backlog, or a stale clock offset after a wall-clock step, either of which otherwise
|
||||
// reads as permanent extra "network" latency for the rest of the session.
|
||||
let mut standing_lat = StandingLatency::new();
|
||||
while !pump_shutdown.load(Ordering::SeqCst) {
|
||||
// The live host↔client offset: re-loaded every iteration so an applied mid-stream
|
||||
// re-sync takes effect on the very next frame's latency math.
|
||||
let clock_offset_ns = pump_clock_offset.load(Ordering::Relaxed);
|
||||
// An applied re-sync invalidates the staleness run measured under the OLD offset:
|
||||
// reset the counters and re-arm the clock-based detector if a step had disarmed it.
|
||||
let gen = pump_clock_gen.load(Ordering::Relaxed);
|
||||
if gen != seen_clock_gen {
|
||||
seen_clock_gen = gen;
|
||||
stale_since = None;
|
||||
noop_clock_flushes = 0;
|
||||
// Every OWD reading shifted with the offset — the standing-latency floor and
|
||||
// any elevation measured under the old one are meaningless now. If a stale
|
||||
// offset WAS the elevation, this is also the moment it gets fixed.
|
||||
standing_lat.rebase();
|
||||
if !clock_detector_armed {
|
||||
clock_detector_armed = true;
|
||||
tracing::info!("clock re-sync applied — clock-based jump-to-live re-armed");
|
||||
}
|
||||
}
|
||||
// Mirror the reassembler's unrecoverable-drop count for the client's keyframe-recovery
|
||||
// loop, and (during a speed test) the packet-level receive counters for the throughput
|
||||
// measurement. Updated every iteration (not just on a produced frame) so they stay current
|
||||
// through a total-loss drought where no AU completes. Cheap: a few relaxed atomic loads.
|
||||
let st = session.stats();
|
||||
frames_dropped.store(st.frames_dropped, Ordering::Relaxed);
|
||||
fec_recovered.store(st.fec_recovered_shards, Ordering::Relaxed);
|
||||
let probe_active = {
|
||||
let mut p = pump_probe.lock().unwrap();
|
||||
if p.active && !p.done {
|
||||
p.rx_packets_now = st.packets_received;
|
||||
p.rx_bytes_now = st.bytes_received;
|
||||
p.base_packets.get_or_insert(st.packets_received);
|
||||
p.base_bytes.get_or_insert(st.bytes_received);
|
||||
}
|
||||
p.active && !p.done
|
||||
};
|
||||
// A probe just ended (either kind): rebase EVERY window anchor past the burst. Its
|
||||
// FLAG_PROBE filler landed in `bytes_received`/`packets_received` (session.rs counts
|
||||
// every accepted datagram) but never reached the decoder, and the report tick was
|
||||
// suppressed for the whole burst, so `last_*` still points before it. Without this the
|
||||
// first post-burst window reads the burst rate as `actual_kbps` and poisons the ABR's
|
||||
// monotone proven-throughput high-water mark — which never decays — and divides the
|
||||
// window's loss by a packet count inflated with filler.
|
||||
if was_probing && !probe_active {
|
||||
last_recovered = st.fec_recovered_shards;
|
||||
last_late = st.fec_late_shards;
|
||||
last_received = st.packets_received;
|
||||
last_dropped = st.frames_dropped;
|
||||
last_bytes = st.bytes_received;
|
||||
last_report = Instant::now();
|
||||
}
|
||||
// Arm a watchdog on the leading edge of ANY probe, so a host that silently ignores
|
||||
// `ProbeRequest` (an old build — anticipated, see the capacity-probe timeout below)
|
||||
// cannot latch `active` forever and suppress the report tick for the whole session.
|
||||
if !was_probing && probe_active {
|
||||
let burst = Duration::from_millis(pump_probe.lock().unwrap().duration_ms as u64);
|
||||
probe_watchdog = Some(Instant::now() + burst + CAPACITY_PROBE_TIMEOUT);
|
||||
}
|
||||
if !probe_active {
|
||||
probe_watchdog = None;
|
||||
} else if let Some(deadline) = probe_watchdog {
|
||||
if Instant::now() >= deadline {
|
||||
probe_watchdog = None;
|
||||
pump_probe.lock().unwrap().active = false;
|
||||
tracing::warn!(
|
||||
"speed-test probe unanswered — clearing it so loss reports and ABR resume"
|
||||
);
|
||||
}
|
||||
}
|
||||
was_probing = probe_active;
|
||||
// Fire the startup link-capacity probe once the stream has settled (see the constants
|
||||
// above), and fold its measurement into the ABR ceiling when the result lands.
|
||||
// Never steal the slot from an embedder speed test in flight: there is one `ProbeState`
|
||||
// and no correlation id, so a clobber both wrecks the user's "Test connection" figure
|
||||
// (its base counters get re-snapshotted mid-burst against the full-burst denominator)
|
||||
// and mis-scales our own ceiling. Retry once it finishes.
|
||||
if capacity_probe_at.is_some_and(|at| Instant::now() >= at) && probe_active {
|
||||
capacity_probe_at = Some(Instant::now() + CAPACITY_PROBE_DELAY);
|
||||
} else if capacity_probe_at.is_some_and(|at| Instant::now() >= at) {
|
||||
capacity_probe_at = None;
|
||||
*pump_probe.lock().unwrap() = ProbeState {
|
||||
active: true,
|
||||
duration_ms: CAPACITY_PROBE_MS,
|
||||
..Default::default()
|
||||
};
|
||||
if ctrl_tx
|
||||
.try_send(CtrlRequest::Probe(ProbeRequest {
|
||||
target_kbps: CAPACITY_PROBE_KBPS,
|
||||
duration_ms: CAPACITY_PROBE_MS,
|
||||
}))
|
||||
.is_ok()
|
||||
{
|
||||
capacity_probe_deadline = Some(Instant::now() + CAPACITY_PROBE_TIMEOUT);
|
||||
tracing::info!(
|
||||
target_kbps = CAPACITY_PROBE_KBPS,
|
||||
duration_ms = CAPACITY_PROBE_MS,
|
||||
"adaptive bitrate: startup link-capacity probe"
|
||||
);
|
||||
} else {
|
||||
pump_probe.lock().unwrap().active = false; // ctrl queue full — skip
|
||||
}
|
||||
}
|
||||
if let Some(deadline) = capacity_probe_deadline {
|
||||
let mut p = pump_probe.lock().unwrap();
|
||||
if p.done {
|
||||
capacity_probe_deadline = None;
|
||||
// An all-zero reply is a decline (old host / probe-less build) — keep the
|
||||
// negotiated ceiling. Otherwise: delivered wire kbps × 0.7.
|
||||
if p.host_duration_ms > 0 && p.delivered_bytes > 0 {
|
||||
let delivered_kbps = (p.delivered_bytes.saturating_mul(8)
|
||||
/ p.host_duration_ms.max(1) as u64)
|
||||
as u32;
|
||||
let ceiling = delivered_kbps.saturating_mul(7) / 10;
|
||||
abr.set_ceiling(ceiling);
|
||||
tracing::info!(
|
||||
delivered_kbps,
|
||||
ceiling_kbps = ceiling,
|
||||
"adaptive bitrate: link-capacity probe done — climb ceiling set"
|
||||
);
|
||||
} else {
|
||||
tracing::info!(
|
||||
"adaptive bitrate: capacity probe declined — keeping negotiated ceiling"
|
||||
);
|
||||
}
|
||||
// The probe's FLAG_PROBE filler landed in `bytes_received` but never reached
|
||||
// the decoder — rebase the ABR window's byte counter past it, or the next
|
||||
// window's "actual throughput" reads as the burst rate and poisons the
|
||||
// controller's proven-throughput high-water mark with the LINK rate.
|
||||
last_bytes = st.bytes_received;
|
||||
} else if Instant::now() >= deadline {
|
||||
// The host never answered (a build that ignores ProbeRequest): clear the
|
||||
// stuck-active state so LossReports resume, keep the negotiated ceiling.
|
||||
p.active = false;
|
||||
capacity_probe_deadline = None;
|
||||
tracing::info!(
|
||||
"adaptive bitrate: capacity probe timed out (old host?) — keeping negotiated ceiling"
|
||||
);
|
||||
}
|
||||
}
|
||||
if !probe_active && last_report.elapsed() >= ADAPT_REPORT_INTERVAL {
|
||||
// A no-op clock flush earlier in this window suspected a wall-clock step: fire
|
||||
// the mid-stream re-sync now (once — the 60 s periodic covers everything else).
|
||||
if resync_wanted {
|
||||
resync_wanted = false;
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::ClockResync);
|
||||
}
|
||||
let window_dropped = st.frames_dropped.wrapping_sub(last_dropped);
|
||||
let loss_ppm = window_loss_ppm(
|
||||
st.fec_recovered_shards.wrapping_sub(last_recovered),
|
||||
st.fec_late_shards.wrapping_sub(last_late),
|
||||
st.packets_received.wrapping_sub(last_received),
|
||||
window_dropped,
|
||||
);
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::Loss(LossReport { loss_ppm }));
|
||||
// Standing-latency bleed: close the detector's window with this report's loss
|
||||
// verdict and run its escalation ladder — re-sync first (free; a stale offset
|
||||
// from a stepped wall clock produces exactly this signature and the applied
|
||||
// re-sync rebases the floor), then a bounded flush+keyframe (drains a real
|
||||
// sub-threshold standing backlog the jump-to-live thresholds tolerate), then a
|
||||
// loud disarm (the path latency itself changed; nothing local fixes that).
|
||||
match standing_lat.on_window(loss_ppm == 0 && window_dropped == 0) {
|
||||
StandingLatAction::None => {}
|
||||
StandingLatAction::Resync { above_ms } => {
|
||||
tracing::info!(
|
||||
above_ms,
|
||||
"standing latency above the session floor with zero loss — \
|
||||
requesting a clock re-sync first (a stale offset reads exactly \
|
||||
like this)"
|
||||
);
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::ClockResync);
|
||||
}
|
||||
StandingLatAction::Bleed { above_ms } => {
|
||||
// Shares the jump-to-live cooldown: an unexecuted bleed simply re-arms
|
||||
// over the next windows (the detector's run rebuilds).
|
||||
if last_flush.is_none_or(|t| t.elapsed() >= FLUSH_COOLDOWN) {
|
||||
last_flush = Some(Instant::now());
|
||||
// Deliberately NOT `flush_in_window = true`: that flag is the ABR's
|
||||
// SEVERE verdict (an immediate ×0.7 back-off), and the bleed fires
|
||||
// only after ~6 provably loss-free windows with a sub-25ms elevation
|
||||
// the controller itself scores as fine. The bleed's effect reaches
|
||||
// the ABR through the window's own honest signals (OWD/loss/decode);
|
||||
// the flag stays exclusive to the jump-to-live path below.
|
||||
let flushed = session.flush_backlog().unwrap_or(0);
|
||||
let dropped = frames.clear();
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::Keyframe);
|
||||
standing_lat.bled();
|
||||
tracing::warn!(
|
||||
above_ms,
|
||||
flushed_datagrams = flushed,
|
||||
dropped_frames = dropped,
|
||||
"standing latency survived a clock re-sync — bled the local \
|
||||
backlog (flush + keyframe)"
|
||||
);
|
||||
}
|
||||
}
|
||||
StandingLatAction::Disarm { above_ms } => {
|
||||
tracing::warn!(
|
||||
above_ms,
|
||||
"standing latency persists after a re-sync and every bleed — not \
|
||||
local, not clock; the path latency changed. Leaving it be \
|
||||
(reconnect re-baselines)"
|
||||
);
|
||||
}
|
||||
}
|
||||
// Adaptive bitrate: drain any host ack first (its clamp is authoritative), then
|
||||
// feed the controller this window's congestion signals; a decision becomes a
|
||||
// SetBitrate on the control stream.
|
||||
if let Some(acked) = bitrate_ack.lock().unwrap().take() {
|
||||
abr.on_ack(acked);
|
||||
}
|
||||
let owd_mean_us =
|
||||
(owd_frames > 0).then(|| (owd_sum_ns / owd_frames as i128 / 1000) as i64);
|
||||
(owd_sum_ns, owd_frames) = (0, 0);
|
||||
// Drain the embedder's decode-latency window (always, so it stays bounded even when
|
||||
// the controller is disabled) → the mean feeds the decode signal; `None` when the
|
||||
// embedder reported nothing this window (old embedder / no decoded frames).
|
||||
let decode_mean_us = {
|
||||
let mut acc = pump_decode_lat.lock().unwrap();
|
||||
let (sum, count) = (acc.sum_us, acc.count);
|
||||
*acc = DecodeLatAcc::default();
|
||||
(count > 0).then(|| (sum / count as u64) as i64)
|
||||
};
|
||||
// The window's ACTUAL delivered throughput — what the pipeline really carried, vs
|
||||
// the target it was allowed. Wire bytes (headers + FEC) slightly overstate the
|
||||
// media rate the decoder ingests; acceptable for the climb gate / proven-mark
|
||||
// semantics (both compare against targets with their own headroom).
|
||||
let window_ms = last_report.elapsed().as_millis().max(1) as u64;
|
||||
let actual_kbps = (st.bytes_received.wrapping_sub(last_bytes).saturating_mul(8)
|
||||
/ window_ms) as u32;
|
||||
if let Some(kbps) = abr.on_window(
|
||||
Instant::now(),
|
||||
window_dropped,
|
||||
loss_ppm,
|
||||
owd_mean_us,
|
||||
decode_mean_us,
|
||||
actual_kbps,
|
||||
flush_in_window,
|
||||
) {
|
||||
// Log the window's signals alongside the decision so an on-glass session can
|
||||
// tell a decode-driven re-target (the new signal — decode_mean_us elevated with
|
||||
// loss/OWD flat) from a network-driven one.
|
||||
tracing::info!(
|
||||
kbps,
|
||||
loss_ppm,
|
||||
owd_mean_us = owd_mean_us.unwrap_or(-1),
|
||||
decode_mean_us = decode_mean_us.unwrap_or(-1),
|
||||
actual_kbps,
|
||||
flushed = flush_in_window,
|
||||
"adaptive bitrate: requesting encoder re-target"
|
||||
);
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::SetBitrate(kbps));
|
||||
}
|
||||
flush_in_window = false;
|
||||
last_report = Instant::now();
|
||||
last_recovered = st.fec_recovered_shards;
|
||||
last_late = st.fec_late_shards;
|
||||
last_received = st.packets_received;
|
||||
last_dropped = st.frames_dropped;
|
||||
last_bytes = st.bytes_received;
|
||||
if pump_perf_on {
|
||||
if let Some(p) = session.take_pump_perf() {
|
||||
let per_pkt_ns = |ns: u64| ns.checked_div(p.packets).unwrap_or(0);
|
||||
tracing::info!(
|
||||
recv_ms = p.recv_ns / 1_000_000,
|
||||
decrypt_ms = p.decrypt_ns / 1_000_000,
|
||||
reasm_ms = p.reasm_ns / 1_000_000,
|
||||
packets = p.packets,
|
||||
batches = p.batches,
|
||||
pkts_per_batch = p.packets.checked_div(p.batches).unwrap_or(0),
|
||||
decrypt_ns_pkt = per_pkt_ns(p.decrypt_ns),
|
||||
reasm_ns_pkt = per_pkt_ns(p.reasm_ns),
|
||||
"pump stage split (window)"
|
||||
);
|
||||
}
|
||||
// Inter-arrival jitter over the window's completed AUs. `late` counts gaps
|
||||
// over 2× the window median — the "a frame arrived visibly off-beat" tally.
|
||||
if arrivals_us.len() >= 8 {
|
||||
arrivals_us.sort_unstable();
|
||||
let pct = |q: usize| arrivals_us[(arrivals_us.len() - 1) * q / 100];
|
||||
let (p50, p95) = (pct(50), pct(95));
|
||||
let late = arrivals_us.iter().filter(|&&d| d > p50 * 2).count();
|
||||
tracing::info!(
|
||||
frames = arrivals_us.len() + 1,
|
||||
arrival_p50_us = p50,
|
||||
arrival_p95_us = p95,
|
||||
arrival_max_us = arrivals_us.last().copied().unwrap_or(0),
|
||||
late,
|
||||
"frame inter-arrival jitter (window)"
|
||||
);
|
||||
}
|
||||
arrivals_us.clear();
|
||||
}
|
||||
}
|
||||
match session.poll_frame() {
|
||||
Ok(frame) => {
|
||||
if frame.flags & FLAG_PROBE as u32 != 0 {
|
||||
continue; // speed-test filler, not video — measured via the counters above
|
||||
}
|
||||
if pump_perf_on {
|
||||
let now = Instant::now();
|
||||
if let Some(prev) = last_arrival.replace(now) {
|
||||
// 4096 ≈ 17 s at 240 fps — a stuck window can't grow it unbounded.
|
||||
if arrivals_us.len() < 4096 {
|
||||
arrivals_us
|
||||
.push((now - prev).as_micros().min(u32::MAX as u128) as u32);
|
||||
}
|
||||
}
|
||||
}
|
||||
// Jump-to-live guard. A standing receive/hand-off queue never drains by itself —
|
||||
// the pump consumes strictly in order at the arrival rate, so once behind, the
|
||||
// stream stays behind for good (observed live: stuck 6–7 s). Pre-decode AUs are
|
||||
// reference-chained (infinite GOP), so we can NOT drop a frame mid-stream to catch
|
||||
// up; the only safe recovery is to discard the whole backlog and re-anchor decode
|
||||
// on a fresh keyframe. Two independent "we're behind" signals arm it, both gated by
|
||||
// FLUSH_COOLDOWN, both suspended during a speed test (the probe MEASURES a saturated
|
||||
// queue; flushing would corrupt its counters):
|
||||
// * clock-based — completed frames sit > FLUSH_LATENCY behind the skew-corrected
|
||||
// capture clock continuously for FLUSH_AFTER. Needs the skew handshake, and
|
||||
// also catches kernel/reassembler backlog the hand-off queue hasn't reached yet.
|
||||
// * clock-free — the pre-decode hand-off queue stopped draining: its depth stayed
|
||||
// ≥ QUEUE_HIGH (never falling to QUEUE_LOW, still high at the trip) for
|
||||
// STANDING_TIME. Works with no handshake / a same-clock session (where the
|
||||
// clock path is disarmed), and is the direct signal that the embedder can't
|
||||
// keep up. A transient Wi-Fi clump drains within ~100 ms and never trips it.
|
||||
if probe_active {
|
||||
// Keep both detectors disarmed across a speed test so its (deliberately)
|
||||
// saturated queue doesn't leave a primed run that fires the moment it ends.
|
||||
stale_since = None;
|
||||
standing_since = None;
|
||||
} else {
|
||||
let lat_ns = if clock_offset_ns != 0 {
|
||||
now_realtime_ns() + clock_offset_ns as i128 - frame.pts_ns as i128
|
||||
} else {
|
||||
0
|
||||
};
|
||||
// Feed the adaptive-bitrate controller's OWD window (mean capture→received
|
||||
// delay): rising delay under zero loss is queue growth — the pre-loss
|
||||
// congestion signal. Only meaningful with a clock handshake.
|
||||
if clock_offset_ns != 0 && lat_ns > 0 {
|
||||
owd_sum_ns += lat_ns;
|
||||
owd_frames += 1;
|
||||
// The standing-latency detector rides the same signal, but off the
|
||||
// window MINIMUM (robust against jitter/burst spikes — a standing
|
||||
// state elevates the floor itself). Same 10 s plausibility clamp as
|
||||
// the hn stats use.
|
||||
if lat_ns < 10_000_000_000 {
|
||||
standing_lat.note_frame(lat_ns);
|
||||
}
|
||||
}
|
||||
if clock_detector_armed
|
||||
&& clock_offset_ns != 0
|
||||
&& lat_ns > FLUSH_LATENCY.as_nanos() as i128
|
||||
{
|
||||
stale_since.get_or_insert_with(Instant::now);
|
||||
} else {
|
||||
stale_since = None;
|
||||
}
|
||||
let depth = frames.depth();
|
||||
if depth >= QUEUE_HIGH {
|
||||
standing_since.get_or_insert_with(Instant::now);
|
||||
} else if depth <= QUEUE_LOW {
|
||||
standing_since = None;
|
||||
}
|
||||
// The queue trip additionally requires the depth to still be high NOW, so
|
||||
// a run that started ≥ high but is hovering in the hysteresis band (a
|
||||
// clump mid-drain) never fires on elapsed time alone.
|
||||
let clock_behind = stale_since.is_some_and(|t| t.elapsed() >= FLUSH_AFTER);
|
||||
let queue_behind = depth >= QUEUE_HIGH
|
||||
&& standing_since.is_some_and(|t| t.elapsed() >= STANDING_TIME);
|
||||
if (clock_behind || queue_behind)
|
||||
&& last_flush.is_none_or(|t| t.elapsed() >= FLUSH_COOLDOWN)
|
||||
{
|
||||
stale_since = None;
|
||||
standing_since = None;
|
||||
last_flush = Some(Instant::now());
|
||||
flush_in_window = true; // strongest "link can't hold the rate" signal
|
||||
let flushed = session.flush_backlog().unwrap_or(0);
|
||||
let dropped = frames.clear();
|
||||
let _ = ctrl_tx.try_send(CtrlRequest::Keyframe);
|
||||
tracing::warn!(
|
||||
behind_ms = if clock_behind { lat_ns / 1_000_000 } else { -1 },
|
||||
queue_depth = depth,
|
||||
flushed_datagrams = flushed,
|
||||
dropped_frames = dropped,
|
||||
"receive backlog stopped draining — jumped to live (flush + keyframe)"
|
||||
);
|
||||
// Clock-detector health check: a clock-only trigger whose flush found
|
||||
// no local backlog is a false "behind" reading (a wall-clock step, or
|
||||
// an upstream queue a local flush can't drain) — repeated, it would
|
||||
// cost a recovery IDR every cooldown forever. Disarm after two in a
|
||||
// row; the clock-free queue detector keeps covering real backlogs.
|
||||
if clock_behind
|
||||
&& !queue_behind
|
||||
&& flushed < NOOP_FLUSH_DATAGRAMS
|
||||
&& dropped == 0
|
||||
{
|
||||
noop_clock_flushes += 1;
|
||||
if noop_clock_flushes == 1 {
|
||||
// First no-op flush = a wall-clock step is the prime
|
||||
// suspect: ask for an immediate re-sync (sent on the next
|
||||
// report tick). Applied, it resets these counters and
|
||||
// re-arms the detector before the disarm below triggers.
|
||||
resync_wanted = true;
|
||||
}
|
||||
if noop_clock_flushes >= NOOP_CLOCK_FLUSHES_TO_DISARM {
|
||||
clock_detector_armed = false;
|
||||
tracing::warn!(
|
||||
"clock-based jump-to-live disarmed — its flushes found no \
|
||||
local backlog (clock step or upstream queueing suspected); \
|
||||
the queue-depth detector stays armed"
|
||||
);
|
||||
}
|
||||
} else {
|
||||
noop_clock_flushes = 0;
|
||||
}
|
||||
continue; // this frame is part of the stale past — don't render it
|
||||
}
|
||||
}
|
||||
frames.push(frame);
|
||||
}
|
||||
Err(PunktfunkError::NoFrame) => {
|
||||
std::thread::sleep(Duration::from_micros(300));
|
||||
}
|
||||
Err(_) => break,
|
||||
}
|
||||
}
|
||||
// The pump exited (shutdown / fatal session error) — wake any consumer blocked in
|
||||
// `next_frame` with a Closed signal instead of a spurious timeout (the old mpsc did this
|
||||
// implicitly when the sender dropped).
|
||||
frames.close();
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user