The macOS device-change recovery answered itself — mic-on streams cut audio and input every ~2.5 s #221

Merged
enricobuehler merged 1 commits from worktree-macos-mic-rebuild-loop into main 2026-08-14 10:49:39 +00:00
7 changed files with 315 additions and 24 deletions
+25
View File
@@ -456,6 +456,31 @@ otherwise had silently run stage-4, because the gate keyed on build config.
renders code 16 as grey rather than black. Now gated on `connection.isHDR` as well; layout re-runs
it, so a session that flips to HDR mid-stream still picks the mode up.
### Apple — the macOS device-change recovery could answer itself forever (mic on)
**Streaming from a Mac with the microphone enabled cut audio AND input on a ~2.5 s metronome
while video ran untouched** (field, 2026-08-14: a Mac Studio whose default input is a 6-channel
device). The chain: the voice-processing engine cannot start on that mic, every rebuild re-tried
it, and the failed attempt's HAL churn (VPIO builds and tears down an aggregate device) stopped
the healthy fallback engines — which posted the `AVAudioEngineConfigurationChange` that scheduled
the next rebuild. Each ~1.9 s rebuild runs on the main thread, where macOS input capture and
sending live, so input froze on the same beat — and since audio, input and mic share the QUIC
datagram plane while video rides its own socket, the wire signature read as a network fault and
the host's METRONOMIC heuristic pointed at the display stack. Three defenses, layered because no
single one covers every feedback shape:
- **A voice-processing start failure latches per input device** (`CombinedTopologyGate`): a
rebuild goes straight to the split topology instead of re-running a failure that is a property
of the device. A different default input earns exactly one fresh attempt.
- **A configuration change posted by an engine that is RUNNING is the rebuild's own echo, and is
ignored**: an engine stops itself before posting, so a live poster was already restarted.
- **Rebuilds that chain anyway back off exponentially** (`RebuildBackoff`: 0.5 s floor doubling
to a 30 s cap, reset by 10 s of quiet) — an unforeseen loop costs one blip per half-minute
instead of a metronome, and the chaining itself logs a WARN that names the condition.
iOS/tvOS behaviour is untouched (routes are session-managed there; nothing is latched). Until a
client carries this, the field workaround is turning the client microphone off.
### Apple gamepad UI — a host menu, and About becomes a page
**UP on a saved tile opens Wake / Copy link / Edit… / Forget pairing / Remove.** The desktop and
@@ -32,8 +32,11 @@ final class AudioDeviceWatcher {
/// posts one last change as it is torn down, and other AVAudioEngines in the process are not
/// ours to restart.
private let isOurs: (AnyObject?) -> Bool
/// Delivered on the main queue.
private let onChange: (Reason) -> Void
/// Delivered on the main queue. The second argument is the engine that posted the change
/// (`.engineConfiguration` only; nil for the HAL listener) the owner needs the OBJECT, not
/// just the reason, because an engine that is RUNNING when the notification lands is one the
/// owner already restarted: acting on that echo is how a rebuild loop starts.
private let onChange: (Reason, AnyObject?) -> Void
private let lock = NSLock()
private var configObserver: NSObjectProtocol?
@@ -41,7 +44,7 @@ final class AudioDeviceWatcher {
private var defaultOutputListener: AudioObjectPropertyListenerBlock?
#endif
init(isOurs: @escaping (AnyObject?) -> Bool, onChange: @escaping (Reason) -> Void) {
init(isOurs: @escaping (AnyObject?) -> Bool, onChange: @escaping (Reason, AnyObject?) -> Void) {
self.isOurs = isOurs
self.onChange = onChange
}
@@ -63,7 +66,7 @@ final class AudioDeviceWatcher {
let posted = note.object as AnyObject?
DispatchQueue.main.async {
guard let self, self.isOurs(posted) else { return }
self.onChange(.engineConfiguration)
self.onChange(.engineConfiguration, posted)
}
}
lock.lock()
@@ -77,7 +80,8 @@ final class AudioDeviceWatcher {
// (the voice-processing engine, which is the DEFAULT macOS configuration and which no Mac
// here can even initialize). The HAL is told either way.
let block: AudioObjectPropertyListenerBlock = { [weak self] _, _ in
self?.onChange(.defaultOutputDevice) // on the main queue registered against it below
// On the main queue registered against it below. No engine posted this, so nil.
self?.onChange(.defaultOutputDevice, nil)
}
var address = Self.defaultOutputAddress()
let status = AudioObjectAddPropertyListenerBlock(
@@ -42,7 +42,10 @@ public enum AudioDevices {
return channelCount(id, scope: kAudioObjectPropertyScopeInput)
}
private static func defaultInputDevice() -> AudioDeviceID? {
/// The device the system is currently capturing from the key `SessionAudio`'s
/// voice-processing gate latches a start failure against (the failure is a property of the
/// input device, so a new device earns a fresh attempt).
static func defaultInputDevice() -> AudioDeviceID? {
systemDevice(kAudioHardwarePropertyDefaultInputDevice)
}
@@ -0,0 +1,89 @@
// The two policy decisions of the device-change recovery, extracted where a unit test can reach
// them. Both exist because of one field incident (2026-08-14, Mac Studio): the voice-processing
// engine could not start on a 6-channel input device, every rebuild re-tried it, and the failed
// attempt's HAL churn (VPIO builds and tears down an aggregate device) re-stopped the fallback
// engines which posted the configuration change that scheduled the next rebuild. A ~2.5 s
// metronome of audio gaps, forever, with each rebuild also stalling the main thread (where macOS
// input capture lives), so the stream's INPUT cut out on the same beat. The session-side wiring
// lives in `SessionAudio`; the decisions live here because the loop shipped precisely because
// they could not be tested without a mic and a session.
#if os(macOS)
import CoreAudio
#endif
import Foundation
#if os(macOS)
/// Should a rebuild try the combined (voice-processing) topology again?
///
/// A VPIO start failure is a property of the INPUT DEVICE (its channel count and format), not of
/// the moment: retrying it on the same device fails the same way, and the attempt is not free
/// engaging and abandoning the voice processor churns the HAL hard enough to stop the healthy
/// fallback engines. So a failure latches until the default input actually changes; a new device
/// earns exactly one fresh attempt (it may well support VPIO), and its own failure latches again.
struct CombinedTopologyGate {
private var failed = false
/// The default input device the failure was observed on nil is a real value here ("failed
/// with no resolvable input device"), which is why `failed` is tracked separately.
private var failedInput: AudioDeviceID?
/// The combined topology failed with `input` as the default input device.
mutating func noteFailure(input: AudioDeviceID?) {
failed = true
failedInput = input
}
/// True when the combined topology is worth attempting with `input` as the default input
/// device. A device change clears the latch the answer is about the CURRENT hardware, and
/// coming back to a device that failed before earns a fresh attempt too (the failure may have
/// been the mid-transition kind, and one attempt per device change cannot loop).
mutating func shouldTry(input: AudioDeviceID?) -> Bool {
guard failed else { return true }
guard input == failedInput else {
failed = false
failedInput = nil
return true
}
return false
}
}
#endif
/// The delay before the next engine rebuild the base debounce/floor behaviour, plus an
/// escalating floor when rebuilds CHAIN (each one retriggered by its predecessor's own fallout).
///
/// One device switch produces one rebuild: its trigger burst is coalesced upstream, so the next
/// trigger normally arrives minutes later and gets the base floor. A trigger that arrives hard on
/// the heels of the last rebuild, again and again, is a rebuild answering itself and since the
/// recovery cannot always identify its own echo, the backstop is to keep answering but at a
/// doubling floor, so an unforeseen feedback shape costs one audio blip per half-minute instead
/// of a metronome. A quiet stretch resets the ladder to full responsiveness.
struct RebuildBackoff {
/// Let the burst of triggers from one switch land before rebuilding.
static let debounce: TimeInterval = 0.15
/// Floor between two rebuilds.
static let floor: TimeInterval = 0.5
/// The escalated floor's cap: looping recoveries settle at one attempt per this interval.
static let floorCap: TimeInterval = 30
/// A trigger this long after the last rebuild is unrelated to it the chain resets.
static let chainWindow: TimeInterval = 10
/// Consecutive rebuilds whose trigger arrived within `chainWindow` of the previous rebuild.
private(set) var chain = 0
private var lastRebuildAt: TimeInterval = -.infinity
/// The delay to schedule the next rebuild with, for a trigger arriving at `now`
/// (`systemUptime`). Mutates the chain accounting: call once per SCHEDULED rebuild, not per
/// coalesced trigger.
mutating func delay(now: TimeInterval) -> TimeInterval {
let since = now - lastRebuildAt
chain = since < Self.chainWindow ? chain + 1 : 0
let floor = min(Self.floor * pow(2, Double(min(chain, 6))), Self.floorCap)
return max(Self.debounce, floor - since)
}
/// The rebuild actually ran at `now` the reference the next trigger's `delay` measures from.
mutating func noteRebuild(at now: TimeInterval) {
lastRebuildAt = now
}
}
@@ -117,13 +117,16 @@ public final class SessionAudio {
/// A rebuild is already on the main queue one device switch produces a burst of triggers
/// and they must collapse into one restart. Main-thread confined.
private var rebuildQueued = false
/// `systemUptime` of the last rebuild, so a device that renegotiates in a loop cannot spin
/// the session. Main-thread confined.
private var lastRebuildAt: TimeInterval = 0
/// Let the burst of triggers from one switch land before rebuilding.
private static let rebuildDebounce: TimeInterval = 0.15
/// Floor between two rebuilds.
private static let rebuildFloor: TimeInterval = 0.5
/// Debounce/floor for the next rebuild, with an escalating floor when rebuilds chain (each
/// retriggered by its predecessor see `RebuildBackoff`). Main-thread confined.
private var rebuildBackoff = RebuildBackoff()
#if os(macOS)
/// Latches a voice-processing start failure per input device, so a rebuild never re-attempts
/// a topology that deterministically fails the retry is what turned one failure into a
/// rebuild loop (see `CombinedTopologyGate` and the note on `installDeviceChangeRecovery`).
/// Main-thread confined, like the start paths that consult it.
private var combinedGate = CombinedTopologyGate()
#endif
/// Retries when a rebuild's `start()` loses the race with a device that is still going away
/// (0.3 s, 0.6 s, 1.2 s). A failed rebuild leaves no engine to post the next notification,
/// so this ladder and, on macOS, the HAL listener is all that stands between a mistimed
@@ -356,9 +359,25 @@ public final class SessionAudio {
startPlayback(speakerUID: speakerUID)
return
}
#if os(macOS)
// A rebuild must not re-attempt a voice-processing start that already failed on this
// input device: the failure repeats, and the failed attempt's HAL churn stops the healthy
// fallback engines the 2026-08-14 rebuild loop (see `CombinedTopologyGate`).
var combined = wantsCombined(
speakerUID: speakerUID, micUID: micUID, micChannel: micChannel,
echoCancel: echoCancel)
if combined, !combinedGate.shouldTry(input: AudioDevices.defaultInputDevice()) {
log.info("""
voice processing already failed on this input device split engines, no echo \
cancellation
""")
combined = false
}
#else
let combined = wantsCombined(
speakerUID: speakerUID, micUID: micUID, micChannel: micChannel,
echoCancel: echoCancel)
#endif
switch AVCaptureDevice.authorizationStatus(for: .audio) {
case .authorized:
if combined {
@@ -513,6 +532,17 @@ public final class SessionAudio {
/// - the route-change and media-services-reset notifications, iOS/tvOS, where the session and
/// not the device is what moves.
///
/// And three defenses keep the recovery from ANSWERING ITSELF a rebuild is not a silent
/// act (a voice-processing start builds and tears down HAL aggregates, and every fresh engine
/// renegotiates its IO), so its own fallout can retrigger it. The 2026-08-14 field loop was
/// exactly that: VPIO failed on a 6-channel mic, every rebuild re-tried it, and the failure's
/// churn stopped the fallback engines audio and (via the main thread) INPUT cutting out
/// every ~2.5 s for the whole session. The defenses: a configuration change from an engine
/// that is RUNNING is a rebuild's echo and is ignored (`hardwareMoved`); a VPIO failure is
/// latched per input device and never re-attempted on it (`CombinedTopologyGate`); and
/// rebuilds that chain anyway back off exponentially instead of metronoming
/// (`RebuildBackoff`).
///
/// `micEnabled` only decides whether the mic-bearing session observers are worth installing.
/// Main thread.
private func installDeviceChangeRecovery(micEnabled: Bool) {
@@ -523,7 +553,7 @@ public final class SessionAudio {
let watcher = AudioDeviceWatcher(
isOurs: { [weak self] posted in self?.ownsEngine(posted) ?? false },
onChange: { [weak self] reason in self?.hardwareMoved(reason) })
onChange: { [weak self] reason, posted in self?.hardwareMoved(reason, posted: posted) })
stateLock.lock()
deviceWatcher = watcher
stateLock.unlock()
@@ -549,10 +579,17 @@ public final class SessionAudio {
/// question is playback still where it should be but they answer it differently: an engine
/// that told us it stopped is definitive, while the default device moving might not concern us
/// at all.
private func hardwareMoved(_ reason: AudioDeviceWatcher.Reason) {
private func hardwareMoved(_ reason: AudioDeviceWatcher.Reason, posted: AnyObject?) {
guard !flag.isStopped else { return }
switch reason {
case .engineConfiguration:
// The engine stops itself BEFORE posting this so an engine that is RUNNING when the
// notification lands on the main queue is one a rebuild already replaced or restarted:
// the notification is the rebuild's own echo, and answering it is how the recovery
// loops. A change that stops the engine again after this posts again, and the HAL
// backstop checks placement independently, so ignoring a live engine's echo can never
// strand a stopped one.
if let engine = posted as? AVAudioEngine, engine.isRunning { return }
scheduleEngineRebuild(reason: reason.rawValue)
case .defaultOutputDevice:
#if os(macOS)
@@ -594,9 +631,18 @@ public final class SessionAudio {
private func scheduleEngineRebuild(reason: String) {
guard !rebuildQueued else { return }
rebuildQueued = true
let since = ProcessInfo.processInfo.systemUptime - lastRebuildAt
let delay = max(Self.rebuildDebounce, Self.rebuildFloor - since)
log.info("\(reason) — restarting the audio engines in \(Int(delay * 1000)) ms")
let delay = rebuildBackoff.delay(now: ProcessInfo.processInfo.systemUptime)
if rebuildBackoff.chain >= 2 {
// Each rebuild is retriggering the next a feedback shape the echo guard and the
// topology gate did not identify. Keep answering (a real recovery must not be
// abandoned), but say what is happening: this line repeating IS the diagnosis.
log.warning("""
audio engine rebuilds are chaining (\(self.rebuildBackoff.chain) in a row \
\(reason)); backing off \(Int(delay * 1000)) ms
""")
} else {
log.info("\(reason) — restarting the audio engines in \(Int(delay * 1000)) ms")
}
DispatchQueue.main.asyncAfter(deadline: .now() + delay) { [weak self] in
self?.rebuildEngines(attempt: 0)
}
@@ -613,7 +659,7 @@ public final class SessionAudio {
private func rebuildEngines(attempt: Int) {
rebuildQueued = false
guard !flag.isStopped, let config = startConfig else { return }
lastRebuildAt = ProcessInfo.processInfo.systemUptime
rebuildBackoff.noteRebuild(at: ProcessInfo.processInfo.systemUptime)
tearDownEngines()
startEngines(
speakerUID: config.speakerUID, micUID: config.micUID, micChannel: config.micChannel,
@@ -638,7 +684,7 @@ public final class SessionAudio {
return
}
rebuildQueued = true // holds off a trigger that would only race this ladder
let delay = Self.rebuildDebounce * Double(1 << (attempt + 1))
let delay = RebuildBackoff.debounce * Double(1 << (attempt + 1))
DispatchQueue.main.asyncAfter(deadline: .now() + delay) { [weak self] in
self?.rebuildEngines(attempt: attempt + 1)
}
@@ -983,6 +1029,17 @@ public final class SessionAudio {
// MARK: - Mic (mic host)
#if !os(tvOS)
/// The combined topology failed to come up. On macOS, latch the input device it failed on so
/// a rebuild goes straight to the split topology instead of re-running the failure the
/// failed attempt is what churns the HAL and retriggers the recovery (see
/// `CombinedTopologyGate`). On iOS routes are session-managed and a VPIO failure is the
/// transient route-transition kind, so nothing is latched there.
private func noteCombinedFailure() {
#if os(macOS)
combinedGate.noteFailure(input: AudioDevices.defaultInputDevice())
#endif
}
/// One engine, both directions: engage the system voice processor on the shared IO unit
/// (AEC + noise suppression + AGC), hang the playback source off its render side and the
/// mic tap off its capture side. Every failure falls back to a WORKING configuration
@@ -1001,6 +1058,7 @@ public final class SessionAudio {
voice processing unavailable (\(error.localizedDescription)) separate \
engines, no echo cancellation
""")
noteCombinedFailure()
startPlayback(speakerUID: speakerUID)
startCapture(micUID: micUID, micChannel: micChannel)
return
@@ -1054,6 +1112,7 @@ public final class SessionAudio {
// processor won't engage at all, already does exactly this; this arm used to give up
// on the mic instead, which is how a whole session could go silent uplink-only.)
engine.stop()
noteCombinedFailure()
startPlayback(speakerUID: speakerUID)
startCapture(micUID: micUID, micChannel: micChannel)
return
@@ -1064,6 +1123,7 @@ public final class SessionAudio {
log.error("combined engine failed to start: \(error.localizedDescription)")
engine.inputNode.removeTap(onBus: 0)
engine.stop()
noteCombinedFailure()
// Same rule: a working mic without echo cancellation beats no mic at all.
startPlayback(speakerUID: speakerUID)
startCapture(micUID: micUID, micChannel: micChannel)
@@ -32,7 +32,7 @@ final class AudioDeviceWatcherTests: XCTestCase {
let engine = AVAudioEngine()
var reasons: [AudioDeviceWatcher.Reason] = []
let watcher = AudioDeviceWatcher(
isOurs: { $0 === engine }, onChange: { reasons.append($0) })
isOurs: { $0 === engine }, onChange: { reason, _ in reasons.append(reason) })
watcher.start()
defer { watcher.stop() }
@@ -51,7 +51,7 @@ final class AudioDeviceWatcherTests: XCTestCase {
let stranger = AVAudioEngine()
var reasons: [AudioDeviceWatcher.Reason] = []
let watcher = AudioDeviceWatcher(
isOurs: { $0 === ours }, onChange: { reasons.append($0) })
isOurs: { $0 === ours }, onChange: { reason, _ in reasons.append(reason) })
watcher.start()
defer { watcher.stop() }
@@ -66,7 +66,7 @@ final class AudioDeviceWatcherTests: XCTestCase {
let engine = AVAudioEngine()
var reasons: [AudioDeviceWatcher.Reason] = []
let watcher = AudioDeviceWatcher(
isOurs: { $0 === engine }, onChange: { reasons.append($0) })
isOurs: { $0 === engine }, onChange: { reason, _ in reasons.append(reason) })
watcher.start()
watcher.stop()
@@ -93,7 +93,7 @@ final class AudioDeviceWatcherTests: XCTestCase {
}
var reasons: [AudioDeviceWatcher.Reason] = []
let watcher = AudioDeviceWatcher(isOurs: { _ in false }, onChange: { reasons.append($0) })
let watcher = AudioDeviceWatcher(isOurs: { _ in false }, onChange: { reason, _ in reasons.append(reason) })
watcher.start()
defer {
_ = Self.setDefaultOutput(original)
@@ -0,0 +1,110 @@
// The two decisions that ended the 2026-08-14 rebuild loop, driven with a synthetic clock.
//
// The loop's shape, for the plant-the-defect cases below: the voice-processing engine fails to
// start (~1.9 s spent trying), the fallback comes up, and its own HAL fallout retriggers the
// recovery ~0.6 s later forever. Restore either defect (retry the failed topology, or keep the
// flat 0.5 s floor) and the session pays an audio gap every ~2.5 s for as long as it lives.
import XCTest
@testable import PunktfunkKit
final class AudioRebuildPolicyTests: XCTestCase {
// MARK: - RebuildBackoff
/// The first trigger of a session keeps the old behaviour: the burst-coalescing debounce.
func testFirstTriggerWaitsOnlyTheDebounce() {
var backoff = RebuildBackoff()
XCTAssertEqual(backoff.delay(now: 1000), RebuildBackoff.debounce)
}
/// One rebuild, then quiet: the next real device switch minutes later is answered at full
/// responsiveness the ladder must never make a HEALTHY recovery sluggish.
func testAnIsolatedSwitchLongAfterTheLastRebuildResetsTheChain() {
var backoff = RebuildBackoff()
_ = backoff.delay(now: 1000)
backoff.noteRebuild(at: 1000.2)
// Chained once (a second switch soon after legitimate, e.g. AirPods out then back in).
_ = backoff.delay(now: 1001)
backoff.noteRebuild(at: 1002)
// Minutes of quiet, then a fresh switch: base debounce again, chain forgotten.
XCTAssertEqual(backoff.delay(now: 1300), RebuildBackoff.debounce)
XCTAssertEqual(backoff.chain, 0)
}
/// THE FIELD LOOP, against the real constants: a trigger 0.6 s after every rebuild, ten
/// minutes long. The flat 0.5 s floor produced a rebuild every ~2.5 s ~240 audio gaps.
/// The ladder must cut that by an order of magnitude and settle at the floor cap.
func testAChainedLoopBacksOffToTheFloorCap() {
var backoff = RebuildBackoff()
var now: TimeInterval = 0
var rebuilds = 0
var lastDelay: TimeInterval = 0
let end: TimeInterval = 600
while now < end {
lastDelay = backoff.delay(now: now)
now += lastDelay // the scheduled rebuild fires...
backoff.noteRebuild(at: now)
rebuilds += 1
now += 0.6 // ...and its fallout retriggers the recovery 0.6 s later.
}
XCTAssertEqual(
lastDelay, RebuildBackoff.floorCap - 0.6, accuracy: 0.01,
"a persistent loop should settle at one rebuild per floorCap")
XCTAssertLessThanOrEqual(
rebuilds, 30,
"\(rebuilds) rebuilds in 10 min — the ladder is not escalating (the shipped flat "
+ "floor produced ~240)")
// And the loop's END must restore responsiveness: quiet, then a real switch.
XCTAssertEqual(backoff.delay(now: now + 120), RebuildBackoff.debounce)
}
/// The ladder's exponent is clamped a loop that runs for hours must neither overflow nor
/// push the interval past the cap.
func testTheFloorNeverExceedsTheCap() {
var backoff = RebuildBackoff()
var now: TimeInterval = 0
for _ in 0..<1000 {
let delay = backoff.delay(now: now)
XCTAssertLessThanOrEqual(delay, RebuildBackoff.floorCap)
now += delay
backoff.noteRebuild(at: now)
now += 0.1
}
}
#if os(macOS)
// MARK: - CombinedTopologyGate
/// The loop's fuel: re-attempting the voice-processing start that just failed. Same input
/// device never again.
func testAFailureLatchesForTheDeviceItFailedOn() {
var gate = CombinedTopologyGate()
XCTAssertTrue(gate.shouldTry(input: 42), "an unfailed gate must allow the attempt")
gate.noteFailure(input: 42)
XCTAssertFalse(gate.shouldTry(input: 42))
XCTAssertFalse(gate.shouldTry(input: 42), "the latch must hold across rebuilds")
}
/// The failure is a property of the DEVICE: a different default input earns a fresh attempt,
/// and its own failure latches again one attempt per device change can never loop.
func testADifferentInputDeviceEarnsOneFreshAttempt() {
var gate = CombinedTopologyGate()
gate.noteFailure(input: 42)
XCTAssertTrue(gate.shouldTry(input: 7))
gate.noteFailure(input: 7)
XCTAssertFalse(gate.shouldTry(input: 7))
// Back to the first device: the earlier failure may have been mid-transition one fresh
// attempt again, not a permanent ban.
XCTAssertTrue(gate.shouldTry(input: 42))
}
/// "No resolvable input device" is a real failure key too, distinct from "never failed".
func testFailingWithNoInputDeviceLatchesForNoInputDevice() {
var gate = CombinedTopologyGate()
gate.noteFailure(input: nil)
XCTAssertFalse(gate.shouldTry(input: nil))
XCTAssertTrue(gate.shouldTry(input: 42), "a device appearing is a device change")
}
#endif
}