Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -111,7 +111,8 @@ jobs:
exit 1
fi
done
wait "$suite"; rc=$?
# `|| rc=$?` keeps errexit from skipping the log print when the suite fails.
rc=0; wait "$suite" || rc=$?
cat test-output.log
exit $rc

Expand Down
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Controller command errors distinguish `controllerStateUnavailable` from `controllerCommandUnsupported`.

### Changed
- Queue buffers are sized by duration so pipeline depth no longer depends on bit depth or sample rate.
- `PairingCodeEmission.languages` supplies the server's language priority list for host-provided speech.
- `openPairingWindow(for:)` resets the emitted-round budget with one operator gesture, including an already-open window.

Expand Down
35 changes: 20 additions & 15 deletions Sources/SendspinKit/Audio/AudioPlayer.swift
Original file line number Diff line number Diff line change
Expand Up @@ -47,15 +47,20 @@ let audioQueueTeardownSlowThresholdUs: Int64 = 500_000
/// investigating CoreAudio start stalls. Set `SENDSPIN_NO_PREWARM=1` to enable it.
let audioQueuePrewarmDisabled = ProcessInfo.processInfo.environment["SENDSPIN_NO_PREWARM"] == "1"

/// Byte size of each prepared AudioQueue buffer.
/// Shared with startup latency estimation so priming and correction use the same model.
let audioQueueBufferByteSize: UInt32 = 16_384
/// Startup lead, pad ceiling and modelled fallback describe depth in time,
/// so buffer duration stays independent of bit depth and sample rate.
let audioQueueBufferDuration: Duration = .milliseconds(50)

func audioQueueBufferSize(for format: AudioFormatSpec) -> (frames: Int, bytes: UInt32) {
let components = audioQueueBufferDuration.components
let seconds = Double(components.seconds) + Double(components.attoseconds) / 1e18
let frames = max(1, Int(seconds * Double(format.sampleRate)))
let bytesPerFrame = format.channels * (format.effectiveOutputBitDepth / 8)
return (frames, UInt32(frames * bytesPerFrame))
}

/// Buffers allocated and primed by `prepare()`, and so the pipeline depth once running.
///
/// Must stay the single source of truth for both the allocation loop and the latency
/// model: a mismatch is a constant offset that `graceExpiryRebaselineCursor` bakes in
/// permanently at grace expiry (~85ms per buffer at 48kHz/stereo/16-bit).
/// Allocated and primed buffer count matches the latency model, because
/// `graceExpiryRebaselineCursor` permanently absorbs any mismatch at grace expiry.
let audioQueueBufferCount = 3

private let volumeRampStepCount = 5
Expand Down Expand Up @@ -445,7 +450,7 @@ actor AudioPlayer {
state.correctionGraceFrames = Int64(format.sampleRate)
}

try allocateAndPrewarm(queue: queue)
try allocateAndPrewarm(queue: queue, format: format)
}

private func prepareHardwareQueue(format: AudioFormatSpec) throws {
Expand All @@ -458,11 +463,12 @@ actor AudioPlayer {
/// begin producing and the figure varies by ~100ms between starts, so it cannot be led by an
/// estimate — but paid during the window already spent buffering, it is spent before any
/// audio depends on it. The ring is empty, so every buffer enqueued below is silence.
private func allocateAndPrewarm(queue: AudioQueueRef) throws {
private func allocateAndPrewarm(queue: AudioQueueRef, format: AudioFormatSpec) throws {
let bufferSize = audioQueueBufferSize(for: format)
var allocated: [AudioQueueBufferRef] = []
for _ in 0 ..< audioQueueBufferCount {
var buffer: AudioQueueBufferRef?
let allocStatus = AudioQueueAllocateBuffer(queue, audioQueueBufferByteSize, &buffer)
let allocStatus = AudioQueueAllocateBuffer(queue, bufferSize.bytes, &buffer)
guard allocStatus == noErr, let buffer else {
AudioQueueDispose(queue, true)
audioQueue = nil
Expand Down Expand Up @@ -517,10 +523,9 @@ actor AudioPlayer {
/// Correction reads the device; this conservative estimate is zero before `prepare()`.
func pipelineLatencyMicroseconds() -> Int64 {
guard let format = currentFormat else { return 0 }
let bytesPerFrame = format.channels * (format.effectiveOutputBitDepth / 8)
guard bytesPerFrame > 0, format.sampleRate > 0 else { return 0 }
let queueDepthUs = Int64(audioQueueBufferCount) * Int64(audioQueueBufferByteSize) * 1_000_000
/ Int64(format.sampleRate * bytesPerFrame)
let bufferSize = audioQueueBufferSize(for: format)
let queueDepthUs = Int64(audioQueueBufferCount) * Int64(bufferSize.frames) * 1_000_000
/ Int64(format.sampleRate)
return queueDepthUs + lockedState.withLock { $0.deviceLatencyUs }
}

Expand Down
6 changes: 2 additions & 4 deletions Tests/SendspinKitTests/Audio/AudioEngineTests.swift
Original file line number Diff line number Diff line change
Expand Up @@ -191,10 +191,8 @@ actor SpyAudioOutput: AudioOutput {

func pipelineLatencyMicroseconds() -> Int64 {
guard let format = preparedFormat else { return 0 }
let bytesPerFrame = format.channels * (format.effectiveOutputBitDepth / 8)
guard bytesPerFrame > 0, format.sampleRate > 0 else { return 0 }
let depth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferByteSize) * 1_000_000
/ Int64(format.sampleRate * bytesPerFrame)
let depth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferSize(for: format).frames) * 1_000_000
/ Int64(format.sampleRate)
return depth + stubDeviceLatencyUs
}

Expand Down
35 changes: 30 additions & 5 deletions Tests/SendspinKitTests/Audio/AudioPlayerTests.swift
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,33 @@ extension Tag {

@Suite(.tags(.hardware))
struct AudioPlayerTests {
@Test
func bufferSizingUsesDurationRateAndOutputWidth() throws {
let components = audioQueueBufferDuration.components
let milliseconds = components.seconds * 1_000 + components.attoseconds / 1_000_000_000_000_000
for rate in [44_100, 48_000] {
let expectedFrames = Int(milliseconds) * rate / 1_000
for codec in [AudioCodec.pcm, .flac, .opus] {
for bitDepth in [16, 24, 32] {
let format = try AudioFormatSpec(codec: codec, channels: 2, sampleRate: rate, bitDepth: bitDepth)
let size = audioQueueBufferSize(for: format)
// Only 16-bit PCM passes through; 24-bit PCM is unpacked and compressed codecs decode to Int32.
let sampleBytes = codec == .pcm && bitDepth == 16
? MemoryLayout<Int16>.size : MemoryLayout<Int32>.size
#expect(size.frames == expectedFrames)
#expect(Int(size.bytes) == expectedFrames * format.channels * sampleBytes)
}
}
}

// A rate too low for one frame per buffer duration still allocates a single frame.
let subFrameRate = Int(1_000 / milliseconds) / 2
let tiny = try AudioFormatSpec(codec: .pcm, channels: 1, sampleRate: subFrameRate, bitDepth: 16)
let tinySize = audioQueueBufferSize(for: tiny)
#expect(tinySize.frames == 1)
#expect(Int(tinySize.bytes) == MemoryLayout<Int16>.size)
}

@Test
func initializeAudioPlayerWithDependencies() async {
let player = AudioPlayer()
Expand All @@ -29,8 +56,7 @@ struct AudioPlayerTests {
let player = AudioPlayer()
let format = try AudioFormatSpec(codec: .pcm, channels: 2, sampleRate: 48_000, bitDepth: 16)
try await player.prepare(format: format, codecHeader: nil)
let bytesPerFrame = format.channels * (format.effectiveOutputBitDepth / 8)
let expected = Int64(audioQueueBufferCount) * (Int64(audioQueueBufferByteSize) / Int64(bytesPerFrame))
let expected = Int64(audioQueueBufferCount) * Int64(audioQueueBufferSize(for: format).frames)
let primedFrames = await player.totalFramesEnqueued
await player.stop()
#expect(primedFrames == expected)
Expand Down Expand Up @@ -177,9 +203,8 @@ struct AudioPlayerTests {
try await player.prepare(format: format, codecHeader: nil)
defer { Task { await player.stop() } }

let bytesPerFrame = format.channels * (format.effectiveOutputBitDepth / 8)
let depth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferByteSize) * 1_000_000
/ Int64(format.sampleRate * bytesPerFrame)
let depth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferSize(for: format).frames) * 1_000_000
/ Int64(format.sampleRate)
let reported = await player.pipelineLatencyMicroseconds()

#expect(reported >= depth, "latency must at least cover the buffers we prime")
Expand Down
9 changes: 3 additions & 6 deletions Tests/SendspinKitTests/Audio/AudioProcessCallbackTests.swift
Original file line number Diff line number Diff line change
Expand Up @@ -40,9 +40,8 @@ struct AudioProcessCallbackTests {

try await player.start(format: Self.stereo16, codecHeader: nil)

// Feed enough PCM data that the AudioQueue callback must fire.
// At 48kHz stereo 16-bit, each frame is 4 bytes. 16384-byte buffers
// need ~4096 frames. Feed 2 seconds to ensure at least one callback.
// Two seconds of PCM comfortably exceeds the duration-sized queue depth,
// so the callback receives audio rather than only primed silence.
let bytesPerFrame = Self.stereo16.channels * (Self.stereo16.bitDepth / 8)
let twoSeconds = Self.stereo16.sampleRate * bytesPerFrame * 2
let pcmData = Data(repeating: 0, count: twoSeconds)
Expand Down Expand Up @@ -177,9 +176,7 @@ struct AudioProcessCallbackTests {
#expect(fired)

let byteCounts = invoked.byteCounts
// Import the source constant rather than duplicating it, so retuning the
// buffer size cannot silently decouple this assertion from production.
let expectedBufferSize = Int(audioQueueBufferByteSize)
let expectedBufferSize = Int(audioQueueBufferSize(for: Self.stereo16).bytes)
for byteCount in byteCounts {
#expect(
byteCount == expectedBufferSize,
Expand Down
12 changes: 7 additions & 5 deletions Tests/SendspinKitTests/Audio/CallbackDepthTelemetryTests.swift
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ import Testing
struct CallbackDepthTelemetryTests {
@Test func skipsDoNotChangeDepthAndMissedPositionKeepsLatch() throws {
let format = try AudioFormatSpec(codec: .pcm, channels: 2, sampleRate: 48_000, bitDepth: 16)
let buffer = Int64(audioQueueBufferByteSize) / Int64(format.channels * (format.bitDepth / 8))
let buffer = Int64(audioQueueBufferSize(for: format).frames)
let total = buffer * Int64(audioQueueBufferCount)
var telemetry = CallbackDepthTelemetry()
telemetry.record(total: total, played: buffer, bufferFrames: buffer, costUs: buffer, prewarming: true)
Expand All @@ -24,8 +24,9 @@ struct CallbackDepthTelemetryTests {
#expect(telemetry.timeCostUs == buffer)
}

@Test func aggregatesMinimumMaximumLastDelayAndCost() {
let buffer = Int64(audioQueueBufferByteSize)
@Test func aggregatesMinimumMaximumLastDelayAndCost() throws {
let format = try AudioFormatSpec(codec: .flac, channels: 2, sampleRate: 44_100, bitDepth: 16)
let buffer = Int64(audioQueueBufferSize(for: format).frames)
let total = buffer * Int64(audioQueueBufferCount)
var telemetry = CallbackDepthTelemetry()
telemetry.record(total: total, played: buffer, bufferFrames: buffer, costUs: buffer, prewarming: false)
Expand All @@ -40,8 +41,9 @@ struct CallbackDepthTelemetryTests {
#expect(telemetry.prewarmSkipped == 0 && telemetry.zeroPlayedSkipped == 0)
}

@Test func depthClampsAndMeasuresDeliveryDelay() {
let bufferFrames = Int64(audioQueueBufferByteSize)
@Test func depthClampsAndMeasuresDeliveryDelay() throws {
let format = try AudioFormatSpec(codec: .pcm, channels: 2, sampleRate: 48_000, bitDepth: 16)
let bufferFrames = Int64(audioQueueBufferSize(for: format).frames)
let modelled = Int64(audioQueueBufferCount) * bufferFrames
let afterCompletion = modelled - bufferFrames
let delivered = CallbackDepthTelemetry.depth(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,8 @@ struct CorrectionPipelineLatencyTests {
let format = try AudioFormatSpec(codec: .pcm, channels: 2, sampleRate: 48_000, bitDepth: 16)
let frames = Int64(format.sampleRate / format.channels)
let deviceLatency = Int64(1_000_000 / format.sampleRate)
let modelledDepth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferByteSize) * 1_000_000
/ Int64(format.sampleRate * format.channels * (format.bitDepth / 8))
let modelledDepth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferSize(for: format).frames) * 1_000_000
/ Int64(format.sampleRate)
let pipeline = AudioPlayer.correctionPipelineLatencyUs(
inFlightFrames: frames, sampleRate: format.sampleRate,
modelledQueueDepthUs: modelledDepth, deviceLatencyUs: deviceLatency
Expand All @@ -20,8 +20,8 @@ struct CorrectionPipelineLatencyTests {
let format = try AudioFormatSpec(codec: .pcm, channels: 2, sampleRate: 48_000, bitDepth: 16)
let frames = Int64(format.sampleRate / format.channels)
let deviceLatency = Int64(1_000_000 / format.sampleRate)
let modelledDepth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferByteSize) * 1_000_000
/ Int64(format.sampleRate * format.channels * (format.bitDepth / 8))
let modelledDepth = Int64(audioQueueBufferCount) * Int64(audioQueueBufferSize(for: format).frames) * 1_000_000
/ Int64(format.sampleRate)
var previous: Int64?
let absent = CallbackDepthTelemetry.measuredDepth(total: frames, played: 0, previous: &previous)
#expect(absent == nil)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -1770,9 +1770,16 @@ struct ClientIntegrationTests {
func applyUnderrunTransition_toErrorIsIgnoredDuringExternalSource() async throws {
let client = try makeTestClient()
let mock = try await connectClient(client)
// Sync first so the external-source flip is wire-visible as available
// true -> false; waiting for that flip to read back pins the baseline
// after every client/state the flip produces.
try await establishClockSync(client, via: mock)
#expect(await waitUntil { await lastClientState(from: mock, after: 0)?.available == true })
let countSynced = await mock.sentTextMessages.count

try await client.enterExternalSource()
#expect(client.clientOperationalState == .externalSource)
#expect(await waitUntil { await lastClientState(from: mock, after: countSynced)?.available == false })

let countBefore = await mock.sentTextMessages.count
await client.applyUnderrunTransition(.toError)
Expand Down
9 changes: 7 additions & 2 deletions docs/audio-timing-model.md
Original file line number Diff line number Diff line change
Expand Up @@ -161,8 +161,13 @@ once a position is observed for the queue, a missed or zero read holds its previ
Only successful enqueues enter the cumulative frame count; refused and withheld buffers do not.
Startup leads and silence-pad ceilings remain conservative allocated-depth estimates.

Three 30-second tone runs on a MacBook Air's built-in speakers use 44,100Hz/stereo/32-bit output:
16,384 bytes per buffer / 8 bytes per frame = 2,048 frames; three buffers model 6,144 frames.
Each queue buffer contains whole frames from `audioQueueBufferDuration × sampleRate`, with at
least one frame; three buffers model `audioQueueBufferCount × frames / sampleRate` seconds.
The byte allocation uses the effective output width, so depth in time does not depend on bit depth
or sample rate. At 50 ms, 44,100Hz output contains 2,205 frames per buffer and models 6,615 frames.

Three 30-second tone runs on a MacBook Air's built-in speakers use 44,100Hz/stereo/32-bit output
(measured with 2,048-frame buffers, modelling 6,144 frames).
Callback depth spans 4,622–5,090 frames (2.26–2.49 buffers), leaving a 1,054–1,522-frame model gap.
The interval-last medians are 4,801.5 / 4,932.5 / 4,926 frames. The delivery-delay formula clamps to
zero: measured depth exceeds two buffers rather than simply being two minus callback delay.
Expand Down
Loading