Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Sources/App/VoicePipeline+Processing.swift
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ extension VoicePipeline {
targetApp: NSRunningApplication?
) async {
defer { audioCapture.cleanupLastRecording() }
AudioCaptureDiagnostics.log(audioActivity)

do {
guard audioActivity.hasMeaningfulAudio else {
Expand Down
30 changes: 30 additions & 0 deletions Sources/Audio/AudioActivityThresholds.swift
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
import Foundation

/// Tunable audio-activity thresholds, split into two independent groups so the
/// recording gate and the weak-speech heuristic can be calibrated separately.
///
/// `gate` drives `AudioCaptureActivity.hasMeaningfulAudio` (whether a recording
/// is handed to ASR at all); `weakSpeechEvidence` drives
/// `AudioCaptureActivity.hasWeakSpeechEvidence` (whether `TranscriptionSanitizer`
/// may collapse repetition or strip hallucinated tails). Loosening one group
/// must never move the other's decision.
struct AudioActivityThresholds: Equatable, Sendable {
struct Gate: Equatable, Sendable {
var minimumAverageRMS: Float
var minimumPeakRMS: Float
}

struct WeakSpeechEvidence: Equatable, Sendable {
var averageRMS: Float
var peakRMS: Float
}

var gate: Gate
var weakSpeechEvidence: WeakSpeechEvidence

/// Production values, identical to the constants they replaced.
static let `default` = AudioActivityThresholds(
gate: Gate(minimumAverageRMS: 0.0015, minimumPeakRMS: 0.005),
weakSpeechEvidence: WeakSpeechEvidence(averageRMS: 0.004, peakRMS: 0.012)
)
}
52 changes: 52 additions & 0 deletions Sources/Audio/AudioCaptureDiagnostics.swift
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
import Foundation

/// Numeric-only audio activity record used to calibrate the recording gate and
/// the weak-speech heuristic. It never carries audio samples or transcript
/// content.
struct AudioCaptureDiagnostic: Equatable, Sendable {
let averageRMS: Float
let maxRMS: Float
let frameCount: Int
let gateRejected: Bool

init(activity: AudioCaptureActivity) {
self.averageRMS = activity.averageRMS
self.maxRMS = activity.maxRMS
self.frameCount = activity.frameCount
self.gateRejected = !activity.hasMeaningfulAudio
}

/// Stable, greppable single-line format for later corpus correlation.
var logLine: String {
String(
format: "audio-activity averageRMS=%.6f maxRMS=%.6f frames=%d gateRejected=",
Double(averageRMS),
Double(maxRMS),
frameCount
) + (gateRejected ? "true" : "false")
}
}

enum AudioCaptureDiagnostics {
static let environmentKey = "UTTER_AUDIO_DIAGNOSTICS"

/// Disabled by default. Only debug builds opt in, via the environment
/// variable, so release builds never log recording activity.
static var isEnabled: Bool {
#if DEBUG
resolveEnabled(environment: ProcessInfo.processInfo.environment, isDebugBuild: true)
#else
false
#endif
}

static func resolveEnabled(environment: [String: String], isDebugBuild: Bool) -> Bool {
guard isDebugBuild else { return false }
return environment[environmentKey] == "1"
}

static func log(_ activity: AudioCaptureActivity) {
guard isEnabled else { return }
Log.info("[AudioCapture] \(AudioCaptureDiagnostic(activity: activity).logLine)")
}
}
15 changes: 9 additions & 6 deletions Sources/Audio/AudioCaptureManager.swift
Original file line number Diff line number Diff line change
Expand Up @@ -3,29 +3,32 @@ import CoreAudio
import AudioToolbox

struct AudioCaptureActivity: Equatable {
private static let minimumAverageRMS: Float = 0.0015
private static let minimumPeakRMS: Float = 0.005
private static let lowConfidenceAverageRMS: Float = 0.004
private static let lowConfidencePeakRMS: Float = 0.012
let thresholds: AudioActivityThresholds

private(set) var bufferCount = 0
private(set) var frameCount = 0
private(set) var maxRMS: Float = 0
private var weightedRMSSum: Double = 0

init(thresholds: AudioActivityThresholds = .default) {
self.thresholds = thresholds
}

var averageRMS: Float {
guard frameCount > 0 else { return 0 }
return Float(weightedRMSSum / Double(frameCount))
}

var hasMeaningfulAudio: Bool {
guard frameCount > 0 else { return false }
return averageRMS >= Self.minimumAverageRMS || maxRMS >= Self.minimumPeakRMS
let gate = thresholds.gate
return averageRMS >= gate.minimumAverageRMS || maxRMS >= gate.minimumPeakRMS
}

var hasWeakSpeechEvidence: Bool {
guard frameCount > 0 else { return true }
return averageRMS < Self.lowConfidenceAverageRMS && maxRMS < Self.lowConfidencePeakRMS
let weak = thresholds.weakSpeechEvidence
return averageRMS < weak.averageRMS && maxRMS < weak.peakRMS
}

mutating func record(rms: Float, frameCount: Int) {
Expand Down
1 change: 1 addition & 0 deletions Sources/Integration/InputSessionCoordinator.swift
Original file line number Diff line number Diff line change
Expand Up @@ -155,6 +155,7 @@ final class InputSessionCoordinator {
}

private func processRecording(_ active: ActiveSession) async throws -> (transcript: String, text: String) {
AudioCaptureDiagnostics.log(audioCapture.lastActivity)
guard audioCapture.lastActivity.hasMeaningfulAudio else {
throw IntegrationError.noSpeechDetected
}
Expand Down
105 changes: 105 additions & 0 deletions Tests/OpenTypeTests/AudioCaptureActivityTests.swift
Original file line number Diff line number Diff line change
@@ -0,0 +1,105 @@
import XCTest
@testable import OpenType

final class AudioCaptureActivityTests: XCTestCase {
private let sampleRMS: Float = 0.002
private let sampleFrames = 16_000

private func activity(
gate: AudioActivityThresholds.Gate,
weak: AudioActivityThresholds.WeakSpeechEvidence,
rms: Float
) -> AudioCaptureActivity {
var activity = AudioCaptureActivity(
thresholds: AudioActivityThresholds(gate: gate, weakSpeechEvidence: weak)
)
activity.record(rms: rms, frameCount: sampleFrames)
return activity
}

func testDefaultThresholdsMatchPreviousConstants() {
XCTAssertEqual(AudioActivityThresholds.default.gate.minimumAverageRMS, 0.0015)
XCTAssertEqual(AudioActivityThresholds.default.gate.minimumPeakRMS, 0.005)
XCTAssertEqual(AudioActivityThresholds.default.weakSpeechEvidence.averageRMS, 0.004)
XCTAssertEqual(AudioActivityThresholds.default.weakSpeechEvidence.peakRMS, 0.012)
}

func testGateAndWeakSpeechEvidenceUseIndependentThresholdGroups() {
let defaultWeak = AudioActivityThresholds.default.weakSpeechEvidence

let strictGate = activity(
gate: .init(minimumAverageRMS: 0.5, minimumPeakRMS: 0.5),
weak: defaultWeak,
rms: sampleRMS
)
XCTAssertFalse(strictGate.hasMeaningfulAudio)
XCTAssertTrue(strictGate.hasWeakSpeechEvidence)

let looseGate = activity(
gate: .init(minimumAverageRMS: 0.001, minimumPeakRMS: 0.001),
weak: defaultWeak,
rms: sampleRMS
)
XCTAssertTrue(looseGate.hasMeaningfulAudio)
XCTAssertTrue(looseGate.hasWeakSpeechEvidence)

let strictWeak = activity(
gate: .init(minimumAverageRMS: 0.001, minimumPeakRMS: 0.001),
weak: .init(averageRMS: 0.0001, peakRMS: 0.0001),
rms: sampleRMS
)
XCTAssertTrue(strictWeak.hasMeaningfulAudio)
XCTAssertFalse(strictWeak.hasWeakSpeechEvidence)
}

func testDiagnosticRecordFormatIsStable() {
var activity = AudioCaptureActivity()
activity.record(rms: 0.002, frameCount: 16_000)

let diagnostic = AudioCaptureDiagnostic(activity: activity)
XCTAssertEqual(diagnostic.averageRMS, 0.002, accuracy: 1e-6)
XCTAssertEqual(diagnostic.maxRMS, 0.002, accuracy: 1e-6)
XCTAssertEqual(diagnostic.frameCount, 16_000)
XCTAssertFalse(diagnostic.gateRejected)
XCTAssertEqual(
diagnostic.logLine,
"audio-activity averageRMS=0.002000 maxRMS=0.002000 frames=16000 gateRejected=false"
)
}

func testDiagnosticFlagsGateRejection() {
var activity = AudioCaptureActivity()
activity.record(rms: 0.0001, frameCount: 16_000)

let diagnostic = AudioCaptureDiagnostic(activity: activity)
XCTAssertTrue(diagnostic.gateRejected)
XCTAssertEqual(
diagnostic.logLine,
"audio-activity averageRMS=0.000100 maxRMS=0.000100 frames=16000 gateRejected=true"
)
}

func testDiagnosticResolutionRequiresDebugBuildAndOptIn() {
XCTAssertFalse(
AudioCaptureDiagnostics.resolveEnabled(environment: [:], isDebugBuild: true)
)
XCTAssertFalse(
AudioCaptureDiagnostics.resolveEnabled(
environment: ["UTTER_AUDIO_DIAGNOSTICS": "0"],
isDebugBuild: true
)
)
XCTAssertTrue(
AudioCaptureDiagnostics.resolveEnabled(
environment: ["UTTER_AUDIO_DIAGNOSTICS": "1"],
isDebugBuild: true
)
)
XCTAssertFalse(
AudioCaptureDiagnostics.resolveEnabled(
environment: ["UTTER_AUDIO_DIAGNOSTICS": "1"],
isDebugBuild: false
)
)
}
}
Loading