diff --git a/Sources/App/VoiceInputSettings.swift b/Sources/App/VoiceInputSettings.swift index ccba5bc..6f49b82 100644 --- a/Sources/App/VoiceInputSettings.swift +++ b/Sources/App/VoiceInputSettings.swift @@ -19,10 +19,18 @@ struct VoiceInputSettings { var espressoModelPath: String { processing.espressoModelPath } @MainActor - init(settings: AppSettings, inputLanguage: InputLanguage? = nil) { + init( + settings: AppSettings, + inputLanguage: InputLanguage? = nil, + bundleIdentifier: String? = nil + ) { processing = TextProcessingOptions(settings: settings, inputLanguage: inputLanguage) speech = SpeechEngineProvider.Selection(settings: settings, inputLanguage: inputLanguage) - dictionary = PersonalDictionary.shared.snapshot(settings: settings) + dictionary = PersonalDictionary.shared.snapshot( + settings: settings, + bundleIdentifier: bundleIdentifier, + languageCode: (inputLanguage ?? settings.inputLanguage).whisperCode + ) outputMode = settings.outputMode enableInstantInsert = settings.enableInstantInsert enableMemory = settings.enableMemory diff --git a/Sources/App/VoicePipeline+Processing.swift b/Sources/App/VoicePipeline+Processing.swift index 6bc98c4..ba3d14c 100644 --- a/Sources/App/VoicePipeline+Processing.swift +++ b/Sources/App/VoicePipeline+Processing.swift @@ -19,6 +19,10 @@ extension VoicePipeline { showNoSpeechDetected(reason: "recorded audio energy below threshold") return } + guard await recordingContainsSpeech(audioURL) else { + showNoSpeechDetected(reason: "recorded audio has no speech evidence") + return + } let preparedRaw = try await transcribePreparedText( audioURL: audioURL, @@ -86,6 +90,15 @@ extension VoicePipeline { } } + private func recordingContainsSpeech(_ audioURL: URL?) async -> Bool { + #if DEBUG + if let speechActivityOverrideForTesting { + return await speechActivityOverrideForTesting(audioURL) + } + #endif + return await SpeechActivityClassifier.containsSpeech(at: audioURL) + } + private func transcribePreparedText( audioURL: URL?, audioActivity: AudioCaptureActivity, @@ -103,8 +116,11 @@ extension VoicePipeline { Log.info("[VoicePipeline] ASR stage finished in \(String(format: "%.2f", elapsed))s") try Task.checkCancellation() - guard let prepared = TranscriptionSanitizer.prepare(raw, audioActivity: audioActivity) else { - showNoSpeechDetected(reason: "transcription has no meaningful content: \(raw)") + guard let prepared = TranscriptionSanitizer.prepare( + raw, audioActivity: audioActivity, + recognitionPhrases: settings.dictionary.recognitionPhrases + ) else { + showNoSpeechDetected(reason: "transcription has no meaningful content") throw VoicePipelineStop.noSpeech } diff --git a/Sources/App/VoicePipeline+Recording.swift b/Sources/App/VoicePipeline+Recording.swift index be3aa4c..a3637c3 100644 --- a/Sources/App/VoicePipeline+Recording.swift +++ b/Sources/App/VoicePipeline+Recording.swift @@ -24,7 +24,9 @@ extension VoicePipeline { sessionLease = lease var started = false defer { if !started { releaseSession(lease) } } - let snapshot = VoiceInputSettings(settings: appState.settings) + let snapshot = VoiceInputSettings( + settings: appState.settings, bundleIdentifier: targetApp?.bundleIdentifier + ) sessionSettings = snapshot recordingLanguage = snapshot.inputLanguage.whisperCode recordingStreaming = snapshot.streamingEnabled @@ -89,7 +91,9 @@ extension VoicePipeline { let streamingEnabled = recordingStreaming && (currentEngine?.supportsStreaming ?? false) recordingStreaming = streamingEnabled currentEngine?.configureRecognition( - context: SpeechRecognitionContext(phrases: vocabularySnapshot.recognitionPhrases) + context: SpeechRecognitionContext(phrases: currentEngine is QwenNativeASREngine + ? vocabularySnapshot.personalRecognitionPhrases + : vocabularySnapshot.recognitionPhrases) ) if streamingEnabled { currentEngine?.startListening(language: language) { [weak self] partialText in diff --git a/Sources/App/VoicePipeline.swift b/Sources/App/VoicePipeline.swift index 78547a6..e530ac1 100644 --- a/Sources/App/VoicePipeline.swift +++ b/Sources/App/VoicePipeline.swift @@ -43,6 +43,9 @@ final class VoicePipeline { var engineLoadBarrier: (() async -> Void)? /// Test-only observation point for whether the remote capture path is used. var remoteCaptureSpy: RemoteMicCaptureSpy? + #if DEBUG + var speechActivityOverrideForTesting: ((URL?) async -> Bool)? + #endif var currentEngine: (any SpeechEngine)? { engineOverride ?? sessionEngine } diff --git a/Sources/Audio/SpeechActivityClassifier.swift b/Sources/Audio/SpeechActivityClassifier.swift new file mode 100644 index 0000000..633f167 --- /dev/null +++ b/Sources/Audio/SpeechActivityClassifier.swift @@ -0,0 +1,115 @@ +import AVFoundation +import Foundation +import SoundAnalysis + +enum SpeechActivityClassifier { + static let minimumSpeechConfidence = 0.6 + private static let windowSeconds = 0.5 + private static let minimumFileSeconds = 0.75 + + static func containsSpeech(at audioURL: URL?) async -> Bool { + guard let audioURL, !Task.isCancelled else { return false } + var paddedURL: URL? + defer { + if let paddedURL { try? FileManager.default.removeItem(at: paddedURL) } + } + + do { + let analysisURL = try preparedURL(audioURL, paddedURL: &paddedURL) + let request = try SNClassifySoundRequest(classifierIdentifier: .version1) + request.windowDuration = CMTime(seconds: windowSeconds, preferredTimescale: 16_000) + let observer = SpeechClassificationObserver() + let analyzer = try SNAudioFileAnalyzer(url: analysisURL) + try analyzer.add(request, withObserver: observer) + let completed = await withTaskCancellationHandler { + await analyzer.analyze() + } onCancel: { + analyzer.cancelAnalysis() + } + let result = observer.result + return completed && !Task.isCancelled && !result.failed + && result.windows > 0 && result.maxSpeech >= minimumSpeechConfidence + } catch { + Log.error("[SpeechActivity] classification failed") + return false + } + } + + private static func preparedURL(_ url: URL, paddedURL: inout URL?) throws -> URL { + let source = try AVAudioFile(forReading: url) + let format = source.processingFormat + guard format.sampleRate > 0, source.length > 0 else { throw SpeechActivityError.invalidAudio } + let requiredFrames = Int(ceil(minimumFileSeconds * format.sampleRate)) + guard source.length < requiredFrames else { return url } + guard requiredFrames <= Int(UInt32.max), + let buffer = AVAudioPCMBuffer(pcmFormat: format, frameCapacity: AVAudioFrameCount(requiredFrames)) else { + throw SpeechActivityError.invalidAudio + } + try source.read(into: buffer) + let readFrames = Int(buffer.frameLength) + guard readFrames > 0, readFrames < requiredFrames else { throw SpeechActivityError.invalidAudio } + + let channels = Int(format.channelCount) + if let data = buffer.floatChannelData { + for channel in 0.. SpeechRecognitionContext { + return SpeechRecognitionContext(phrases: engine is QwenNativeASREngine + ? snapshot.personalRecognitionPhrases + : snapshot.recognitionPhrases) + } + + func recordingContainsSpeech(_ audioURL: URL?) async -> Bool { + #if DEBUG + if let speechActivityOverrideForTesting { + return await speechActivityOverrideForTesting(audioURL) + } + #endif + return await SpeechActivityClassifier.containsSpeech(at: audioURL) + } +} diff --git a/Sources/Integration/InputSessionCoordinator+Lifecycle.swift b/Sources/Integration/InputSessionCoordinator+Lifecycle.swift index 9ad72a1..2f9fd2a 100644 --- a/Sources/Integration/InputSessionCoordinator+Lifecycle.swift +++ b/Sources/Integration/InputSessionCoordinator+Lifecycle.swift @@ -10,7 +10,11 @@ extension InputSessionCoordinator { let reservation = try ownership.acquire() lease = reservation owner = (sessionID, clientID) - requestSettings = VoiceInputSettings(settings: settings, inputLanguage: session.request.language) + requestSettings = VoiceInputSettings( + settings: settings, + inputLanguage: session.request.language, + bundleIdentifier: service.integrationClient(id: clientID)?.bundleIdentifier + ) return session } diff --git a/Sources/Integration/InputSessionCoordinator.swift b/Sources/Integration/InputSessionCoordinator.swift index d1af5e4..e6bd71b 100644 --- a/Sources/Integration/InputSessionCoordinator.swift +++ b/Sources/Integration/InputSessionCoordinator.swift @@ -30,6 +30,9 @@ final class InputSessionCoordinator { var activeSession: ActiveSession? var requestSettings: VoiceInputSettings? var pendingHistory: (() -> Void)? + #if DEBUG + var speechActivityOverrideForTesting: ((URL?) async -> Bool)? + #endif var isBusy: Bool { lease != nil } @@ -66,7 +69,7 @@ final class InputSessionCoordinator { let engine = try await loadSpeechEngine() let vocabularySnapshot = snapshot.dictionary engine.configureRecognition( - context: SpeechRecognitionContext(phrases: vocabularySnapshot.recognitionPhrases) + context: recognitionContext(engine: engine, snapshot: vocabularySnapshot) ) if effective.streamingEnabled, engine.supportsStreaming { @@ -170,6 +173,10 @@ final class InputSessionCoordinator { guard audioCapture.lastActivity.hasMeaningfulAudio else { throw IntegrationError.noSpeechDetected } + guard await recordingContainsSpeech(audioCapture.lastRecordingURL) else { + throw IntegrationError.noSpeechDetected + } + try checkCurrent() let raw: String if active.streamingEnabled { @@ -185,7 +192,10 @@ final class InputSessionCoordinator { } try checkCurrent() - let transcript = try prepareTranscript(raw, audioActivity: audioCapture.lastActivity) + let transcript = try prepareTranscript( + raw, audioActivity: audioCapture.lastActivity, + dictionarySnapshot: active.snapshot?.dictionary + ) try service.emitTranscriptFinal( sessionID: active.sessionID, @@ -202,8 +212,16 @@ final class InputSessionCoordinator { audioCapture.cleanupLastRecording() } - func prepareTranscript(_ raw: String, audioActivity: AudioCaptureActivity?) throws -> String { - guard let transcript = TranscriptionSanitizer.prepare(raw, audioActivity: audioActivity) else { + func prepareTranscript( + _ raw: String, + audioActivity: AudioCaptureActivity?, + dictionarySnapshot: PersonalDictionarySnapshot? = nil + ) throws -> String { + let vocabulary = dictionarySnapshot ?? requestSettings?.dictionary + guard let transcript = TranscriptionSanitizer.prepare( + raw, audioActivity: audioActivity, + recognitionPhrases: vocabulary?.recognitionPhrases ?? [] + ) else { throw IntegrationError.noSpeechDetected } return transcript diff --git a/Sources/Output/CorrectionCaptureService.swift b/Sources/Output/CorrectionCaptureService.swift index ac9e6fd..da987e5 100644 --- a/Sources/Output/CorrectionCaptureService.swift +++ b/Sources/Output/CorrectionCaptureService.swift @@ -54,8 +54,7 @@ final class CorrectionCaptureService { inserted: session.seed.insertedText, userFinal: session.latestFinalText, sourceRecordID: session.recordID, - languageCode: session.seed.context.inputLanguage.whisperCode - ?? session.seed.context.inputLanguage.rawValue, + languageCode: session.seed.context.inputLanguage.whisperCode, bundleIdentifier: session.seed.context.bundleIdentifier ) { PersonalDictionary.shared.recordLearnedCandidate(candidate) diff --git a/Sources/Output/TextInserter.swift b/Sources/Output/TextInserter.swift index 80f9ffc..cbddaea 100644 --- a/Sources/Output/TextInserter.swift +++ b/Sources/Output/TextInserter.swift @@ -11,9 +11,15 @@ enum InsertResult { @MainActor final class TextInserter { var recentInsertionAnchor: RecentInsertionAnchor? + #if DEBUG + var insertOverrideForTesting: ((String) -> InsertResult)? + #endif func insert(text: String, targetApp: NSRunningApplication? = nil) async -> InsertResult { guard !Task.isCancelled else { return .probablyFailed(reason: L("error.operation_failed")) } + #if DEBUG + if let insertOverrideForTesting { return insertOverrideForTesting(text) } + #endif guard AXIsProcessTrusted() else { Log.error("[TextInserter] no AX trust") return .probablyFailed(reason: "Accessibility permission not granted") diff --git a/Sources/Processing/CorrectionCandidateClassifier.swift b/Sources/Processing/CorrectionCandidateClassifier.swift index 8e4edbf..b78e6c3 100644 --- a/Sources/Processing/CorrectionCandidateClassifier.swift +++ b/Sources/Processing/CorrectionCandidateClassifier.swift @@ -69,6 +69,7 @@ enum CorrectionCandidateClassifier { languageCode: String?, bundleIdentifier: String? ) -> LearnedCorrectionCandidate? { + guard !LearnedCorrectionPolicy.isUnsafeSource(inserted) else { return nil } guard let diff = CorrectionEditDiff.between(inserted, userFinal) else { return nil } var original = diff.beforeSegment.trimmingCharacters(in: .whitespacesAndNewlines) var replacement = diff.afterSegment.trimmingCharacters(in: .whitespacesAndNewlines) diff --git a/Sources/Processing/LearnedCorrectionPolicy.swift b/Sources/Processing/LearnedCorrectionPolicy.swift new file mode 100644 index 0000000..51e4b90 --- /dev/null +++ b/Sources/Processing/LearnedCorrectionPolicy.swift @@ -0,0 +1,38 @@ +import Foundation + +enum LearnedCorrectionPolicy { + private static let isolatedFillers: Set = [ + "嗯", "呃", "额", "唔", "啊", "哦", "um", "uh", "hmm", "hm" + ] + + static func isUnsafeSource(_ text: String) -> Bool { + let lexical = String(text.lowercased().filter { $0.isLetter || $0.isNumber }) + return isolatedFillers.contains(lexical) + || TranscriptionSanitizer.isNonSpeechArtifact(text) + } +} + +extension DictionaryEntry { + func applies(bundleIdentifier: String?, languageCode: String?) -> Bool { + guard isEffective else { return false } + if origin == .learned { + guard !LearnedCorrectionPolicy.isUnsafeSource(original), + !appScopes.isEmpty || (self.languageCode != nil + && self.languageCode != InputLanguage.auto.rawValue) else { return false } + } + if !appScopes.isEmpty { + guard let bundleIdentifier, + appScopes.contains(where: { $0.caseInsensitiveCompare(bundleIdentifier) == .orderedSame }) else { + return false + } + } + if let expected = self.languageCode, expected != InputLanguage.auto.rawValue { + guard let languageCode, + expected.split(separator: "-").first?.lowercased() + == languageCode.split(separator: "-").first?.lowercased() else { + return false + } + } + return true + } +} diff --git a/Sources/Processing/PersonalDictionary+Learning.swift b/Sources/Processing/PersonalDictionary+Learning.swift index abc7d3f..c04d576 100644 --- a/Sources/Processing/PersonalDictionary+Learning.swift +++ b/Sources/Processing/PersonalDictionary+Learning.swift @@ -1,6 +1,14 @@ import Foundation extension PersonalDictionary { + func suspendLearnedMappings(for original: String, excluding id: UUID) { + for index in entries.indices where entries[index].id != id + && entries[index].origin == .learned + && entries[index].original.caseInsensitiveCompare(original) == .orderedSame { + entries[index].status = .pending + } + } + func clearLearnedEntries() { entries.removeAll { $0.origin == .learned } save() @@ -8,6 +16,8 @@ extension PersonalDictionary { @discardableResult func recordLearnedCandidate(_ candidate: LearnedCorrectionCandidate) -> UUID? { + guard !LearnedCorrectionPolicy.isUnsafeSource(candidate.original), + candidate.languageCode != nil || candidate.bundleIdentifier != nil else { return nil } removePreviousEvidence(for: candidate) if entries.contains(where: { @@ -23,19 +33,23 @@ extension PersonalDictionary { $0.origin == .learned && $0.original.caseInsensitiveCompare(candidate.original) == .orderedSame && $0.replacement.caseInsensitiveCompare(candidate.replacement) != .orderedSame + && $0.languageCode == candidate.languageCode + && $0.appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) } if let index = entries.firstIndex(where: { $0.origin == .learned && $0.original.caseInsensitiveCompare(candidate.original) == .orderedSame && $0.replacement.caseInsensitiveCompare(candidate.replacement) == .orderedSame + && $0.languageCode == candidate.languageCode + && $0.appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) }) { merge(candidate, intoEntryAt: index, now: now) - if hasConflict { markLearnedMappingsPending(for: candidate.original) } + if hasConflict { markLearnedMappingsPending(for: candidate) } save() return entries[index].id } - if hasConflict { markLearnedMappingsPending(for: candidate.original) } + if hasConflict { markLearnedMappingsPending(for: candidate) } let entry = DictionaryEntry( original: candidate.original, @@ -66,11 +80,6 @@ private extension PersonalDictionary { ) entries[index].confidence = max(entries[index].confidence, candidate.confidence) entries[index].lastSeenAt = now - entries[index].languageCode = candidate.languageCode ?? entries[index].languageCode - if let bundleIdentifier = candidate.bundleIdentifier, - !entries[index].appScopes.contains(bundleIdentifier) { - entries[index].appScopes.append(bundleIdentifier) - } if entries[index].confidence >= 0.92 || entries[index].evidenceCount >= 2 { entries[index].status = .active } @@ -92,9 +101,11 @@ private extension PersonalDictionary { } } - func markLearnedMappingsPending(for original: String) { + func markLearnedMappingsPending(for candidate: LearnedCorrectionCandidate) { for index in entries.indices where entries[index].origin == .learned - && entries[index].original.caseInsensitiveCompare(original) == .orderedSame { + && entries[index].original.caseInsensitiveCompare(candidate.original) == .orderedSame + && entries[index].languageCode == candidate.languageCode + && entries[index].appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) { entries[index].status = .pending } } diff --git a/Sources/Processing/PersonalDictionary.swift b/Sources/Processing/PersonalDictionary.swift index 472ba85..0b86d01 100644 --- a/Sources/Processing/PersonalDictionary.swift +++ b/Sources/Processing/PersonalDictionary.swift @@ -14,9 +14,13 @@ struct PersonalDictionarySnapshot: Sendable { init( entries: [DictionaryEntry], editRules: [EditRule], - industryLexicon: IndustryLexiconSnapshot = .empty + industryLexicon: IndustryLexiconSnapshot = .empty, + bundleIdentifier: String? = nil, + languageCode: String? = nil ) { - self.entries = entries + self.entries = entries.filter { + $0.applies(bundleIdentifier: bundleIdentifier, languageCode: languageCode) + } self.editRules = editRules self.industryLexicon = industryLexicon } @@ -69,10 +73,13 @@ struct PersonalDictionarySnapshot: Sendable { .joined(separator: "\n") } + var personalRecognitionPhrases: [String] { + SpeechRecognitionContext(dictionaryEntries: entries).phrases + } + var recognitionPhrases: [String] { - let personal = SpeechRecognitionContext(dictionaryEntries: entries).phrases return SpeechRecognitionContext( - phrases: personal + industryLexicon.recognitionPhrases + phrases: personalRecognitionPhrases + industryLexicon.recognitionPhrases ).phrases } @@ -93,7 +100,6 @@ struct PersonalDictionarySnapshot: Sendable { } return industryTerms + personalTerms } - } final class PersonalDictionary: ObservableObject { @@ -129,19 +135,31 @@ final class PersonalDictionary: ObservableObject { snapshot().activeRulesDescription } - func snapshot(industryLexicon: IndustryLexiconSnapshot = .empty) -> PersonalDictionarySnapshot { + func snapshot( + industryLexicon: IndustryLexiconSnapshot = .empty, + bundleIdentifier: String? = nil, + languageCode: String? = nil + ) -> PersonalDictionarySnapshot { PersonalDictionarySnapshot( entries: entries, editRules: editRules, - industryLexicon: industryLexicon + industryLexicon: industryLexicon, + bundleIdentifier: bundleIdentifier, + languageCode: languageCode ) } - func snapshot(settings: AppSettings) -> PersonalDictionarySnapshot { + func snapshot( + settings: AppSettings, + bundleIdentifier: String? = nil, + languageCode: String? = nil + ) -> PersonalDictionarySnapshot { snapshot( industryLexicon: IndustryLexiconCatalog.shared.snapshot( for: settings.industryLexicon - ) + ), + bundleIdentifier: bundleIdentifier, + languageCode: languageCode ) } @@ -151,9 +169,10 @@ final class PersonalDictionary: ObservableObject { let replacement = normalized(replacement) guard !original.isEmpty, !replacement.isEmpty, original != replacement else { return nil } - if let index = entries.firstIndex(where: { - $0.original.caseInsensitiveCompare(original) == .orderedSame - }) { + let matches = entries.indices.filter { + entries[$0].original.caseInsensitiveCompare(original) == .orderedSame + } + if let index = matches.first(where: { entries[$0].origin == .manual }) ?? matches.first { entries[index].original = original entries[index].replacement = replacement entries[index].enabled = true @@ -161,12 +180,16 @@ final class PersonalDictionary: ObservableObject { entries[index].status = .active entries[index].confidence = 1 entries[index].evidenceCount = max(1, entries[index].evidenceCount) + entries[index].languageCode = nil + entries[index].appScopes = [] + suspendLearnedMappings(for: original, excluding: entries[index].id) save() return entries[index].id } let entry = DictionaryEntry(original: original, replacement: replacement) entries.append(entry) + suspendLearnedMappings(for: original, excluding: entry.id) save() return entry.id } @@ -188,6 +211,11 @@ final class PersonalDictionary: ObservableObject { guard !original.isEmpty, !replacement.isEmpty, original != replacement else { return } entries[index].original = original entries[index].replacement = replacement + entries[index].origin = .manual + entries[index].status = .active + entries[index].languageCode = nil + entries[index].appScopes = [] + suspendLearnedMappings(for: original, excluding: entries[index].id) save() } @@ -199,11 +227,23 @@ final class PersonalDictionary: ObservableObject { func approveEntry(id: UUID) { guard let index = entries.firstIndex(where: { $0.id == id }) else { return } + if entries[index].origin == .learned, + LearnedCorrectionPolicy.isUnsafeSource(entries[index].original) { + entries[index].origin = .manual + entries[index].languageCode = nil + entries[index].appScopes = [] + } let original = entries[index].original - for otherIndex in entries.indices where otherIndex != index - && entries[otherIndex].origin == .learned - && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame { - entries[otherIndex].status = .pending + if entries[index].origin == .manual { + suspendLearnedMappings(for: original, excluding: entries[index].id) + } else { + for otherIndex in entries.indices where otherIndex != index + && entries[otherIndex].origin == .learned + && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame + && entries[otherIndex].languageCode == entries[index].languageCode + && entries[otherIndex].appScopes == entries[index].appScopes { + entries[otherIndex].status = .pending + } } entries[index].status = .active entries[index].enabled = true @@ -237,7 +277,14 @@ final class PersonalDictionary: ObservableObject { decoder.dateDecodingStrategy = .iso8601 if let data = try? Data(contentsOf: entriesURL), let decoded = try? decoder.decode([DictionaryEntry].self, from: data) { - entries = decoded + entries = decoded.map { entry in + var entry = entry + if entry.origin == .learned, + LearnedCorrectionPolicy.isUnsafeSource(entry.original) { + entry.status = .pending + } + return entry + } } if let data = try? Data(contentsOf: rulesURL), let decoded = try? decoder.decode([EditRule].self, from: data) { @@ -246,8 +293,7 @@ final class PersonalDictionary: ObservableObject { } private func normalized(_ text: String) -> String { - text.replacingOccurrences(of: "\\s+", with: " ", options: .regularExpression) - .trimmingCharacters(in: .whitespacesAndNewlines) + text.replacingOccurrences(of: "\\s+", with: " ", options: .regularExpression).trimmingCharacters(in: .whitespacesAndNewlines) } } diff --git a/Sources/Processing/TranscriptionSanitizer.swift b/Sources/Processing/TranscriptionSanitizer.swift index f21d593..35b5707 100644 --- a/Sources/Processing/TranscriptionSanitizer.swift +++ b/Sources/Processing/TranscriptionSanitizer.swift @@ -11,9 +11,16 @@ enum TranscriptionSanitizer { .trimmingCharacters(in: .whitespacesAndNewlines) } - static func prepare(_ text: String, audioActivity: AudioCaptureActivity? = nil) -> String? { + static func prepare( + _ text: String, + audioActivity: AudioCaptureActivity? = nil, + recognitionPhrases: [String] = [] + ) -> String? { let normalized = normalizeTranscript(text) guard !isNonSpeechArtifact(normalized) else { return nil } + guard !isVocabularyEcho(normalized, audioActivity: audioActivity, phrases: recognitionPhrases) else { + return nil + } // Whole-transcript repetition is a hallucination pattern that shows up // when the model has little real speech to work with. Deliberate spoken @@ -124,6 +131,41 @@ enum TranscriptionSanitizer { return isNonSpeechArtifact(cleaned) ? nil : cleaned } + private static func isVocabularyEcho( + _ text: String, + audioActivity: AudioCaptureActivity?, + phrases: [String] + ) -> Bool { + guard audioActivity?.hasWeakSpeechEvidence == true, phrases.count >= 8 else { return false } + let trimSet = CharacterSet.whitespacesAndNewlines.union(.punctuationCharacters) + let expected = phrases.map { $0.trimmingCharacters(in: trimSet).lowercased() } + var indices: [String: Int] = [:] + for (index, phrase) in expected.enumerated() where !phrase.isEmpty { + if indices[phrase] == nil { indices[phrase] = index } + } + let segments = text.components(separatedBy: CharacterSet(charactersIn: ",,、;;\n")) + .map { $0.trimmingCharacters(in: trimSet).lowercased() } + .filter { !$0.isEmpty } + guard segments.count >= 8 else { return false } + + var matching = 0 + var run = 0 + var longestRun = 0 + var previousIndex: Int? + for segment in segments { + guard let index = indices[segment] else { + run = 0 + previousIndex = nil + continue + } + matching += 1 + run = previousIndex.map { $0 + 1 == index } == true ? run + 1 : 1 + longestRun = max(longestRun, run) + previousIndex = index + } + return longestRun >= 8 && matching * 2 >= segments.count + } + private static func normalizedPhrase(_ text: String) -> String { String(String.UnicodeScalarView(text.unicodeScalars.filter { scalar in !CharacterSet.punctuationCharacters.contains(scalar) diff --git a/Sources/Speech/QwenNativeASREngine.swift b/Sources/Speech/QwenNativeASREngine.swift index 3b30d71..4b2f8ed 100644 --- a/Sources/Speech/QwenNativeASREngine.swift +++ b/Sources/Speech/QwenNativeASREngine.swift @@ -7,7 +7,7 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { private let modelDirectory: URL private let tailPaddingFrames: AVAudioFrameCount private let runtime = QwenNativeASRRuntime() - private let recognitionContextLock = NSLock() + private let contextLock = NSLock() private var recognitionContext = SpeechRecognitionContext.empty init(modelPath: String, modelID: String = QwenASRModel.defaultID) { @@ -24,17 +24,7 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { } func configureRecognition(context: SpeechRecognitionContext) { - recognitionContextLock.lock() - recognitionContext = context - recognitionContextLock.unlock() - } - - /// The exact context string handed to `Qwen3ASRModel.generate(context:)`. - /// Exposed for tests so vocabulary injection can be asserted without a model. - func currentContextPrompt() -> String? { - recognitionContextLock.lock() - defer { recognitionContextLock.unlock() } - return recognitionContext.contextualPrompt() + contextLock.withLock { recognitionContext = context } } func prepare() async { @@ -50,19 +40,25 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { guard isReady else { throw QwenNativeASRError.notConfigured } guard let audioURL else { throw QwenNativeASRError.noAudioFile } - let contextPrompt = currentContextPrompt() + let prompt = contextLock.withLock { + QwenRecognitionPrompt(phrases: recognitionContext.phrases) + } + let started = CFAbsoluteTimeGetCurrent() let result = try await QwenAudioPreprocessor.withPreparedAudio( from: audioURL, tailPaddingFrames: tailPaddingFrames ) { preparedURL in - try await runtime.transcribe( - audioURL: preparedURL, - modelDirectory: modelDirectory, - language: language, - context: contextPrompt - ) + try await QwenContextRecovery.run(prompt: prompt) { context in + try await runtime.transcribe( + audioURL: preparedURL, + modelDirectory: modelDirectory, + language: language, + context: context + ) + } text: { $0.text } } + guard let result else { return "" } let elapsed = CFAbsoluteTimeGetCurrent() - started Log.info( "[Qwen3ASRNative] transcribed \(result.text.count) chars in " diff --git a/Sources/Speech/QwenRecognitionPrompt.swift b/Sources/Speech/QwenRecognitionPrompt.swift new file mode 100644 index 0000000..2bb298b --- /dev/null +++ b/Sources/Speech/QwenRecognitionPrompt.swift @@ -0,0 +1,64 @@ +import Foundation + +struct QwenRecognitionPrompt: Sendable { + static let maximumPhrases = 8 + static let maximumTermsLength = 160 + + let phrases: [String] + let text: String + + init(phrases: [String]) { + var accepted: [String] = [] + var seen = Set() + for raw in phrases { + let phrase = raw.trimmingCharacters(in: .whitespacesAndNewlines) + guard !phrase.isEmpty, phrase.count <= 80, + seen.insert(phrase.lowercased()).inserted else { continue } + let proposed = (accepted + [phrase]).joined(separator: ", ") + guard accepted.count < Self.maximumPhrases else { break } + guard proposed.count <= Self.maximumTermsLength else { continue } + accepted.append(phrase) + } + self.phrases = accepted + text = accepted.isEmpty ? "" : "Vocabulary: \(accepted.joined(separator: ", "))." + } +} + +enum QwenPromptEcho { + static func matches(_ transcript: String, prompt: QwenRecognitionPrompt) -> Bool { + guard !prompt.text.isEmpty else { return false } + let text = transcript.trimmingCharacters(in: .whitespacesAndNewlines) + let lowered = text.lowercased() + if lowered.hasPrefix("vocabulary:") || lowered.hasPrefix("terms:") { return true } + + let trim = CharacterSet.whitespacesAndNewlines.union(.punctuationCharacters) + let segments = text.components(separatedBy: CharacterSet(charactersIn: ",,、;;\n")) + .map { $0.trimmingCharacters(in: trim).lowercased() } + .filter { !$0.isEmpty } + guard segments.count >= 4 else { return false } + let expected = prompt.phrases.map { $0.trimmingCharacters(in: trim).lowercased() } + guard expected.count >= 4 else { return false } + for expectedStart in 0...(expected.count - 4) { + for transcriptStart in 0...(segments.count - 4) { + if Array(segments[transcriptStart..<(transcriptStart + 4)]) + == Array(expected[expectedStart..<(expectedStart + 4)]) { + return true + } + } + } + return false + } +} + +enum QwenContextRecovery { + static func run( + prompt: QwenRecognitionPrompt, + recognize: (String) async throws -> Result, + text: (Result) -> String + ) async throws -> Result? { + let first = try await recognize(prompt.text) + guard QwenPromptEcho.matches(text(first), prompt: prompt) else { return first } + let retry = try await recognize("") + return QwenPromptEcho.matches(text(retry), prompt: prompt) ? nil : retry + } +} diff --git a/Sources/Speech/SpeechRecognitionContext.swift b/Sources/Speech/SpeechRecognitionContext.swift index 32538e5..a599c9a 100644 --- a/Sources/Speech/SpeechRecognitionContext.swift +++ b/Sources/Speech/SpeechRecognitionContext.swift @@ -92,23 +92,4 @@ struct SpeechRecognitionContext: Equatable, Sendable { } return bestTokens } - - /// Free-text biasing prompt for engines whose API accepts a text context - /// (e.g. Qwen3-ASR `context`, injected into the system prompt). Terms are - /// added in rank order until the character budget is reached, so a large - /// dictionary cannot grow the prompt without bound. - func contextualPrompt(maximumCharacters: Int = 600) -> String? { - guard maximumCharacters > 0 else { return nil } - let prefix = "Terms: " - var accepted: [String] = [] - var characterCount = prefix.count - for phrase in phrases { - let addition = (accepted.isEmpty ? 0 : 2) + phrase.count - guard characterCount + addition <= maximumCharacters else { continue } - accepted.append(phrase) - characterCount += addition - } - guard !accepted.isEmpty else { return nil } - return prefix + accepted.joined(separator: ", ") - } } diff --git a/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift b/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift index 040a642..c862074 100644 --- a/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift +++ b/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift @@ -2,6 +2,16 @@ import XCTest @testable import OpenType final class CorrectionCandidateClassifierTests: XCTestCase { + func testDoesNotLearnStandaloneFillerAsReusableCorrection() { + XCTAssertNil(CorrectionCandidateClassifier.candidate( + inserted: "嗯。", + userFinal: "Do anything", + sourceRecordID: UUID(), + languageCode: "zh", + bundleIdentifier: "com.apple.Notes" + )) + } + func testLearnsCaseAndSpacingCorrectionAsHighConfidenceTerm() throws { let candidate = try XCTUnwrap(CorrectionCandidateClassifier.candidate( inserted: "Please use open type today.", diff --git a/Tests/OpenTypeTests/InputSessionOwnershipTests.swift b/Tests/OpenTypeTests/InputSessionOwnershipTests.swift index 767edc6..96ef2a4 100644 --- a/Tests/OpenTypeTests/InputSessionOwnershipTests.swift +++ b/Tests/OpenTypeTests/InputSessionOwnershipTests.swift @@ -220,6 +220,7 @@ private final class SessionFixture { settings: .init(developerInterfaceEnabled: true, httpToken: "token"), registry: registry ) coordinator = InputSessionCoordinator(service: service, settings: settings, ownership: ownership) + coordinator.speechActivityOverrideForTesting = { _ in true } } func create() async throws -> InputSession { diff --git a/Tests/OpenTypeTests/IntegrationOutputTests.swift b/Tests/OpenTypeTests/IntegrationOutputTests.swift index fbf7733..f7bc3dd 100644 --- a/Tests/OpenTypeTests/IntegrationOutputTests.swift +++ b/Tests/OpenTypeTests/IntegrationOutputTests.swift @@ -5,6 +5,50 @@ import XCTest @MainActor final class IntegrationOutputTests: XCTestCase { + func testImportedAudioRejectsNoSpeechBeforeTranscriptionOrFinalEvent() async throws { + let store = registry() + defer { store.cleanup() } + store.registry.approve(IntegrationClient.localHTTP(tokenID: "token")) + let service = makeService(registry: store.registry) + let session = try await service.createSession(request(mode: .direct), clientID: clientID) + let coordinator = InputSessionCoordinator(service: service) + let engine = TestSpeechEngine(transcript: "Vocabulary: Alpha, Beta, Gamma, Delta") + coordinator.engineLoader = { _ in engine } + coordinator.speechActivityOverrideForTesting = { _ in false } + + await assertThrowsIntegrationError(.noSpeechDetected) { + _ = try await coordinator.processAudioFile( + sessionID: session.id, clientID: clientID, + audioURL: URL(fileURLWithPath: "/tmp/utter-missing-audio.wav"), cleanup: false + ) + } + XCTAssertEqual(engine.transcribeCount, 0) + XCTAssertEqual(try service.session(session.id, clientID: clientID)?.state, .failed) + } + + func testCoordinatorRejectsWeakAudioVocabularyEcho() { + let store = registry() + defer { store.cleanup() } + let suite = "IntegrationVocabularyEcho-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + settings.industryLexicon = .technology + let coordinator = InputSessionCoordinator( + service: makeService(registry: store.registry), + settings: settings + ) + coordinator.requestSettings = VoiceInputSettings(settings: settings) + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + let echo = ["云原生", "容器编排", "微服务", "服务网格", "持续集成", "CI", "持续交付", "CD"] + .joined(separator: ", ") + + XCTAssertThrowsError(try coordinator.prepareTranscript(echo, audioActivity: activity)) { error in + XCTAssertEqual(error as? IntegrationError, .noSpeechDetected) + } + } + func testAppDelegateSharesTextProcessorWithIntegrationCoordinator() { let delegate = AppDelegate() @@ -127,6 +171,7 @@ private extension IntegrationOutputTests { private final class TestSpeechEngine: SpeechEngine, @unchecked Sendable { let transcript: String + private(set) var transcribeCount = 0 var isReady: Bool { true } init(transcript: String) { @@ -134,7 +179,8 @@ private final class TestSpeechEngine: SpeechEngine, @unchecked Sendable { } func transcribe(audioURL: URL?, language: String?) async throws -> String { - transcript + transcribeCount += 1 + return transcript } } diff --git a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift index 766639d..6252260 100644 --- a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift +++ b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift @@ -3,6 +3,91 @@ import XCTest @testable import OpenType final class PersonalDictionaryLearningTests: XCTestCase { + func testExistingLearnedFillerIsInertButManualRuleStillWorks() { + let bad = DictionaryEntry( + original: "嗯。", replacement: "Do anything", origin: .learned, + languageCode: "zh", appScopes: ["com.apple.Notes"] + ) + let snapshot = PersonalDictionarySnapshot(entries: [bad], editRules: []) + XCTAssertEqual(snapshot.applyReplacements(to: "嗯。"), "嗯。") + XCTAssertFalse(snapshot.recognitionPhrases.contains("Do anything")) + + let manual = PersonalDictionarySnapshot(entries: [ + DictionaryEntry(original: "嗯。", replacement: "Do anything") + ], editRules: []) + XCTAssertEqual(manual.applyReplacements(to: "嗯。"), "Do anything") + } + + func testUnsafePersistedRuleAppearsPendingWithoutRewritingFile() throws { + let directory = FileManager.default.temporaryDirectory + .appendingPathComponent("OpenTypeDictionaryReload-\(UUID().uuidString)", isDirectory: true) + addTeardownBlock { try? FileManager.default.removeItem(at: directory) } + let store = PersonalDictionary(directoryURL: directory) + let entry = DictionaryEntry( + original: "嗯。", replacement: "Do anything", origin: .learned, + languageCode: "zh", appScopes: ["com.apple.Notes"] + ) + store.entries = [entry] + store.save() + let file = directory.appendingPathComponent("dictionary.json") + let originalData = try Data(contentsOf: file) + + let reloaded = PersonalDictionary(directoryURL: directory) + XCTAssertEqual(reloaded.entries.first?.status, .pending) + XCTAssertEqual(try Data(contentsOf: file), originalData) + reloaded.approveEntry(id: entry.id) + XCTAssertEqual(reloaded.entries.first?.origin, .manual) + XCTAssertEqual(reloaded.applyReplacements(to: "嗯。"), "Do anything") + } + + func testLearnedScopeDoesNotLeakAcrossAppsOrLanguages() { + let entry = DictionaryEntry( + original: "open type", replacement: "OpenType", origin: .learned, + languageCode: "en", appScopes: ["com.apple.Notes"] + ) + let store = makeStore() + store.entries = [entry] + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "en") + .applyReplacements(to: "open type"), "OpenType") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.TextEdit", languageCode: "en") + .applyReplacements(to: "open type"), "open type") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "open type"), "open type") + XCTAssertEqual(store.snapshot().applyReplacements(to: "open type"), "open type") + } + + func testIndependentAppsDoNotMergeLearnedEvidence() throws { + let store = makeStore() + let first = LearnedCorrectionCandidate( + original: "open tape", replacement: "OpenType", confidence: 0.82, + sourceRecordID: UUID(), languageCode: "en", bundleIdentifier: "com.apple.Notes" + ) + let second = LearnedCorrectionCandidate( + original: "open tape", replacement: "OpenType", confidence: 0.82, + sourceRecordID: UUID(), languageCode: "en", bundleIdentifier: "com.apple.TextEdit" + ) + store.recordLearnedCandidate(first) + store.recordLearnedCandidate(second) + XCTAssertEqual(store.entries.count, 2) + XCTAssertEqual(store.entries.map(\.status), [.pending, .pending]) + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "en") + .applyReplacements(to: "open tape"), "open tape") + } + + func testGlobalManualEntrySuspendsConflictingScopedLearnedRules() { + let store = makeStore() + store.entries = [ + DictionaryEntry(original: "open type", replacement: "Wrong A", origin: .learned, + languageCode: "en", appScopes: ["com.apple.Notes"]), + DictionaryEntry(original: "open type", replacement: "Wrong B", origin: .learned, + languageCode: "en", appScopes: ["com.apple.TextEdit"]), + ] + store.addEntry(original: "open type", replacement: "OpenType") + XCTAssertEqual(store.entries.filter { $0.origin == .learned }.map(\.status), [.pending]) + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.TextEdit", languageCode: "en") + .applyReplacements(to: "open type"), "OpenType") + } + func testLegacyEntryDecodesAsActiveManualTerm() throws { let data = Data(#"{"original":"open type","replacement":"OpenType","enabled":true}"#.utf8) let entry = try JSONDecoder().decode(DictionaryEntry.self, from: data) @@ -27,12 +112,14 @@ final class PersonalDictionaryLearningTests: XCTestCase { let entryID = try XCTUnwrap(store.recordLearnedCandidate(first)) XCTAssertEqual(store.entries.first(where: { $0.id == entryID })?.status, .pending) - XCTAssertEqual(store.applyReplacements(to: "菜单蓝"), "菜单蓝") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "菜单蓝"), "菜单蓝") XCTAssertTrue(SpeechRecognitionContext(dictionaryEntries: store.entries).phrases.isEmpty) store.recordLearnedCandidate(second) XCTAssertEqual(store.entries.first(where: { $0.id == entryID })?.status, .active) - XCTAssertEqual(store.applyReplacements(to: "菜单蓝"), "菜单栏") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "菜单蓝"), "菜单栏") } func testHighConfidenceLearnedTermActivatesOnceAndManualEntryWins() throws { @@ -76,7 +163,7 @@ final class PersonalDictionaryLearningTests: XCTestCase { XCTAssertEqual(store.applyReplacements(to: "open tape"), "open tape") store.approveEntry(id: competingID) - XCTAssertEqual(store.applyReplacements(to: "open tape"), "Open Tape") + XCTAssertEqual(store.snapshot(languageCode: "en").applyReplacements(to: "open tape"), "Open Tape") XCTAssertEqual(store.entries.filter(\.isEffective).count, 1) } diff --git a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift index 32f2a37..5eadb4a 100644 --- a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift +++ b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift @@ -1,3 +1,4 @@ +import AVFoundation import Foundation import XCTest @testable import OpenType @@ -75,6 +76,89 @@ final class QwenNativeASREngineTests: XCTestCase { XCTAssertEqual(weightAfter.contentModificationDate, weightBefore.contentModificationDate) } + func testExistingModelUsesBoundedRareTermHint() async throws { + guard ProcessInfo.processInfo.environment["OPENTYPE_QWEN_NATIVE_INTEGRATION"] == "1" else { + throw XCTSkip("Set OPENTYPE_QWEN_NATIVE_INTEGRATION=1 to run native Qwen integration tests") + } + let modelPath = try XCTUnwrap(ProcessInfo.processInfo.environment["OPENTYPE_QWEN_MODEL_PATH"]) + let audioURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-rare-term-\(UUID().uuidString).aiff") + defer { try? FileManager.default.removeItem(at: audioURL) } + let speech = Process() + speech.executableURL = URL(fileURLWithPath: "/usr/bin/say") + speech.arguments = ["-v", "Samantha", "-o", audioURL.path, + "We shipped Zyralith to production today."] + try speech.run() + speech.waitUntilExit() + XCTAssertEqual(speech.terminationStatus, 0) + + let engine = QwenNativeASREngine(modelPath: modelPath) + engine.configureRecognition(context: .empty) + let withoutContext = try await engine.transcribe(audioURL: audioURL, language: "en") + engine.configureRecognition(context: SpeechRecognitionContext(phrases: ["Zyralith"])) + let withContext = try await engine.transcribe(audioURL: audioURL, language: "en") + engine.configureRecognition(context: SpeechRecognitionContext( + phrases: ["Zyralith", "Roleva", "Utter"] + )) + let withBoundedContext = try await engine.transcribe(audioURL: audioURL, language: "en") + print("QWEN_RARE_TERM_WITHOUT_CONTEXT=\(withoutContext)") + print("QWEN_RARE_TERM_WITH_CONTEXT=\(withContext)") + print("QWEN_RARE_TERM_WITH_BOUNDED_CONTEXT=\(withBoundedContext)") + XCTAssertTrue(withContext.contains("Zyralith"), withContext) + XCTAssertTrue(withBoundedContext.contains("Zyralith"), withBoundedContext) + } + + func testExistingModelRejectsSyntheticNonSpeechNoise() async throws { + guard ProcessInfo.processInfo.environment["OPENTYPE_QWEN_NATIVE_INTEGRATION"] == "1" else { + throw XCTSkip("Set OPENTYPE_QWEN_NATIVE_INTEGRATION=1 to run the native Qwen integration test") + } + let modelPath = try XCTUnwrap(ProcessInfo.processInfo.environment["OPENTYPE_QWEN_MODEL_PATH"]) + let engine = QwenNativeASREngine(modelPath: modelPath) + XCTAssertTrue(engine.isReady) + let phrases = IndustryLexiconCatalog.shared.snapshot(for: .technology).recognitionPhrases + engine.configureRecognition(context: SpeechRecognitionContext(phrases: phrases)) + + let audioURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-noise-\(UUID().uuidString).wav") + defer { try? FileManager.default.removeItem(at: audioURL) } + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: 32_000)) + buffer.frameLength = 32_000 + var seed: UInt32 = 1 + for index in 0.. Set { let files = try FileManager.default.contentsOfDirectory( at: directory, diff --git a/Tests/OpenTypeTests/QwenRecognitionContextTests.swift b/Tests/OpenTypeTests/QwenRecognitionContextTests.swift new file mode 100644 index 0000000..252090b --- /dev/null +++ b/Tests/OpenTypeTests/QwenRecognitionContextTests.swift @@ -0,0 +1,65 @@ +import XCTest +@testable import OpenType + +final class QwenRecognitionContextTests: XCTestCase { + func testPromptIsBoundedAndDoesNotContainWholeLexicon() { + let phrases = (1...30).map { "Term\($0)" } + let prompt = QwenRecognitionPrompt(phrases: phrases) + XCTAssertLessThanOrEqual(prompt.phrases.count, 8) + XCTAssertLessThanOrEqual(prompt.text.count, 180) + XCTAssertFalse(prompt.text.contains("Term30")) + let deduplicated = QwenRecognitionPrompt(phrases: ["Alpha", "alpha", "Beta"]) + XCTAssertEqual(deduplicated.phrases, ["Alpha", "Beta"]) + } + + func testEchoRetriesOnceWithoutContextAndRejectsRepeatedEcho() async throws { + let prompt = QwenRecognitionPrompt(phrases: ["Alpha", "Beta", "Gamma", "Delta"]) + var contexts: [String] = [] + let accepted = try await QwenContextRecovery.run(prompt: prompt) { context in + contexts.append(context) + return context.isEmpty ? "We shipped Alpha today." : "Vocabulary: Alpha, Beta, Gamma, Delta." + } text: { $0 } + XCTAssertEqual(accepted, "We shipped Alpha today.") + XCTAssertEqual(contexts, [prompt.text, ""]) + + contexts.removeAll() + let rejected = try await QwenContextRecovery.run(prompt: prompt) { context in + contexts.append(context) + return "Alpha, Beta, Gamma, Delta" + } text: { $0 } + XCTAssertNil(rejected) + XCTAssertEqual(contexts, [prompt.text, ""]) + } + + func testSingleSpokenTermIsNotClassedAsEcho() { + let prompt = QwenRecognitionPrompt(phrases: ["Zyralith"]) + XCTAssertFalse(QwenPromptEcho.matches("We shipped Zyralith today.", prompt: prompt)) + XCTAssertFalse(QwenPromptEcho.matches("Zyralith", prompt: prompt)) + XCTAssertTrue(QwenPromptEcho.matches("Vocabulary: Zyralith.", prompt: prompt)) + } + + func testDetectsOrderedEchoAfterLeadingText() { + let prompt = QwenRecognitionPrompt(phrases: ["Alpha", "Beta", "Gamma", "Delta"]) + XCTAssertTrue(QwenPromptEcho.matches( + "Here are the terms, Alpha, Beta, Gamma, Delta", prompt: prompt + )) + } + + func testRetryFailureNeverReturnsPromptEcho() async { + enum RetryFailure: Error { case failed } + let prompt = QwenRecognitionPrompt(phrases: ["Alpha", "Beta", "Gamma", "Delta"]) + var attempts = 0 + do { + _ = try await QwenContextRecovery.run(prompt: prompt) { _ -> String in + attempts += 1 + if attempts == 2 { throw RetryFailure.failed } + return "Alpha, Beta, Gamma, Delta" + } text: { $0 } + XCTFail("A failed retry must not return the first transcript") + } catch RetryFailure.failed { + XCTAssertEqual(attempts, 2) + } catch { + XCTFail("Unexpected error: \(error)") + } + } +} diff --git a/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift b/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift new file mode 100644 index 0000000..929a7f0 --- /dev/null +++ b/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift @@ -0,0 +1,121 @@ +import AVFoundation +import XCTest +@testable import OpenType + +final class SpeechActivityClassifierTests: XCTestCase { + func testGeneratedAmbientSoundsAreNotSpeech() async throws { + for amplitude: Float in [0.007, 0.04, 0.2] { + for tone in [false, true] { + let url = try makeAmbientAudio(amplitude: amplitude, tone: tone) + defer { try? FileManager.default.removeItem(at: url) } + let hasSpeech = await SpeechActivityClassifier.containsSpeech(at: url) + XCTAssertFalse(hasSpeech, "amplitude \(amplitude), tone \(tone)") + } + } + } + + func testDetectsRepositorySpeechAndShortClips() async throws { + for sample in ["en-sample.m4a", "zh-sample.m4a"] { + let sourceURL = repositorySample(sample) + let fullSpeech = await SpeechActivityClassifier.containsSpeech(at: sourceURL) + XCTAssertTrue(fullSpeech, sample) + + for seconds in [0.8, 0.25] { + for gain: Float in [1, 0.1] { + let clipURL = try makeClip(from: sourceURL, seconds: seconds, gain: gain) + defer { try? FileManager.default.removeItem(at: clipURL) } + let shortSpeech = await SpeechActivityClassifier.containsSpeech(at: clipURL) + XCTAssertTrue(shortSpeech, "\(sample), \(seconds)s, gain \(gain)") + } + } + } + } + + func testMissingEmptyAndCancelledAudioFailClosed() async throws { + let missing = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-missing-\(UUID().uuidString).wav") + let missingResult = await SpeechActivityClassifier.containsSpeech(at: missing) + XCTAssertFalse(missingResult) + let nilResult = await SpeechActivityClassifier.containsSpeech(at: nil) + XCTAssertFalse(nilResult) + + let emptyURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-empty-\(UUID().uuidString).wav") + defer { try? FileManager.default.removeItem(at: emptyURL) } + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + _ = try AVAudioFile(forWriting: emptyURL, settings: format.settings) + let emptyResult = await SpeechActivityClassifier.containsSpeech(at: emptyURL) + XCTAssertFalse(emptyResult) + + let cancelledResult = await Task { + withUnsafeCurrentTask { $0?.cancel() } + return await SpeechActivityClassifier.containsSpeech(at: repositorySample("en-sample.m4a")) + }.value + XCTAssertFalse(cancelledResult) + } + + private func repositorySample(_ filename: String) -> URL { + URL(fileURLWithPath: #filePath) + .deletingLastPathComponent() + .deletingLastPathComponent() + .deletingLastPathComponent() + .appendingPathComponent("docs/assets/demos/\(filename)") + } + + private func makeAmbientAudio(amplitude: Float, tone: Bool) throws -> URL { + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: 32_000)) + buffer.frameLength = 32_000 + var seed: UInt32 = 1 + for index in 0.. URL { + let source = try AVAudioFile(forReading: sourceURL) + let format = source.processingFormat + source.framePosition = AVAudioFramePosition(format.sampleRate) + let frames = AVAudioFrameCount(seconds * format.sampleRate) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: frames)) + try source.read(into: buffer, frameCount: frames) + for channel in 0.. String { transcript } +} + +@MainActor +final class VoicePipelineSilentInsertionTests: XCTestCase { + func testNonSpeechGateStopsHallucinatedShortWordBeforeASR() async { + let suite = "VoicePipelineNoSpeech-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + let state = AppState() + let pipeline = VoicePipeline(appState: state) + let engine = VocabularyEchoEngine(transcript: "嗯。") + pipeline.engineOverride = engine + pipeline.speechActivityOverrideForTesting = { _ in false } + var insertionCount = 0 + pipeline.textInserter.insertOverrideForTesting = { _ in + insertionCount += 1 + return .success + } + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + + await pipeline.processRecording( + audioURL: nil, + audioActivity: activity, + language: nil, + settings: VoiceInputSettings(settings: settings), + inputMode: .dictation, + targetApp: nil + ) + + XCTAssertEqual(insertionCount, 0) + XCTAssertEqual(state.phase, .idle) + XCTAssertTrue(state.rawTranscription.isEmpty) + } + + func testVocabularyEchoNeverReachesInsertionAcrossOutputModes() async { + let suite = "VoicePipelineSilentInsertionTests-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + settings.industryLexicon = .technology + let terms = ["云原生", "容器编排", "微服务", "服务网格", "持续集成", "CI", "持续交付", "CD"] + let transcript = terms.joined(separator: ", ") + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + XCTAssertTrue(activity.hasMeaningfulAudio) + + let cases: [(OutputMode, VoiceInputMode, Bool)] = [ + (.direct, .dictation, false), + (.processed, .dictation, false), + (.processed, .dictation, true), + (.command, .dictation, false), + (.direct, .translation(.english), false), + ] + for (outputMode, inputMode, instantInsert) in cases { + settings.outputMode = outputMode + settings.enableInstantInsert = instantInsert + let state = AppState() + let pipeline = VoicePipeline(appState: state) + pipeline.engineOverride = VocabularyEchoEngine(transcript: transcript) + pipeline.speechActivityOverrideForTesting = { _ in true } + var insertionCount = 0 + pipeline.textInserter.insertOverrideForTesting = { _ in + insertionCount += 1 + return .success + } + + await pipeline.processRecording( + audioURL: nil, + audioActivity: activity, + language: nil, + settings: VoiceInputSettings(settings: settings), + inputMode: inputMode, + targetApp: nil + ) + + XCTAssertEqual(insertionCount, 0, "mode: \(outputMode), instant: \(instantInsert)") + XCTAssertEqual(state.phase, .idle) + XCTAssertTrue(state.lastInsertedText.isEmpty) + } + } +} diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md new file mode 100644 index 0000000..2f6940a --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md @@ -0,0 +1,40 @@ +# Intent: Prevent unsolicited text insertion when no one speaks + +**Status:** approved +**Approved-by:** User (conversation) +**Approved-date:** 2026-09-23 +**Upstream:** User report in this conversation, 2026-09-23 + +## Problem + +The user observed Utter inserting a long list of unrelated technical terms into the focused input field while the user was not speaking. The list included “Do anything” and terms that resemble the technology industry lexicon. This is an unsolicited output and can expose text to whichever application has focus. The user identified the local Qwen recognition model and computer microphone; recording mode remains uncertain. + +## Outcome + +An input session without user speech ends with no text insertion, clipboard write, spoken-edit action, or history entry. Intended speech still produces text, including legitimately spoken technical terms. + +## Scope + +- Trace local and remote microphone capture, streaming and recorded recognition, vocabulary context, transcript preparation, and every insertion path reachable from voice input. +- Reproduce the reported vocabulary-list output with a deterministic test or a privacy-safe captured trace before changing behavior. +- Fix the shared boundary responsible for unsolicited output and remove any bypass that permits the same failure. +- Keep microphone audio and transcript content out of diagnostic logs and repository fixtures unless the user explicitly supplies a safe sample. +- No change to cloud deployment, billing, or unrelated product features. + +## Constraints + +- Preserve correctly spoken short phrases, deliberate repetition, and vocabulary terms. +- Treat unintended insertion as a privacy-sensitive failure; avoid a broad phrase blacklist that would erase legitimate dictation. +- Follow the repository's staged approval, regression-test, independent-verification, and rollback requirements for a high-risk change. + +## Acceptance criteria + +- A deterministic regression check reproduces a no-speech session that currently inserts the reported kind of technical-term list; it passes after the fix and verifies that insertion and clipboard side effects do not occur. +- Silence and ambient non-speech audio cannot trigger final text insertion in direct, processed, command, or translation mode, including streaming and remote-microphone paths where applicable. +- Tests show that audible, intentionally spoken “Do anything” and representative technology terms remain insertable. +- The affected privacy and permission paths, repository checks, and a release-style app build pass; an independent verifier reviews the high-risk change before PR approval. + +## Open questions + +- Which output mode and trigger produced the observed list? The user identified the computer microphone and local Qwen model; real-time recognition remains unconfirmed. +- Did the list appear in the Utter overlay before insertion, or only in the target application? diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md new file mode 100644 index 0000000..e828c9a --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md @@ -0,0 +1,30 @@ +# Plan: Prevent unsolicited Qwen vocabulary insertion + +**Status:** approved +**Approved-by:** User (conversation; confirmed revised implementation plan) +**Approved-date:** 2026-09-24 +**Upstream:** [spec.md](spec.md) + +This plan supersedes the 2026-09-23 plan. Existing code and tests remain in the draft PR; the items below describe only the redesigned work. + +## Implementation sequence + +- [ ] First establish red contract tests at the real seams: all three recorded-audio entry points, Qwen context and echo handling, dictionary snapshot and learning. Represent the reported long-list and `嗯。 -> Do anything` failures without copying private data into fixtures. Assert zero insertion, clipboard, command, and history effects on rejection. +- [x] Trace every snapshot consumer and source of target app/language. Make one effective snapshot carry the same scoped personal entries through ASR context, replacement, formatter hints, and protected terms. Preserve global manual terms and industry post-processing. Test unknown scope and evidence from different scopes. +- [x] Restore a bounded Qwen term prompt with effective personal entries only. Keep the established ASR interface where possible. Retain the exact per-call context for echo comparison. Test priority, deduplication, budget, and empty context; compare rare-word recognition with the installed model. +- [x] Route menu-bar, live integration, and imported audio through one recorded-audio speech decision and final transcript acceptance boundary. Reuse the existing classifier. Ensure provisional streaming text cannot become final output after rejection. +- [x] On Qwen context echo, retry the same audio once without context. Reject a remaining echo or retry failure before any external effect. Delete stale duplicate guards and obsolete full-list context code after the common contract passes. +- [x] Tighten correction-candidate and persisted learned-entry eligibility for short filler/non-speech sources. Prevent cross-scope evidence merges. Keep the user's local dictionary intact; test that the observed learned mapping is inert and a manual mapping still works. +- [x] Add fault-injection and end-to-end tests for classifier failure/no window/cancellation, retry failure, each output mode, short intentional phrases, English/Chinese speech, scoped vocabulary, and imported audio. Review privacy logging and split touched Swift files at real responsibility boundaries to keep them below 300 lines. + +## Verification and delivery + +- [x] Run focused red/green tests, then `bash scripts/sdlc-checks.sh`, `bash scripts/ci-basic-checks.sh`, and `swift test`. +- [x] Replay generated silence/noise and repository-owned speech through installed Qwen. Measure recognition with zero, one, and bounded relevant terms; test prompt echo and fallback. Do not save private microphone recordings. +- [x] Build a release-style app with `bash scripts/build-app.sh` and verify the assembled artifact. +- [ ] Perform real-window QA with the computer microphone once competing Utter instances can be avoided without disrupting the user's work. Record unavailable paths as unverified. +- [ ] Update `verification.md` with fresh evidence and residual risks. Check the PR against current `main` and resolve conflicts. Keep it draft pending independent verification, PR approval, and the separate signed-release/production decision. + +## Human gate + +The user approved this revised plan on 2026-09-24. Verification, independent review, and release retain their separate gates. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md new file mode 100644 index 0000000..04f7ffd --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md @@ -0,0 +1,45 @@ +# Spec: Prevent unsolicited Qwen vocabulary insertion + +**Status:** approved +**Approved-by:** User (conversation; confirmed the three-part redesign) +**Approved-date:** 2026-09-24 +**Upstream:** [intent.md](intent.md) + +The user rejected the earlier design after it disabled Qwen recognition hints and left a harmful learned rule active, then confirmed a three-part redesign on 2026-09-24. The earlier implementation and verification remain historical evidence, not acceptance of this revision. + +## Context + +The user identified the local Qwen recognition model and computer microphone. Real-time recognition and overlay behavior remain unknown. The reported output starts with terms present in the user's personal dictionary and continues with the bundled technology lexicon in its source order. In the incident-era implementation, `VoicePipeline` combined those terms into `SpeechRecognitionContext`, and `QwenNativeASREngine` passed the resulting `Terms: ...` string into `Qwen3ASRModel.generate(context:)`. The current draft PR has removed that context. RMS loudness does not distinguish speech from other sound. The final transcript passes through `TranscriptionSanitizer.prepare` and may reach direct or processed insertion, including instant insert. + +The user clarified that no words were spoken and questioned whether the later formatting AI added the list. Local `input_history.json` retains both stages. Four processed menu-bar records (2026-09-21 and 2026-09-23 UTC) show the long list already present in `rawText`, beginning with the literal `Terms: ` prefix; `processedText` removes that prefix and changes only a few terms and punctuation. The old `SpeechRecognitionContext.contextualPrompt()` generated precisely a `Terms: ` vocabulary prompt, which `QwenNativeASREngine` passed to `Qwen3ASRModel.generate(context:)`. This strongly identifies Qwen's echo of its recognition context as the long-list source. The original audio and exact runtime context snapshot were not retained, so the triggering sound and exact prompt bytes cannot be replayed. The overlay state is also unknown. + +Three older processed records show a separate failure: the two-character ASR result “嗯。” became “Do anything”. Read-only inspection of the local personal dictionary found an active learned `嗯。 -> Do anything` rule with confidence 0.82 and two evidence records. `PersonalDictionarySnapshot.applyReplacements` runs before formatting and deterministically makes this substitution. The previous attribution to the formatting AI was wrong. The rule was created by correction capture and activated after two observations; the original edits were pruned from the 500-record history, so their intent is unknown. A no-speech gate blocks a non-speech source, but the learned-rule eligibility and scope also need review before the dictionary is considered safe. + +Before the first patch, the sanitizer only rejected the isolated phrase “Do anything” on weak audio. A temporary command that compiled that sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduced the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. + +After the first implementation, a local Qwen replay with generated non-speech noise passed the RMS gate and returned “嗯。”; `TranscriptionSanitizer.prepare` accepted it. The first design therefore cannot satisfy the no-speech criterion. A separate read-only prototype of Apple's built-in Sound Analysis classifier gave maximum `speech` confidence 0.357 for that two-second noise clip, 0.91–0.97 for the repository's English and Chinese speech samples, and 0.91–0.94 for 0.8-second speech clips using 0.5-second analysis windows. A 0.25-second speech crop padded to 0.75 seconds scored 0.63–0.65. These are feasibility results, not a calibrated acceptance threshold. + +## Superseding design + +The four saved sessions already contain the ordered `Terms: ...` list in Qwen's raw transcript. Formatting changed only its presentation. The original audio and exact runtime prompt are unavailable. Separate saved sessions show raw `嗯。` becoming `Do anything` because of an active learned dictionary rule, before formatting. Its source edits were pruned, so there is no evidence that it was an intended reusable correction. A synthetic rare-name sample was recognized correctly with a one-term Qwen hint and incorrectly without it; this is a narrow benefit to preserve, not a measured production success rate. + +1. **One recorded-audio speech decision.** Menu-bar capture, live integration capture, and integration audio-file import must call the same classifier before any final transcript or output is committed. Keep RMS as a cheap early filter where available. Missing/unreadable audio, no classification window, cancellation, or classifier failure produces no output and a recoverable status. Streaming partials remain provisional. +2. **Bounded Qwen recognition hints.** Restore Qwen context using only effective personal-dictionary replacement terms for the current target app and selected language. Prefer manual entries, then recent/high-evidence learned entries; deduplicate and impose a small phrase-count and character budget. Do not include the full dictionary, bundled industry lexicon, edit rules, or whole correction pairs. Preserve post-ASR personal and industry correction. Other recognizers keep their established mechanism but receive effective scoped personal entries. +3. **Pre-output prompt-echo handling.** Compare Qwen's candidate with the exact context supplied to that recognition call. On a prompt-prefix or ordered-term echo, discard it and retry the same audio once with empty context. Revalidate the retry at the shared transcript boundary. If it still echoes, lacks speech evidence, or errors, finish without insertion, clipboard, command, or history. Short deliberately spoken terms and ordinary sentences remain eligible. +4. **Safe learned corrections.** Enforce learned-entry app/language scope at the dictionary snapshot used by recognition, replacement, formatter hints, and protected terms. A scoped learned entry is ineligible when the required context is unknown; unscoped manual entries stay global. Reject automatic learning from a short filler/interjection or non-speech artifact, including the observed `嗯。` source. Apply the same eligibility to persisted learned entries so the existing unsafe rule cannot act. Do not silently delete or rewrite the local dictionary. Do not merge evidence from different app/language scopes into a global rule. User-entered manual corrections remain possible. + +Recorded audio is the source for speech evidence, the session snapshot is the source for effective terms, and final transcript acceptance is the transaction boundary before external effects. Replace the narrow coordinator seam if it cannot carry exact context and scope; do not add three unrelated path guards. + +## Failure modes and limits + +- The classifier can reject quiet speech or accept humanlike ambient sound. Calibrate with short English/Chinese speech, quiet speech, silence, and several noise types; verify the computer microphone. Nearby human speech is outside this no-speech guarantee. +- A small Qwen prompt can still echo. One empty-context retry contains detected echoes but costs an extra inference. Rare names outside the term budget may be less accurate; measure the tradeoff. +- Scope filtering may leave a learned correction unavailable when app or language is unknown. Prefer a missed correction to an unintended replacement. Manual global entries remain available. +- Do not log audio, transcript contents, or dictionary terms. Preserve existing user data. Roll back to the prior signed artifact if necessary; do not restore the full Qwen term list without separate review. + +## Acceptance and verification + +- A deterministic regression reaches all three recorded-audio paths with non-speech audio and fabricated vocabulary output and observes no final text, clipboard, command, or history side effect. +- Local Qwen replay shows a bounded relevant term helps a rare-word sample, and prompt echo triggers at most one empty-context retry that cannot insert the list. Deliberately spoken short terms and ordinary speech remain eligible. +- A regression proves `嗯。 -> Do anything` cannot arise from automatic learning or an existing learned rule, while a manual correction and valid scoped learned correction work in their proper app/language. +- Fault injection covers classifier errors, cancellation, missing audio, retry failure, and unknown scope. Run repository checks, full tests, release-style build, real-window microphone QA, independent high-risk review, conflict-free PR review, and separate protected release gate. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md new file mode 100644 index 0000000..8b9b8f4 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -0,0 +1,74 @@ +# Verification: Prevent unsolicited Qwen vocabulary insertion + +**Status:** draft +**Approved-by:** — +**Approved-date:** — +**Upstream:** [plan.md](plan.md) + +The prior verification was approved for the superseded design. Results below describe the existing draft PR and do not establish acceptance of the revised design approved on 2026-09-24. Fresh verification is required after implementation. + +## Revised-design evidence (2026-09-24, pending independent review) + +| Check | Result | Evidence | +|---|---|---| +| Session dictionary scope | Pass locally | Snapshots filter learned entries by app and language; menu-bar and integration admission capture one effective snapshot. Global manual entries remain available. Tests cover unknown context, cross-app evidence, and valid scoped replacements. | +| Unsafe existing learned rule | Pass locally | The saved-type `嗯。 -> Do anything` rule is inert on use and shown as pending when loaded, without rewriting the dictionary file. A user approval or edit explicitly converts it to a manual rule. New correction capture rejects the filler source. | +| Qwen context | Pass locally | Qwen receives at most eight effective personal replacement terms within a 160-character terms budget; the expanded industry lexicon remains available for post-ASR correction without being sent whole to Qwen. Tests cover budget and deduplication. | +| Qwen echo recovery | Pass locally | A prompt-prefix or ordered-term echo triggers one empty-context retry. A repeated echo produces no transcript; a retry error cannot return the first echo. Short spoken terms remain eligible. | +| All recorded-audio entry paths | Pass locally | Menu-bar, live integration, and imported audio call the same Sound Analysis classifier before final transcript and output. Imported-audio regression asserts no ASR call or final session when the classifier rejects. | +| Existing session lifecycle | Pass locally | Integrated current `main` ownership/cancellation changes without removing their transaction guard. Ten ownership tests and the mode-specific insertion tests pass. | +| Full `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass locally | 793 XCTest cases, 18 skipped, zero failures; one Swift Testing case passed. Ran after integration with current `main`. | +| `bash scripts/sdlc-checks.sh` and `bash scripts/ci-basic-checks.sh` | Pass locally | Both completed after the merge resolution. | +| Installed local Qwen replay | Pass locally | Generated noise was rejected; repository-owned English and Chinese speech samples were transcribed. The synthetic rare name was rendered as “Zerolith” without a hint and “Zyralith” with one or three bounded hints. This is one generated voice sample, not a general accuracy estimate. | +| Release-style app build | Pass locally | `bash scripts/build-app.sh --app-only --sign=-` built the app and CLI helper, bundled Metal resources, and passed artifact verification. Ad-hoc signing is for local checking only. | +| PR mergeability and remote CI | Pass at observed head | GitHub reported draft PR #112 `MERGEABLE` at `9f9ccc8`; Contract & Tests, Release-style App Build, and SDLC Gate all succeeded. | +| Real computer-microphone silence QA | Pass for live integration route | Temporarily stopped both original instances, launched the ad-hoc local build, granted its microphone permission through the onboarding UI, and recorded three seconds from the default MacBook Pro Microphone. The live recording API returned `no_speech_detected` with no transcript or final text. | +| Real computer-microphone speech QA | Inconclusive | Two attempts to play a Chinese technical sentence through the MacBook speakers into the microphone returned `no_speech_detected`; the system output was subsequently observed muted. No valid audible speech sample was confirmed at the microphone. The menu-bar hotkey path was not exercised because this ad-hoc build lacked Accessibility authorization. | +| Original app restoration | Pass | Stopped the test build and restarted the two exact original bundles, one from `/Applications/Utter.app` and one from the prior worktree. Restored `hasCompletedOnboarding=true` and `activationMode=longPress`; both original processes were observed running. The speaker mute state was left as the user set it. | +| Independent verification, signed release, production observation | Pending | The author cannot satisfy the independent high-risk review or protected production gate. | + +These results verify the revised local implementation only. The exact incident audio was not retained, and Sound Analysis cannot determine who spoke. The draft PR stays unreleased until the remaining gates are met. + +## Evidence + +| Check | Result | Evidence | +|---|---|---| +| Original sanitizer counterexample | Red before change | Weak-audio rehearsal passed the user-reported 351-character vocabulary list unchanged. | +| Qwen synthetic-noise counterexample | Red before speech gate | Installed Qwen model returned an insertable “嗯。” from generated non-speech noise. | +| Incident-stage provenance | Read-only local history | Four processed menu-bar records on 2026-09-21/23 UTC had 357-character `rawText` beginning `Terms: ` and 351-character `processedText` without that prefix. The literal prefix and ordered vocabulary match the legacy Qwen context construction. No private dictionary contents or audio were copied into the repository. | +| Separate learned-dictionary substitution | Read-only local history and dictionary | Three older records had “嗯。” as ASR text and “Do anything” as processed text. An active learned `嗯。 -> Do anything` rule deterministically explains the change; it became active after two correction observations. The original edit records have been pruned, so user intent cannot be inferred. | +| `TranscriptionSanitizerTests` | Pass | Ordered vocabulary echo rejected; short terms and strong deliberate dictation retained. | +| `VoicePipelineSilentInsertionTests` | Pass | Direct, processed, instant insert, command, and translation paths made zero insertion calls for rejected audio. | +| `IntegrationOutputTests/testCoordinatorRejectsWeakAudioVocabularyEcho` | Pass | Live integration's shared transcript preparation rejected the list. | +| `SpeechActivityClassifierTests` | Pass | Generated white noise and tones at three amplitudes rejected; repository English/Chinese speech, 0.8-second clips, padded 0.25-second clips, and 10%-volume short clips accepted; missing, empty, and cancelled audio rejected. | +| Local Qwen noise replay | Pass | `OPENTYPE_QWEN_NATIVE_INTEGRATION=1` test with installed model rejected generated non-speech noise after independent speech classification. | +| Qwen prompt-echo regression | Pass | The same installed-model noise replay now asserts that ASR output does not begin with `Terms: ` even when a technology recognition context was configured on the engine. | +| Local Qwen repository speech samples | Pass | Installed model transcribed English and Chinese repository samples without downloading weights. | +| Formatting-output guard | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms and an unsupported short-ASR expansion under both fidelity policies. The short historical output came from a learned replacement before formatting, so this guard test does not reproduce that incident. | +| `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass | 754 XCTest cases, 15 skipped, no failures; one Swift Testing case passed before the later formatting-AI regression test was added. That focused new test passed separately. Scratch path used because a fresh MLX submodule checkout stalled. | +| `bash scripts/sdlc-checks.sh` | Pass | Stage artifacts valid after the user confirmed updated verification. | +| `bash scripts/ci-basic-checks.sh` | Pass | SDLC, localization, resources, lexicon evaluation, and repository checks passed. | +| `bash scripts/build-app.sh --app-only --sign=-` | Pass | Xcode Release build, Metal shader bundle, CLI helper, bundle assembly, ad-hoc codesign, and artifact verification passed. No usable Utter signing identity was installed; this is a local build, not a trusted release. | +| Rebase onto `origin/main` | Pass | Commit `686becb` is based on `60e7ed4` (Confucius4-R2T2 support). Qwen's new `modelID` and tail-padding path remain intact; vocabulary context injection remains removed. | +| Post-rebase checks | Pass | `bash scripts/ci-basic-checks.sh`; `swift test --scratch-path /tmp/utter-silent-insertion-build` (763 XCTest cases, 17 skipped, no failures; one Swift Testing case passed); release-style ad-hoc app build and artifact verification. | +| Real microphone/window check | Partial | The revised-design table above records the new three-second silence result, inconclusive speaker playback, and restoration of both original instances. This older evidence table describes the superseded design. | +| Independent verification | Pending | — | + +## Acceptance criteria + +- The reported vocabulary echo is rejected before insertion, clipboard, edit-command, and history paths — automated menu-bar and integration tests pass; independent review pending. +- Generated non-speech audio is rejected by an independent classifier even when Qwen emits text — local model replay passes; a three-second real-microphone silence session was also rejected. This does not isolate which of the RMS gate and classifier rejected that session. +- Clear English/Chinese speech and short clips remain accepted — automated file fixtures pass, including quieter short clips; real microphone speech acceptance remains pending. + +## Residual risk + +- The built-in classifier cannot identify who spoke. Nearby human speech may still be transcribed; this is outside the no-speech and non-speech-noise acceptance criterion. +- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. The separately observed learned `嗯。 -> Do anything` rule is now inert unless manually approved or edited. +- The 0.6 speech-confidence threshold is calibrated against fixtures, not a diverse microphone corpus. A real-microphone silence session passed, but an audible speech acceptance test and independent review remain required. +- The two original Utter instances were restored after the test. The ad-hoc test build lacked Accessibility authorization, so menu-bar hotkey insertion was not verified with real audio. +- The branch includes `origin/main` at `7139f54`; compare against the live PR base again before review or merge, since main can advance. +- Remote PR CI passed at `9f9ccc8`; trusted release signing and production behavior have not been verified. + +## Decision + +The user confirmed this historical verification on 2026-09-24, then rejected the design because Qwen vocabulary help was lost and the unsafe learned rule remained active. It no longer approves the change. Independent high-risk review and fresh verification remain pending. Do not release until a conflict-free PR, signed build, and protected production approval are complete.