From 686becb73af6230d9a691cde30c28560036ef3e5 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 00:34:26 +0800 Subject: [PATCH 01/10] fix: block silent Qwen vocabulary insertion --- Sources/App/VoicePipeline+Processing.swift | 22 +++- Sources/App/VoicePipeline+WarmUp.swift | 43 +++++++ Sources/App/VoicePipeline.swift | 42 +----- Sources/Audio/SpeechActivityClassifier.swift | 115 +++++++++++++++++ .../Integration/InputSessionCoordinator.swift | 10 +- Sources/Output/TextInserter.swift | 6 + .../Processing/TranscriptionSanitizer.swift | 44 ++++++- Sources/Speech/QwenNativeASREngine.swift | 19 +-- Sources/Speech/SpeechRecognitionContext.swift | 19 --- .../IntegrationOutputTests.swift | 22 ++++ .../QwenNativeASREngineTests.swift | 64 +++++++-- .../SpeechActivityClassifierTests.swift | 121 ++++++++++++++++++ .../SpeechRecognitionQualityTests.swift | 14 -- .../TranscriptFidelityGuardTests.swift | 19 +++ .../TranscriptionSanitizerTests.swift | 41 ++++++ .../VoicePipelineSilentInsertionTests.swift | 94 ++++++++++++++ .../intent.md | 40 ++++++ .../2026-09-23-silent-input-insertion/plan.md | 35 +++++ .../2026-09-23-silent-input-insertion/spec.md | 49 +++++++ .../verification.md | 45 +++++++ 20 files changed, 757 insertions(+), 107 deletions(-) create mode 100644 Sources/App/VoicePipeline+WarmUp.swift create mode 100644 Sources/Audio/SpeechActivityClassifier.swift create mode 100644 Tests/OpenTypeTests/SpeechActivityClassifierTests.swift create mode 100644 Tests/OpenTypeTests/VoicePipelineSilentInsertionTests.swift create mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md create mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md create mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md create mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md diff --git a/Sources/App/VoicePipeline+Processing.swift b/Sources/App/VoicePipeline+Processing.swift index 82bfa5a5..fa16fda9 100644 --- a/Sources/App/VoicePipeline+Processing.swift +++ b/Sources/App/VoicePipeline+Processing.swift @@ -19,6 +19,10 @@ extension VoicePipeline { showNoSpeechDetected(reason: "recorded audio energy below threshold") return } + guard await recordingContainsSpeech(audioURL) else { + showNoSpeechDetected(reason: "recorded audio has no speech evidence") + return + } let preparedRaw = try await transcribePreparedText( audioURL: audioURL, @@ -86,6 +90,15 @@ extension VoicePipeline { } } + private func recordingContainsSpeech(_ audioURL: URL?) async -> Bool { + #if DEBUG + if let speechActivityOverrideForTesting { + return await speechActivityOverrideForTesting(audioURL) + } + #endif + return await SpeechActivityClassifier.containsSpeech(at: audioURL) + } + private func transcribePreparedText( audioURL: URL?, audioActivity: AudioCaptureActivity, @@ -102,8 +115,13 @@ extension VoicePipeline { let elapsed = CFAbsoluteTimeGetCurrent() - started Log.info("[VoicePipeline] ASR stage finished in \(String(format: "%.2f", elapsed))s") - guard let prepared = TranscriptionSanitizer.prepare(raw, audioActivity: audioActivity) else { - showNoSpeechDetected(reason: "transcription has no meaningful content: \(raw)") + let recognitionPhrases = PersonalDictionary.shared.snapshot(settings: settings).recognitionPhrases + guard let prepared = TranscriptionSanitizer.prepare( + raw, + audioActivity: audioActivity, + recognitionPhrases: recognitionPhrases + ) else { + showNoSpeechDetected(reason: "transcription has no meaningful content") throw VoicePipelineStop.noSpeech } diff --git a/Sources/App/VoicePipeline+WarmUp.swift b/Sources/App/VoicePipeline+WarmUp.swift new file mode 100644 index 00000000..fb04d6ad --- /dev/null +++ b/Sources/App/VoicePipeline+WarmUp.swift @@ -0,0 +1,43 @@ +import Foundation + +@MainActor +extension VoicePipeline { + func warmUp() async { + let settings = appState.settings + let catalog = ModelCatalog.shared + catalog.refreshStatus(recheckingErrors: true) + let llmStatus = catalog.llmModels.first(where: { $0.id == settings.llmModel })?.status + let formattingModelID = settings.localLLMBackend == .espresso + ? settings.espressoModelPath + : settings.llmModel + let formattingModelAvailable = settings.localLLMBackend == .espresso + ? FileManager.default.fileExists(atPath: NSString(string: settings.espressoModelPath).expandingTildeInPath) + : (llmStatus == .downloaded || llmStatus == .ready) + let shouldLoadSpeech = StartupModelPreloadPolicy.shouldPreloadSpeechModel( + enabled: settings.preloadSpeechModelOnLaunch, + speechEngine: settings.speechEngine, + modelDownloaded: catalog.isWhisperDownloaded(settings.whisperModel) + ) + let shouldLoadFormatting = StartupModelPreloadPolicy.shouldPreloadFormattingModel( + enabled: settings.preloadFormattingModelOnLaunch, + useRemoteLLM: settings.useRemoteLLM, + modelID: formattingModelID, + modelDownloaded: formattingModelAvailable + ) + + if shouldLoadSpeech { + await ensureEngineLoaded(requestPermission: false) + } + + if shouldLoadFormatting { + let espressoOutcome = await enqueueFormattingModelPreload( + showFailureInStatus: false + ).value + if espressoOutcome != nil { + return + } + } + + markReadyIfPossible() + } +} diff --git a/Sources/App/VoicePipeline.swift b/Sources/App/VoicePipeline.swift index 436009e4..106bde86 100644 --- a/Sources/App/VoicePipeline.swift +++ b/Sources/App/VoicePipeline.swift @@ -40,6 +40,9 @@ final class VoicePipeline { var engineLoadBarrier: (() async -> Void)? /// Test-only observation point for whether the remote capture path is used. var remoteCaptureSpy: RemoteMicCaptureSpy? + #if DEBUG + var speechActivityOverrideForTesting: ((URL?) async -> Bool)? + #endif var currentEngine: (any SpeechEngine)? { if let engineOverride { return engineOverride } @@ -57,45 +60,6 @@ final class VoicePipeline { self.textProcessor = textProcessor } - func warmUp() async { - let settings = appState.settings - let catalog = ModelCatalog.shared - catalog.refreshStatus(recheckingErrors: true) - let llmStatus = catalog.llmModels.first(where: { $0.id == settings.llmModel })?.status - let formattingModelID = settings.localLLMBackend == .espresso - ? settings.espressoModelPath - : settings.llmModel - let formattingModelAvailable = settings.localLLMBackend == .espresso - ? FileManager.default.fileExists(atPath: NSString(string: settings.espressoModelPath).expandingTildeInPath) - : (llmStatus == .downloaded || llmStatus == .ready) - let shouldLoadSpeech = StartupModelPreloadPolicy.shouldPreloadSpeechModel( - enabled: settings.preloadSpeechModelOnLaunch, - speechEngine: settings.speechEngine, - modelDownloaded: catalog.isWhisperDownloaded(settings.whisperModel) - ) - let shouldLoadFormatting = StartupModelPreloadPolicy.shouldPreloadFormattingModel( - enabled: settings.preloadFormattingModelOnLaunch, - useRemoteLLM: settings.useRemoteLLM, - modelID: formattingModelID, - modelDownloaded: formattingModelAvailable - ) - - if shouldLoadSpeech { - await ensureEngineLoaded(requestPermission: false) - } - - if shouldLoadFormatting { - let espressoOutcome = await enqueueFormattingModelPreload( - showFailureInStatus: false - ).value - if espressoOutcome != nil { - return - } - } - - markReadyIfPossible() - } - // MARK: - Recording func start( diff --git a/Sources/Audio/SpeechActivityClassifier.swift b/Sources/Audio/SpeechActivityClassifier.swift new file mode 100644 index 00000000..633f1670 --- /dev/null +++ b/Sources/Audio/SpeechActivityClassifier.swift @@ -0,0 +1,115 @@ +import AVFoundation +import Foundation +import SoundAnalysis + +enum SpeechActivityClassifier { + static let minimumSpeechConfidence = 0.6 + private static let windowSeconds = 0.5 + private static let minimumFileSeconds = 0.75 + + static func containsSpeech(at audioURL: URL?) async -> Bool { + guard let audioURL, !Task.isCancelled else { return false } + var paddedURL: URL? + defer { + if let paddedURL { try? FileManager.default.removeItem(at: paddedURL) } + } + + do { + let analysisURL = try preparedURL(audioURL, paddedURL: &paddedURL) + let request = try SNClassifySoundRequest(classifierIdentifier: .version1) + request.windowDuration = CMTime(seconds: windowSeconds, preferredTimescale: 16_000) + let observer = SpeechClassificationObserver() + let analyzer = try SNAudioFileAnalyzer(url: analysisURL) + try analyzer.add(request, withObserver: observer) + let completed = await withTaskCancellationHandler { + await analyzer.analyze() + } onCancel: { + analyzer.cancelAnalysis() + } + let result = observer.result + return completed && !Task.isCancelled && !result.failed + && result.windows > 0 && result.maxSpeech >= minimumSpeechConfidence + } catch { + Log.error("[SpeechActivity] classification failed") + return false + } + } + + private static func preparedURL(_ url: URL, paddedURL: inout URL?) throws -> URL { + let source = try AVAudioFile(forReading: url) + let format = source.processingFormat + guard format.sampleRate > 0, source.length > 0 else { throw SpeechActivityError.invalidAudio } + let requiredFrames = Int(ceil(minimumFileSeconds * format.sampleRate)) + guard source.length < requiredFrames else { return url } + guard requiredFrames <= Int(UInt32.max), + let buffer = AVAudioPCMBuffer(pcmFormat: format, frameCapacity: AVAudioFrameCount(requiredFrames)) else { + throw SpeechActivityError.invalidAudio + } + try source.read(into: buffer) + let readFrames = Int(buffer.frameLength) + guard readFrames > 0, readFrames < requiredFrames else { throw SpeechActivityError.invalidAudio } + + let channels = Int(format.channelCount) + if let data = buffer.floatChannelData { + for channel in 0.. String { - guard let transcript = TranscriptionSanitizer.prepare(raw, audioActivity: audioActivity) else { + let recognitionPhrases = PersonalDictionary.shared.snapshot(settings: settings).recognitionPhrases + guard let transcript = TranscriptionSanitizer.prepare( + raw, + audioActivity: audioActivity, + recognitionPhrases: recognitionPhrases + ) else { throw IntegrationError.noSpeechDetected } return transcript diff --git a/Sources/Output/TextInserter.swift b/Sources/Output/TextInserter.swift index 5784c64d..befa5f8b 100644 --- a/Sources/Output/TextInserter.swift +++ b/Sources/Output/TextInserter.swift @@ -11,8 +11,14 @@ enum InsertResult { @MainActor final class TextInserter { var recentInsertionAnchor: RecentInsertionAnchor? + #if DEBUG + var insertOverrideForTesting: ((String) -> InsertResult)? + #endif func insert(text: String, targetApp: NSRunningApplication? = nil) async -> InsertResult { + #if DEBUG + if let insertOverrideForTesting { return insertOverrideForTesting(text) } + #endif guard AXIsProcessTrusted() else { Log.error("[TextInserter] no AX trust") return .probablyFailed(reason: "Accessibility permission not granted") diff --git a/Sources/Processing/TranscriptionSanitizer.swift b/Sources/Processing/TranscriptionSanitizer.swift index f21d593d..35b57076 100644 --- a/Sources/Processing/TranscriptionSanitizer.swift +++ b/Sources/Processing/TranscriptionSanitizer.swift @@ -11,9 +11,16 @@ enum TranscriptionSanitizer { .trimmingCharacters(in: .whitespacesAndNewlines) } - static func prepare(_ text: String, audioActivity: AudioCaptureActivity? = nil) -> String? { + static func prepare( + _ text: String, + audioActivity: AudioCaptureActivity? = nil, + recognitionPhrases: [String] = [] + ) -> String? { let normalized = normalizeTranscript(text) guard !isNonSpeechArtifact(normalized) else { return nil } + guard !isVocabularyEcho(normalized, audioActivity: audioActivity, phrases: recognitionPhrases) else { + return nil + } // Whole-transcript repetition is a hallucination pattern that shows up // when the model has little real speech to work with. Deliberate spoken @@ -124,6 +131,41 @@ enum TranscriptionSanitizer { return isNonSpeechArtifact(cleaned) ? nil : cleaned } + private static func isVocabularyEcho( + _ text: String, + audioActivity: AudioCaptureActivity?, + phrases: [String] + ) -> Bool { + guard audioActivity?.hasWeakSpeechEvidence == true, phrases.count >= 8 else { return false } + let trimSet = CharacterSet.whitespacesAndNewlines.union(.punctuationCharacters) + let expected = phrases.map { $0.trimmingCharacters(in: trimSet).lowercased() } + var indices: [String: Int] = [:] + for (index, phrase) in expected.enumerated() where !phrase.isEmpty { + if indices[phrase] == nil { indices[phrase] = index } + } + let segments = text.components(separatedBy: CharacterSet(charactersIn: ",,、;;\n")) + .map { $0.trimmingCharacters(in: trimSet).lowercased() } + .filter { !$0.isEmpty } + guard segments.count >= 8 else { return false } + + var matching = 0 + var run = 0 + var longestRun = 0 + var previousIndex: Int? + for segment in segments { + guard let index = indices[segment] else { + run = 0 + previousIndex = nil + continue + } + matching += 1 + run = previousIndex.map { $0 + 1 == index } == true ? run + 1 : 1 + longestRun = max(longestRun, run) + previousIndex = index + } + return longestRun >= 8 && matching * 2 >= segments.count + } + private static func normalizedPhrase(_ text: String) -> String { String(String.UnicodeScalarView(text.unicodeScalars.filter { scalar in !CharacterSet.punctuationCharacters.contains(scalar) diff --git a/Sources/Speech/QwenNativeASREngine.swift b/Sources/Speech/QwenNativeASREngine.swift index 3b30d715..f0b06a0f 100644 --- a/Sources/Speech/QwenNativeASREngine.swift +++ b/Sources/Speech/QwenNativeASREngine.swift @@ -7,8 +7,6 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { private let modelDirectory: URL private let tailPaddingFrames: AVAudioFrameCount private let runtime = QwenNativeASRRuntime() - private let recognitionContextLock = NSLock() - private var recognitionContext = SpeechRecognitionContext.empty init(modelPath: String, modelID: String = QwenASRModel.defaultID) { modelDirectory = URL(fileURLWithPath: modelPath).standardizedFileURL @@ -23,20 +21,6 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { modelDirectory == URL(fileURLWithPath: modelPath).standardizedFileURL } - func configureRecognition(context: SpeechRecognitionContext) { - recognitionContextLock.lock() - recognitionContext = context - recognitionContextLock.unlock() - } - - /// The exact context string handed to `Qwen3ASRModel.generate(context:)`. - /// Exposed for tests so vocabulary injection can be asserted without a model. - func currentContextPrompt() -> String? { - recognitionContextLock.lock() - defer { recognitionContextLock.unlock() } - return recognitionContext.contextualPrompt() - } - func prepare() async { guard isReady else { return } do { @@ -50,7 +34,6 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { guard isReady else { throw QwenNativeASRError.notConfigured } guard let audioURL else { throw QwenNativeASRError.noAudioFile } - let contextPrompt = currentContextPrompt() let started = CFAbsoluteTimeGetCurrent() let result = try await QwenAudioPreprocessor.withPreparedAudio( from: audioURL, @@ -60,7 +43,7 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { audioURL: preparedURL, modelDirectory: modelDirectory, language: language, - context: contextPrompt + context: "" ) } let elapsed = CFAbsoluteTimeGetCurrent() - started diff --git a/Sources/Speech/SpeechRecognitionContext.swift b/Sources/Speech/SpeechRecognitionContext.swift index 32538e5f..a599c9a0 100644 --- a/Sources/Speech/SpeechRecognitionContext.swift +++ b/Sources/Speech/SpeechRecognitionContext.swift @@ -92,23 +92,4 @@ struct SpeechRecognitionContext: Equatable, Sendable { } return bestTokens } - - /// Free-text biasing prompt for engines whose API accepts a text context - /// (e.g. Qwen3-ASR `context`, injected into the system prompt). Terms are - /// added in rank order until the character budget is reached, so a large - /// dictionary cannot grow the prompt without bound. - func contextualPrompt(maximumCharacters: Int = 600) -> String? { - guard maximumCharacters > 0 else { return nil } - let prefix = "Terms: " - var accepted: [String] = [] - var characterCount = prefix.count - for phrase in phrases { - let addition = (accepted.isEmpty ? 0 : 2) + phrase.count - guard characterCount + addition <= maximumCharacters else { continue } - accepted.append(phrase) - characterCount += addition - } - guard !accepted.isEmpty else { return nil } - return prefix + accepted.joined(separator: ", ") - } } diff --git a/Tests/OpenTypeTests/IntegrationOutputTests.swift b/Tests/OpenTypeTests/IntegrationOutputTests.swift index fbf7733d..4d772464 100644 --- a/Tests/OpenTypeTests/IntegrationOutputTests.swift +++ b/Tests/OpenTypeTests/IntegrationOutputTests.swift @@ -5,6 +5,28 @@ import XCTest @MainActor final class IntegrationOutputTests: XCTestCase { + func testCoordinatorRejectsWeakAudioVocabularyEcho() { + let store = registry() + defer { store.cleanup() } + let suite = "IntegrationVocabularyEcho-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + settings.industryLexicon = .technology + let coordinator = InputSessionCoordinator( + service: makeService(registry: store.registry), + settings: settings + ) + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + let echo = ["云原生", "容器编排", "微服务", "服务网格", "持续集成", "CI", "持续交付", "CD"] + .joined(separator: ", ") + + XCTAssertThrowsError(try coordinator.prepareTranscript(echo, audioActivity: activity)) { error in + XCTAssertEqual(error as? IntegrationError, .noSpeechDetected) + } + } + func testAppDelegateSharesTextProcessorWithIntegrationCoordinator() { let delegate = AppDelegate() diff --git a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift index 32f2a37c..75f84f55 100644 --- a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift +++ b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift @@ -1,3 +1,4 @@ +import AVFoundation import Foundation import XCTest @testable import OpenType @@ -75,6 +76,56 @@ final class QwenNativeASREngineTests: XCTestCase { XCTAssertEqual(weightAfter.contentModificationDate, weightBefore.contentModificationDate) } + func testExistingModelRejectsSyntheticNonSpeechNoise() async throws { + guard ProcessInfo.processInfo.environment["OPENTYPE_QWEN_NATIVE_INTEGRATION"] == "1" else { + throw XCTSkip("Set OPENTYPE_QWEN_NATIVE_INTEGRATION=1 to run the native Qwen integration test") + } + let modelPath = try XCTUnwrap(ProcessInfo.processInfo.environment["OPENTYPE_QWEN_MODEL_PATH"]) + let engine = QwenNativeASREngine(modelPath: modelPath) + XCTAssertTrue(engine.isReady) + let phrases = IndustryLexiconCatalog.shared.snapshot(for: .technology).recognitionPhrases + engine.configureRecognition(context: SpeechRecognitionContext(phrases: phrases)) + + let audioURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-noise-\(UUID().uuidString).wav") + defer { try? FileManager.default.removeItem(at: audioURL) } + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: 32_000)) + buffer.frameLength = 32_000 + var seed: UInt32 = 1 + for index in 0.. Set { let files = try FileManager.default.contentsOfDirectory( at: directory, diff --git a/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift b/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift new file mode 100644 index 00000000..929a7f0c --- /dev/null +++ b/Tests/OpenTypeTests/SpeechActivityClassifierTests.swift @@ -0,0 +1,121 @@ +import AVFoundation +import XCTest +@testable import OpenType + +final class SpeechActivityClassifierTests: XCTestCase { + func testGeneratedAmbientSoundsAreNotSpeech() async throws { + for amplitude: Float in [0.007, 0.04, 0.2] { + for tone in [false, true] { + let url = try makeAmbientAudio(amplitude: amplitude, tone: tone) + defer { try? FileManager.default.removeItem(at: url) } + let hasSpeech = await SpeechActivityClassifier.containsSpeech(at: url) + XCTAssertFalse(hasSpeech, "amplitude \(amplitude), tone \(tone)") + } + } + } + + func testDetectsRepositorySpeechAndShortClips() async throws { + for sample in ["en-sample.m4a", "zh-sample.m4a"] { + let sourceURL = repositorySample(sample) + let fullSpeech = await SpeechActivityClassifier.containsSpeech(at: sourceURL) + XCTAssertTrue(fullSpeech, sample) + + for seconds in [0.8, 0.25] { + for gain: Float in [1, 0.1] { + let clipURL = try makeClip(from: sourceURL, seconds: seconds, gain: gain) + defer { try? FileManager.default.removeItem(at: clipURL) } + let shortSpeech = await SpeechActivityClassifier.containsSpeech(at: clipURL) + XCTAssertTrue(shortSpeech, "\(sample), \(seconds)s, gain \(gain)") + } + } + } + } + + func testMissingEmptyAndCancelledAudioFailClosed() async throws { + let missing = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-missing-\(UUID().uuidString).wav") + let missingResult = await SpeechActivityClassifier.containsSpeech(at: missing) + XCTAssertFalse(missingResult) + let nilResult = await SpeechActivityClassifier.containsSpeech(at: nil) + XCTAssertFalse(nilResult) + + let emptyURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-empty-\(UUID().uuidString).wav") + defer { try? FileManager.default.removeItem(at: emptyURL) } + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + _ = try AVAudioFile(forWriting: emptyURL, settings: format.settings) + let emptyResult = await SpeechActivityClassifier.containsSpeech(at: emptyURL) + XCTAssertFalse(emptyResult) + + let cancelledResult = await Task { + withUnsafeCurrentTask { $0?.cancel() } + return await SpeechActivityClassifier.containsSpeech(at: repositorySample("en-sample.m4a")) + }.value + XCTAssertFalse(cancelledResult) + } + + private func repositorySample(_ filename: String) -> URL { + URL(fileURLWithPath: #filePath) + .deletingLastPathComponent() + .deletingLastPathComponent() + .deletingLastPathComponent() + .appendingPathComponent("docs/assets/demos/\(filename)") + } + + private func makeAmbientAudio(amplitude: Float, tone: Bool) throws -> URL { + let format = try XCTUnwrap(AVAudioFormat( + commonFormat: .pcmFormatFloat32, + sampleRate: 16_000, + channels: 1, + interleaved: false + )) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: 32_000)) + buffer.frameLength = 32_000 + var seed: UInt32 = 1 + for index in 0.. URL { + let source = try AVAudioFile(forReading: sourceURL) + let format = source.processingFormat + source.framePosition = AVAudioFramePosition(format.sampleRate) + let frames = AVAudioFrameCount(seconds * format.sampleRate) + let buffer = try XCTUnwrap(AVAudioPCMBuffer(pcmFormat: format, frameCapacity: frames)) + try source.read(into: buffer, frameCount: frames) + for channel in 0.. String { transcript } +} + +@MainActor +final class VoicePipelineSilentInsertionTests: XCTestCase { + func testNonSpeechGateStopsHallucinatedShortWordBeforeASR() async { + let suite = "VoicePipelineNoSpeech-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + let state = AppState() + let pipeline = VoicePipeline(appState: state) + let engine = VocabularyEchoEngine(transcript: "嗯。") + pipeline.engineOverride = engine + pipeline.speechActivityOverrideForTesting = { _ in false } + var insertionCount = 0 + pipeline.textInserter.insertOverrideForTesting = { _ in + insertionCount += 1 + return .success + } + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + + await pipeline.processRecording( + audioURL: nil, + audioActivity: activity, + language: nil, + settings: settings, + inputMode: .dictation, + targetApp: nil + ) + + XCTAssertEqual(insertionCount, 0) + XCTAssertEqual(state.phase, .idle) + XCTAssertTrue(state.rawTranscription.isEmpty) + } + + func testVocabularyEchoNeverReachesInsertionAcrossOutputModes() async { + let suite = "VoicePipelineSilentInsertionTests-\(UUID().uuidString)" + let defaults = UserDefaults(suiteName: suite)! + defer { defaults.removePersistentDomain(forName: suite) } + let settings = AppSettings(defaults: defaults) + settings.industryLexicon = .technology + let terms = ["云原生", "容器编排", "微服务", "服务网格", "持续集成", "CI", "持续交付", "CD"] + let transcript = terms.joined(separator: ", ") + var activity = AudioCaptureActivity() + activity.record(rms: 0.002, frameCount: 16_000) + XCTAssertTrue(activity.hasMeaningfulAudio) + + let cases: [(OutputMode, VoiceInputMode, Bool)] = [ + (.direct, .dictation, false), + (.processed, .dictation, false), + (.processed, .dictation, true), + (.command, .dictation, false), + (.direct, .translation(.english), false), + ] + for (outputMode, inputMode, instantInsert) in cases { + settings.outputMode = outputMode + settings.enableInstantInsert = instantInsert + let state = AppState() + let pipeline = VoicePipeline(appState: state) + pipeline.engineOverride = VocabularyEchoEngine(transcript: transcript) + pipeline.speechActivityOverrideForTesting = { _ in true } + var insertionCount = 0 + pipeline.textInserter.insertOverrideForTesting = { _ in + insertionCount += 1 + return .success + } + + await pipeline.processRecording( + audioURL: nil, + audioActivity: activity, + language: nil, + settings: settings, + inputMode: inputMode, + targetApp: nil + ) + + XCTAssertEqual(insertionCount, 0, "mode: \(outputMode), instant: \(instantInsert)") + XCTAssertEqual(state.phase, .idle) + XCTAssertTrue(state.lastInsertedText.isEmpty) + } + } +} diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md new file mode 100644 index 00000000..1dfebb29 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md @@ -0,0 +1,40 @@ +# Intent: Prevent unsolicited text insertion when no one speaks + +**Status:** approved +**Approved-by:** User (conversation) +**Approved-date:** 2026-09-23 +**Upstream:** User report in this conversation, 2026-09-23 + +## Problem + +The user observed Utter inserting a long list of unrelated technical terms into the focused input field while the user was not speaking. The list included “Do anything” and terms that resemble the technology industry lexicon. This is an unsolicited output and can expose text to whichever application has focus. The specific microphone, recognition engine, and recording mode involved have not yet been identified. + +## Outcome + +An input session without user speech ends with no text insertion, clipboard write, spoken-edit action, or history entry. Intended speech still produces text, including legitimately spoken technical terms. + +## Scope + +- Trace local and remote microphone capture, streaming and recorded recognition, vocabulary context, transcript preparation, and every insertion path reachable from voice input. +- Reproduce the reported vocabulary-list output with a deterministic test or a privacy-safe captured trace before changing behavior. +- Fix the shared boundary responsible for unsolicited output and remove any bypass that permits the same failure. +- Keep microphone audio and transcript content out of diagnostic logs and repository fixtures unless the user explicitly supplies a safe sample. +- No change to cloud deployment, billing, or unrelated product features. + +## Constraints + +- Preserve correctly spoken short phrases, deliberate repetition, and vocabulary terms. +- Treat unintended insertion as a privacy-sensitive failure; avoid a broad phrase blacklist that would erase legitimate dictation. +- Follow the repository's staged approval, regression-test, independent-verification, and rollback requirements for a high-risk change. + +## Acceptance criteria + +- A deterministic regression check reproduces a no-speech session that currently inserts the reported kind of technical-term list; it passes after the fix and verifies that insertion and clipboard side effects do not occur. +- Silence and ambient non-speech audio cannot trigger final text insertion in direct, processed, command, or translation mode, including streaming and remote-microphone paths where applicable. +- Tests show that audible, intentionally spoken “Do anything” and representative technology terms remain insertable. +- The affected privacy and permission paths, repository checks, and a release-style app build pass; an independent verifier reviews the high-risk change before PR approval. + +## Open questions + +- Which microphone source, recognition engine, output mode, and trigger produced the observed list? Diagnosis can begin without these details, but they are needed to match the user's exact runtime path. +- Did the list appear in the Utter overlay before insertion, or only in the target application? diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md new file mode 100644 index 00000000..0860ac77 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md @@ -0,0 +1,35 @@ +# Plan: Prevent unsolicited Qwen vocabulary insertion + +**Status:** approved +**Approved-by:** User (conversation; revised plan) +**Approved-date:** 2026-09-23 +**Upstream:** [spec.md](spec.md) + +The user approved an earlier plan on 2026-09-23. Generated non-speech noise produced an insertable Qwen transcript, invalidating that plan. The revised spec has been approved; this updated plan requires its own approval before implementation resumes. + +## Work items + +- [x] Establish a red regression at transcript preparation: the original sanitizer passed the user's 351-character list on weak audio. Add a safe pipeline insertion spy and green cases for intended speech. +- [x] Replay generated non-speech noise through the installed Qwen model. It returned an insertable “嗯。” rather than the reported list; retain the deterministic vocabulary-echo regression and the new noise regression as separate contracts. +- [ ] Add one recorded-audio speech classifier using the macOS Sound Analysis built-in model. Run it after the RMS gate and before an ASR transcript can reach output in the menu-bar and live integration paths. Use a 0.5-second analysis window, calibrate the acceptance threshold on the listed fixtures, and pad very short recordings in a temporary copy so they receive a classification result. Treat missing files, no results, classifier errors, and cancellation as no speech; clean up temporary copies on every path. +- [x] Remove Qwen's free-text recognition-context injection in `Sources/Speech/QwenNativeASREngine.swift`; retain existing personal and industry replacement behavior after ASR. Delete obsolete context-only code and tests. +- [x] Add a narrow ordered-vocabulary-echo decision to `Sources/Processing/TranscriptionSanitizer.swift` and feed it the session vocabulary at both `VoicePipeline` and `InputSessionCoordinator`. Keep rejection before edit commands, instant insert, formatting, history, and clipboard effects; targeted tests pass. +- [ ] Add fault-injection tests for weak audio and fabricated ASR output; verify every reachable output mode rejects before insertion, clipboard, edit-command, or history side effects. Test classifier success, failure, no-window, and cancellation paths. Probe louder ambient sound and real microphone noise; if non-speech bypasses the gate, revise the design rather than claiming the broad no-speech criterion passed. Test clear intentionally spoken vocabulary, short phrases, and protected dictionary replacements. +- [ ] Review the complete diff for duplicate guards, stale Qwen context paths, privacy-sensitive logs, and changed contracts. Keep each touched Swift file under 300 lines or split at a real responsibility boundary. +- [ ] Prepare `verification.md` with command results, synthetic Qwen replay, affected permission/privacy paths, remaining limits, and rollback. Obtain independent high-risk verification and PR approval before release. + +## Verification plan + +- [ ] Focused red/green tests: `swift test --filter TranscriptionSanitizerTests`, Qwen engine tests, and pipeline/integration output tests. +- [ ] Real Qwen and Sound Analysis replay: generated silence, the existing two-second noise fixture that produced “嗯。”, several louder/noisier synthetic fixtures, English and Chinese sample speech, 0.8-second speech clips, and padded sub-window speech clips. Keep only generated or repository-owned audio fixtures. +- [ ] `bash scripts/sdlc-checks.sh` +- [ ] `bash scripts/ci-basic-checks.sh` +- [ ] `swift test` +- [ ] `bash scripts/build-app.sh` for a release-style app build with bundled Metal shaders. +- [ ] Real-window check of no-speech handling and intentional dictation, including local computer microphone and relevant output modes; repeat with remote microphone only if available. +- [ ] Verify the PR branch has no merge conflicts and distinguish local results from GitHub CI, signed release, and production observation. + +## Human gates + +- The user must approve this plan before implementation begins, as required by `docs/sdlc/README.md`. +- An independent verifier and PR reviewer must approve the high-risk privacy-sensitive change. A protected production release requires its separate human approval. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md new file mode 100644 index 00000000..3ccd1177 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md @@ -0,0 +1,49 @@ +# Spec: Prevent unsolicited Qwen vocabulary insertion + +**Status:** approved +**Approved-by:** User (conversation; revised design) +**Approved-date:** 2026-09-23 +**Upstream:** [intent.md](intent.md) + +The user approved the earlier design on 2026-09-23. A real Qwen counterexample invalidated it; the user then approved this revised design on 2026-09-23. + +## Context + +The user identified the local Qwen recognition model and computer microphone. Real-time recognition and overlay behavior remain unknown. The reported output starts with terms present in the user's personal dictionary and continues with the bundled technology lexicon in its source order. At recording start, `VoicePipeline` combines those terms into `SpeechRecognitionContext`. `QwenNativeASREngine` passes the resulting `Terms: ...` string into `Qwen3ASRModel.generate(context:)`. The recording gate uses RMS loudness, which does not distinguish speech from other sound. The final transcript then passes through `TranscriptionSanitizer.prepare` and may reach direct or processed insertion, including instant insert. + +The user clarified that no words were spoken and the list may have been added by the later formatting AI. That path is plausible: `TextProcessor.systemPromptWithPersonalContext` also includes the personal dictionary and active industry lexicon in the formatting prompt. The incident has no captured per-stage transcript and output, so its generating stage is unknown. The output-mode setting and whether the list appeared in the overlay before formatting are also unknown. The no-speech gate blocks both model paths before they can produce insertable text; the existing `TranscriptFidelityGuard` additionally compares formatting output against the ASR transcript in processed mode. + +The existing sanitizer only rejects the isolated phrase “Do anything” on weak audio. A temporary command that compiled the real sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduces the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. + +After the first implementation, a local Qwen replay with generated non-speech noise passed the RMS gate and returned “嗯。”; `TranscriptionSanitizer.prepare` accepted it. The first design therefore cannot satisfy the no-speech criterion. A separate read-only prototype of Apple's built-in Sound Analysis classifier gave maximum `speech` confidence 0.357 for that two-second noise clip, 0.91–0.97 for the repository's English and Chinese speech samples, and 0.91–0.94 for 0.8-second speech clips using 0.5-second analysis windows. A 0.25-second speech crop padded to 0.75 seconds scored 0.63–0.65. These are feasibility results, not a calibrated acceptance threshold. + +## Design + +1. **Require independent speech evidence before ASR output can be committed.** Use the macOS built-in Sound Analysis classifier on the recorded audio file, with a short supported analysis window. Reject when no speech window reaches a threshold calibrated on generated noise, repository speech samples, short utterances, and real microphone captures. For recordings shorter than the classifier window, analyze a temporary padded copy so short spoken words are not automatically rejected. A classifier error or unavailable file fails closed with no insertion and a clear status; numeric confidence may be logged only in debug builds. +2. **Remove free-text vocabulary priming from Qwen ASR.** Qwen receives an empty recognition context. Keep personal-dictionary and industry-lexicon corrections after ASR, so existing text cleanup still applies. Other recognizers retain their existing vocabulary mechanisms unless a replay shows the same failure there. +3. **Add a shared prompt-echo rejection at transcript preparation.** Pass the session's effective vocabulary snapshot into the common preparation boundary used by menu-bar voice input and integrations. Reject a transcript when it substantially reproduces a long ordered run of that supplied vocabulary and audio evidence is weak. Do not reject a single matching term or a short ordinary sentence. The policy operates before spoken-edit resolution, formatting, clipboard writes, history, and instant insert. +4. **Keep the RMS gate as a cheap first filter.** Its readings remain useful for absolute silence but cannot alone prove speech. Do not simply raise its global threshold: that would discard quiet speech without reliably excluding ambient noise. + +The module boundary is one recorded-audio speech decision, the ASR context supplied to Qwen, and the shared transcript-preparation decision. The recorded audio is the source of truth for speech; the vocabulary list is a recognition hint, never content to insert. The final transcript decision is the transaction boundary before externally visible output. No second insertion pipeline or model-specific output blacklist is introduced. + +Compared with repeatedly adding phrases to `weakAudioWholeTranscriptHallucinations`, this design addresses the source and the common gate. Replacing the whole speech module or app would multiply model, permission, migration, and validation work without evidence of broader architectural failure. + +## Safety and failure modes + +- **False rejection:** A user may deliberately read a long vocabulary list aloud. Reject only when weak audio evidence accompanies a strong ordered echo; test clear speech and short technical phrases. The Qwen context removal may reduce recognition of rare names; post-ASR replacements continue, and this tradeoff must be checked with speech fixtures. +- **False acceptance:** Sound classification is probabilistic. An unrelated Qwen hallucination or humanlike background sound might still pass. Calibrate and test with multiple noise and speech fixtures, plus the actual microphone. Do not claim universal prevention from the current small prototype. +- **Classifier failure:** A missing or unreadable file, no results, cancellation, or analysis error must not permit an ASR transcript to reach output. Preserve a recoverable user-facing status and do not retain an extra audio copy. +- **Privacy:** Do not persist captured audio or user dictionary contents in fixtures or diagnostics. A rejected transcript must cause no input insertion, clipboard write, edit command, or history entry. +- **Compatibility:** Keep the public `SpeechEngine` behavior and non-Qwen recognition paths stable unless the shared gate requires a narrow signature change. Preserve remote microphone and integration behavior with contract tests. + +## Test strategy + +- First write a failing regression test for the reported ordered vocabulary echo at the shared transcript-preparation seam, including a simulated weak recording that passes the RMS gate. Assert no downstream insertion or clipboard call through the pipeline's test seam. +- Turn the generated-noise Qwen replay that returned “嗯。” into a green regression by gating on independent speech evidence, not by blacklisting that valid utterance. Test English, Chinese, 0.8-second clips, and padded sub-window clips; measure false acceptance and rejection at the chosen threshold. +- Test single technical terms, “Do anything” with clear speech, ordinary mixed-language sentences, and deliberate long-list dictation with strong audio evidence. +- Replay synthetic silence and several non-speech noises through the installed local Qwen model with the technology lexicon enabled; compare the current and proposed context behavior without storing private audio. +- Exercise direct, processed, command, translation, instant-insert, streaming, and integration paths at their shared preparation boundary. Run `bash scripts/sdlc-checks.sh`, `bash scripts/ci-basic-checks.sh`, `swift test`, a release-style build, and independent verification. Document any runtime path that cannot be exercised. + +## Rollout and rollback + +Ship behind the existing signed-release and PR approvals. Observe no-speech rejections and legitimate technical dictation with privacy-safe counts only. Stop rollout if clear speech is lost or unsolicited text still inserts. Roll back the release to the previous signed artifact; restore the prior Qwen context behavior only after a new reviewed design addresses prompt echo. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md new file mode 100644 index 00000000..334eb6b2 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -0,0 +1,45 @@ +# Verification: Prevent unsolicited Qwen vocabulary insertion + +**Status:** approved +**Approved-by:** User (conversation; confirmation of updated verification) +**Approved-date:** 2026-09-24 +**Upstream:** [plan.md](plan.md) + +## Evidence + +| Check | Result | Evidence | +|---|---|---| +| Original sanitizer counterexample | Red before change | Weak-audio rehearsal passed the user-reported 351-character vocabulary list unchanged. | +| Qwen synthetic-noise counterexample | Red before speech gate | Installed Qwen model returned an insertable “嗯。” from generated non-speech noise. | +| `TranscriptionSanitizerTests` | Pass | Ordered vocabulary echo rejected; short terms and strong deliberate dictation retained. | +| `VoicePipelineSilentInsertionTests` | Pass | Direct, processed, instant insert, command, and translation paths made zero insertion calls for rejected audio. | +| `IntegrationOutputTests/testCoordinatorRejectsWeakAudioVocabularyEcho` | Pass | Live integration's shared transcript preparation rejected the list. | +| `SpeechActivityClassifierTests` | Pass | Generated white noise and tones at three amplitudes rejected; repository English/Chinese speech, 0.8-second clips, padded 0.25-second clips, and 10%-volume short clips accepted; missing, empty, and cancelled audio rejected. | +| Local Qwen noise replay | Pass | `OPENTYPE_QWEN_NATIVE_INTEGRATION=1` test with installed model rejected generated non-speech noise after independent speech classification. | +| Local Qwen repository speech samples | Pass | Installed model transcribed English and Chinese repository samples without downloading weights. | +| Later formatting-AI vocabulary echo | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms when the ASR source did not contain those terms, under both faithful-correction and bounded-custom-transformation policies. This is a deterministic guard test, not a replay of the original incident. | +| `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass | 754 XCTest cases, 15 skipped, no failures; one Swift Testing case passed before the later formatting-AI regression test was added. That focused new test passed separately. Scratch path used because a fresh MLX submodule checkout stalled. | +| `bash scripts/sdlc-checks.sh` | Pass | Stage artifacts valid after the user confirmed updated verification. | +| `bash scripts/ci-basic-checks.sh` | Pass | SDLC, localization, resources, lexicon evaluation, and repository checks passed. | +| `bash scripts/build-app.sh --app-only --sign=-` | Pass | Xcode Release build, Metal shader bundle, CLI helper, bundle assembly, ad-hoc codesign, and artifact verification passed. No usable Utter signing identity was installed; this is a local build, not a trusted release. | +| Real microphone/window check | Not run | Two other Utter instances are active on this Mac. Launching a third copy would conflict with hotkeys and would not provide reliable input-path evidence. | +| Independent verification | Pending | — | + +## Acceptance criteria + +- The reported vocabulary echo is rejected before insertion, clipboard, edit-command, and history paths — automated menu-bar and integration tests pass; independent review pending. +- Generated non-speech audio is rejected by an independent classifier even when Qwen emits text — local model replay passes; real microphone check pending. +- Clear English/Chinese speech and short clips remain accepted — automated file fixtures pass, including quieter short clips; real microphone check pending. + +## Residual risk + +- The built-in classifier cannot identify who spoke. Nearby human speech may still be transcribed; this is outside the no-speech and non-speech-noise acceptance criterion. +- The incident's generating stage is unknown. Both Qwen ASR and the later formatting AI previously received the vocabulary list, and no per-stage transcript/output was retained for this incident. The formatting-output guard is exercised by a focused test, but command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. +- The 0.6 speech-confidence threshold is calibrated against the listed fixtures, not a diverse microphone corpus. Real microphone validation and independent review remain required. +- Two other Utter instances prevent a trustworthy real-window test of this local build without interrupting the user's running apps. +- `origin/main` advanced after this worktree branched and changed Qwen files. Resolve those overlapping changes and confirm the future PR is conflict-free before opening it. +- Full release signing, GitHub CI, and production behavior have not been verified. + +## Decision + +The user confirmed the updated verification on 2026-09-24 with the listed evidence gaps visible. Independent high-risk review remains pending. Do not release until a conflict-free PR, signed build, and protected production approval are complete. From 45399638e075b08c190d044b10322554e1d2745c Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 00:36:43 +0800 Subject: [PATCH 02/10] docs: record post-rebase silent input verification --- .../changes/2026-09-23-silent-input-insertion/verification.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 334eb6b2..65fc0b91 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -22,6 +22,8 @@ | `bash scripts/sdlc-checks.sh` | Pass | Stage artifacts valid after the user confirmed updated verification. | | `bash scripts/ci-basic-checks.sh` | Pass | SDLC, localization, resources, lexicon evaluation, and repository checks passed. | | `bash scripts/build-app.sh --app-only --sign=-` | Pass | Xcode Release build, Metal shader bundle, CLI helper, bundle assembly, ad-hoc codesign, and artifact verification passed. No usable Utter signing identity was installed; this is a local build, not a trusted release. | +| Rebase onto `origin/main` | Pass | Commit `686becb` is based on `60e7ed4` (Confucius4-R2T2 support). Qwen's new `modelID` and tail-padding path remain intact; vocabulary context injection remains removed. | +| Post-rebase checks | Pass | `bash scripts/ci-basic-checks.sh`; `swift test --scratch-path /tmp/utter-silent-insertion-build` (763 XCTest cases, 17 skipped, no failures; one Swift Testing case passed); release-style ad-hoc app build and artifact verification. | | Real microphone/window check | Not run | Two other Utter instances are active on this Mac. Launching a third copy would conflict with hotkeys and would not provide reliable input-path evidence. | | Independent verification | Pending | — | @@ -37,7 +39,7 @@ - The incident's generating stage is unknown. Both Qwen ASR and the later formatting AI previously received the vocabulary list, and no per-stage transcript/output was retained for this incident. The formatting-output guard is exercised by a focused test, but command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. - The 0.6 speech-confidence threshold is calibrated against the listed fixtures, not a diverse microphone corpus. Real microphone validation and independent review remain required. - Two other Utter instances prevent a trustworthy real-window test of this local build without interrupting the user's running apps. -- `origin/main` advanced after this worktree branched and changed Qwen files. Resolve those overlapping changes and confirm the future PR is conflict-free before opening it. +- The branch includes `origin/main` at `60e7ed4`; compare against the live PR base again before review or merge, since main can advance. - Full release signing, GitHub CI, and production behavior have not been verified. ## Decision From c6096c41f8db754b481d96f729c606f88ca0cab0 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 01:18:47 +0800 Subject: [PATCH 03/10] test: document Qwen vocabulary echo root cause --- .../OpenTypeTests/QwenNativeASREngineTests.swift | 1 + .../TranscriptFidelityGuardTests.swift | 15 +++++++++++++++ .../2026-09-23-silent-input-insertion/spec.md | 4 +++- .../verification.md | 7 +++++-- 4 files changed, 24 insertions(+), 3 deletions(-) diff --git a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift index 75f84f55..8817cd07 100644 --- a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift +++ b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift @@ -116,6 +116,7 @@ final class QwenNativeASREngineTests: XCTestCase { activity.record(rms: 0.002, frameCount: Int(buffer.frameLength)) XCTAssertTrue(activity.hasMeaningfulAudio) let raw = try await engine.transcribe(audioURL: audioURL, language: "zh") + XCTAssertFalse(raw.hasPrefix("Terms: "), "Qwen must not echo a vocabulary prompt") let hasSpeech = await SpeechActivityClassifier.containsSpeech(at: audioURL) XCTAssertFalse(hasSpeech) let prepared = hasSpeech ? TranscriptionSanitizer.prepare( diff --git a/Tests/OpenTypeTests/TranscriptFidelityGuardTests.swift b/Tests/OpenTypeTests/TranscriptFidelityGuardTests.swift index d69eed75..b1b56986 100644 --- a/Tests/OpenTypeTests/TranscriptFidelityGuardTests.swift +++ b/Tests/OpenTypeTests/TranscriptFidelityGuardTests.swift @@ -112,6 +112,21 @@ final class TranscriptFidelityGuardTests: XCTestCase { } } + func testRejectsDoAnythingExpandedFromShortASROutput() { + for enforceSemanticFidelity in [true, false] { + XCTAssertEqual( + TranscriptFidelityGuard.violation( + source: "嗯。", + candidate: "Do anything。", + protectedTerms: ["Do anything"], + inputLanguage: .auto, + enforceSemanticFidelity: enforceSemanticFidelity + ), + "dictionary_term_change" + ) + } + } + func testRejectsTokenReorderingAndNegationScopeMovement() { XCTAssertEqual( violation("Alice 2, Bob 3", "Alice 3, Bob 2", language: .english), diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md index 3ccd1177..91024783 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md @@ -11,7 +11,9 @@ The user approved the earlier design on 2026-09-23. A real Qwen counterexample i The user identified the local Qwen recognition model and computer microphone. Real-time recognition and overlay behavior remain unknown. The reported output starts with terms present in the user's personal dictionary and continues with the bundled technology lexicon in its source order. At recording start, `VoicePipeline` combines those terms into `SpeechRecognitionContext`. `QwenNativeASREngine` passes the resulting `Terms: ...` string into `Qwen3ASRModel.generate(context:)`. The recording gate uses RMS loudness, which does not distinguish speech from other sound. The final transcript then passes through `TranscriptionSanitizer.prepare` and may reach direct or processed insertion, including instant insert. -The user clarified that no words were spoken and the list may have been added by the later formatting AI. That path is plausible: `TextProcessor.systemPromptWithPersonalContext` also includes the personal dictionary and active industry lexicon in the formatting prompt. The incident has no captured per-stage transcript and output, so its generating stage is unknown. The output-mode setting and whether the list appeared in the overlay before formatting are also unknown. The no-speech gate blocks both model paths before they can produce insertable text; the existing `TranscriptFidelityGuard` additionally compares formatting output against the ASR transcript in processed mode. +The user clarified that no words were spoken and questioned whether the later formatting AI added the list. Local `input_history.json` retains both stages. Four processed menu-bar records (2026-09-21 and 2026-09-23 UTC) show the long list already present in `rawText`, beginning with the literal `Terms: ` prefix; `processedText` removes that prefix and changes only a few terms and punctuation. The old `SpeechRecognitionContext.contextualPrompt()` generated precisely a `Terms: ` vocabulary prompt, which `QwenNativeASREngine` passed to `Qwen3ASRModel.generate(context:)`. This strongly identifies Qwen's echo of its recognition context as the long-list source. The original audio and exact runtime context snapshot were not retained, so the triggering sound and exact prompt bytes cannot be replayed. The overlay state is also unknown. + +Three older processed records show a separate failure: a two-character ASR result was expanded by the formatting stage into a personal-dictionary phrase absent from the ASR text. `TextProcessor.systemPromptWithPersonalContext` includes dictionary and industry vocabulary, so the later AI can also leak prompt content. The existing `TranscriptFidelityGuard` rejects that expansion with current protected terms; the new recorded-audio gate blocks no-speech sessions before either model can commit output. The existing sanitizer only rejects the isolated phrase “Do anything” on weak audio. A temporary command that compiled the real sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduces the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 65fc0b91..3fbd4c42 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -11,13 +11,16 @@ |---|---|---| | Original sanitizer counterexample | Red before change | Weak-audio rehearsal passed the user-reported 351-character vocabulary list unchanged. | | Qwen synthetic-noise counterexample | Red before speech gate | Installed Qwen model returned an insertable “嗯。” from generated non-speech noise. | +| Incident-stage provenance | Read-only local history | Four processed menu-bar records on 2026-09-21/23 UTC had 357-character `rawText` beginning `Terms: ` and 351-character `processedText` without that prefix. The literal prefix and ordered vocabulary match the legacy Qwen context construction. No private dictionary contents or audio were copied into the repository. | +| Separate formatting prompt leak | Read-only local history | Three older records had two-character ASR output expanded to “Do anything” by formatting; this is distinct from the long-list incident. | | `TranscriptionSanitizerTests` | Pass | Ordered vocabulary echo rejected; short terms and strong deliberate dictation retained. | | `VoicePipelineSilentInsertionTests` | Pass | Direct, processed, instant insert, command, and translation paths made zero insertion calls for rejected audio. | | `IntegrationOutputTests/testCoordinatorRejectsWeakAudioVocabularyEcho` | Pass | Live integration's shared transcript preparation rejected the list. | | `SpeechActivityClassifierTests` | Pass | Generated white noise and tones at three amplitudes rejected; repository English/Chinese speech, 0.8-second clips, padded 0.25-second clips, and 10%-volume short clips accepted; missing, empty, and cancelled audio rejected. | | Local Qwen noise replay | Pass | `OPENTYPE_QWEN_NATIVE_INTEGRATION=1` test with installed model rejected generated non-speech noise after independent speech classification. | +| Qwen prompt-echo regression | Pass | The same installed-model noise replay now asserts that ASR output does not begin with `Terms: ` even when a technology recognition context was configured on the engine. | | Local Qwen repository speech samples | Pass | Installed model transcribed English and Chinese repository samples without downloading weights. | -| Later formatting-AI vocabulary echo | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms when the ASR source did not contain those terms, under both faithful-correction and bounded-custom-transformation policies. This is a deterministic guard test, not a replay of the original incident. | +| Later formatting-AI vocabulary echo | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms and the short-ASR “Do anything” expansion, under both faithful-correction and bounded-custom-transformation policies. These are deterministic guard tests, not model replays. | | `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass | 754 XCTest cases, 15 skipped, no failures; one Swift Testing case passed before the later formatting-AI regression test was added. That focused new test passed separately. Scratch path used because a fresh MLX submodule checkout stalled. | | `bash scripts/sdlc-checks.sh` | Pass | Stage artifacts valid after the user confirmed updated verification. | | `bash scripts/ci-basic-checks.sh` | Pass | SDLC, localization, resources, lexicon evaluation, and repository checks passed. | @@ -36,7 +39,7 @@ ## Residual risk - The built-in classifier cannot identify who spoke. Nearby human speech may still be transcribed; this is outside the no-speech and non-speech-noise acceptance criterion. -- The incident's generating stage is unknown. Both Qwen ASR and the later formatting AI previously received the vocabulary list, and no per-stage transcript/output was retained for this incident. The formatting-output guard is exercised by a focused test, but command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. +- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. Separate historical records show formatting AI expansion of a short ASR result. Command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. - The 0.6 speech-confidence threshold is calibrated against the listed fixtures, not a diverse microphone corpus. Real microphone validation and independent review remain required. - Two other Utter instances prevent a trustworthy real-window test of this local build without interrupting the user's running apps. - The branch includes `origin/main` at `60e7ed4`; compare against the live PR base again before review or merge, since main can advance. From 994d9d7ea210f0739f19443207b745918da222c9 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 01:25:52 +0800 Subject: [PATCH 04/10] docs: propose preserving Qwen dictionary recognition --- .../dictionary-preservation-proposal.md | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md new file mode 100644 index 00000000..2a8483e4 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md @@ -0,0 +1,25 @@ +# Proposed revision: keep Qwen dictionary recognition + +This proposal is not yet an approved change to `spec.md` or `plan.md`. The current draft PR removes Qwen's recognition-time vocabulary bias. The user identified this as an unacceptable dictionary regression. + +## Observed contract + +- `QwenNativeASREngine` currently inherits the protocol's no-op `configureRecognition`, then passes `context: ""` to the model. Qwen cannot use the personal or industry vocabulary to recognize an unfamiliar spoken term. +- Exact personal `original -> replacement` rules and industry correction variants still run after ASR in direct and processed output. Processed formatting also receives dictionary context. These paths cannot recover a term when the ASR output contains no supported source phrase. +- The prior Qwen context used one free-text `Terms: ` list. Local history shows that list in the ASR `rawText` before formatting, confirming the echo failure. +- Live integration recordings use the new speech classifier, but imported audio files currently reach ASR without that gate or audio-activity evidence. + +## Revised behavior + +1. Require speech classification before ASR in every Qwen input route, including imported integration audio. Missing, invalid, cancelled, or non-speech audio ends without a transcript, insertion, clipboard write, or history entry. +2. Restore the existing bounded Qwen vocabulary context from the session snapshot, preserving personal-term priority and active industry terms. Use one authoritative context snapshot for both model input and echo checks. +3. Reject a Qwen transcript that starts with the context marker `Terms: ` and reproduces a substantial ordered run of supplied terms, regardless of RMS strength. Keep the existing weak-audio ordered-list check for echoes without the marker. Do not reject an ordinary spoken term or a deliberately spoken list lacking the marker. +4. Retain deterministic post-ASR replacements and formatting fidelity checks. The vocabulary is a recognition hint and correction source, never independently insertable content. + +## Verification before updating the draft PR + +- Red/green tests for imported noise audio, a loud synthetic prompt echo, deliberate strong-audio vocabulary dictation, and every output route's side effects. +- Run the installed local Qwen model on generated/repository-owned speech and noise with the restored context. Compare technical-term recognition against the current context-free branch; report both gains and false insertions. +- Run all repository checks and a release-style build; obtain real computer-microphone and independent high-risk verification before approval to merge. + +The current context-free implementation has lower prompt-echo exposure but loses Qwen recognition bias. A two-pass recognizer or a speech-module rewrite adds latency, model calls, migration work, and more failure modes; the narrow shared gate and echo contract address the observed failure while retaining the existing dictionary feature. If testing shows prompt echo without `Terms: ` under speech-like noise, revise the design again rather than broadening the blacklist blindly. From 4aee498afa4b1470d0a5bcbfc33df7b14d0e0582 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 01:58:41 +0800 Subject: [PATCH 05/10] docs: withdraw unsafe Qwen vocabulary restoration proposal --- .../dictionary-preservation-proposal.md | 25 ------------------- 1 file changed, 25 deletions(-) delete mode 100644 docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md deleted file mode 100644 index 2a8483e4..00000000 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/dictionary-preservation-proposal.md +++ /dev/null @@ -1,25 +0,0 @@ -# Proposed revision: keep Qwen dictionary recognition - -This proposal is not yet an approved change to `spec.md` or `plan.md`. The current draft PR removes Qwen's recognition-time vocabulary bias. The user identified this as an unacceptable dictionary regression. - -## Observed contract - -- `QwenNativeASREngine` currently inherits the protocol's no-op `configureRecognition`, then passes `context: ""` to the model. Qwen cannot use the personal or industry vocabulary to recognize an unfamiliar spoken term. -- Exact personal `original -> replacement` rules and industry correction variants still run after ASR in direct and processed output. Processed formatting also receives dictionary context. These paths cannot recover a term when the ASR output contains no supported source phrase. -- The prior Qwen context used one free-text `Terms: ` list. Local history shows that list in the ASR `rawText` before formatting, confirming the echo failure. -- Live integration recordings use the new speech classifier, but imported audio files currently reach ASR without that gate or audio-activity evidence. - -## Revised behavior - -1. Require speech classification before ASR in every Qwen input route, including imported integration audio. Missing, invalid, cancelled, or non-speech audio ends without a transcript, insertion, clipboard write, or history entry. -2. Restore the existing bounded Qwen vocabulary context from the session snapshot, preserving personal-term priority and active industry terms. Use one authoritative context snapshot for both model input and echo checks. -3. Reject a Qwen transcript that starts with the context marker `Terms: ` and reproduces a substantial ordered run of supplied terms, regardless of RMS strength. Keep the existing weak-audio ordered-list check for echoes without the marker. Do not reject an ordinary spoken term or a deliberately spoken list lacking the marker. -4. Retain deterministic post-ASR replacements and formatting fidelity checks. The vocabulary is a recognition hint and correction source, never independently insertable content. - -## Verification before updating the draft PR - -- Red/green tests for imported noise audio, a loud synthetic prompt echo, deliberate strong-audio vocabulary dictation, and every output route's side effects. -- Run the installed local Qwen model on generated/repository-owned speech and noise with the restored context. Compare technical-term recognition against the current context-free branch; report both gains and false insertions. -- Run all repository checks and a release-style build; obtain real computer-microphone and independent high-risk verification before approval to merge. - -The current context-free implementation has lower prompt-echo exposure but loses Qwen recognition bias. A two-pass recognizer or a speech-module rewrite adds latency, model calls, migration work, and more failure modes; the narrow shared gate and echo contract address the observed failure while retaining the existing dictionary feature. If testing shows prompt echo without `Terms: ` under speech-like noise, revise the design again rather than broadening the blacklist blindly. From de2d36799d6431afa9b3b9bfa862c3e3711b8251 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 02:02:32 +0800 Subject: [PATCH 06/10] docs: correct learned-dictionary root cause --- docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md | 2 +- .../2026-09-23-silent-input-insertion/verification.md | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md index 91024783..41f217ea 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md @@ -13,7 +13,7 @@ The user identified the local Qwen recognition model and computer microphone. Re The user clarified that no words were spoken and questioned whether the later formatting AI added the list. Local `input_history.json` retains both stages. Four processed menu-bar records (2026-09-21 and 2026-09-23 UTC) show the long list already present in `rawText`, beginning with the literal `Terms: ` prefix; `processedText` removes that prefix and changes only a few terms and punctuation. The old `SpeechRecognitionContext.contextualPrompt()` generated precisely a `Terms: ` vocabulary prompt, which `QwenNativeASREngine` passed to `Qwen3ASRModel.generate(context:)`. This strongly identifies Qwen's echo of its recognition context as the long-list source. The original audio and exact runtime context snapshot were not retained, so the triggering sound and exact prompt bytes cannot be replayed. The overlay state is also unknown. -Three older processed records show a separate failure: a two-character ASR result was expanded by the formatting stage into a personal-dictionary phrase absent from the ASR text. `TextProcessor.systemPromptWithPersonalContext` includes dictionary and industry vocabulary, so the later AI can also leak prompt content. The existing `TranscriptFidelityGuard` rejects that expansion with current protected terms; the new recorded-audio gate blocks no-speech sessions before either model can commit output. +Three older processed records show a separate failure: the two-character ASR result “嗯。” became “Do anything”. Read-only inspection of the local personal dictionary found an active learned `嗯。 -> Do anything` rule with confidence 0.82 and two evidence records. `PersonalDictionarySnapshot.applyReplacements` runs before formatting and deterministically makes this substitution. The previous attribution to the formatting AI was wrong. The rule was created by correction capture and activated after two observations; the original edits were pruned from the 500-record history, so their intent is unknown. A no-speech gate blocks a non-speech source, but the learned-rule eligibility and scope also need review before the dictionary is considered safe. The existing sanitizer only rejects the isolated phrase “Do anything” on weak audio. A temporary command that compiled the real sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduces the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 3fbd4c42..961e588f 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -12,7 +12,7 @@ | Original sanitizer counterexample | Red before change | Weak-audio rehearsal passed the user-reported 351-character vocabulary list unchanged. | | Qwen synthetic-noise counterexample | Red before speech gate | Installed Qwen model returned an insertable “嗯。” from generated non-speech noise. | | Incident-stage provenance | Read-only local history | Four processed menu-bar records on 2026-09-21/23 UTC had 357-character `rawText` beginning `Terms: ` and 351-character `processedText` without that prefix. The literal prefix and ordered vocabulary match the legacy Qwen context construction. No private dictionary contents or audio were copied into the repository. | -| Separate formatting prompt leak | Read-only local history | Three older records had two-character ASR output expanded to “Do anything” by formatting; this is distinct from the long-list incident. | +| Separate learned-dictionary substitution | Read-only local history and dictionary | Three older records had “嗯。” as ASR text and “Do anything” as processed text. An active learned `嗯。 -> Do anything` rule deterministically explains the change; it became active after two correction observations. The original edit records have been pruned, so user intent cannot be inferred. | | `TranscriptionSanitizerTests` | Pass | Ordered vocabulary echo rejected; short terms and strong deliberate dictation retained. | | `VoicePipelineSilentInsertionTests` | Pass | Direct, processed, instant insert, command, and translation paths made zero insertion calls for rejected audio. | | `IntegrationOutputTests/testCoordinatorRejectsWeakAudioVocabularyEcho` | Pass | Live integration's shared transcript preparation rejected the list. | @@ -20,7 +20,7 @@ | Local Qwen noise replay | Pass | `OPENTYPE_QWEN_NATIVE_INTEGRATION=1` test with installed model rejected generated non-speech noise after independent speech classification. | | Qwen prompt-echo regression | Pass | The same installed-model noise replay now asserts that ASR output does not begin with `Terms: ` even when a technology recognition context was configured on the engine. | | Local Qwen repository speech samples | Pass | Installed model transcribed English and Chinese repository samples without downloading weights. | -| Later formatting-AI vocabulary echo | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms and the short-ASR “Do anything” expansion, under both faithful-correction and bounded-custom-transformation policies. These are deterministic guard tests, not model replays. | +| Formatting-output guard | Pass | `TranscriptFidelityGuardTests` rejects a list copied from formatting prompt terms and an unsupported short-ASR expansion under both fidelity policies. The short historical output came from a learned replacement before formatting, so this guard test does not reproduce that incident. | | `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass | 754 XCTest cases, 15 skipped, no failures; one Swift Testing case passed before the later formatting-AI regression test was added. That focused new test passed separately. Scratch path used because a fresh MLX submodule checkout stalled. | | `bash scripts/sdlc-checks.sh` | Pass | Stage artifacts valid after the user confirmed updated verification. | | `bash scripts/ci-basic-checks.sh` | Pass | SDLC, localization, resources, lexicon evaluation, and repository checks passed. | @@ -39,7 +39,7 @@ ## Residual risk - The built-in classifier cannot identify who spoke. Nearby human speech may still be transcribed; this is outside the no-speech and non-speech-noise acceptance criterion. -- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. Separate historical records show formatting AI expansion of a short ASR result. Command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. +- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. Separate historical records show an active learned rule replacing a short filler utterance. The current branch does not yet prevent that rule from applying to genuine speech. Command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. - The 0.6 speech-confidence threshold is calibrated against the listed fixtures, not a diverse microphone corpus. Real microphone validation and independent review remain required. - Two other Utter instances prevent a trustworthy real-window test of this local build without interrupting the user's running apps. - The branch includes `origin/main` at `60e7ed4`; compare against the live PR base again before review or merge, since main can advance. From a79f13985030071393fc731b8cd09d17c4197bcc Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 02:20:14 +0800 Subject: [PATCH 07/10] docs: revise silent insertion design and reset verification gate --- .../intent.md | 4 +- .../2026-09-23-silent-input-insertion/plan.md | 44 +++++++--------- .../2026-09-23-silent-input-insertion/spec.md | 50 ++++++++----------- .../verification.md | 10 ++-- 4 files changed, 49 insertions(+), 59 deletions(-) diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md index 1dfebb29..2f6940ad 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/intent.md @@ -7,7 +7,7 @@ ## Problem -The user observed Utter inserting a long list of unrelated technical terms into the focused input field while the user was not speaking. The list included “Do anything” and terms that resemble the technology industry lexicon. This is an unsolicited output and can expose text to whichever application has focus. The specific microphone, recognition engine, and recording mode involved have not yet been identified. +The user observed Utter inserting a long list of unrelated technical terms into the focused input field while the user was not speaking. The list included “Do anything” and terms that resemble the technology industry lexicon. This is an unsolicited output and can expose text to whichever application has focus. The user identified the local Qwen recognition model and computer microphone; recording mode remains uncertain. ## Outcome @@ -36,5 +36,5 @@ An input session without user speech ends with no text insertion, clipboard writ ## Open questions -- Which microphone source, recognition engine, output mode, and trigger produced the observed list? Diagnosis can begin without these details, but they are needed to match the user's exact runtime path. +- Which output mode and trigger produced the observed list? The user identified the computer microphone and local Qwen model; real-time recognition remains unconfirmed. - Did the list appear in the Utter overlay before insertion, or only in the target application? diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md index 0860ac77..4a90595e 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md @@ -1,35 +1,29 @@ # Plan: Prevent unsolicited Qwen vocabulary insertion -**Status:** approved -**Approved-by:** User (conversation; revised plan) -**Approved-date:** 2026-09-23 +**Status:** pending approval +**Approved-by:** — +**Approved-date:** — **Upstream:** [spec.md](spec.md) -The user approved an earlier plan on 2026-09-23. Generated non-speech noise produced an insertable Qwen transcript, invalidating that plan. The revised spec has been approved; this updated plan requires its own approval before implementation resumes. +This plan supersedes the 2026-09-23 plan. Existing code and tests remain in the draft PR; the items below describe only the redesigned work. -## Work items +## Implementation sequence -- [x] Establish a red regression at transcript preparation: the original sanitizer passed the user's 351-character list on weak audio. Add a safe pipeline insertion spy and green cases for intended speech. -- [x] Replay generated non-speech noise through the installed Qwen model. It returned an insertable “嗯。” rather than the reported list; retain the deterministic vocabulary-echo regression and the new noise regression as separate contracts. -- [ ] Add one recorded-audio speech classifier using the macOS Sound Analysis built-in model. Run it after the RMS gate and before an ASR transcript can reach output in the menu-bar and live integration paths. Use a 0.5-second analysis window, calibrate the acceptance threshold on the listed fixtures, and pad very short recordings in a temporary copy so they receive a classification result. Treat missing files, no results, classifier errors, and cancellation as no speech; clean up temporary copies on every path. -- [x] Remove Qwen's free-text recognition-context injection in `Sources/Speech/QwenNativeASREngine.swift`; retain existing personal and industry replacement behavior after ASR. Delete obsolete context-only code and tests. -- [x] Add a narrow ordered-vocabulary-echo decision to `Sources/Processing/TranscriptionSanitizer.swift` and feed it the session vocabulary at both `VoicePipeline` and `InputSessionCoordinator`. Keep rejection before edit commands, instant insert, formatting, history, and clipboard effects; targeted tests pass. -- [ ] Add fault-injection tests for weak audio and fabricated ASR output; verify every reachable output mode rejects before insertion, clipboard, edit-command, or history side effects. Test classifier success, failure, no-window, and cancellation paths. Probe louder ambient sound and real microphone noise; if non-speech bypasses the gate, revise the design rather than claiming the broad no-speech criterion passed. Test clear intentionally spoken vocabulary, short phrases, and protected dictionary replacements. -- [ ] Review the complete diff for duplicate guards, stale Qwen context paths, privacy-sensitive logs, and changed contracts. Keep each touched Swift file under 300 lines or split at a real responsibility boundary. -- [ ] Prepare `verification.md` with command results, synthetic Qwen replay, affected permission/privacy paths, remaining limits, and rollback. Obtain independent high-risk verification and PR approval before release. +- [ ] First establish red contract tests at the real seams: all three recorded-audio entry points, Qwen context and echo handling, dictionary snapshot and learning. Represent the reported long-list and `嗯。 -> Do anything` failures without copying private data into fixtures. Assert zero insertion, clipboard, command, and history effects on rejection. +- [ ] Trace every snapshot consumer and source of target app/language. Make one effective snapshot carry the same scoped personal entries through ASR context, replacement, formatter hints, and protected terms. Preserve global manual terms and industry post-processing. Test unknown scope and evidence from different scopes. +- [ ] Restore a bounded Qwen term prompt with effective personal entries only. Keep the established ASR interface where possible. Retain the exact per-call context for echo comparison. Test priority, deduplication, budget, and empty context; compare rare-word recognition with the installed model. +- [ ] Route menu-bar, live integration, and imported audio through one recorded-audio speech decision and final transcript acceptance boundary. Reuse the existing classifier. Ensure provisional streaming text cannot become final output after rejection. +- [ ] On Qwen context echo, retry the same audio once without context. Reject a remaining echo or retry failure before any external effect. Delete stale duplicate guards and obsolete full-list context code after the common contract passes. +- [ ] Tighten correction-candidate and persisted learned-entry eligibility for short filler/non-speech sources. Prevent cross-scope evidence merges. Keep the user's local dictionary intact; test that the observed learned mapping is inert and a manual mapping still works. +- [ ] Add fault-injection and end-to-end tests for classifier failure/no window/cancellation, retry failure, each output mode, short intentional phrases, English/Chinese speech, scoped vocabulary, and imported audio. Review privacy logging and split touched Swift files at real responsibility boundaries to keep them below 300 lines. -## Verification plan +## Verification and delivery -- [ ] Focused red/green tests: `swift test --filter TranscriptionSanitizerTests`, Qwen engine tests, and pipeline/integration output tests. -- [ ] Real Qwen and Sound Analysis replay: generated silence, the existing two-second noise fixture that produced “嗯。”, several louder/noisier synthetic fixtures, English and Chinese sample speech, 0.8-second speech clips, and padded sub-window speech clips. Keep only generated or repository-owned audio fixtures. -- [ ] `bash scripts/sdlc-checks.sh` -- [ ] `bash scripts/ci-basic-checks.sh` -- [ ] `swift test` -- [ ] `bash scripts/build-app.sh` for a release-style app build with bundled Metal shaders. -- [ ] Real-window check of no-speech handling and intentional dictation, including local computer microphone and relevant output modes; repeat with remote microphone only if available. -- [ ] Verify the PR branch has no merge conflicts and distinguish local results from GitHub CI, signed release, and production observation. +- [ ] Run focused red/green tests, then `bash scripts/sdlc-checks.sh`, `bash scripts/ci-basic-checks.sh`, and `swift test`. +- [ ] Replay generated silence/noise and repository-owned speech through installed Qwen. Measure recognition with zero, one, and bounded relevant terms; test prompt echo and fallback. Do not save private microphone recordings. +- [ ] Build a release-style app with `bash scripts/build-app.sh`; perform real-window QA with the computer microphone once competing Utter instances can be avoided without disrupting the user's work. Record unavailable paths as unverified. +- [ ] Update `verification.md` with fresh evidence and residual risks. Check the PR against current `main` and resolve conflicts. Keep it draft pending independent verification, PR approval, and the separate signed-release/production decision. -## Human gates +## Human gate -- The user must approve this plan before implementation begins, as required by `docs/sdlc/README.md`. -- An independent verifier and PR reviewer must approve the high-risk privacy-sensitive change. A protected production release requires its separate human approval. +`docs/sdlc/README.md` requires explicit approval of this revised plan before implementation resumes. Earlier approval of the old plan does not approve this one. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md index 41f217ea..04f7ffd5 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/spec.md @@ -1,51 +1,45 @@ # Spec: Prevent unsolicited Qwen vocabulary insertion **Status:** approved -**Approved-by:** User (conversation; revised design) -**Approved-date:** 2026-09-23 +**Approved-by:** User (conversation; confirmed the three-part redesign) +**Approved-date:** 2026-09-24 **Upstream:** [intent.md](intent.md) -The user approved the earlier design on 2026-09-23. A real Qwen counterexample invalidated it; the user then approved this revised design on 2026-09-23. +The user rejected the earlier design after it disabled Qwen recognition hints and left a harmful learned rule active, then confirmed a three-part redesign on 2026-09-24. The earlier implementation and verification remain historical evidence, not acceptance of this revision. ## Context -The user identified the local Qwen recognition model and computer microphone. Real-time recognition and overlay behavior remain unknown. The reported output starts with terms present in the user's personal dictionary and continues with the bundled technology lexicon in its source order. At recording start, `VoicePipeline` combines those terms into `SpeechRecognitionContext`. `QwenNativeASREngine` passes the resulting `Terms: ...` string into `Qwen3ASRModel.generate(context:)`. The recording gate uses RMS loudness, which does not distinguish speech from other sound. The final transcript then passes through `TranscriptionSanitizer.prepare` and may reach direct or processed insertion, including instant insert. +The user identified the local Qwen recognition model and computer microphone. Real-time recognition and overlay behavior remain unknown. The reported output starts with terms present in the user's personal dictionary and continues with the bundled technology lexicon in its source order. In the incident-era implementation, `VoicePipeline` combined those terms into `SpeechRecognitionContext`, and `QwenNativeASREngine` passed the resulting `Terms: ...` string into `Qwen3ASRModel.generate(context:)`. The current draft PR has removed that context. RMS loudness does not distinguish speech from other sound. The final transcript passes through `TranscriptionSanitizer.prepare` and may reach direct or processed insertion, including instant insert. The user clarified that no words were spoken and questioned whether the later formatting AI added the list. Local `input_history.json` retains both stages. Four processed menu-bar records (2026-09-21 and 2026-09-23 UTC) show the long list already present in `rawText`, beginning with the literal `Terms: ` prefix; `processedText` removes that prefix and changes only a few terms and punctuation. The old `SpeechRecognitionContext.contextualPrompt()` generated precisely a `Terms: ` vocabulary prompt, which `QwenNativeASREngine` passed to `Qwen3ASRModel.generate(context:)`. This strongly identifies Qwen's echo of its recognition context as the long-list source. The original audio and exact runtime context snapshot were not retained, so the triggering sound and exact prompt bytes cannot be replayed. The overlay state is also unknown. Three older processed records show a separate failure: the two-character ASR result “嗯。” became “Do anything”. Read-only inspection of the local personal dictionary found an active learned `嗯。 -> Do anything` rule with confidence 0.82 and two evidence records. `PersonalDictionarySnapshot.applyReplacements` runs before formatting and deterministically makes this substitution. The previous attribution to the formatting AI was wrong. The rule was created by correction capture and activated after two observations; the original edits were pruned from the 500-record history, so their intent is unknown. A no-speech gate blocks a non-speech source, but the learned-rule eligibility and scope also need review before the dictionary is considered safe. -The existing sanitizer only rejects the isolated phrase “Do anything” on weak audio. A temporary command that compiled the real sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduces the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. +Before the first patch, the sanitizer only rejected the isolated phrase “Do anything” on weak audio. A temporary command that compiled that sanitizer with a minimal audio-activity stub and supplied the user's reported list returned `FAIL: passed through 351 characters` (exit 1). This reproduced the transcript-preparation gap, not the complete microphone-to-Qwen incident. No private microphone recording or dictionary contents were copied into the repository. After the first implementation, a local Qwen replay with generated non-speech noise passed the RMS gate and returned “嗯。”; `TranscriptionSanitizer.prepare` accepted it. The first design therefore cannot satisfy the no-speech criterion. A separate read-only prototype of Apple's built-in Sound Analysis classifier gave maximum `speech` confidence 0.357 for that two-second noise clip, 0.91–0.97 for the repository's English and Chinese speech samples, and 0.91–0.94 for 0.8-second speech clips using 0.5-second analysis windows. A 0.25-second speech crop padded to 0.75 seconds scored 0.63–0.65. These are feasibility results, not a calibrated acceptance threshold. -## Design +## Superseding design -1. **Require independent speech evidence before ASR output can be committed.** Use the macOS built-in Sound Analysis classifier on the recorded audio file, with a short supported analysis window. Reject when no speech window reaches a threshold calibrated on generated noise, repository speech samples, short utterances, and real microphone captures. For recordings shorter than the classifier window, analyze a temporary padded copy so short spoken words are not automatically rejected. A classifier error or unavailable file fails closed with no insertion and a clear status; numeric confidence may be logged only in debug builds. -2. **Remove free-text vocabulary priming from Qwen ASR.** Qwen receives an empty recognition context. Keep personal-dictionary and industry-lexicon corrections after ASR, so existing text cleanup still applies. Other recognizers retain their existing vocabulary mechanisms unless a replay shows the same failure there. -3. **Add a shared prompt-echo rejection at transcript preparation.** Pass the session's effective vocabulary snapshot into the common preparation boundary used by menu-bar voice input and integrations. Reject a transcript when it substantially reproduces a long ordered run of that supplied vocabulary and audio evidence is weak. Do not reject a single matching term or a short ordinary sentence. The policy operates before spoken-edit resolution, formatting, clipboard writes, history, and instant insert. -4. **Keep the RMS gate as a cheap first filter.** Its readings remain useful for absolute silence but cannot alone prove speech. Do not simply raise its global threshold: that would discard quiet speech without reliably excluding ambient noise. +The four saved sessions already contain the ordered `Terms: ...` list in Qwen's raw transcript. Formatting changed only its presentation. The original audio and exact runtime prompt are unavailable. Separate saved sessions show raw `嗯。` becoming `Do anything` because of an active learned dictionary rule, before formatting. Its source edits were pruned, so there is no evidence that it was an intended reusable correction. A synthetic rare-name sample was recognized correctly with a one-term Qwen hint and incorrectly without it; this is a narrow benefit to preserve, not a measured production success rate. -The module boundary is one recorded-audio speech decision, the ASR context supplied to Qwen, and the shared transcript-preparation decision. The recorded audio is the source of truth for speech; the vocabulary list is a recognition hint, never content to insert. The final transcript decision is the transaction boundary before externally visible output. No second insertion pipeline or model-specific output blacklist is introduced. +1. **One recorded-audio speech decision.** Menu-bar capture, live integration capture, and integration audio-file import must call the same classifier before any final transcript or output is committed. Keep RMS as a cheap early filter where available. Missing/unreadable audio, no classification window, cancellation, or classifier failure produces no output and a recoverable status. Streaming partials remain provisional. +2. **Bounded Qwen recognition hints.** Restore Qwen context using only effective personal-dictionary replacement terms for the current target app and selected language. Prefer manual entries, then recent/high-evidence learned entries; deduplicate and impose a small phrase-count and character budget. Do not include the full dictionary, bundled industry lexicon, edit rules, or whole correction pairs. Preserve post-ASR personal and industry correction. Other recognizers keep their established mechanism but receive effective scoped personal entries. +3. **Pre-output prompt-echo handling.** Compare Qwen's candidate with the exact context supplied to that recognition call. On a prompt-prefix or ordered-term echo, discard it and retry the same audio once with empty context. Revalidate the retry at the shared transcript boundary. If it still echoes, lacks speech evidence, or errors, finish without insertion, clipboard, command, or history. Short deliberately spoken terms and ordinary sentences remain eligible. +4. **Safe learned corrections.** Enforce learned-entry app/language scope at the dictionary snapshot used by recognition, replacement, formatter hints, and protected terms. A scoped learned entry is ineligible when the required context is unknown; unscoped manual entries stay global. Reject automatic learning from a short filler/interjection or non-speech artifact, including the observed `嗯。` source. Apply the same eligibility to persisted learned entries so the existing unsafe rule cannot act. Do not silently delete or rewrite the local dictionary. Do not merge evidence from different app/language scopes into a global rule. User-entered manual corrections remain possible. -Compared with repeatedly adding phrases to `weakAudioWholeTranscriptHallucinations`, this design addresses the source and the common gate. Replacing the whole speech module or app would multiply model, permission, migration, and validation work without evidence of broader architectural failure. +Recorded audio is the source for speech evidence, the session snapshot is the source for effective terms, and final transcript acceptance is the transaction boundary before external effects. Replace the narrow coordinator seam if it cannot carry exact context and scope; do not add three unrelated path guards. -## Safety and failure modes +## Failure modes and limits -- **False rejection:** A user may deliberately read a long vocabulary list aloud. Reject only when weak audio evidence accompanies a strong ordered echo; test clear speech and short technical phrases. The Qwen context removal may reduce recognition of rare names; post-ASR replacements continue, and this tradeoff must be checked with speech fixtures. -- **False acceptance:** Sound classification is probabilistic. An unrelated Qwen hallucination or humanlike background sound might still pass. Calibrate and test with multiple noise and speech fixtures, plus the actual microphone. Do not claim universal prevention from the current small prototype. -- **Classifier failure:** A missing or unreadable file, no results, cancellation, or analysis error must not permit an ASR transcript to reach output. Preserve a recoverable user-facing status and do not retain an extra audio copy. -- **Privacy:** Do not persist captured audio or user dictionary contents in fixtures or diagnostics. A rejected transcript must cause no input insertion, clipboard write, edit command, or history entry. -- **Compatibility:** Keep the public `SpeechEngine` behavior and non-Qwen recognition paths stable unless the shared gate requires a narrow signature change. Preserve remote microphone and integration behavior with contract tests. +- The classifier can reject quiet speech or accept humanlike ambient sound. Calibrate with short English/Chinese speech, quiet speech, silence, and several noise types; verify the computer microphone. Nearby human speech is outside this no-speech guarantee. +- A small Qwen prompt can still echo. One empty-context retry contains detected echoes but costs an extra inference. Rare names outside the term budget may be less accurate; measure the tradeoff. +- Scope filtering may leave a learned correction unavailable when app or language is unknown. Prefer a missed correction to an unintended replacement. Manual global entries remain available. +- Do not log audio, transcript contents, or dictionary terms. Preserve existing user data. Roll back to the prior signed artifact if necessary; do not restore the full Qwen term list without separate review. -## Test strategy +## Acceptance and verification -- First write a failing regression test for the reported ordered vocabulary echo at the shared transcript-preparation seam, including a simulated weak recording that passes the RMS gate. Assert no downstream insertion or clipboard call through the pipeline's test seam. -- Turn the generated-noise Qwen replay that returned “嗯。” into a green regression by gating on independent speech evidence, not by blacklisting that valid utterance. Test English, Chinese, 0.8-second clips, and padded sub-window clips; measure false acceptance and rejection at the chosen threshold. -- Test single technical terms, “Do anything” with clear speech, ordinary mixed-language sentences, and deliberate long-list dictation with strong audio evidence. -- Replay synthetic silence and several non-speech noises through the installed local Qwen model with the technology lexicon enabled; compare the current and proposed context behavior without storing private audio. -- Exercise direct, processed, command, translation, instant-insert, streaming, and integration paths at their shared preparation boundary. Run `bash scripts/sdlc-checks.sh`, `bash scripts/ci-basic-checks.sh`, `swift test`, a release-style build, and independent verification. Document any runtime path that cannot be exercised. - -## Rollout and rollback - -Ship behind the existing signed-release and PR approvals. Observe no-speech rejections and legitimate technical dictation with privacy-safe counts only. Stop rollout if clear speech is lost or unsolicited text still inserts. Roll back the release to the previous signed artifact; restore the prior Qwen context behavior only after a new reviewed design addresses prompt echo. +- A deterministic regression reaches all three recorded-audio paths with non-speech audio and fabricated vocabulary output and observes no final text, clipboard, command, or history side effect. +- Local Qwen replay shows a bounded relevant term helps a rare-word sample, and prompt echo triggers at most one empty-context retry that cannot insert the list. Deliberately spoken short terms and ordinary speech remain eligible. +- A regression proves `嗯。 -> Do anything` cannot arise from automatic learning or an existing learned rule, while a manual correction and valid scoped learned correction work in their proper app/language. +- Fault injection covers classifier errors, cancellation, missing audio, retry failure, and unknown scope. Run repository checks, full tests, release-style build, real-window microphone QA, independent high-risk review, conflict-free PR review, and separate protected release gate. diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 961e588f..707e782b 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -1,10 +1,12 @@ # Verification: Prevent unsolicited Qwen vocabulary insertion -**Status:** approved -**Approved-by:** User (conversation; confirmation of updated verification) -**Approved-date:** 2026-09-24 +**Status:** draft +**Approved-by:** — +**Approved-date:** — **Upstream:** [plan.md](plan.md) +The prior verification was approved for the superseded design. Results below describe the existing draft PR and do not establish acceptance of the revised design approved on 2026-09-24. Fresh verification is required after implementation. + ## Evidence | Check | Result | Evidence | @@ -47,4 +49,4 @@ ## Decision -The user confirmed the updated verification on 2026-09-24 with the listed evidence gaps visible. Independent high-risk review remains pending. Do not release until a conflict-free PR, signed build, and protected production approval are complete. +The user confirmed this historical verification on 2026-09-24, then rejected the design because Qwen vocabulary help was lost and the unsafe learned rule remained active. It no longer approves the change. Independent high-risk review and fresh verification remain pending. Do not release until a conflict-free PR, signed build, and protected production approval are complete. From 20923a1cd6eb4d359446c958c6c6cf3835d0a307 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 02:33:55 +0800 Subject: [PATCH 08/10] fix: restore bounded Qwen hints and scope learned corrections --- Sources/App/VoicePipeline+Dictionary.swift | 16 +++++ Sources/App/VoicePipeline+Processing.swift | 19 +++--- Sources/App/VoicePipeline+Replacement.swift | 2 +- Sources/App/VoicePipeline.swift | 13 ++-- .../InputSessionCoordinator+AudioFile.swift | 27 +++++--- .../InputSessionCoordinator+Dictionary.swift | 30 +++++++++ .../InputSessionCoordinator+Output.swift | 4 +- .../Integration/InputSessionCoordinator.swift | 43 ++++++++++--- Sources/Output/CorrectionCaptureService.swift | 3 +- .../CorrectionCandidateClassifier.swift | 1 + .../Processing/LearnedCorrectionPolicy.swift | 38 ++++++++++++ .../PersonalDictionary+Learning.swift | 21 ++++--- Sources/Processing/PersonalDictionary.swift | 41 ++++++++++--- Sources/Speech/QwenNativeASREngine.swift | 27 ++++++-- Sources/Speech/QwenRecognitionPrompt.swift | 61 +++++++++++++++++++ .../CorrectionCandidateClassifierTests.swift | 10 +++ .../IntegrationOutputTests.swift | 25 +++++++- .../PersonalDictionaryLearningTests.swift | 57 ++++++++++++++++- .../QwenNativeASREngineTests.swift | 26 ++++++++ .../QwenRecognitionContextTests.swift | 38 ++++++++++++ .../2026-09-23-silent-input-insertion/plan.md | 8 +-- 21 files changed, 444 insertions(+), 66 deletions(-) create mode 100644 Sources/App/VoicePipeline+Dictionary.swift create mode 100644 Sources/Integration/InputSessionCoordinator+Dictionary.swift create mode 100644 Sources/Processing/LearnedCorrectionPolicy.swift create mode 100644 Sources/Speech/QwenRecognitionPrompt.swift create mode 100644 Tests/OpenTypeTests/QwenRecognitionContextTests.swift diff --git a/Sources/App/VoicePipeline+Dictionary.swift b/Sources/App/VoicePipeline+Dictionary.swift new file mode 100644 index 00000000..4bec2e49 --- /dev/null +++ b/Sources/App/VoicePipeline+Dictionary.swift @@ -0,0 +1,16 @@ +import AppKit + +@MainActor +extension VoicePipeline { + func dictionarySnapshot( + settings: AppSettings, + targetApp: NSRunningApplication? + ) -> PersonalDictionarySnapshot { + if let sessionDictionarySnapshot { return sessionDictionarySnapshot } + return PersonalDictionary.shared.snapshot( + settings: settings, + bundleIdentifier: targetApp?.bundleIdentifier, + languageCode: settings.inputLanguage.whisperCode + ) + } +} diff --git a/Sources/App/VoicePipeline+Processing.swift b/Sources/App/VoicePipeline+Processing.swift index fa16fda9..ffbc8494 100644 --- a/Sources/App/VoicePipeline+Processing.swift +++ b/Sources/App/VoicePipeline+Processing.swift @@ -11,7 +11,10 @@ extension VoicePipeline { inputMode: VoiceInputMode, targetApp: NSRunningApplication? ) async { - defer { audioCapture.cleanupLastRecording() } + defer { + audioCapture.cleanupLastRecording() + sessionDictionarySnapshot = nil + } AudioCaptureDiagnostics.log(audioActivity) do { @@ -28,7 +31,8 @@ extension VoicePipeline { audioURL: audioURL, audioActivity: audioActivity, language: language, - settings: settings + settings: settings, + targetApp: targetApp ) guard !Task.isCancelled else { @@ -103,7 +107,8 @@ extension VoicePipeline { audioURL: URL?, audioActivity: AudioCaptureActivity, language: String?, - settings: AppSettings + settings: AppSettings, + targetApp: NSRunningApplication? ) async throws -> String { let started = CFAbsoluteTimeGetCurrent() let raw: String @@ -115,7 +120,7 @@ extension VoicePipeline { let elapsed = CFAbsoluteTimeGetCurrent() - started Log.info("[VoicePipeline] ASR stage finished in \(String(format: "%.2f", elapsed))s") - let recognitionPhrases = PersonalDictionary.shared.snapshot(settings: settings).recognitionPhrases + let recognitionPhrases = dictionarySnapshot(settings: settings, targetApp: targetApp).recognitionPhrases guard let prepared = TranscriptionSanitizer.prepare( raw, audioActivity: audioActivity, @@ -151,7 +156,7 @@ extension VoicePipeline { return await processCommand(raw, settings: settings, targetApp: targetApp) case .direct: cancelScreenContextCapture() - let dictionarySnapshot = PersonalDictionary.shared.snapshot(settings: settings) + let dictionarySnapshot = dictionarySnapshot(settings: settings, targetApp: targetApp) let context = InputContext.capture( targetApp: targetApp, screenContext: "", @@ -180,7 +185,7 @@ extension VoicePipeline { let started = CFAbsoluteTimeGetCurrent() let processingOptions = TextProcessingOptions(settings: settings) - let dictionarySnapshot = PersonalDictionary.shared.snapshot(settings: settings) + let dictionarySnapshot = dictionarySnapshot(settings: settings, targetApp: targetApp) let enableMemory = settings.enableMemory let memoryWindowMinutes = settings.memoryWindowMinutes let screenContext = await finishScreenContextCapture() @@ -225,7 +230,7 @@ extension VoicePipeline { let started = CFAbsoluteTimeGetCurrent() let processingOptions = TextProcessingOptions(settings: settings) - let dictionarySnapshot = PersonalDictionary.shared.snapshot(settings: settings) + let dictionarySnapshot = dictionarySnapshot(settings: settings, targetApp: targetApp) let enableMemory = settings.enableMemory let memoryWindowMinutes = settings.memoryWindowMinutes let screenContext = await finishScreenContextCapture() diff --git a/Sources/App/VoicePipeline+Replacement.swift b/Sources/App/VoicePipeline+Replacement.swift index 30790894..aca3ea13 100644 --- a/Sources/App/VoicePipeline+Replacement.swift +++ b/Sources/App/VoicePipeline+Replacement.swift @@ -39,7 +39,7 @@ extension VoicePipeline { targetApp: NSRunningApplication? ) async { let processingOptions = TextProcessingOptions(settings: settings) - let dictionarySnapshot = PersonalDictionary.shared.snapshot(settings: settings) + let dictionarySnapshot = dictionarySnapshot(settings: settings, targetApp: targetApp) let enableMemory = settings.enableMemory let memoryWindowMinutes = settings.memoryWindowMinutes let quickText = immediateInsertText( diff --git a/Sources/App/VoicePipeline.swift b/Sources/App/VoicePipeline.swift index 106bde86..dc22615c 100644 --- a/Sources/App/VoicePipeline.swift +++ b/Sources/App/VoicePipeline.swift @@ -26,6 +26,7 @@ final class VoicePipeline { var hideOverlayTask: Task? var formattingModelLifecycleTask: Task? var recordingTargetApp: NSRunningApplication? + var sessionDictionarySnapshot: PersonalDictionarySnapshot? var formattingPreloadGeneration = 0 /// Injectable engine used by tests to drive the real `start` await through a /// controlled model-load barrier. `nil` in production. @@ -135,11 +136,12 @@ final class VoicePipeline { let micID = appState.settings.microphoneID let language = appState.settings.inputLanguage.whisperCode let streamingEnabled = appState.settings.enableStreamingRecognitionBeta - let vocabularySnapshot = PersonalDictionary.shared.snapshot( - settings: appState.settings - ) + let vocabularySnapshot = dictionarySnapshot(settings: appState.settings, targetApp: targetApp) + sessionDictionarySnapshot = vocabularySnapshot currentEngine?.configureRecognition( - context: SpeechRecognitionContext(phrases: vocabularySnapshot.recognitionPhrases) + context: SpeechRecognitionContext(phrases: currentEngine is QwenNativeASREngine + ? vocabularySnapshot.personalRecognitionPhrases + : vocabularySnapshot.recognitionPhrases) ) if streamingEnabled { currentEngine?.startListening(language: language) { [weak self] partialText in @@ -160,6 +162,7 @@ final class VoicePipeline { currentEngine?.cancelListening() cancelScreenContextCapture() recordingTargetApp = nil + sessionDictionarySnapshot = nil appState.phase = .error(L("pipeline.mic_failed_permissions")) appState.statusMessage = L("pipeline.mic_unavailable") overlay.hide() @@ -187,6 +190,7 @@ final class VoicePipeline { currentEngine?.cancelListening() cancelScreenContextCapture() recordingTargetApp = nil + sessionDictionarySnapshot = nil appState.phase = .error(L("pipeline.mic_failed_permissions")) appState.statusMessage = L("pipeline.mic_unavailable") overlay.hide() @@ -251,6 +255,7 @@ final class VoicePipeline { audioCapture.stop() audioCapture.cleanupLastRecording() recordingTargetApp = nil + sessionDictionarySnapshot = nil soundPlayer.playStop() appState.reset() overlay.hide() diff --git a/Sources/Integration/InputSessionCoordinator+AudioFile.swift b/Sources/Integration/InputSessionCoordinator+AudioFile.swift index f5c76b3c..6d997343 100644 --- a/Sources/Integration/InputSessionCoordinator+AudioFile.swift +++ b/Sources/Integration/InputSessionCoordinator+AudioFile.swift @@ -20,21 +20,24 @@ extension InputSessionCoordinator { throw IntegrationError.sessionNotFound } let effective = effectiveSettings(for: session.request) - guard let engine = await engineProvider.engine(settings: settings), engine.isReady else { + guard let engine = await selectedEngine(), engine.isReady else { throw IntegrationError.modelNotReady } - let vocabularySnapshot = PersonalDictionary.shared.snapshot(settings: settings) - engine.configureRecognition( - context: SpeechRecognitionContext(phrases: vocabularySnapshot.recognitionPhrases) - ) + let vocabularySnapshot = dictionarySnapshot(clientID: clientID, languageCode: effective.languageCode) + engine.configureRecognition(context: recognitionContext(engine: engine, snapshot: vocabularySnapshot)) do { try service.emitAudioReceived(sessionID: sessionID, clientID: clientID) try await service.beginProcessing(sessionID: sessionID, clientID: clientID) + guard await recordingContainsSpeech(audioURL) else { + throw IntegrationError.noSpeechDetected + } let transcript = try await transcribeAudioURL( audioURL, engine: engine, - languageCode: effective.languageCode + languageCode: effective.languageCode, + clientID: clientID, + dictionarySnapshot: vocabularySnapshot ) try service.emitTranscriptFinal(sessionID: sessionID, clientID: clientID, text: transcript) let active = ActiveSession( @@ -50,7 +53,8 @@ extension InputSessionCoordinator { mode: effective.mode, useScreenContext: effective.useScreenContext ), - client: service.integrationClient(id: clientID) + client: service.integrationClient(id: clientID), + dictionarySnapshot: vocabularySnapshot ) let text = try await outputText(for: transcript, active: active) try await service.completeSession(sessionID: sessionID, clientID: clientID, finalText: text) @@ -71,9 +75,14 @@ extension InputSessionCoordinator { private func transcribeAudioURL( _ audioURL: URL, engine: any SpeechEngine, - languageCode: String? + languageCode: String?, + clientID: String, + dictionarySnapshot: PersonalDictionarySnapshot ) async throws -> String { let raw = try await engine.transcribe(audioURL: audioURL, language: languageCode) - return try prepareTranscript(raw, audioActivity: nil) + return try prepareTranscript( + raw, audioActivity: nil, clientID: clientID, + languageCode: languageCode, dictionarySnapshot: dictionarySnapshot + ) } } diff --git a/Sources/Integration/InputSessionCoordinator+Dictionary.swift b/Sources/Integration/InputSessionCoordinator+Dictionary.swift new file mode 100644 index 00000000..70e31fa4 --- /dev/null +++ b/Sources/Integration/InputSessionCoordinator+Dictionary.swift @@ -0,0 +1,30 @@ +import Foundation + +@MainActor +extension InputSessionCoordinator { + func dictionarySnapshot(clientID: String, languageCode: String?) -> PersonalDictionarySnapshot { + PersonalDictionary.shared.snapshot( + settings: settings, + bundleIdentifier: service.integrationClient(id: clientID)?.bundleIdentifier, + languageCode: languageCode + ) + } + + func recognitionContext( + engine: any SpeechEngine, + snapshot: PersonalDictionarySnapshot + ) -> SpeechRecognitionContext { + return SpeechRecognitionContext(phrases: engine is QwenNativeASREngine + ? snapshot.personalRecognitionPhrases + : snapshot.recognitionPhrases) + } + + func recordingContainsSpeech(_ audioURL: URL?) async -> Bool { + #if DEBUG + if let speechActivityOverrideForTesting { + return await speechActivityOverrideForTesting(audioURL) + } + #endif + return await SpeechActivityClassifier.containsSpeech(at: audioURL) + } +} diff --git a/Sources/Integration/InputSessionCoordinator+Output.swift b/Sources/Integration/InputSessionCoordinator+Output.swift index bb183543..2acdfdce 100644 --- a/Sources/Integration/InputSessionCoordinator+Output.swift +++ b/Sources/Integration/InputSessionCoordinator+Output.swift @@ -10,7 +10,9 @@ extension InputSessionCoordinator { private func trackedOutputText(for raw: String, active: ActiveSession) async throws -> String { let options = TextProcessingOptions(settings: settings, inputLanguage: active.inputLanguage) - let dictionarySnapshot = PersonalDictionary.shared.snapshot(settings: settings) + let dictionarySnapshot = active.dictionarySnapshot ?? dictionarySnapshot( + clientID: active.clientID, languageCode: active.languageCode + ) let enableMemory = settings.enableMemory let memoryWindowMinutes = settings.memoryWindowMinutes let text: String diff --git a/Sources/Integration/InputSessionCoordinator.swift b/Sources/Integration/InputSessionCoordinator.swift index 74fb1627..7491ec7b 100644 --- a/Sources/Integration/InputSessionCoordinator.swift +++ b/Sources/Integration/InputSessionCoordinator.swift @@ -14,6 +14,7 @@ final class InputSessionCoordinator { let streamingEnabled: Bool let screenContextTask: Task? let client: IntegrationClient? + var dictionarySnapshot: PersonalDictionarySnapshot? = nil } let service: OpenTypeService @@ -23,6 +24,10 @@ final class InputSessionCoordinator { let settings: AppSettings let isUserWorkflowBusy: @MainActor () -> Bool var activeSession: ActiveSession? + #if DEBUG + var speechActivityOverrideForTesting: ((URL?) async -> Bool)? + var engineOverrideForTesting: (any SpeechEngine)? + #endif var isBusy: Bool { activeSession != nil } @@ -50,13 +55,11 @@ final class InputSessionCoordinator { throw IntegrationError.sessionNotFound } let effective = effectiveSettings(for: session.request) - guard let engine = await engineProvider.engine(settings: settings), engine.isReady else { + guard let engine = await selectedEngine(), engine.isReady else { throw IntegrationError.modelNotReady } - let vocabularySnapshot = PersonalDictionary.shared.snapshot(settings: settings) - engine.configureRecognition( - context: SpeechRecognitionContext(phrases: vocabularySnapshot.recognitionPhrases) - ) + let vocabularySnapshot = dictionarySnapshot(clientID: clientID, languageCode: effective.languageCode) + engine.configureRecognition(context: recognitionContext(engine: engine, snapshot: vocabularySnapshot)) if effective.streamingEnabled, engine.supportsStreaming { engine.startListening(language: effective.languageCode) { [weak service] partialText in @@ -106,7 +109,8 @@ final class InputSessionCoordinator { mode: effective.mode, useScreenContext: effective.useScreenContext ), - client: service.integrationClient(id: clientID) + client: service.integrationClient(id: clientID), + dictionarySnapshot: vocabularySnapshot ) } catch { audioCapture.stop() @@ -161,7 +165,7 @@ final class InputSessionCoordinator { guard audioCapture.lastActivity.hasMeaningfulAudio else { throw IntegrationError.noSpeechDetected } - guard await SpeechActivityClassifier.containsSpeech(at: audioCapture.lastRecordingURL) else { + guard await recordingContainsSpeech(audioCapture.lastRecordingURL) else { throw IntegrationError.noSpeechDetected } @@ -178,7 +182,11 @@ final class InputSessionCoordinator { ) } - let transcript = try prepareTranscript(raw, audioActivity: audioCapture.lastActivity) + let transcript = try prepareTranscript( + raw, audioActivity: audioCapture.lastActivity, + clientID: active.clientID, languageCode: active.languageCode, + dictionarySnapshot: active.dictionarySnapshot + ) try service.emitTranscriptFinal( sessionID: active.sessionID, @@ -190,8 +198,16 @@ final class InputSessionCoordinator { return (transcript, text) } - func prepareTranscript(_ raw: String, audioActivity: AudioCaptureActivity?) throws -> String { - let recognitionPhrases = PersonalDictionary.shared.snapshot(settings: settings).recognitionPhrases + func prepareTranscript( + _ raw: String, + audioActivity: AudioCaptureActivity?, + clientID: String? = nil, + languageCode: String? = nil, + dictionarySnapshot: PersonalDictionarySnapshot? = nil + ) throws -> String { + let recognitionPhrases = (dictionarySnapshot ?? self.dictionarySnapshot( + clientID: clientID ?? "", languageCode: languageCode + )).recognitionPhrases guard let transcript = TranscriptionSanitizer.prepare( raw, audioActivity: audioActivity, @@ -207,6 +223,13 @@ final class InputSessionCoordinator { try? await service.failSession(sessionID: active.sessionID, clientID: active.clientID, error: error) } + func selectedEngine() async -> (any SpeechEngine)? { + #if DEBUG + if let engineOverrideForTesting { return engineOverrideForTesting } + #endif + return await engineProvider.engine(settings: settings) + } + private func release(_ active: ActiveSession) { active.engine.cancelListening() active.screenContextTask?.cancel() diff --git a/Sources/Output/CorrectionCaptureService.swift b/Sources/Output/CorrectionCaptureService.swift index ac9e6fdf..da987e55 100644 --- a/Sources/Output/CorrectionCaptureService.swift +++ b/Sources/Output/CorrectionCaptureService.swift @@ -54,8 +54,7 @@ final class CorrectionCaptureService { inserted: session.seed.insertedText, userFinal: session.latestFinalText, sourceRecordID: session.recordID, - languageCode: session.seed.context.inputLanguage.whisperCode - ?? session.seed.context.inputLanguage.rawValue, + languageCode: session.seed.context.inputLanguage.whisperCode, bundleIdentifier: session.seed.context.bundleIdentifier ) { PersonalDictionary.shared.recordLearnedCandidate(candidate) diff --git a/Sources/Processing/CorrectionCandidateClassifier.swift b/Sources/Processing/CorrectionCandidateClassifier.swift index 8e4edbf0..b78e6c3d 100644 --- a/Sources/Processing/CorrectionCandidateClassifier.swift +++ b/Sources/Processing/CorrectionCandidateClassifier.swift @@ -69,6 +69,7 @@ enum CorrectionCandidateClassifier { languageCode: String?, bundleIdentifier: String? ) -> LearnedCorrectionCandidate? { + guard !LearnedCorrectionPolicy.isUnsafeSource(inserted) else { return nil } guard let diff = CorrectionEditDiff.between(inserted, userFinal) else { return nil } var original = diff.beforeSegment.trimmingCharacters(in: .whitespacesAndNewlines) var replacement = diff.afterSegment.trimmingCharacters(in: .whitespacesAndNewlines) diff --git a/Sources/Processing/LearnedCorrectionPolicy.swift b/Sources/Processing/LearnedCorrectionPolicy.swift new file mode 100644 index 00000000..51e4b901 --- /dev/null +++ b/Sources/Processing/LearnedCorrectionPolicy.swift @@ -0,0 +1,38 @@ +import Foundation + +enum LearnedCorrectionPolicy { + private static let isolatedFillers: Set = [ + "嗯", "呃", "额", "唔", "啊", "哦", "um", "uh", "hmm", "hm" + ] + + static func isUnsafeSource(_ text: String) -> Bool { + let lexical = String(text.lowercased().filter { $0.isLetter || $0.isNumber }) + return isolatedFillers.contains(lexical) + || TranscriptionSanitizer.isNonSpeechArtifact(text) + } +} + +extension DictionaryEntry { + func applies(bundleIdentifier: String?, languageCode: String?) -> Bool { + guard isEffective else { return false } + if origin == .learned { + guard !LearnedCorrectionPolicy.isUnsafeSource(original), + !appScopes.isEmpty || (self.languageCode != nil + && self.languageCode != InputLanguage.auto.rawValue) else { return false } + } + if !appScopes.isEmpty { + guard let bundleIdentifier, + appScopes.contains(where: { $0.caseInsensitiveCompare(bundleIdentifier) == .orderedSame }) else { + return false + } + } + if let expected = self.languageCode, expected != InputLanguage.auto.rawValue { + guard let languageCode, + expected.split(separator: "-").first?.lowercased() + == languageCode.split(separator: "-").first?.lowercased() else { + return false + } + } + return true + } +} diff --git a/Sources/Processing/PersonalDictionary+Learning.swift b/Sources/Processing/PersonalDictionary+Learning.swift index abc7d3ff..5eb0f348 100644 --- a/Sources/Processing/PersonalDictionary+Learning.swift +++ b/Sources/Processing/PersonalDictionary+Learning.swift @@ -8,6 +8,8 @@ extension PersonalDictionary { @discardableResult func recordLearnedCandidate(_ candidate: LearnedCorrectionCandidate) -> UUID? { + guard !LearnedCorrectionPolicy.isUnsafeSource(candidate.original), + candidate.languageCode != nil || candidate.bundleIdentifier != nil else { return nil } removePreviousEvidence(for: candidate) if entries.contains(where: { @@ -23,19 +25,23 @@ extension PersonalDictionary { $0.origin == .learned && $0.original.caseInsensitiveCompare(candidate.original) == .orderedSame && $0.replacement.caseInsensitiveCompare(candidate.replacement) != .orderedSame + && $0.languageCode == candidate.languageCode + && $0.appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) } if let index = entries.firstIndex(where: { $0.origin == .learned && $0.original.caseInsensitiveCompare(candidate.original) == .orderedSame && $0.replacement.caseInsensitiveCompare(candidate.replacement) == .orderedSame + && $0.languageCode == candidate.languageCode + && $0.appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) }) { merge(candidate, intoEntryAt: index, now: now) - if hasConflict { markLearnedMappingsPending(for: candidate.original) } + if hasConflict { markLearnedMappingsPending(for: candidate) } save() return entries[index].id } - if hasConflict { markLearnedMappingsPending(for: candidate.original) } + if hasConflict { markLearnedMappingsPending(for: candidate) } let entry = DictionaryEntry( original: candidate.original, @@ -66,11 +72,6 @@ private extension PersonalDictionary { ) entries[index].confidence = max(entries[index].confidence, candidate.confidence) entries[index].lastSeenAt = now - entries[index].languageCode = candidate.languageCode ?? entries[index].languageCode - if let bundleIdentifier = candidate.bundleIdentifier, - !entries[index].appScopes.contains(bundleIdentifier) { - entries[index].appScopes.append(bundleIdentifier) - } if entries[index].confidence >= 0.92 || entries[index].evidenceCount >= 2 { entries[index].status = .active } @@ -92,9 +93,11 @@ private extension PersonalDictionary { } } - func markLearnedMappingsPending(for original: String) { + func markLearnedMappingsPending(for candidate: LearnedCorrectionCandidate) { for index in entries.indices where entries[index].origin == .learned - && entries[index].original.caseInsensitiveCompare(original) == .orderedSame { + && entries[index].original.caseInsensitiveCompare(candidate.original) == .orderedSame + && entries[index].languageCode == candidate.languageCode + && entries[index].appScopes == (candidate.bundleIdentifier.map { [$0] } ?? []) { entries[index].status = .pending } } diff --git a/Sources/Processing/PersonalDictionary.swift b/Sources/Processing/PersonalDictionary.swift index 2261612e..b692e08a 100644 --- a/Sources/Processing/PersonalDictionary.swift +++ b/Sources/Processing/PersonalDictionary.swift @@ -14,9 +14,13 @@ struct PersonalDictionarySnapshot: Sendable { init( entries: [DictionaryEntry], editRules: [EditRule], - industryLexicon: IndustryLexiconSnapshot = .empty + industryLexicon: IndustryLexiconSnapshot = .empty, + bundleIdentifier: String? = nil, + languageCode: String? = nil ) { - self.entries = entries + self.entries = entries.filter { + $0.applies(bundleIdentifier: bundleIdentifier, languageCode: languageCode) + } self.editRules = editRules self.industryLexicon = industryLexicon } @@ -73,10 +77,13 @@ struct PersonalDictionarySnapshot: Sendable { industryLexicon.promptDescription } + var personalRecognitionPhrases: [String] { + SpeechRecognitionContext(dictionaryEntries: entries).phrases + } + var recognitionPhrases: [String] { - let personal = SpeechRecognitionContext(dictionaryEntries: entries).phrases return SpeechRecognitionContext( - phrases: personal + industryLexicon.recognitionPhrases + phrases: personalRecognitionPhrases + industryLexicon.recognitionPhrases ).phrases } @@ -133,19 +140,31 @@ final class PersonalDictionary: ObservableObject { snapshot().activeRulesDescription } - func snapshot(industryLexicon: IndustryLexiconSnapshot = .empty) -> PersonalDictionarySnapshot { + func snapshot( + industryLexicon: IndustryLexiconSnapshot = .empty, + bundleIdentifier: String? = nil, + languageCode: String? = nil + ) -> PersonalDictionarySnapshot { PersonalDictionarySnapshot( entries: entries, editRules: editRules, - industryLexicon: industryLexicon + industryLexicon: industryLexicon, + bundleIdentifier: bundleIdentifier, + languageCode: languageCode ) } - func snapshot(settings: AppSettings) -> PersonalDictionarySnapshot { + func snapshot( + settings: AppSettings, + bundleIdentifier: String? = nil, + languageCode: String? = nil + ) -> PersonalDictionarySnapshot { snapshot( industryLexicon: IndustryLexiconCatalog.shared.snapshot( for: settings.industryLexicon - ) + ), + bundleIdentifier: bundleIdentifier, + languageCode: languageCode ) } @@ -165,6 +184,8 @@ final class PersonalDictionary: ObservableObject { entries[index].status = .active entries[index].confidence = 1 entries[index].evidenceCount = max(1, entries[index].evidenceCount) + entries[index].languageCode = nil + entries[index].appScopes = [] save() return entries[index].id } @@ -206,7 +227,9 @@ final class PersonalDictionary: ObservableObject { let original = entries[index].original for otherIndex in entries.indices where otherIndex != index && entries[otherIndex].origin == .learned - && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame { + && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame + && entries[otherIndex].languageCode == entries[index].languageCode + && entries[otherIndex].appScopes == entries[index].appScopes { entries[otherIndex].status = .pending } entries[index].status = .active diff --git a/Sources/Speech/QwenNativeASREngine.swift b/Sources/Speech/QwenNativeASREngine.swift index f0b06a0f..870feb06 100644 --- a/Sources/Speech/QwenNativeASREngine.swift +++ b/Sources/Speech/QwenNativeASREngine.swift @@ -7,6 +7,8 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { private let modelDirectory: URL private let tailPaddingFrames: AVAudioFrameCount private let runtime = QwenNativeASRRuntime() + private let contextLock = NSLock() + private var recognitionContext = SpeechRecognitionContext.empty init(modelPath: String, modelID: String = QwenASRModel.defaultID) { modelDirectory = URL(fileURLWithPath: modelPath).standardizedFileURL @@ -21,6 +23,12 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { modelDirectory == URL(fileURLWithPath: modelPath).standardizedFileURL } + func configureRecognition(context: SpeechRecognitionContext) { + contextLock.lock() + recognitionContext = context + contextLock.unlock() + } + func prepare() async { guard isReady else { return } do { @@ -34,18 +42,25 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { guard isReady else { throw QwenNativeASRError.notConfigured } guard let audioURL else { throw QwenNativeASRError.noAudioFile } + contextLock.lock() + let prompt = QwenRecognitionPrompt(phrases: recognitionContext.phrases) + contextLock.unlock() + let started = CFAbsoluteTimeGetCurrent() let result = try await QwenAudioPreprocessor.withPreparedAudio( from: audioURL, tailPaddingFrames: tailPaddingFrames ) { preparedURL in - try await runtime.transcribe( - audioURL: preparedURL, - modelDirectory: modelDirectory, - language: language, - context: "" - ) + try await QwenContextRecovery.run(prompt: prompt) { context in + try await runtime.transcribe( + audioURL: preparedURL, + modelDirectory: modelDirectory, + language: language, + context: context + ) + } text: { $0.text } } + guard let result else { return "" } let elapsed = CFAbsoluteTimeGetCurrent() - started Log.info( "[Qwen3ASRNative] transcribed \(result.text.count) chars in " diff --git a/Sources/Speech/QwenRecognitionPrompt.swift b/Sources/Speech/QwenRecognitionPrompt.swift new file mode 100644 index 00000000..788b7080 --- /dev/null +++ b/Sources/Speech/QwenRecognitionPrompt.swift @@ -0,0 +1,61 @@ +import Foundation + +struct QwenRecognitionPrompt: Sendable { + static let maximumPhrases = 8 + static let maximumTermsLength = 160 + + let phrases: [String] + let text: String + + init(phrases: [String]) { + var accepted: [String] = [] + var seen = Set() + for raw in phrases { + let phrase = raw.trimmingCharacters(in: .whitespacesAndNewlines) + guard !phrase.isEmpty, phrase.count <= 80, + seen.insert(phrase.lowercased()).inserted else { continue } + let proposed = (accepted + [phrase]).joined(separator: ", ") + guard accepted.count < Self.maximumPhrases else { break } + guard proposed.count <= Self.maximumTermsLength else { continue } + accepted.append(phrase) + } + self.phrases = accepted + text = accepted.isEmpty ? "" : "Vocabulary: \(accepted.joined(separator: ", "))." + } +} + +enum QwenPromptEcho { + static func matches(_ transcript: String, prompt: QwenRecognitionPrompt) -> Bool { + guard !prompt.text.isEmpty else { return false } + let text = transcript.trimmingCharacters(in: .whitespacesAndNewlines) + let lowered = text.lowercased() + if lowered.hasPrefix("vocabulary:") || lowered.hasPrefix("terms:") { return true } + + let trim = CharacterSet.whitespacesAndNewlines.union(.punctuationCharacters) + let segments = text.components(separatedBy: CharacterSet(charactersIn: ",,、;;\n")) + .map { $0.trimmingCharacters(in: trim).lowercased() } + .filter { !$0.isEmpty } + guard segments.count >= 4 else { return false } + let expected = prompt.phrases.map { $0.trimmingCharacters(in: trim).lowercased() } + guard expected.count >= 4 else { return false } + for start in 0...(expected.count - 4) { + if Array(segments.prefix(4)) == Array(expected[start..<(start + 4)]) { + return true + } + } + return false + } +} + +enum QwenContextRecovery { + static func run( + prompt: QwenRecognitionPrompt, + recognize: (String) async throws -> Result, + text: (Result) -> String + ) async throws -> Result? { + let first = try await recognize(prompt.text) + guard QwenPromptEcho.matches(text(first), prompt: prompt) else { return first } + let retry = try await recognize("") + return QwenPromptEcho.matches(text(retry), prompt: prompt) ? nil : retry + } +} diff --git a/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift b/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift index 040a6423..c8620742 100644 --- a/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift +++ b/Tests/OpenTypeTests/CorrectionCandidateClassifierTests.swift @@ -2,6 +2,16 @@ import XCTest @testable import OpenType final class CorrectionCandidateClassifierTests: XCTestCase { + func testDoesNotLearnStandaloneFillerAsReusableCorrection() { + XCTAssertNil(CorrectionCandidateClassifier.candidate( + inserted: "嗯。", + userFinal: "Do anything", + sourceRecordID: UUID(), + languageCode: "zh", + bundleIdentifier: "com.apple.Notes" + )) + } + func testLearnsCaseAndSpacingCorrectionAsHighConfidenceTerm() throws { let candidate = try XCTUnwrap(CorrectionCandidateClassifier.candidate( inserted: "Please use open type today.", diff --git a/Tests/OpenTypeTests/IntegrationOutputTests.swift b/Tests/OpenTypeTests/IntegrationOutputTests.swift index 4d772464..24bb1541 100644 --- a/Tests/OpenTypeTests/IntegrationOutputTests.swift +++ b/Tests/OpenTypeTests/IntegrationOutputTests.swift @@ -5,6 +5,27 @@ import XCTest @MainActor final class IntegrationOutputTests: XCTestCase { + func testImportedAudioRejectsNoSpeechBeforeTranscriptionOrFinalEvent() async throws { + let store = registry() + defer { store.cleanup() } + store.registry.approve(IntegrationClient.localHTTP(tokenID: "token")) + let service = makeService(registry: store.registry) + let session = try await service.createSession(request(mode: .direct), clientID: clientID) + let coordinator = InputSessionCoordinator(service: service) + let engine = TestSpeechEngine(transcript: "Vocabulary: Alpha, Beta, Gamma, Delta") + coordinator.engineOverrideForTesting = engine + coordinator.speechActivityOverrideForTesting = { _ in false } + + await assertThrowsIntegrationError(.noSpeechDetected) { + _ = try await coordinator.processAudioFile( + sessionID: session.id, clientID: clientID, + audioURL: URL(fileURLWithPath: "/tmp/utter-missing-audio.wav"), cleanup: false + ) + } + XCTAssertEqual(engine.transcribeCount, 0) + XCTAssertEqual(try service.session(session.id, clientID: clientID)?.state, .failed) + } + func testCoordinatorRejectsWeakAudioVocabularyEcho() { let store = registry() defer { store.cleanup() } @@ -149,6 +170,7 @@ private extension IntegrationOutputTests { private final class TestSpeechEngine: SpeechEngine, @unchecked Sendable { let transcript: String + private(set) var transcribeCount = 0 var isReady: Bool { true } init(transcript: String) { @@ -156,7 +178,8 @@ private final class TestSpeechEngine: SpeechEngine, @unchecked Sendable { } func transcribe(audioURL: URL?, language: String?) async throws -> String { - transcript + transcribeCount += 1 + return transcript } } diff --git a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift index 766639d6..56790841 100644 --- a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift +++ b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift @@ -3,6 +3,55 @@ import XCTest @testable import OpenType final class PersonalDictionaryLearningTests: XCTestCase { + func testExistingLearnedFillerIsInertButManualRuleStillWorks() { + let bad = DictionaryEntry( + original: "嗯。", replacement: "Do anything", origin: .learned, + languageCode: "zh", appScopes: ["com.apple.Notes"] + ) + let snapshot = PersonalDictionarySnapshot(entries: [bad], editRules: []) + XCTAssertEqual(snapshot.applyReplacements(to: "嗯。"), "嗯。") + XCTAssertFalse(snapshot.recognitionPhrases.contains("Do anything")) + + let manual = PersonalDictionarySnapshot(entries: [ + DictionaryEntry(original: "嗯。", replacement: "Do anything") + ], editRules: []) + XCTAssertEqual(manual.applyReplacements(to: "嗯。"), "Do anything") + } + + func testLearnedScopeDoesNotLeakAcrossAppsOrLanguages() { + let entry = DictionaryEntry( + original: "open type", replacement: "OpenType", origin: .learned, + languageCode: "en", appScopes: ["com.apple.Notes"] + ) + let store = makeStore() + store.entries = [entry] + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "en") + .applyReplacements(to: "open type"), "OpenType") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.TextEdit", languageCode: "en") + .applyReplacements(to: "open type"), "open type") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "open type"), "open type") + XCTAssertEqual(store.snapshot().applyReplacements(to: "open type"), "open type") + } + + func testIndependentAppsDoNotMergeLearnedEvidence() throws { + let store = makeStore() + let first = LearnedCorrectionCandidate( + original: "open tape", replacement: "OpenType", confidence: 0.82, + sourceRecordID: UUID(), languageCode: "en", bundleIdentifier: "com.apple.Notes" + ) + let second = LearnedCorrectionCandidate( + original: "open tape", replacement: "OpenType", confidence: 0.82, + sourceRecordID: UUID(), languageCode: "en", bundleIdentifier: "com.apple.TextEdit" + ) + store.recordLearnedCandidate(first) + store.recordLearnedCandidate(second) + XCTAssertEqual(store.entries.count, 2) + XCTAssertEqual(store.entries.map(\.status), [.pending, .pending]) + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "en") + .applyReplacements(to: "open tape"), "open tape") + } + func testLegacyEntryDecodesAsActiveManualTerm() throws { let data = Data(#"{"original":"open type","replacement":"OpenType","enabled":true}"#.utf8) let entry = try JSONDecoder().decode(DictionaryEntry.self, from: data) @@ -27,12 +76,14 @@ final class PersonalDictionaryLearningTests: XCTestCase { let entryID = try XCTUnwrap(store.recordLearnedCandidate(first)) XCTAssertEqual(store.entries.first(where: { $0.id == entryID })?.status, .pending) - XCTAssertEqual(store.applyReplacements(to: "菜单蓝"), "菜单蓝") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "菜单蓝"), "菜单蓝") XCTAssertTrue(SpeechRecognitionContext(dictionaryEntries: store.entries).phrases.isEmpty) store.recordLearnedCandidate(second) XCTAssertEqual(store.entries.first(where: { $0.id == entryID })?.status, .active) - XCTAssertEqual(store.applyReplacements(to: "菜单蓝"), "菜单栏") + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.Notes", languageCode: "zh") + .applyReplacements(to: "菜单蓝"), "菜单栏") } func testHighConfidenceLearnedTermActivatesOnceAndManualEntryWins() throws { @@ -76,7 +127,7 @@ final class PersonalDictionaryLearningTests: XCTestCase { XCTAssertEqual(store.applyReplacements(to: "open tape"), "open tape") store.approveEntry(id: competingID) - XCTAssertEqual(store.applyReplacements(to: "open tape"), "Open Tape") + XCTAssertEqual(store.snapshot(languageCode: "en").applyReplacements(to: "open tape"), "Open Tape") XCTAssertEqual(store.entries.filter(\.isEffective).count, 1) } diff --git a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift index 8817cd07..6404e6a4 100644 --- a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift +++ b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift @@ -76,6 +76,32 @@ final class QwenNativeASREngineTests: XCTestCase { XCTAssertEqual(weightAfter.contentModificationDate, weightBefore.contentModificationDate) } + func testExistingModelUsesBoundedRareTermHint() async throws { + guard ProcessInfo.processInfo.environment["OPENTYPE_QWEN_NATIVE_INTEGRATION"] == "1" else { + throw XCTSkip("Set OPENTYPE_QWEN_NATIVE_INTEGRATION=1 to run native Qwen integration tests") + } + let modelPath = try XCTUnwrap(ProcessInfo.processInfo.environment["OPENTYPE_QWEN_MODEL_PATH"]) + let audioURL = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-rare-term-\(UUID().uuidString).aiff") + defer { try? FileManager.default.removeItem(at: audioURL) } + let speech = Process() + speech.executableURL = URL(fileURLWithPath: "/usr/bin/say") + speech.arguments = ["-v", "Samantha", "-o", audioURL.path, + "We shipped Zyralith to production today."] + try speech.run() + speech.waitUntilExit() + XCTAssertEqual(speech.terminationStatus, 0) + + let engine = QwenNativeASREngine(modelPath: modelPath) + engine.configureRecognition(context: .empty) + let withoutContext = try await engine.transcribe(audioURL: audioURL, language: "en") + engine.configureRecognition(context: SpeechRecognitionContext(phrases: ["Zyralith"])) + let withContext = try await engine.transcribe(audioURL: audioURL, language: "en") + print("QWEN_RARE_TERM_WITHOUT_CONTEXT=\(withoutContext)") + print("QWEN_RARE_TERM_WITH_CONTEXT=\(withContext)") + XCTAssertTrue(withContext.contains("Zyralith"), withContext) + } + func testExistingModelRejectsSyntheticNonSpeechNoise() async throws { guard ProcessInfo.processInfo.environment["OPENTYPE_QWEN_NATIVE_INTEGRATION"] == "1" else { throw XCTSkip("Set OPENTYPE_QWEN_NATIVE_INTEGRATION=1 to run the native Qwen integration test") diff --git a/Tests/OpenTypeTests/QwenRecognitionContextTests.swift b/Tests/OpenTypeTests/QwenRecognitionContextTests.swift new file mode 100644 index 00000000..4192587d --- /dev/null +++ b/Tests/OpenTypeTests/QwenRecognitionContextTests.swift @@ -0,0 +1,38 @@ +import XCTest +@testable import OpenType + +final class QwenRecognitionContextTests: XCTestCase { + func testPromptIsBoundedAndDoesNotContainWholeLexicon() { + let phrases = (1...30).map { "Term\($0)" } + let prompt = QwenRecognitionPrompt(phrases: phrases) + XCTAssertLessThanOrEqual(prompt.phrases.count, 8) + XCTAssertLessThanOrEqual(prompt.text.count, 180) + XCTAssertFalse(prompt.text.contains("Term30")) + } + + func testEchoRetriesOnceWithoutContextAndRejectsRepeatedEcho() async throws { + let prompt = QwenRecognitionPrompt(phrases: ["Alpha", "Beta", "Gamma", "Delta"]) + var contexts: [String] = [] + let accepted = try await QwenContextRecovery.run(prompt: prompt) { context in + contexts.append(context) + return context.isEmpty ? "We shipped Alpha today." : "Vocabulary: Alpha, Beta, Gamma, Delta." + } text: { $0 } + XCTAssertEqual(accepted, "We shipped Alpha today.") + XCTAssertEqual(contexts, [prompt.text, ""]) + + contexts.removeAll() + let rejected = try await QwenContextRecovery.run(prompt: prompt) { context in + contexts.append(context) + return "Alpha, Beta, Gamma, Delta" + } text: { $0 } + XCTAssertNil(rejected) + XCTAssertEqual(contexts, [prompt.text, ""]) + } + + func testSingleSpokenTermIsNotClassedAsEcho() { + let prompt = QwenRecognitionPrompt(phrases: ["Zyralith"]) + XCTAssertFalse(QwenPromptEcho.matches("We shipped Zyralith today.", prompt: prompt)) + XCTAssertFalse(QwenPromptEcho.matches("Zyralith", prompt: prompt)) + XCTAssertTrue(QwenPromptEcho.matches("Vocabulary: Zyralith.", prompt: prompt)) + } +} diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md index 4a90595e..36dadbfd 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/plan.md @@ -1,8 +1,8 @@ # Plan: Prevent unsolicited Qwen vocabulary insertion -**Status:** pending approval -**Approved-by:** — -**Approved-date:** — +**Status:** approved +**Approved-by:** User (conversation; confirmed revised implementation plan) +**Approved-date:** 2026-09-24 **Upstream:** [spec.md](spec.md) This plan supersedes the 2026-09-23 plan. Existing code and tests remain in the draft PR; the items below describe only the redesigned work. @@ -26,4 +26,4 @@ This plan supersedes the 2026-09-23 plan. Existing code and tests remain in the ## Human gate -`docs/sdlc/README.md` requires explicit approval of this revised plan before implementation resumes. Earlier approval of the old plan does not approve this one. +The user approved this revised plan on 2026-09-24. Verification, independent review, and release retain their separate gates. From 9f9ccc89355fa23d04874c586172796e2eaf08c9 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 02:48:49 +0800 Subject: [PATCH 09/10] fix: keep manual dictionary rules authoritative across app scopes --- .../PersonalDictionary+Learning.swift | 8 +++++ Sources/Processing/PersonalDictionary.swift | 30 +++++++++++-------- .../PersonalDictionaryLearningTests.swift | 14 +++++++++ .../verification.md | 3 +- 4 files changed, 42 insertions(+), 13 deletions(-) diff --git a/Sources/Processing/PersonalDictionary+Learning.swift b/Sources/Processing/PersonalDictionary+Learning.swift index 5eb0f348..c04d5761 100644 --- a/Sources/Processing/PersonalDictionary+Learning.swift +++ b/Sources/Processing/PersonalDictionary+Learning.swift @@ -1,6 +1,14 @@ import Foundation extension PersonalDictionary { + func suspendLearnedMappings(for original: String, excluding id: UUID) { + for index in entries.indices where entries[index].id != id + && entries[index].origin == .learned + && entries[index].original.caseInsensitiveCompare(original) == .orderedSame { + entries[index].status = .pending + } + } + func clearLearnedEntries() { entries.removeAll { $0.origin == .learned } save() diff --git a/Sources/Processing/PersonalDictionary.swift b/Sources/Processing/PersonalDictionary.swift index f5e06e02..0b86d017 100644 --- a/Sources/Processing/PersonalDictionary.swift +++ b/Sources/Processing/PersonalDictionary.swift @@ -100,7 +100,6 @@ struct PersonalDictionarySnapshot: Sendable { } return industryTerms + personalTerms } - } final class PersonalDictionary: ObservableObject { @@ -170,9 +169,10 @@ final class PersonalDictionary: ObservableObject { let replacement = normalized(replacement) guard !original.isEmpty, !replacement.isEmpty, original != replacement else { return nil } - if let index = entries.firstIndex(where: { - $0.original.caseInsensitiveCompare(original) == .orderedSame - }) { + let matches = entries.indices.filter { + entries[$0].original.caseInsensitiveCompare(original) == .orderedSame + } + if let index = matches.first(where: { entries[$0].origin == .manual }) ?? matches.first { entries[index].original = original entries[index].replacement = replacement entries[index].enabled = true @@ -182,12 +182,14 @@ final class PersonalDictionary: ObservableObject { entries[index].evidenceCount = max(1, entries[index].evidenceCount) entries[index].languageCode = nil entries[index].appScopes = [] + suspendLearnedMappings(for: original, excluding: entries[index].id) save() return entries[index].id } let entry = DictionaryEntry(original: original, replacement: replacement) entries.append(entry) + suspendLearnedMappings(for: original, excluding: entry.id) save() return entry.id } @@ -213,6 +215,7 @@ final class PersonalDictionary: ObservableObject { entries[index].status = .active entries[index].languageCode = nil entries[index].appScopes = [] + suspendLearnedMappings(for: original, excluding: entries[index].id) save() } @@ -231,12 +234,16 @@ final class PersonalDictionary: ObservableObject { entries[index].appScopes = [] } let original = entries[index].original - for otherIndex in entries.indices where otherIndex != index - && entries[otherIndex].origin == .learned - && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame - && entries[otherIndex].languageCode == entries[index].languageCode - && entries[otherIndex].appScopes == entries[index].appScopes { - entries[otherIndex].status = .pending + if entries[index].origin == .manual { + suspendLearnedMappings(for: original, excluding: entries[index].id) + } else { + for otherIndex in entries.indices where otherIndex != index + && entries[otherIndex].origin == .learned + && entries[otherIndex].original.caseInsensitiveCompare(original) == .orderedSame + && entries[otherIndex].languageCode == entries[index].languageCode + && entries[otherIndex].appScopes == entries[index].appScopes { + entries[otherIndex].status = .pending + } } entries[index].status = .active entries[index].enabled = true @@ -286,8 +293,7 @@ final class PersonalDictionary: ObservableObject { } private func normalized(_ text: String) -> String { - text.replacingOccurrences(of: "\\s+", with: " ", options: .regularExpression) - .trimmingCharacters(in: .whitespacesAndNewlines) + text.replacingOccurrences(of: "\\s+", with: " ", options: .regularExpression).trimmingCharacters(in: .whitespacesAndNewlines) } } diff --git a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift index d2bc8a6d..62522601 100644 --- a/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift +++ b/Tests/OpenTypeTests/PersonalDictionaryLearningTests.swift @@ -74,6 +74,20 @@ final class PersonalDictionaryLearningTests: XCTestCase { .applyReplacements(to: "open tape"), "open tape") } + func testGlobalManualEntrySuspendsConflictingScopedLearnedRules() { + let store = makeStore() + store.entries = [ + DictionaryEntry(original: "open type", replacement: "Wrong A", origin: .learned, + languageCode: "en", appScopes: ["com.apple.Notes"]), + DictionaryEntry(original: "open type", replacement: "Wrong B", origin: .learned, + languageCode: "en", appScopes: ["com.apple.TextEdit"]), + ] + store.addEntry(original: "open type", replacement: "OpenType") + XCTAssertEqual(store.entries.filter { $0.origin == .learned }.map(\.status), [.pending]) + XCTAssertEqual(store.snapshot(bundleIdentifier: "com.apple.TextEdit", languageCode: "en") + .applyReplacements(to: "open type"), "OpenType") + } + func testLegacyEntryDecodesAsActiveManualTerm() throws { let data = Data(#"{"original":"open type","replacement":"OpenType","enabled":true}"#.utf8) let entry = try JSONDecoder().decode(DictionaryEntry.self, from: data) diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 8122433e..3b1e3266 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -17,10 +17,11 @@ The prior verification was approved for the superseded design. Results below des | Qwen echo recovery | Pass locally | A prompt-prefix or ordered-term echo triggers one empty-context retry. A repeated echo produces no transcript; a retry error cannot return the first echo. Short spoken terms remain eligible. | | All recorded-audio entry paths | Pass locally | Menu-bar, live integration, and imported audio call the same Sound Analysis classifier before final transcript and output. Imported-audio regression asserts no ASR call or final session when the classifier rejects. | | Existing session lifecycle | Pass locally | Integrated current `main` ownership/cancellation changes without removing their transaction guard. Ten ownership tests and the mode-specific insertion tests pass. | -| Full `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass locally | 791 XCTest cases, 18 skipped, zero failures; one Swift Testing case passed. Ran after integration with current `main`. | +| Full `swift test --scratch-path /tmp/utter-silent-insertion-build` | Pass locally | 793 XCTest cases, 18 skipped, zero failures; one Swift Testing case passed. Ran after integration with current `main`. | | `bash scripts/sdlc-checks.sh` and `bash scripts/ci-basic-checks.sh` | Pass locally | Both completed after the merge resolution. | | Installed local Qwen replay | Pass locally | Generated noise was rejected; repository-owned English and Chinese speech samples were transcribed. The synthetic rare name was rendered as “Zerolith” without a hint and “Zyralith” with one or three bounded hints. This is one generated voice sample, not a general accuracy estimate. | | Release-style app build | Pass locally | `bash scripts/build-app.sh --app-only --sign=-` built the app and CLI helper, bundled Metal resources, and passed artifact verification. Ad-hoc signing is for local checking only. | +| PR mergeability | Pass at observed head | Merged `origin/main` at `7139f54`; GitHub reported PR #112 `MERGEABLE` at head `c82e77f`. Final branch updates require another mergeability check. Remote CI was still in progress. | | Real computer-microphone/window QA | Pending | Two other Utter instances are active. Launching this build concurrently would not give trustworthy hotkey/permission evidence. | | Independent verification, signed release, production observation | Pending | The author cannot satisfy the independent high-risk review or protected production gate. | From 577bf787000cab7f50e13add11ff1e6bd43e2a07 Mon Sep 17 00:00:00 2001 From: idevlab Date: Thu, 24 Sep 2026 11:38:33 +0800 Subject: [PATCH 10/10] Document real microphone silence check and app restoration --- .../verification.md | 22 ++++++++++--------- 1 file changed, 12 insertions(+), 10 deletions(-) diff --git a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md index 3b1e3266..8b9b8f45 100644 --- a/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md +++ b/docs/sdlc/changes/2026-09-23-silent-input-insertion/verification.md @@ -21,8 +21,10 @@ The prior verification was approved for the superseded design. Results below des | `bash scripts/sdlc-checks.sh` and `bash scripts/ci-basic-checks.sh` | Pass locally | Both completed after the merge resolution. | | Installed local Qwen replay | Pass locally | Generated noise was rejected; repository-owned English and Chinese speech samples were transcribed. The synthetic rare name was rendered as “Zerolith” without a hint and “Zyralith” with one or three bounded hints. This is one generated voice sample, not a general accuracy estimate. | | Release-style app build | Pass locally | `bash scripts/build-app.sh --app-only --sign=-` built the app and CLI helper, bundled Metal resources, and passed artifact verification. Ad-hoc signing is for local checking only. | -| PR mergeability | Pass at observed head | Merged `origin/main` at `7139f54`; GitHub reported PR #112 `MERGEABLE` at head `c82e77f`. Final branch updates require another mergeability check. Remote CI was still in progress. | -| Real computer-microphone/window QA | Pending | Two other Utter instances are active. Launching this build concurrently would not give trustworthy hotkey/permission evidence. | +| PR mergeability and remote CI | Pass at observed head | GitHub reported draft PR #112 `MERGEABLE` at `9f9ccc8`; Contract & Tests, Release-style App Build, and SDLC Gate all succeeded. | +| Real computer-microphone silence QA | Pass for live integration route | Temporarily stopped both original instances, launched the ad-hoc local build, granted its microphone permission through the onboarding UI, and recorded three seconds from the default MacBook Pro Microphone. The live recording API returned `no_speech_detected` with no transcript or final text. | +| Real computer-microphone speech QA | Inconclusive | Two attempts to play a Chinese technical sentence through the MacBook speakers into the microphone returned `no_speech_detected`; the system output was subsequently observed muted. No valid audible speech sample was confirmed at the microphone. The menu-bar hotkey path was not exercised because this ad-hoc build lacked Accessibility authorization. | +| Original app restoration | Pass | Stopped the test build and restarted the two exact original bundles, one from `/Applications/Utter.app` and one from the prior worktree. Restored `hasCompletedOnboarding=true` and `activationMode=longPress`; both original processes were observed running. The speaker mute state was left as the user set it. | | Independent verification, signed release, production observation | Pending | The author cannot satisfy the independent high-risk review or protected production gate. | These results verify the revised local implementation only. The exact incident audio was not retained, and Sound Analysis cannot determine who spoke. The draft PR stays unreleased until the remaining gates are met. @@ -49,23 +51,23 @@ These results verify the revised local implementation only. The exact incident a | `bash scripts/build-app.sh --app-only --sign=-` | Pass | Xcode Release build, Metal shader bundle, CLI helper, bundle assembly, ad-hoc codesign, and artifact verification passed. No usable Utter signing identity was installed; this is a local build, not a trusted release. | | Rebase onto `origin/main` | Pass | Commit `686becb` is based on `60e7ed4` (Confucius4-R2T2 support). Qwen's new `modelID` and tail-padding path remain intact; vocabulary context injection remains removed. | | Post-rebase checks | Pass | `bash scripts/ci-basic-checks.sh`; `swift test --scratch-path /tmp/utter-silent-insertion-build` (763 XCTest cases, 17 skipped, no failures; one Swift Testing case passed); release-style ad-hoc app build and artifact verification. | -| Real microphone/window check | Not run | Two other Utter instances are active on this Mac. Launching a third copy would conflict with hotkeys and would not provide reliable input-path evidence. | +| Real microphone/window check | Partial | The revised-design table above records the new three-second silence result, inconclusive speaker playback, and restoration of both original instances. This older evidence table describes the superseded design. | | Independent verification | Pending | — | ## Acceptance criteria - The reported vocabulary echo is rejected before insertion, clipboard, edit-command, and history paths — automated menu-bar and integration tests pass; independent review pending. -- Generated non-speech audio is rejected by an independent classifier even when Qwen emits text — local model replay passes; real microphone check pending. -- Clear English/Chinese speech and short clips remain accepted — automated file fixtures pass, including quieter short clips; real microphone check pending. +- Generated non-speech audio is rejected by an independent classifier even when Qwen emits text — local model replay passes; a three-second real-microphone silence session was also rejected. This does not isolate which of the RMS gate and classifier rejected that session. +- Clear English/Chinese speech and short clips remain accepted — automated file fixtures pass, including quieter short clips; real microphone speech acceptance remains pending. ## Residual risk - The built-in classifier cannot identify who spoke. Nearby human speech may still be transcribed; this is outside the no-speech and non-speech-noise acceptance criterion. -- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. Separate historical records show an active learned rule replacing a short filler utterance. The current branch does not yet prevent that rule from applying to genuine speech. Command mode has a different output contract; the new audio gate remains the primary protection for no-speech recordings. -- The 0.6 speech-confidence threshold is calibrated against the listed fixtures, not a diverse microphone corpus. Real microphone validation and independent review remain required. -- Two other Utter instances prevent a trustworthy real-window test of this local build without interrupting the user's running apps. -- The branch includes `origin/main` at `60e7ed4`; compare against the live PR base again before review or merge, since main can advance. -- Full release signing, GitHub CI, and production behavior have not been verified. +- The long-list generating stage is strongly identified as Qwen ASR context echo by the saved `rawText` and legacy `Terms: ` construction. The exact incident audio and runtime prompt snapshot were not retained, so the sound that crossed the original RMS gate is unknown. The separately observed learned `嗯。 -> Do anything` rule is now inert unless manually approved or edited. +- The 0.6 speech-confidence threshold is calibrated against fixtures, not a diverse microphone corpus. A real-microphone silence session passed, but an audible speech acceptance test and independent review remain required. +- The two original Utter instances were restored after the test. The ad-hoc test build lacked Accessibility authorization, so menu-bar hotkey insertion was not verified with real audio. +- The branch includes `origin/main` at `7139f54`; compare against the live PR base again before review or merge, since main can advance. +- Remote PR CI passed at `9f9ccc8`; trusted release signing and production behavior have not been verified. ## Decision