diff --git a/Sources/App/VoicePipeline+Models.swift b/Sources/App/VoicePipeline+Models.swift index ad292c4c..b3a8a752 100644 --- a/Sources/App/VoicePipeline+Models.swift +++ b/Sources/App/VoicePipeline+Models.swift @@ -161,7 +161,10 @@ extension VoicePipeline { } let modelPath = catalog.asrModelPath(for: settings.qwenASRModel) if qwenSpeechEngine?.usesModel(at: modelPath) == true { return } - let engine = QwenNativeASREngine(modelPath: modelPath) + let engine = QwenNativeASREngine( + modelPath: modelPath, + modelID: settings.qwenASRModel + ) qwenSpeechEngine = engine Task { await engine.prepare() } case .firered, .megaASR: diff --git a/Sources/Config/ConfuciusModelDownloader.swift b/Sources/Config/ConfuciusModelDownloader.swift new file mode 100644 index 00000000..25a75ca5 --- /dev/null +++ b/Sources/Config/ConfuciusModelDownloader.swift @@ -0,0 +1,182 @@ +import CryptoKit +import Foundation + +enum ConfuciusModelDownloader { + enum Failure: Error { + case invalidRangeResponse + case checksumMismatch + } + + private struct ModelFile { + let name: String + let size: Int64 + let sha256: String + } + + private static let smallFiles: [ModelFile] = [ + .init(name: ".gitattributes", size: 1_570, sha256: "34448b82c17d60fec9b65b1f093c115ddbaadc04beb1b0140b6bfed2e012a930"), + .init(name: "LICENSE", size: 11_071, sha256: "4d9321cdad58182faa878b015de7d60069881614ddd7571de70f751a9b8e3811"), + .init(name: "MODEL_LICENSE_zh", size: 7_733, sha256: "18b438311ebb842c15a7c91d9dbb59034efdaf6fff0d105031ef7baf5eeed275"), + .init(name: "NOTICE", size: 847, sha256: "473880a027d736ca2c9efb3da01cfd30b745d8259e0666b3dea946d4be4b58ef"), + .init(name: "README.md", size: 2_808, sha256: "7dfc5b13dc96aceb9803ab31c13d5011c6cb6f916ee4472ef96f5d1c8f0abab2"), + .init(name: "added_tokens.json", size: 1_566, sha256: "de40784677cbd1843cabe5fbee078c7e042cd0b62155f0810af5a13842e5722a"), + .init(name: "chat_template.json", size: 1_161, sha256: "75a8cfca24f00de72d796fbfed6858fc9614ef3dabd8696684cc3bc03a9c58ff"), + .init(name: "config.json", size: 7_188, sha256: "1b76b3b6c655fc54595da025f7a96474ad9fa86363303fbdd61a7d8483ccfaf7"), + .init(name: "generation_config.json", size: 142, sha256: "1da527824d81e07118facff437e03f2e24a23311e3bdeb2368973fe77e5f275c"), + .init(name: "merges.txt", size: 1_671_853, sha256: "8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5"), + .init(name: "model.safetensors.index.json", size: 78_928, sha256: "b62daee0e37a7bedb8675f69c733eef18a398235ea513f0999f3805d5ac4dedf"), + .init(name: "preprocessor_config.json", size: 330, sha256: "45e120a4eda2c20c5d7f2ea9354e63536bf35e27aa573fb7cdf78017b378770d"), + .init(name: "special_tokens_map.json", size: 1_008, sha256: "7b376c510ccf9d88bb9bbee41dfc5052122e16e0dec1124a8d8983c59259a9f3"), + .init(name: "tokenizer.json", size: 11_429_499, sha256: "0499602714160467f2d68b910651d6216020689f1e016be87a2d0019ee3baeab"), + .init(name: "tokenizer_config.json", size: 12_487, sha256: "4942d005604266809309cabc9f4e9cb89ce855d59b14681fdc0e1cc62ea26c4c"), + .init(name: "vocab.json", size: 2_776_833, sha256: "ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910"), + ] + + static func downloadRepository(to directory: URL, onProgress: @escaping @Sendable (Int64) -> Void) async throws { + let base = "https://huggingface.co/\(QwenASRModel.confuciusR2T2ID)/resolve/\(QwenASRModel.confuciusRevision)" + var completedBytes: Int64 = 0 + for file in smallFiles { + let previousBytes = completedBytes + try await download( + to: directory.appendingPathComponent(file.name), + source: URL(string: "\(base)/\(file.name)")!, + size: file.size, + sha256: file.sha256 + ) { onProgress(previousBytes + $0) } + completedBytes += file.size + } + let previousBytes = completedBytes + try await download(to: directory.appendingPathComponent("model.safetensors")) { + onProgress(previousBytes + $0) + } + } + + static func download( + to destination: URL, + session: URLSession? = nil, + source: URL = URL(string: "https://huggingface.co/\(QwenASRModel.confuciusR2T2ID)/resolve/\(QwenASRModel.confuciusRevision)/model.safetensors")!, + size: Int64 = QwenASRModel.confuciusWeightBytes, + sha256: String = QwenASRModel.confuciusWeightSHA256, + partSize: Int64 = 2 * 1_024 * 1_024, + maxConcurrent: Int = 16, + onProgress: @escaping @Sendable (Int64) -> Void + ) async throws { + try FileManager.default.createDirectory( + at: destination.deletingLastPathComponent(), + withIntermediateDirectories: true + ) + let writer = try WeightWriter(destination: destination, size: size) + let partCount = Int((size + partSize - 1) / partSize) + // Separate sessions avoid this CDN throttling ranges on one HTTP/2 connection. + let sessions = session.map { Array(repeating: $0, count: maxConcurrent) } + ?? (0.. Data { + var lastError: Error = Failure.invalidRangeResponse + for attempt in 0..<5 { + try Task.checkCancellation() + do { + var request = URLRequest(url: source) + request.cachePolicy = .reloadIgnoringLocalCacheData + request.setValue("bytes=\(start)-\(end)", forHTTPHeaderField: "Range") + let (data, response) = try await session.data(for: request) + guard let http = response as? HTTPURLResponse, + http.statusCode == 206, + http.value(forHTTPHeaderField: "Content-Range") == "bytes \(start)-\(end)/\(size)", + Int64(data.count) == end - start + 1 else { + throw Failure.invalidRangeResponse + } + return data + } catch { + lastError = error + if attempt < 4 { + try await Task.sleep(for: .seconds(1 << attempt)) + } + } + } + throw lastError + } + + private static func fileSHA256(_ url: URL) throws -> String { + let file = try FileHandle(forReadingFrom: url) + defer { try? file.close() } + var hash = SHA256() + while let data = try file.read(upToCount: 4 * 1_024 * 1_024), !data.isEmpty { + hash.update(data: data) + } + return hash.finalize().map { String(format: "%02x", $0) }.joined() + } +} + +private actor WeightWriter { + private let file: FileHandle + private var completedBytes: Int64 = 0 + + init(destination: URL, size: Int64) throws { + FileManager.default.createFile(atPath: destination.path, contents: nil) + file = try FileHandle(forWritingTo: destination) + try file.truncate(atOffset: UInt64(size)) + } + + func write(_ data: Data, at offset: Int64) throws -> Int64 { + try file.seek(toOffset: UInt64(offset)) + try file.write(contentsOf: data) + completedBytes += Int64(data.count) + return completedBytes + } + + func close() throws { + try file.synchronize() + try file.close() + } +} diff --git a/Sources/Config/ModelCatalogASR.swift b/Sources/Config/ModelCatalogASR.swift index 26f7510a..05817bc3 100644 --- a/Sources/Config/ModelCatalogASR.swift +++ b/Sources/Config/ModelCatalogASR.swift @@ -11,6 +11,11 @@ extension ModelCatalog { "Qwen3-ASR 1.7B", L("model.qwen3_asr_quality") ), + ( + QwenASRModel.confuciusR2T2ID, + "Confucius4-R2T2 8-bit", + L("model.confucius_r2t2") + ), ( "mlx-community/FireRedASR2-AED-mlx", "FireRedASR2-AED", @@ -33,7 +38,9 @@ extension ModelCatalog { func asrModels(for engine: SpeechEngineType) -> [ModelEntry] { switch engine { case .qwen3: - return asrModels.filter { $0.id == QwenASRModel.defaultID } + return asrModels.filter { + $0.id == QwenASRModel.defaultID || $0.id == QwenASRModel.confuciusR2T2ID + } case .firered: return asrModels.filter { $0.id == "mlx-community/FireRedASR2-AED-mlx" } case .megaASR: @@ -97,10 +104,6 @@ extension ModelCatalog { do { try ModelStorage.prepareGeneration(staging) - let api = HubApi( - downloadBase: staging.downloadBase, - cache: staging.hubCache - ) let tracker = DownloadProgressTracker( startDate: Date(), initialBytes: asrRepoSize(id, downloadBase: staging.downloadBase) @@ -108,31 +111,50 @@ extension ModelCatalog { let estimatedTotalBytes = estimatedASRDownloadBytes(id) ?? 0 let repositories = asrRequiredRepoIDs(for: id) for (repositoryIndex, repositoryID) in repositories.enumerated() { - _ = try await api.snapshot(from: ModelStorage.hubModelRepo(repositoryID)) { [weak self] progress in - Task { @MainActor in - guard let self, - self.downloadTasks.isCurrent(key, token: token), - let i = self.asrModels.firstIndex(where: { $0.id == id }) else { return } - let repositoryFraction = - (Double(repositoryIndex) + progress.fractionCompleted) / Double(repositories.count) - let downloadedBytes = self.asrRepoSize( - id, - downloadBase: staging.downloadBase - ) - let info = tracker.update( - completedBytes: downloadedBytes, - totalBytes: estimatedTotalBytes, - fraction: repositoryFraction - ) - if signal.advanced( - completedBytes: max(downloadedBytes, progress.completedUnitCount), - fraction: info.fraction - ) { - watchdog.noteProgress() + if repositoryID == QwenASRModel.confuciusR2T2ID { + let directory = ModelStorage.asrRepoDir(repositoryID, downloadBase: staging.downloadBase) + try await ConfuciusModelDownloader.downloadRepository(to: directory) { [weak self] bytes in + Task { @MainActor in + guard let self, + self.downloadTasks.isCurrent(key, token: token), + let i = self.asrModels.firstIndex(where: { $0.id == id }) else { return } + let info = tracker.update( + completedBytes: bytes, + totalBytes: estimatedTotalBytes + ) + if signal.advanced(completedBytes: bytes, fraction: info.fraction) { + watchdog.noteProgress() + } + self.asrModels[i].downloadProgress = info.fraction + self.asrModels[i].downloadDetail = info.detailText + onProgress?(info) + } + } + } else { + let api = HubApi(downloadBase: staging.downloadBase, cache: staging.hubCache) + _ = try await api.snapshot(from: ModelStorage.hubModelRepo(repositoryID)) { [weak self] progress in + Task { @MainActor in + guard let self, + self.downloadTasks.isCurrent(key, token: token), + let i = self.asrModels.firstIndex(where: { $0.id == id }) else { return } + let repositoryFraction = + (Double(repositoryIndex) + progress.fractionCompleted) / Double(repositories.count) + let downloadedBytes = self.asrRepoSize(id, downloadBase: staging.downloadBase) + let info = tracker.update( + completedBytes: downloadedBytes, + totalBytes: estimatedTotalBytes, + fraction: repositoryFraction + ) + if signal.advanced( + completedBytes: max(downloadedBytes, progress.completedUnitCount), + fraction: info.fraction + ) { + watchdog.noteProgress() + } + self.asrModels[i].downloadProgress = info.fraction + self.asrModels[i].downloadDetail = info.detailText + onProgress?(info) } - self.asrModels[i].downloadProgress = info.fraction - self.asrModels[i].downloadDetail = info.detailText - onProgress?(info) } } } @@ -239,65 +261,7 @@ extension ModelCatalog { } } - private func asrRepoSize(_ id: String, downloadBase: URL) -> Int64 { - asrRequiredRepoIDs(for: id).reduce(0) { total, repositoryID in - total + ModelStorage.directorySize( - at: ModelStorage.asrRepoDir(repositoryID, downloadBase: downloadBase) - ) - } - } - - private func asrMissingStatus(size: Int64) -> ModelStatus { - size > 0 ? .error(L("model.asr_incomplete")) : .notDownloaded - } - - func asrRepoSize(_ id: String) -> Int64 { - asrRequiredRepoIDs(for: id).reduce(0) { total, repositoryID in - guard let directory = ModelStorage.asrRepoDir(repositoryID) else { return total } - return total + ModelStorage.directorySize(at: directory) - } - } - - private func asrRequiredRepoIDs(for id: String) -> [String] { + func asrRequiredRepoIDs(for id: String) -> [String] { [id] } - - static func asrRepoContainsRequiredFiles(_ id: String, at dir: URL?) -> Bool { - guard let dir else { return false } - return asrRequiredFiles(for: id).allSatisfy { relativePath in - let file = dir.appendingPathComponent(relativePath) - var isDirectory = ObjCBool(false) - guard FileManager.default.fileExists(atPath: file.path, isDirectory: &isDirectory), - !isDirectory.boolValue else { return false } - let attributes = try? FileManager.default.attributesOfItem(atPath: file.path) - return (attributes?[.size] as? NSNumber)?.int64Value ?? 0 > 0 - } - } - - nonisolated static func asrRequiredFiles(for id: String) -> [String] { - switch id { - case QwenASRModel.defaultID: - return [ - "config.json", - "model.safetensors", - "model.safetensors.index.json", - "preprocessor_config.json", - "tokenizer_config.json", - "vocab.json", - "merges.txt", - ] - case "mlx-community/FireRedASR2-AED-mlx": - return [ - "config.json", - "tokenizer.json", - ] - case "mlx-community/Mega-ASR-6bit": - return [ - "config.json", - "tokenizer_config.json", - ] - default: - return ["config.json"] - } - } } diff --git a/Sources/Config/ModelCatalogASRFiles.swift b/Sources/Config/ModelCatalogASRFiles.swift new file mode 100644 index 00000000..c663a1a7 --- /dev/null +++ b/Sources/Config/ModelCatalogASRFiles.swift @@ -0,0 +1,40 @@ +import Foundation + +extension ModelCatalog { + static func asrRepoContainsRequiredFiles(_ id: String, at dir: URL?) -> Bool { + guard let dir else { return false } + return asrRequiredFiles(for: id).allSatisfy { relativePath in + let file = dir.appendingPathComponent(relativePath) + var isDirectory = ObjCBool(false) + guard FileManager.default.fileExists(atPath: file.path, isDirectory: &isDirectory), + !isDirectory.boolValue else { return false } + let attributes = try? FileManager.default.attributesOfItem(atPath: file.path) + return (attributes?[.size] as? NSNumber)?.int64Value ?? 0 > 0 + } + } + + nonisolated static func asrRequiredFiles(for id: String) -> [String] { + switch id { + case QwenASRModel.defaultID, QwenASRModel.confuciusR2T2ID: + var required = [ + "config.json", + "model.safetensors", + "model.safetensors.index.json", + "preprocessor_config.json", + "tokenizer_config.json", + "vocab.json", + "merges.txt", + ] + if id == QwenASRModel.confuciusR2T2ID { + required += ["LICENSE", "MODEL_LICENSE_zh", "NOTICE"] + } + return required + case "mlx-community/FireRedASR2-AED-mlx": + return ["config.json", "tokenizer.json"] + case "mlx-community/Mega-ASR-6bit": + return ["config.json", "tokenizer_config.json"] + default: + return ["config.json"] + } + } +} diff --git a/Sources/Config/ModelCatalogASRStorage.swift b/Sources/Config/ModelCatalogASRStorage.swift new file mode 100644 index 00000000..ebaeea83 --- /dev/null +++ b/Sources/Config/ModelCatalogASRStorage.swift @@ -0,0 +1,22 @@ +import Foundation + +extension ModelCatalog { + func asrRepoSize(_ id: String, downloadBase: URL) -> Int64 { + asrRequiredRepoIDs(for: id).reduce(0) { total, repositoryID in + total + ModelStorage.directorySize( + at: ModelStorage.asrRepoDir(repositoryID, downloadBase: downloadBase) + ) + } + } + + func asrMissingStatus(size: Int64) -> ModelStatus { + size > 0 ? .error(L("model.asr_incomplete")) : .notDownloaded + } + + func asrRepoSize(_ id: String) -> Int64 { + asrRequiredRepoIDs(for: id).reduce(0) { total, repositoryID in + guard let directory = ModelStorage.asrRepoDir(repositoryID) else { return total } + return total + ModelStorage.directorySize(at: directory) + } + } +} diff --git a/Sources/Config/ModelCatalogDownloadEstimates.swift b/Sources/Config/ModelCatalogDownloadEstimates.swift index 006eed13..9660b671 100644 --- a/Sources/Config/ModelCatalogDownloadEstimates.swift +++ b/Sources/Config/ModelCatalogDownloadEstimates.swift @@ -53,6 +53,7 @@ extension ModelCatalog { "mlx-community/Llama-4-Scout-17B-16E-Instruct-4bit": 61_143_654_248, "mlx-community/Llama-4-Maverick-17B-128E-Instruct-4bit": 225_923_469_800, QwenASRModel.defaultID: 4_080_707_826, + QwenASRModel.confuciusR2T2ID: 2_479_312_565, "mlx-community/FireRedASR2-AED-mlx": 4_570_000_000, "mlx-community/Mega-ASR-6bit": 2_040_000_000, ] diff --git a/Sources/Resources/en.lproj/Localizable.strings b/Sources/Resources/en.lproj/Localizable.strings index 791b816a..88640345 100644 --- a/Sources/Resources/en.lproj/Localizable.strings +++ b/Sources/Resources/en.lproj/Localizable.strings @@ -516,8 +516,12 @@ "volc.resource_id" = "Resource ID"; /* ── Local ASR ── */ -"qwen_asr.config_hint" = "Qwen3-ASR runs locally with native Swift and MLX. Download the model once, then recognition works offline."; +"qwen_asr.config_hint" = "Qwen-compatible speech models run locally with Swift and MLX. Download a model once for offline recognition."; "model.qwen3_asr_quality" = "Local ASR through MLX, ~4.1 GB"; +"model.confucius_r2t2" = "Local MLX speech recognition, ~2.5 GB"; +"model.confucius_license" = "Model license (Chinese)"; +"model.confucius_english_license" = "English translation"; +"model.confucius_notice" = "Conversion notice"; "model.firered_asr" = "Robust Chinese ASR, ~4.6 GB"; "model.mega_asr" = "Noise-robust ASR (Qwen3-ASR + LoRA), ~2.0 GB"; "model.asr_incomplete" = "Only part of the model was downloaded. Select Resume to finish."; diff --git a/Sources/Resources/zh-Hans.lproj/Localizable.strings b/Sources/Resources/zh-Hans.lproj/Localizable.strings index d623ca92..31156df6 100644 --- a/Sources/Resources/zh-Hans.lproj/Localizable.strings +++ b/Sources/Resources/zh-Hans.lproj/Localizable.strings @@ -516,8 +516,12 @@ "volc.resource_id" = "Resource ID(资源 ID)"; /* ── 本地语音识别 ── */ -"qwen_asr.config_hint" = "Qwen3-ASR 使用原生 Swift 和 MLX 在本机运行。模型下载一次后即可离线识别。"; +"qwen_asr.config_hint" = "兼容 Qwen 的语音模型使用 Swift 和 MLX 在本机运行。模型下载一次后即可离线识别。"; "model.qwen3_asr_quality" = "本地 MLX 语音识别,约 4.1 GB"; +"model.confucius_r2t2" = "本地 MLX 语音识别,约 2.5 GB"; +"model.confucius_license" = "模型许可协议(中文)"; +"model.confucius_english_license" = "英文译本"; +"model.confucius_notice" = "转换声明"; "model.firered_asr" = "强鲁棒中文 ASR,约 4.6 GB"; "model.mega_asr" = "抗噪 ASR(Qwen3-ASR + LoRA),约 2.0 GB"; "model.asr_incomplete" = "模型只下载了一部分。点击“继续下载”即可接着完成"; diff --git a/Sources/Speech/QwenASRModel.swift b/Sources/Speech/QwenASRModel.swift index 6698c703..7860735b 100644 --- a/Sources/Speech/QwenASRModel.swift +++ b/Sources/Speech/QwenASRModel.swift @@ -1,3 +1,7 @@ enum QwenASRModel { static let defaultID = "mlx-community/Qwen3-ASR-1.7B-bf16" + static let confuciusR2T2ID = "mlx-community/Confucius4-R2T2-8bit" + static let confuciusRevision = "2d6d997c3e09c65a65b1b2576b6b9b7728df8eab" + static let confuciusWeightBytes: Int64 = 2_463_307_541 + static let confuciusWeightSHA256 = "49b41186fbd139d0c9b3ec50f31834e47b93a6617c9271cb7c28793822ac6a98" } diff --git a/Sources/Speech/QwenAudioPreprocessor.swift b/Sources/Speech/QwenAudioPreprocessor.swift index 252826f6..78a2bc19 100644 --- a/Sources/Speech/QwenAudioPreprocessor.swift +++ b/Sources/Speech/QwenAudioPreprocessor.swift @@ -6,6 +6,7 @@ enum QwenAudioPreprocessor { static func withPreparedAudio( from sourceURL: URL, + tailPaddingFrames: AVAudioFrameCount = 0, operation: (URL) async throws -> T ) async throws -> T { let preparedURL = FileManager.default.temporaryDirectory @@ -14,7 +15,11 @@ enum QwenAudioPreprocessor { defer { try? FileManager.default.removeItem(at: preparedURL) } do { - try convertToPCM16kMono(from: sourceURL, to: preparedURL) + try convertToPCM16kMono( + from: sourceURL, + to: preparedURL, + tailPaddingFrames: tailPaddingFrames + ) } catch { Log.error("[Qwen3ASR] audio preprocessing failed: \(error.localizedDescription)") throw QwenAudioPreprocessorError.conversionFailed @@ -23,7 +28,11 @@ enum QwenAudioPreprocessor { return try await operation(preparedURL) } - private static func convertToPCM16kMono(from sourceURL: URL, to outputURL: URL) throws { + private static func convertToPCM16kMono( + from sourceURL: URL, + to outputURL: URL, + tailPaddingFrames: AVAudioFrameCount + ) throws { let sourceFile = try AVAudioFile(forReading: sourceURL) let sourceFormat = sourceFile.processingFormat guard sourceFormat.sampleRate > 0, sourceFormat.channelCount > 0 else { @@ -55,6 +64,17 @@ enum QwenAudioPreprocessor { outputFormat: outputFormat, converter: converter ) + if tailPaddingFrames > 0 { + guard let silence = AVAudioPCMBuffer( + pcmFormat: outputFormat, + frameCapacity: tailPaddingFrames + ), let samples = silence.int16ChannelData else { + throw AudioConversionError.outputBufferCreationFailed + } + samples[0].update(repeating: 0, count: Int(tailPaddingFrames)) + silence.frameLength = tailPaddingFrames + try outputFile.write(from: silence) + } } private static func convert( diff --git a/Sources/Speech/QwenNativeASREngine.swift b/Sources/Speech/QwenNativeASREngine.swift index b243b8a3..3b30d715 100644 --- a/Sources/Speech/QwenNativeASREngine.swift +++ b/Sources/Speech/QwenNativeASREngine.swift @@ -1,15 +1,18 @@ +import AVFoundation import Foundation import MLXAudioCore import MLXAudioSTT final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { private let modelDirectory: URL + private let tailPaddingFrames: AVAudioFrameCount private let runtime = QwenNativeASRRuntime() private let recognitionContextLock = NSLock() private var recognitionContext = SpeechRecognitionContext.empty - init(modelPath: String) { + init(modelPath: String, modelID: String = QwenASRModel.defaultID) { modelDirectory = URL(fileURLWithPath: modelPath).standardizedFileURL + tailPaddingFrames = modelID == QwenASRModel.confuciusR2T2ID ? 8_000 : 0 } var isReady: Bool { @@ -49,7 +52,10 @@ final class QwenNativeASREngine: SpeechEngine, @unchecked Sendable { let contextPrompt = currentContextPrompt() let started = CFAbsoluteTimeGetCurrent() - let result = try await QwenAudioPreprocessor.withPreparedAudio(from: audioURL) { preparedURL in + let result = try await QwenAudioPreprocessor.withPreparedAudio( + from: audioURL, + tailPaddingFrames: tailPaddingFrames + ) { preparedURL in try await runtime.transcribe( audioURL: preparedURL, modelDirectory: modelDirectory, diff --git a/Sources/Speech/SpeechEngineProvider.swift b/Sources/Speech/SpeechEngineProvider.swift index 718d1e29..f91a006f 100644 --- a/Sources/Speech/SpeechEngineProvider.swift +++ b/Sources/Speech/SpeechEngineProvider.swift @@ -50,7 +50,10 @@ final class SpeechEngineProvider { } let modelPath = ModelCatalog.shared.asrModelPath(for: settings.qwenASRModel) if qwenSpeechEngine?.usesModel(at: modelPath) == true { return } - qwenSpeechEngine = QwenNativeASREngine(modelPath: modelPath) + qwenSpeechEngine = QwenNativeASREngine( + modelPath: modelPath, + modelID: settings.qwenASRModel + ) case .firered, .megaASR: guard let modelID = settings.speechEngine.asrModelID else { return } guard localASRIsAvailable(modelID) else { diff --git a/Sources/UI/ModelManagementASRSection.swift b/Sources/UI/ModelManagementASRSection.swift new file mode 100644 index 00000000..10f54b66 --- /dev/null +++ b/Sources/UI/ModelManagementASRSection.swift @@ -0,0 +1,32 @@ +import SwiftUI + +extension ModelManagementView { + var qwenASRSection: some View { + let engineType = settings.speechEngine + let models = catalog.asrModels(for: engineType) + let activeID = engineType == .qwen3 + ? settings.qwenASRModel + : (engineType.asrModelID ?? settings.qwenASRModel) + return VStack(alignment: .leading, spacing: 8) { + Text(L("qwen_asr.config_hint")) + .font(.system(size: 11)) + .foregroundStyle(.secondary) + + modelList(models, activeID: activeID, type: .asr) + if engineType == .qwen3 { + HStack(spacing: 12) { + Link(L("model.confucius_license"), destination: URL( + string: "https://huggingface.co/mlx-community/Confucius4-R2T2-8bit/blob/main/MODEL_LICENSE_zh" + )!) + Link(L("model.confucius_english_license"), destination: URL( + string: "https://huggingface.co/mlx-community/Confucius4-R2T2-8bit/blob/main/LICENSE" + )!) + Link(L("model.confucius_notice"), destination: URL( + string: "https://huggingface.co/mlx-community/Confucius4-R2T2-8bit/blob/main/NOTICE" + )!) + } + .font(.system(size: 10)) + } + } + } +} diff --git a/Sources/UI/ModelManagementSections.swift b/Sources/UI/ModelManagementSections.swift index 859d0557..2e7ba4ed 100644 --- a/Sources/UI/ModelManagementSections.swift +++ b/Sources/UI/ModelManagementSections.swift @@ -119,23 +119,6 @@ extension ModelManagementView { } } - var qwenASRSection: some View { - let engineType = settings.speechEngine - let models = catalog.asrModels(for: engineType) - let activeID = engineType.asrModelID ?? settings.qwenASRModel - return VStack(alignment: .leading, spacing: 8) { - Text(L("qwen_asr.config_hint")) - .font(.system(size: 11)) - .foregroundStyle(.secondary) - - modelList( - models, - activeID: activeID, - type: .asr - ) - } - } - var llmSection: some View { VStack(alignment: .leading, spacing: 12) { if appState.lastFormattingDurationSeconds > 0 { diff --git a/Tests/OpenTypeTests/ConfigurationTests.swift b/Tests/OpenTypeTests/ConfigurationTests.swift index f6e655d6..551c1367 100644 --- a/Tests/OpenTypeTests/ConfigurationTests.swift +++ b/Tests/OpenTypeTests/ConfigurationTests.swift @@ -222,6 +222,7 @@ final class ConfigurationTests: XCTestCase { let models = ModelCatalog.defaultASRModels XCTAssertEqual(models.map(\.id), [ QwenASRModel.defaultID, + QwenASRModel.confuciusR2T2ID, "mlx-community/FireRedASR2-AED-mlx", "mlx-community/Mega-ASR-6bit", ]) diff --git a/Tests/OpenTypeTests/ConfuciusASRTests.swift b/Tests/OpenTypeTests/ConfuciusASRTests.swift new file mode 100644 index 00000000..5a90831e --- /dev/null +++ b/Tests/OpenTypeTests/ConfuciusASRTests.swift @@ -0,0 +1,136 @@ +import AVFoundation +import Foundation +import XCTest +@testable import OpenType + +@MainActor +final class ConfuciusASRTests: XCTestCase { + func testQwenCompatibleModelsShareOneSpeechEngine() { + let catalog = ModelCatalog(startupCleanup: { _ in Task { 0 } }) + XCTAssertEqual(catalog.asrModels(for: .qwen3).map(\.id), [ + QwenASRModel.defaultID, + QwenASRModel.confuciusR2T2ID, + ]) + } + + func testModelRequiresCompleteMLXFilesAndLicenseNotices() throws { + let dir = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) + try FileManager.default.createDirectory(at: dir, withIntermediateDirectories: true) + defer { try? FileManager.default.removeItem(at: dir) } + + let id = QwenASRModel.confuciusR2T2ID + let required = ModelCatalog.asrRequiredFiles(for: id) + XCTAssertTrue(required.contains("model.safetensors")) + XCTAssertTrue(required.contains("model.safetensors.index.json")) + XCTAssertTrue(required.contains("MODEL_LICENSE_zh")) + XCTAssertTrue(required.contains("NOTICE")) + for file in required where file != "model.safetensors" { + try Data([1]).write(to: dir.appendingPathComponent(file)) + } + XCTAssertFalse(ModelCatalog.asrRepoContainsRequiredFiles(id, at: dir)) + + try Data().write(to: dir.appendingPathComponent("model.safetensors")) + XCTAssertFalse(ModelCatalog.asrRepoContainsRequiredFiles(id, at: dir)) + try Data([1]).write(to: dir.appendingPathComponent("model.safetensors")) + XCTAssertTrue(ModelCatalog.asrRepoContainsRequiredFiles(id, at: dir)) + + try FileManager.default.removeItem(at: dir.appendingPathComponent("NOTICE")) + XCTAssertFalse(ModelCatalog.asrRepoContainsRequiredFiles(id, at: dir)) + } + + func testTailPaddingAddsHalfSecondOfSilence() async throws { + let source = URL(fileURLWithPath: #filePath) + .deletingLastPathComponent() + .deletingLastPathComponent() + .deletingLastPathComponent() + .appendingPathComponent("docs/assets/demos/confucius-mid-sentence.wav") + let baseline = try await QwenAudioPreprocessor.withPreparedAudio(from: source) { url in + try AVAudioFile(forReading: url).length + } + let padded = try await QwenAudioPreprocessor.withPreparedAudio( + from: source, + tailPaddingFrames: 8_000 + ) { url in + try AVAudioFile(forReading: url).length + } + XCTAssertEqual(padded - baseline, 8_000) + } + + func testDownloadedModelTranscribesChineseAndEnglish() async throws { + guard let modelPath = ProcessInfo.processInfo.environment["OPENTYPE_CONFUCIUS_MODEL_PATH"] else { + throw XCTSkip("Set OPENTYPE_CONFUCIUS_MODEL_PATH to run the model integration test") + } + XCTAssertTrue(ModelCatalog.asrRepoContainsRequiredFiles( + QwenASRModel.confuciusR2T2ID, + at: URL(fileURLWithPath: modelPath) + )) + let engine = QwenNativeASREngine(modelPath: modelPath, modelID: QwenASRModel.confuciusR2T2ID) + XCTAssertTrue(engine.isReady) + let root = URL(fileURLWithPath: #filePath) + .deletingLastPathComponent() + .deletingLastPathComponent() + .deletingLastPathComponent() + + let english = try await engine.transcribe( + audioURL: root.appendingPathComponent("docs/assets/demos/en-sample.m4a"), + language: "en" + ) + let chinese = try await engine.transcribe( + audioURL: root.appendingPathComponent("docs/assets/demos/zh-sample.m4a"), + language: "zh" + ) + print("CONFUCIUS_EN_TEXT=\(english)") + print("CONFUCIUS_ZH_TEXT=\(chinese)") + XCTAssertTrue(english.localizedCaseInsensitiveContains("design doc"), english) + XCTAssertTrue(chinese.contains("周五"), chinese) + + let tail = try await engine.transcribe( + audioURL: root.appendingPathComponent("docs/assets/demos/confucius-mid-sentence.wav"), + language: "en" + ) + print("CONFUCIUS_TAIL_TEXT=\(tail)") + XCTAssertTrue(tail.localizedCaseInsensitiveContains("follow"), tail) + XCTAssertFalse(tail.hasSuffix("|"), tail) + } + + func testApplicationCatalogDownloadsAndDeletesModel() async throws { + guard ProcessInfo.processInfo.environment["OPENTYPE_CONFUCIUS_LIVE_DOWNLOAD"] == "1" else { + throw XCTSkip("Set OPENTYPE_CONFUCIUS_LIVE_DOWNLOAD=1 for the live catalog download") + } + + let root = FileManager.default.temporaryDirectory + .appendingPathComponent("utter-confucius-catalog-\(UUID().uuidString)") + let settings = AppSettings.shared + let previousPath = settings.modelStoragePath + settings.modelStoragePath = root.path + defer { + settings.modelStoragePath = previousPath + try? FileManager.default.removeItem(at: root) + } + + let catalog = ModelCatalog(startupStorageRoot: root) + let id = QwenASRModel.confuciusR2T2ID + guard let index = catalog.asrModels.firstIndex(where: { $0.id == id }) else { + return XCTFail("Confucius must appear in the application catalog") + } + XCTAssertEqual(catalog.asrModels[index].status, .notDownloaded) + + var lastProgressLog = -10 + await catalog.downloadASR(id) { info in + let elapsed = Int(info.elapsedSeconds) + guard elapsed >= lastProgressLog + 10 else { return } + lastProgressLog = elapsed + print("[CONFUCIUS-DOWNLOAD] seconds=\(elapsed) fraction=\(info.fraction) bytes=\(info.completedBytes)") + fflush(stdout) + } + XCTAssertEqual(catalog.asrModels[index].status, .downloaded) + let modelPath = catalog.asrModelPath(for: id) + XCTAssertFalse(modelPath.isEmpty) + XCTAssertTrue(ModelCatalog.asrRepoContainsRequiredFiles(id, at: URL(fileURLWithPath: modelPath))) + print("CONFUCIUS_CATALOG_DOWNLOAD_BYTES=\(catalog.asrRepoSize(id))") + + await catalog.deleteASR(id) + XCTAssertEqual(catalog.asrModels[index].status, .notDownloaded) + XCTAssertTrue(catalog.asrModelPath(for: id).isEmpty) + } +} diff --git a/Tests/OpenTypeTests/ConfuciusModelDownloaderTests.swift b/Tests/OpenTypeTests/ConfuciusModelDownloaderTests.swift new file mode 100644 index 00000000..82dfb8c3 --- /dev/null +++ b/Tests/OpenTypeTests/ConfuciusModelDownloaderTests.swift @@ -0,0 +1,128 @@ +import CryptoKit +import Foundation +import XCTest +@testable import OpenType + +final class ConfuciusModelDownloaderTests: XCTestCase { + func testParallelRangesReassembleExactWeightAndReportProgress() async throws { + let bytes = Data((0..<103).map { UInt8($0) }) + let session = rangeSession(bytes: bytes) + let output = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) + defer { try? FileManager.default.removeItem(at: output) } + let progress = ProgressValues() + + try await ConfuciusModelDownloader.download( + to: output, + session: session, + source: URL(string: "https://example.test/model.safetensors")!, + size: Int64(bytes.count), + sha256: digest(bytes), + partSize: 16, + maxConcurrent: 4 + ) { value in + progress.append(value) + } + + XCTAssertEqual(try Data(contentsOf: output), bytes) + XCTAssertEqual(progress.last, Int64(bytes.count)) + } + + func testCorruptRangeCannotPublishAWeight() async throws { + let bytes = Data((0..<32).map { UInt8($0) }) + let session = rangeSession(bytes: Data(repeating: 0, count: bytes.count)) + let output = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) + defer { try? FileManager.default.removeItem(at: output) } + + do { + try await ConfuciusModelDownloader.download( + to: output, + session: session, + source: URL(string: "https://example.test/model.safetensors")!, + size: Int64(bytes.count), + sha256: digest(bytes), + partSize: 8, + maxConcurrent: 4 + ) { _ in } + XCTFail("Corrupted content must fail checksum validation") + } catch ConfuciusModelDownloader.Failure.checksumMismatch { + } + } + + func testFullResponseIsRejectedWhenRangeWasRequested() async throws { + let bytes = Data((0..<16).map { UInt8($0) }) + let session = rangeSession(bytes: bytes, statusCode: 200) + let output = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) + defer { try? FileManager.default.removeItem(at: output) } + + do { + try await ConfuciusModelDownloader.download( + to: output, + session: session, + source: URL(string: "https://example.test/model.safetensors")!, + size: Int64(bytes.count), + sha256: digest(bytes), + partSize: 8, + maxConcurrent: 2 + ) { _ in } + XCTFail("A full response must not be written as a ranged part") + } catch ConfuciusModelDownloader.Failure.invalidRangeResponse { + } + } + + private func rangeSession(bytes: Data, statusCode: Int = 206) -> URLSession { + RangeFixtureProtocol.bytes = bytes + RangeFixtureProtocol.statusCode = statusCode + let configuration = URLSessionConfiguration.ephemeral + configuration.protocolClasses = [RangeFixtureProtocol.self] + return URLSession(configuration: configuration) + } + + private func digest(_ data: Data) -> String { + SHA256.hash(data: data).map { String(format: "%02x", $0) }.joined() + } +} + +private final class ProgressValues: @unchecked Sendable { + private let lock = NSLock() + private var values: [Int64] = [] + + func append(_ value: Int64) { + lock.lock() + values.append(value) + lock.unlock() + } + + var last: Int64? { + lock.lock() + defer { lock.unlock() } + return values.last + } +} + +private final class RangeFixtureProtocol: URLProtocol { + static var bytes = Data() + static var statusCode = 206 + + override class func canInit(with request: URLRequest) -> Bool { true } + override class func canonicalRequest(for request: URLRequest) -> URLRequest { request } + + override func startLoading() { + let range = request.value(forHTTPHeaderField: "Range")! + .replacingOccurrences(of: "bytes=", with: "") + .split(separator: "-") + let start = Int(range[0])! + let end = Int(range[1])! + let bytes = Self.bytes + let response = HTTPURLResponse( + url: request.url!, + statusCode: Self.statusCode, + httpVersion: nil, + headerFields: ["Content-Range": "bytes \(start)-\(end)/\(bytes.count)"] + )! + client?.urlProtocol(self, didReceive: response, cacheStoragePolicy: .notAllowed) + client?.urlProtocol(self, didLoad: bytes.subdata(in: start..<(end + 1))) + client?.urlProtocolDidFinishLoading(self) + } + + override func stopLoading() {} +} diff --git a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift index fecc47ed..32f2a37c 100644 --- a/Tests/OpenTypeTests/QwenNativeASREngineTests.swift +++ b/Tests/OpenTypeTests/QwenNativeASREngineTests.swift @@ -9,6 +9,7 @@ final class QwenNativeASREngineTests: XCTestCase { ModelCatalog.defaultASRModels.map(\.id), [ QwenASRModel.defaultID, + QwenASRModel.confuciusR2T2ID, "mlx-community/FireRedASR2-AED-mlx", "mlx-community/Mega-ASR-6bit", ] diff --git a/docs/assets/demos/confucius-mid-sentence.wav b/docs/assets/demos/confucius-mid-sentence.wav new file mode 100644 index 00000000..78ca8c01 Binary files /dev/null and b/docs/assets/demos/confucius-mid-sentence.wav differ diff --git a/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/intent.md b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/intent.md new file mode 100644 index 00000000..0738e4e6 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/intent.md @@ -0,0 +1,62 @@ +# Intent: Confucius4-R2T2 MLX speech recognition in Utter + +**Status:** approved +**Approved-by:** user (chat: "ok,批准") +**Approved-date:** 2026-09-23 +**Upstream:** — + +## Problem + +Utter cannot select or run the user-requested Confucius4-R2T2 speech-recognition +model. The original linked repository contains GGUF weights, whereas Utter's +existing local Qwen3-ASR engine loads MLX weights. + +## Outcome + +A user can select Confucius4-R2T2 as a local speech engine, download +`mlx-community/Confucius4-R2T2-8bit` from within Utter, and transcribe a +completed recording without sending audio to a remote service. + +## Scope + +Affected: local ASR model catalog and download management, speech-engine +selection, model status, and localized UI text. Reuse the existing Qwen3-ASR +MLX runtime if a real load and transcription test confirms compatibility. + +The initial outcome covers completed recordings. Live partial transcripts are +outside this change unless the design review identifies a supported runtime +and the user explicitly expands the scope. + +## Constraints + +- Use the MLX conversion of the same upstream + `netease-youdao/Confucius4-R2T2` model, not the GGUF files. +- Keep recognition on device and preserve the current recording, cancellation, + model-switching, and model-deletion behavior. +- Confirm compatibility with Utter's pinned `mlx-audio-swift` version 0.1.3 + through actual inference; the model card's compatibility claim alone is not + sufficient. +- The repository applies the NetEase Model Use License Agreement. Review + distribution and attribution requirements before shipping binaries or + enabling automated downloads in a release. + +## Acceptance criteria + +- The speech-engine picker identifies Confucius4-R2T2 as a local ASR option, + with English and Chinese UI text. +- Download, completion checks, cancellation, retry, and deletion work for the + MLX repository; missing or empty required files cannot appear ready. +- Selecting the model and ending a recording produces text through Utter's + existing output pipeline, entirely on device. +- Switching away from this model leaves existing speech engines functional. +- A short Chinese clip, an English clip, and a clip ending mid-sentence are + checked against the actual runtime; the latter must retain the final words. +- Contract and failure-path tests cover model readiness and runtime errors; + repository checks, unit tests, and a release-style app build pass. + +## Open questions + +- Does the pinned Swift runtime decode this specific 8-bit conversion correctly, + including the model's stable-prefix end marker and final words? +- Does the model license permit the intended in-app distribution and download + flow, and what attribution is required? diff --git a/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/plan.md b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/plan.md new file mode 100644 index 00000000..cef7b5ce --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/plan.md @@ -0,0 +1,58 @@ +# Plan: Confucius4-R2T2 MLX speech recognition in Utter + +**Status:** approved +**Approved-by:** user (chat: "批准") +**Approved-date:** 2026-09-23 +**Upstream:** spec.md + +## Work items + +- [x] Check the pinned `mlx-audio-swift` loader against the actual 8-bit + checkpoint in an isolated local probe using existing Chinese and English + samples. Record load result, output, time, memory, and any `|` suffix. If + loading fails because the pinned dependency lacks a required feature, stop + for a design amendment before changing dependencies. +- [x] Add focused contract tests first for the Qwen-compatible model list, + selected-model identity, required MLX files, and size/hint consistency. + Reuse the existing Qwen integration test seam for the runtime probe. +- [x] Add `mlx-community/Confucius4-R2T2-8bit` to the ASR catalog and Qwen3 + model list, with a matching required-file contract and repository-size + estimate. Keep the existing model selection and download transaction path. +- [x] Make the Qwen3 model list's active row follow `settings.qwenASRModel`. + Add the English and Chinese model hint and update the section description + so it covers both Qwen-compatible MLX checkpoints. +- [x] Run a mid-sentence stop sample. If final words are lost, add the + smallest model-specific silence-padding change and a regression test that + distinguishes recovered text from merely removing `|`. +- [x] Review the exact NetEase license and conversion notices for the in-app + download flow. Provide the required model-license link and retain notices; + record the release owner's license decision before any distribution. +- [x] Verify the parallel HTTP application catalog download publishes the full + Confucius repository, then deletes it through the same catalog path. +- [x] Remove any temporary probe code and keep only a single active Qwen + runtime path. Review the final diff for stale selection assumptions and + changes outside this model's boundary. + +## Verification plan + +- [x] Focused catalog, settings, model-readiness, and failure-path tests. +- [x] Actual 8-bit model inference on Chinese, English, and abruptly ended + clips; capture output and resource measurements. +- [ ] Cancel/retry and active-model delete/switch checks through the existing + model-management path. +- [x] `bash scripts/sdlc-checks.sh` +- [x] `bash scripts/ci-basic-checks.sh` +- [x] `swift test` +- [x] `bash scripts/build-app.sh --app-only` (release-style Metal packaging). +- [x] Real-window model-selection and download UI review in light and dark + appearances, including the active-row state after switching models. +- [x] Record results, skipped checks, and residual risk in `verification.md`. + +## Human gates + +- This plan needs explicit approval before implementation begins, per + `docs/sdlc/README.md`. +- A failed compatibility probe that requires a dependency or architecture + change returns to design approval. +- Verification and license/distribution decisions need owner review before + PR merge or release. No model weights are bundled in the app artifact. diff --git a/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/spec.md b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/spec.md new file mode 100644 index 00000000..36cfe29d --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/spec.md @@ -0,0 +1,126 @@ +# Spec: Confucius4-R2T2 MLX speech recognition in Utter + +**Status:** approved +**Approved-by:** user (chat: "继续,并批准,并继续。") +**Approved-date:** 2026-09-23 +**Upstream:** intent.md + +## Context + +The selected repository is `mlx-community/Confucius4-R2T2-8bit`, an MLX +conversion of `netease-youdao/Confucius4-R2T2`. Its published `config.json` +declares `Qwen3ASRForConditionalGeneration` and 8-bit affine quantization. It +contains `model.safetensors`, its index, tokenizer files, and an audio +preprocessor configuration. The repository totals about 2.48 GB. + +Utter already has one Qwen3-ASR path: `AppSettings.qwenASRModel` selects a +repository ID; `ModelCatalog` downloads and validates it; `VoicePipeline` and +`SpeechEngineProvider` create `QwenNativeASREngine` for that model directory; +the engine calls `Qwen3ASRModel.fromModelDirectory` from the pinned +`mlx-audio-swift` 0.1.3 dependency. Completed audio is converted to 16 kHz +mono PCM and sent through the existing output pipeline. There is no GGUF +runtime. The model card asserts Swift loader compatibility, but this exact +repository and pinned dependency have not been tested together in Utter. + +The Qwen3 model list currently only returns the default model. Its active-row +indicator uses `SpeechEngineType.asrModelID`, which is always the default Qwen +ID even when `settings.qwenASRModel` changes. That must be corrected when a +second Qwen-compatible model is offered. + +## Design + +- Register the exact MLX repository ID in `ModelCatalog.defaultASRModels` and + show it in the existing Qwen3-ASR model list. Keep one persisted selection, + `settings.qwenASRModel`, as the source of truth. The Qwen3 list's active row + reads that setting. No second runtime state or speech-engine enum case is + needed. +- Require the same complete MLX file set as the existing Qwen3-ASR model, + including weights, index, config, preprocessor, and tokenizer files. Reuse + the generation-staged Hub download, cancellation, retry, and deletion path. + Add an explicit download-size estimate near the repository metadata value. +- Reuse `QwenNativeASREngine` and its path-based reload behavior. First prove + an actual load and Chinese/English transcription with the pinned Swift + dependency. If 8-bit loading fails, diagnose that compatibility issue before + considering a dependency update; do not silently route to another model. +- Keep whole-recording recognition. The R2T2 model is trained for stable-prefix + streaming, but Utter's current Swift Qwen path calls whole-clip `generate`. + Check a recording stopped mid-sentence for a trailing `|` or lost final + words. If it reproduces, add model-specific tail silence to the prepared + audio and verify recovery; stripping `|` alone is insufficient. +- Localize the model name/hint and make the Qwen3 section text accurately + describe both selectable MLX models. The model remains an ASR option, never + an LLM option. + +This is an extension of the existing model boundary. Repeated small patches to +callers would duplicate selection state; a replacement of the Qwen engine would +add migration and regression cost without a demonstrated need. A whole-app or +whole-ASR rewrite would require substantially more validation and has no +benefit for this model addition. + +### Download amendment, approved 2026-09-23 + +The first live `ModelCatalog.downloadASR` test found that Hub snapshot progress +counts completed files, so it remained unchanged while the 2.46 GB weight was +transferring and the shared 120-second watchdog cancelled the transfer. The +user approved an upstream Xet probe, but it stalled near the tail after 17 +minutes. The user then approved an application-side parallel HTTP fallback for +this model. A later live test also found Hub could stall on a small file before +reaching the weight. Download the model's files from one pinned revision in +bounded HTTP ranges, with 16 independent URL sessions because one shared +HTTP/2 connection was much slower in a local probe. Require HTTP 206 with the +exact `Content-Range` and length and verify each file against its pinned +SHA-256 before publishing. Report actual completed bytes to the existing +progress tracker and stall watchdog. Retain generation staging, cancellation, +deletion, and the commit path. Other models continue using the existing Hub +download. A complete live catalog download is required before merge. + +## Safety and failure modes + +- Audio remains local. Downloaded weights come from the named Hugging Face + repository through the existing Hub client. No audio is sent to that host. +- A partial or empty weight/config/tokenizer file must remain incomplete. + A load failure is surfaced as an error, not a ready model or a silent + substitute. Existing download operations remain serialized by model ID. +- Switching between the two Qwen-compatible models creates the engine from + the selected path. Deleting the active model unloads it and leaves the + selection visibly unavailable until another model is chosen or downloaded. +- The NetEase license is not MIT. Its English text requires retaining notices + and the agreement in copies of the model, and imposes downstream terms. + Distribution and in-app download presentation require owner/legal review + before release. Keep a link to the model license and conversion notice in + the model UI or accompanying documentation; do not bundle weights in the app. + +## Test strategy + +- Contract tests: model list and persisted selection, active-row selection, + required-file validation (including missing/zero-byte weights), and + download-size hint consistency. +- Failure tests: incomplete download, cancellation/retry, failed model load, + and switch/delete of the active model. Reuse existing download tests where + their behavior is already covered; add only missing assertions. +- Runtime probe on Apple Silicon: load the downloaded 8-bit model using the + pinned Swift package, transcribe one Chinese and one English clip, then a + clip ending mid-sentence. Record exact output, elapsed time, memory use, + and whether tail padding is needed. +- Run `bash scripts/sdlc-checks.sh`, `bash scripts/ci-basic-checks.sh`, + `swift test`, a release-style build, and real-window light/dark checks of + the model selection/download UI after implementation. Record evidence in + `verification.md`. + +## Rollout and rollback + +Do not release until the runtime probe and license review pass. The feature +adds a selectable model and does not change the default. If load or output +quality fails, remove or hide only the new catalog entry and keep existing +models and their stored files untouched. If a release is rolled back, the +prior app continues to use its existing Qwen model; the new cached repository +can be removed through model management or storage cleanup. + +## Design approval decisions + +- Approve placing Confucius4 beneath the existing Qwen3-ASR engine picker + rather than adding a separate engine segment. +- Approve whole-recording recognition as the initial scope. Streaming partial + text is a separate feature with a different runtime contract. +- Release remains gated on successful pinned-runtime inference and model + license review. diff --git a/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/verification.md b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/verification.md new file mode 100644 index 00000000..b40e53f1 --- /dev/null +++ b/docs/sdlc/changes/2026-09-23-confucius4-r2t2-mlx/verification.md @@ -0,0 +1,38 @@ +# Verification: Confucius4-R2T2 MLX speech recognition in Utter + +**Status:** approved +**Approved-by:** user (chat: "批准") +**Approved-date:** 2026-09-23 +**Upstream:** plan.md + +## Evidence + +| Check | Result | Evidence | +|---|---|---| +| Pinned Swift MLX runtime | Pass | `mlx-audio-swift` 0.1.3 loaded the actual `mlx-community/Confucius4-R2T2-8bit` repository. English output: `Hey, so um, I wanted to, I wanted to follow up on the design doc we talked about.` Chinese output contained `周五之前能review一下?`. | +| Mid-sentence stop | Pass | A 3.1 s cut of `en-sample.m4a` ended in `I wanted to|` without padding; with 0.5 s of silence it continued to `I wanted to follow` without a trailing marker. The regression clip is `docs/assets/demos/confucius-mid-sentence.wav`. | +| Focused model tests | Pass | `OPENTYPE_CONFUCIUS_MODEL_PATH=/tmp/utter-confucius4-r2t2-8bit swift test --filter ConfuciusASRTests`: 4 tests, 0 failures. A final timed run took 14.19 s including SwiftPM setup; XCTest took 1.83 s. `/usr/bin/time -l` reported 2,682,716,160 bytes maximum resident set size for the command and child test process, not an isolated model-only peak. | +| `bash scripts/sdlc-checks.sh` | Pass | `SDLC checks passed.` | +| `bash scripts/ci-basic-checks.sh` | Pass | `Basic CI checks passed.` Localization and repository invariants passed. | +| `swift test` | Pass | Latest run: 756 XCTest executed, 16 environment-gated skips, 0 failures; one Swift Testing test passed. The full live network download was run separately below. | +| Release-style app build | Pass | Final `bash scripts/build-app.sh --app-only` passed after the HTTP fallback; `dist/Utter.app` assembled with `default.metallib` and passed artifact verification. This local build used ad-hoc signing and is not a production release. | +| Real-window UI render | Pass | An AppKit `NSWindow` hosted the model-management view at 760 × 680 in light and dark appearances. Reviewed `review-evidence/2026-09-23-confucius4-r2t2-mlx/light.png` and `dark.png`: Confucius4 row, active badge, download action, hint, Chinese license, English license, and conversion notice links are visible without clipping or contrast loss. The native interaction channel failed to start, so button clicks were not observed in the running app. | +| Parallel HTTP catalog download | Pass | `OPENTYPE_CONFUCIUS_LIVE_DOWNLOAD=1 swift test --filter ConfuciusASRTests/testApplicationCatalogDownloadsAndDeletesModel` completed in 1704.575 s. The application catalog downloaded 2,479,312,565 bytes from the pinned revision, verified SHA-256 for every file, published a complete model directory, then deleted it through `deleteASR`. The test used an isolated temporary storage root. | +| Range integrity and failure injection | Pass | `swift test --filter ConfuciusModelDownloaderTests`: exact range reconstruction, full-response rejection, and corrupt-content checksum rejection passed. | + +## Acceptance criteria + +- Model selection and localized labels — pass for catalog contract, localization parity, and rendered selected-model state. English strings were validated by parity checks; the window captures used Chinese UI language. +- Download completion — pass for the full application catalog network path, file contract, and SHA-256 integrity of all files, including weights and license notices. +- Cancellation, retry, deletion, and switching — generic model-management and download-recovery tests passed in the full suite; exact-model deletion passed in the live catalog test. Native UI clicks were not observed. +- On-device transcription — pass for English, Chinese, and mid-sentence clips with the pinned runtime and model-specific tail silence. +- Existing engines remain functional — full suite passed; the default Qwen model is not modified by the tail-padding branch. + +## Residual risk + +- The user confirmed the model-license decision in chat on 2026-09-23. The repository's Chinese agreement prevails and says the separate-license revenue threshold is RMB 100 million, while its English copy says RMB 1 billion. The app links both licenses and the conversion notice and requires the downloaded `LICENSE`, `MODEL_LICENSE_zh`, and `NOTICE` files; weights are not bundled. +- Native UI button interaction was not observed because the computer-use native pipe failed. The exact `ModelCatalog.downloadASR` and `deleteASR` path used by the UI passed its live network test. + +## Decision + +The full application catalog download and delete check passed. Release remains subject to the normal PR and production gates; native UI button interaction was not observed. diff --git a/review-evidence/2026-09-23-confucius4-r2t2-mlx/dark.png b/review-evidence/2026-09-23-confucius4-r2t2-mlx/dark.png new file mode 100644 index 00000000..db2e31bd Binary files /dev/null and b/review-evidence/2026-09-23-confucius4-r2t2-mlx/dark.png differ diff --git a/review-evidence/2026-09-23-confucius4-r2t2-mlx/light.png b/review-evidence/2026-09-23-confucius4-r2t2-mlx/light.png new file mode 100644 index 00000000..9151a376 Binary files /dev/null and b/review-evidence/2026-09-23-confucius4-r2t2-mlx/light.png differ