diff --git a/ios/Podfile.lock b/ios/Podfile.lock index 77f53258d..c9050a312 100644 --- a/ios/Podfile.lock +++ b/ios/Podfile.lock @@ -77,9 +77,9 @@ PODS: - QCloudTrack/Beacon (6.4.7) - record_darwin (1.0.0): - Flutter - - SDWebImage (5.21.1): - - SDWebImage/Core (= 5.21.1) - - SDWebImage/Core (5.21.1) + - SDWebImage (5.21.0): + - SDWebImage/Core (= 5.21.0) + - SDWebImage/Core (5.21.0) - SDWebImageWebPCoder (0.14.6): - libwebp (~> 1.0) - SDWebImage/Core (~> 5.17) @@ -200,7 +200,7 @@ SPEC CHECKSUMS: QCloudCOSXML: 7205e76aa9cf613468222615483c9d7c074e4298 QCloudTrack: 3b53a7fc4fe3920e407f2aa73f2452992a61f7f3 record_darwin: fb1f375f1d9603714f55b8708a903bbb91ffdb0a - SDWebImage: f29024626962457f3470184232766516dee8dfea + SDWebImage: f84b0feeb08d2d11e6a9b843cb06d75ebf5b8868 SDWebImageWebPCoder: e38c0a70396191361d60c092933e22c20d5b1380 spotify_sdk: a48400bb29f70c4fe251ebfdc9135c37097ac5ca tencentcloud_cos_sdk_plugin: 7bed564dbe72df23e7b5cacb1e51b1a20cb1639c diff --git a/ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme b/ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme index c3fedb29c..95d6e55f2 100644 --- a/ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme +++ b/ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme @@ -1,7 +1,7 @@ + version = "1.7"> diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt index bd6ffc884..36891f2b8 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -669,7 +669,9 @@ class AzureAsrHelper(private val context: Context) { } - +/** + * 开启音频写入线程 + */ private fun startWriteThread() { writeThread = Thread { try { @@ -745,6 +747,9 @@ class AzureAsrHelper(private val context: Context) { } + /** + * 开始音频输入 + */ fun startAudioRecord() { isWriting.set(true) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index f35a6561e..37db68014 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -5,7 +5,7 @@ import MicrosoftCognitiveServicesSpeech import speech import os.log - +// MARK: - 基于微软Azure语音服务的ASR实现 /** * Azure ASR Helper * @@ -48,8 +48,8 @@ public class AzureAsrHelper: NSObject { // 音频处理 public var audioStream: AudioStream? - - public var recordfile: RecordFile? + //录音文件 + public var recordfile: RecordFile? /** * 初始化Azure语音服务 * @@ -74,7 +74,6 @@ public class AzureAsrHelper: NSObject { // 释放之前的资源 dispose() - print("w[w[w]]") // 保存配置 self.subscriptionKey = subscriptionKey self.region = region @@ -108,8 +107,8 @@ public class AzureAsrHelper: NSObject { // 设置指定的识别语言 speechConfig?.speechRecognitionLanguage = currentLanguage } - // 录音文件类 - recordfile = RecordFile() + // 录音文件类 + recordfile = RecordFile() // 创建识别器 return true } catch { @@ -147,12 +146,10 @@ public class AzureAsrHelper: NSObject { // callback.onError("重置识别器失败") return false } - return false + return false } - - // 启动音频处理 - audioStream?.startAudioRecord() - + // 启动音频处理 + audioStream?.startAudioInput() // 开始连续识别 try recognizer?.startContinuousRecognition() @@ -162,9 +159,8 @@ public class AzureAsrHelper: NSObject { return true } catch { - audioStream?.stopMicrophoneCapture() - // stopAudioProcessing() - _isContinuousRecognitionActive = false + audioStream?.stopAudioCapture() + _isContinuousRecognitionActive = false //callback.onError("启动连续识别失败: \(error.localizedDescription)") return false } @@ -176,42 +172,40 @@ public class AzureAsrHelper: NSObject { * @return 是否成功停止 */ public func stopContinuousRecognition() -> Bool { - print("stopContinuousRecognition") + print("stopContinuousRecognition") guard speechConfig != nil else { os_log("语音服务未初始化", log: log, type: .error) return false } - guard let recognizer = recognizer else { - os_log("识别器为空,重置状态", log: log, type: .info) - return true - } - + guard let recognizer = recognizer else { + os_log("识别器为空,重置状态", log: log, type: .info) + return true + } + if !_isContinuousRecognitionActive { return true } do { - _isContinuousRecognitionActive = false - + _isContinuousRecognitionActive = false + if audioSourceType == .external { //pushAudioData(data: Data()) } - + + audioStream?.stopAudioCapture() // 停止连续识别 try recognizer.stopContinuousRecognition() - - // 停止音频处理 - //stopAudioProcessing() - audioStream?.stopMicrophoneCapture() + // 会话结束事件会设置_isContinuousRecognitionActive = false return true } catch { // 强制重置状态 _isContinuousRecognitionActive = false - os_log("强制停止识别失败", log: log, type: .error) - + os_log("强制停止识别失败", log: log, type: .error) + // 停止音频处理 - audioStream?.stopMicrophoneCapture() + audioStream?.stopAudioCapture() return false } @@ -229,7 +223,7 @@ public class AzureAsrHelper: NSObject { */ public func dispose() { do { - print("释放所有资源:") + print("释放所有资源:") // 如果正在进行连续识别,先停止 if _isContinuousRecognitionActive { // 直接停止,不等待结果 @@ -238,7 +232,7 @@ public class AzureAsrHelper: NSObject { } // 停止音频处理 - stopAudioProcessing() + audioStream?.releaseAudioResources() // 释放资源 recognizer = nil @@ -257,8 +251,7 @@ public class AzureAsrHelper: NSObject { speechConfig = nil } } - - // MARK: - 私有方法 + /** * 设置识别器 @@ -294,29 +287,29 @@ public class AzureAsrHelper: NSObject { return true } catch { os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) - stopAudioProcessing() + try? recognizer?.stopContinuousRecognition() return false } } - + /** * 设置麦克风流 - 使用推流方式 */ private func setupMicrophoneStream() { do { - - if (audioStream == nil) { + + if (audioStream == nil) { // 创建外部音频拉流对象 - // 明确指定使用外部音频源(推流模式) - audioStream = AudioStream(audioSourceType: audioSourceType) + // 明确指定使用外部音频源(推流模式) + audioStream = AudioStream(audioSourceType: audioSourceType) audioStream?.initAudioRecord() } - + // 创建音频配置 if let pushStream = audioStream?.pushAudioStream { // Safely unwrap optional audioConfig = SPXAudioConfiguration(streamInput: pushStream) - os_log("设置麦克风流:") + os_log("设置麦克风流:") } } catch { os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription) @@ -328,13 +321,13 @@ public class AzureAsrHelper: NSObject { * 设置事件监听器 */ public func setupEventListeners(callback: ContinuousRecognizeCallback) -> Bool{ - print("设置ssssss监听器:${speechConfig}") - + print("设置ssssss监听器:${speechConfig}") + // 重设识别器 if (!setupRecognizer()) { return false } - guard let recognizer = recognizer else { return false} + guard let recognizer = recognizer else { return false} // 识别中事件 recognizer.addRecognizingEventHandler { [weak self] (sender, event) in guard let self = self else { return } @@ -349,7 +342,7 @@ public class AzureAsrHelper: NSObject { // 识别完成事件 recognizer.addRecognizedEventHandler { [weak self] (sender, event) in guard let self = self else { return } - print("识别完成事件:") + print("识别完成事件:") let result = event.result if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) @@ -362,13 +355,13 @@ public class AzureAsrHelper: NSObject { recognizer.addSessionStartedEventHandler { (sender, event) in // 直接在当前线程调用回调 callback.onSessionStarted() - print("会话开始事件:") + print("会话开始事件:") } // 会话结束事件 recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in guard let self = self else { return } - print("会话结束事件:") + print("会话结束事件:") // 直接在当前线程调用回调 // callback.onSessionStopped() // self._isContinuousRecognitionActive = false @@ -378,7 +371,7 @@ public class AzureAsrHelper: NSObject { // 取消事件 recognizer.addCanceledEventHandler { [weak self] (sender, event) in guard let self = self else { return } - print("取消事件:") + print("取消事件:") let errorDetails = event.errorDetails ?? "未知错误" let reason = String(describing: event.reason.rawValue) @@ -411,73 +404,87 @@ public class AzureAsrHelper: NSObject { return "" } } - - + /** - * 停止音频处理 + * 禁用蓝牙音频功能,切换回正常音频模式 */ - private func stopAudioProcessing() { -print("stopAudioProcessing") - // if let stream = externalAudioStream { - // stream.close() - // externalAudioStream = nil - // // os_log("外部音频流已关闭", log: log, type: .info) - // } + public func disableBluetoothAudio() { + + do { + audioStream?.setAudioOutputRoute(.speaker) + } catch { + print("disableBluetoothAudio") + } + } + /** + * 恢复原始音频设备状态(通常是重新启用蓝牙) + */ + public func restoreOriginalAudioState() { + + do { + audioStream?.setAudioOutputRoute(.receiver) + } catch { + print("restoreOriginalAudioState") + } } - //录音音频文件 + + // MARK: - 录音音频文件方法 + /** * 开启录音 */ public func enableRecord(filePath: String) { - - + + recordfile?.closeFile(isSave: true) recordfile?.creatingFiles(atPath: filePath) // Fixed method call - + } - + /** * 移动文件到新路径 */ public func moveFile(sourcePath: String, destPath: String)-> Bool { - - - recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call + + + recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call return true } - - + + /** * 重命名指定路径的音频文件 */ public func renameFile(filePath: String, newName: String)-> Bool { - - - - recordfile?.renameFile(at: filePath, to: newName) // Fixed method call + + + + recordfile?.renameFile(at: filePath, to: newName) // Fixed method call return true } - + /** * 停止录音 */ public func pauseRecord() { - - + + recordfile?.isPause = true } - - + + /** * 关闭录音 */ public func stopRecord(isSave: Bool) { - - + + recordfile?.isPause = false - recordfile?.closeFile(isSave: true) + recordfile?.closeFile(isSave: true) } + // MARK: - 音频流类 + public class AudioStream: NSObject { public private(set) var pushAudioStream: SPXPushAudioInputStream? private let writeQueue = LinkedBlockingQueue() @@ -485,18 +492,24 @@ print("stopAudioProcessing") private var audioFormat: AVAudioFormat? private let audioSourceType: AudioSourceType // 添加引用 private var isRunning = false - private var isWriting = false + public var isWriting = false private let bufferSize: Int = 4096 private let writeThread = DispatchQueue(label: "audio.stream.writer") + init(audioSourceType: AudioSourceType) { self.audioSourceType = audioSourceType } + /** + * 初始化 + */ public func initAudioRecord() { audioFormat = getOptimalAudioFormat() pushAudioStream = SPXPushAudioInputStream() isRunning=true startWriteThread() + } + /// 获取最佳音频格式 (iOS 通常支持标准采样率) public func getOptimalAudioFormat() -> AVAudioFormat? { let sampleRate: Double = 16000 // iOS 通常支持 16kHz @@ -507,35 +520,41 @@ print("stopAudioProcessing") interleaved: true ) } - public func startAudioRecord() { + /** + * 开始音频输入 + */ + public func startAudioInput() { isWriting=true - print("startAudioRecord=audioSourceType\(audioSourceType)") + print("startAudioInput=audioSourceType\(audioSourceType)") switch audioSourceType { case .microphone: - startMicrophoneCapture() + runMicrophoneCapture() case .external: - startExternalCapture() + runExternalCapture() } } - - public func startWriteThread() { - writeThread.async { [weak self] in - guard let self = self else { return } - - while self.isRunning { - if !self.isWriting { - usleep(10_000) - continue + + /** + * 开启音频写入线程 + */ + public func startWriteThread() { + writeThread.async { [weak self] in + guard let self = self else { return } + + while self.isRunning { + if !self.isWriting { + usleep(10_000) + continue + } + + guard let dataToWrite = self.writeQueue.take() else { continue } + + //print("写入数据长度: \(dataToWrite.count)") + self.pushAudioStream?.write(dataToWrite) + } } - - guard let dataToWrite = self.writeQueue.take() else { continue } - - print("写入数据长度: \(dataToWrite.count)") - self.pushAudioStream?.write(dataToWrite) } - } -} - + /** * 向音频流写入音频数据 * 仅当音频源设置为external时有效 @@ -550,198 +569,178 @@ print("stopAudioProcessing") // 放入队列,由写线程写入 writeQueue.put(data) } - private func startMicrophoneCapture() { - do { - let audioSession = AVAudioSession.sharedInstance() - // 改用 playAndRecord 模式(同时支持播放和录音) - try audioSession.setCategory( - .playAndRecord, - mode: .default, - options: [.allowBluetooth, .defaultToSpeaker] - ) - try audioSession.setActive(true, options: .notifyOthersOnDeactivation) - - audioEngine = AVAudioEngine() - guard let inputNode = audioEngine?.inputNode else { - throw NSError(domain: "AudioSetup", code: 1) - } - - // Use hardware's native format - let hardwareFormat = inputNode.inputFormat(forBus: 0) - - // Create converter to target format - guard let targetFormat = audioFormat, - let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { - throw NSError(domain: "AudioSetup", code: 2) - } - - inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) { - [weak self] buffer, time in - guard let self = self, self.isWriting else { return } - - // Convert to target format - let convertedBuffer = AVAudioPCMBuffer( - pcmFormat: targetFormat, - frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate) -)! -///print("进入 tap 回调,frameLength: \(buffer.frameLength)") - var error: NSError? - let status = converter.convert( - to: convertedBuffer, - error: &error, - withInputFrom: { inNumPackets, outStatus in - outStatus.pointee = .haveData - return buffer - } - ) - //print("转换状态: \(status.rawValue), 错误: \(String(describing: error))") - if status == .haveData, error == nil { - let data = self.audioBufferToData(convertedBuffer) - // print("准备写入数据,大小: \(data.count)") - - self.writeQueue.put(data) + private func runMicrophoneCapture() { + do { + //self.setAudioOutputRoute(.speaker) + audioEngine = AVAudioEngine() + guard let inputNode = audioEngine?.inputNode else { + throw NSError(domain: "AudioSetup", code: 1) + } + // 推荐直接用系统 format + let hardwareFormat = inputNode.inputFormat(forBus: 0) + // // ===== 新增:启用专业级语音处理 ===== + if #available(iOS 13.0, *) { + try inputNode.setVoiceProcessingEnabled(true) + print("Voice processing enabled") + + } + + // Create converter to target format + guard let targetFormat = audioFormat, + let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { + throw NSError(domain: "AudioSetup", code: 2) + } + + inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) { + [weak self] buffer, time in + guard let self = self, self.isWriting else { return } + + // Convert to target format + let convertedBuffer = AVAudioPCMBuffer( + pcmFormat: targetFormat, + frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate) + )! + ///print("进入 tap 回调,frameLength: \(buffer.frameLength)") + var error: NSError? + let status = converter.convert( + to: convertedBuffer, + error: &error, + withInputFrom: { inNumPackets, outStatus in + outStatus.pointee = .haveData + return buffer + } + ) + //print("转换状态: \(status.rawValue), 错误: \(String(describing: error))") + + if status == .haveData, error == nil { + let data = self.audioBufferToData(convertedBuffer) + // print("准备写入数据,大小: \(data.count)") + + self.writeQueue.put(data) + + } + } + + try audioEngine?.start() + } catch { + print("麦克风启动失败: \(error)") } } - - try audioEngine?.start() - } catch { - print("麦克风启动失败: \(error)") - } -} - - private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { - let frameLength = Int(buffer.frameLength) - let channelCount = 1 - let dataLength = frameLength * channelCount * MemoryLayout.size - - // Handle 16-bit integer format - if let int16Data = buffer.int16ChannelData { - return Data( - bytes: int16Data.pointee, - count: dataLength - ) - } - // Handle float format - else if let floatData = buffer.floatChannelData { - var int16Array = [Int16](repeating: 0, count: frameLength) - let floatBuffer = floatData.pointee - - for i in 0.. Data { + let frameLength = Int(buffer.frameLength) + let channelCount = 1 + let dataLength = frameLength * channelCount * MemoryLayout.size + + // Handle 16-bit integer format + if let int16Data = buffer.int16ChannelData { + return Data( + bytes: int16Data.pointee, + count: dataLength + ) + } + // Handle float format + else if let floatData = buffer.floatChannelData { + var int16Array = [Int16](repeating: 0, count: frameLength) + let floatBuffer = floatData.pointee + + for i in 0..() - private var closed = false - - override init() { - super.init() - - pullStream = SPXPullAudioInputStream( - readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in - guard let self = self else { return 0 } - return self.read(buffer: data, size: Int(size)) - }, - closeHandler: { [weak self] in - self?.close() + + // 音频路由管理 + public enum AudioOutputRoute { + case speaker + case receiver + case bluetooth + } + public func setAudioOutputRoute(_ route: AudioOutputRoute) { + let audioSession = AVAudioSession.sharedInstance() + do { + isWriting=false + try audioEngine?.stop() + // 先停用以避免冲突 + try audioSession.setActive(false) + + switch route { + case .speaker: + // 使用扬声器时必须用videoChat模式 + try audioSession.setCategory( + .playAndRecord, + mode: .videoChat, + options: [.allowBluetooth, .mixWithOthers, .defaultToSpeaker] + ) + try audioSession.overrideOutputAudioPort(.speaker) + + case .receiver: + // 听筒模式使用voiceChat节省资源 + try audioSession.setCategory( + .playAndRecord, + mode: .voiceChat, + options: [.allowBluetooth] + ) + try audioSession.overrideOutputAudioPort(.none) + + case .bluetooth: + // 完整蓝牙设备支持 + try audioSession.setCategory( + .playAndRecord, + mode: .voiceChat, + options: [.allowBluetooth, .allowBluetoothA2DP] + ) + try audioSession.overrideOutputAudioPort(.none) + // 不需要override,系统自动路由 } - ) - } - - /** - * 外部调用:推送音频数据到队列 - * @param data 音频数据 - */ - func pushAudio(_ data: Data) { - if !closed { - queue.put(data) - } - } - - /** - * SDK调用:从队列中拉取数据 - * @param buffer SDK提供的缓冲区 - * @param size 缓冲区大小 - * @return 读取的字节数,0表示流结束 - */ - private func read(buffer: NSMutableData, size: Int) -> Int { - // 阻塞等待下一块数据 - guard let chunk = queue.take() else { - return 0 // 队列已关闭 - } + + // 重新激活 + try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) - // 如果是空数据,表示流结束 - if chunk.isEmpty { - return 0 + try audioEngine?.start() + isWriting=true + } catch { + print("路由切换失败: \(error)") } - - let toCopy = min(chunk.count, size) - buffer.append(chunk.prefix(toCopy)) - return toCopy - } - - /** - * SDK调用:关闭流 - */ - func close() { - closed = true - queue.close() } + } + // MARK: - iOS版LinkedBlockingQueue实现 /** * iOS版LinkedBlockingQueue实现 @@ -819,7 +818,7 @@ print("stopAudioProcessing") } } } - + // MARK: - 回调函数 /** * 连续识别回调接口 */ diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index daa11d423..955bc4466 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -37,39 +37,39 @@ import os.log private var currentAsrCallback: AsrCallbackWrapper? // 插件注册 - public static func register(with registrar: FlutterPluginRegistrar) { - let instance = AzureSpeechPlugin() - - // 初始化ASR通道 - let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) - registrar.addMethodCallDelegate(instance, channel: asrChannel) - instance.asrChannel = asrChannel - - // 设置ASR通道处理器 - 使用实例方法 - asrChannel.setMethodCallHandler { [weak instance] (call, result) in - instance?.handleAsrMethodCall(call, result: result) - } - - // 初始化TTS通道 - let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) - registrar.addMethodCallDelegate(instance, channel: ttsChannel) - instance.ttsChannel = ttsChannel - - // 设置TTS通道处理器 - 使用实例方法 - ttsChannel.setMethodCallHandler { [weak instance] (call, result) in - instance?.handleTtsMethodCall(call, result: result) + public static func register(with registrar: FlutterPluginRegistrar) { + let instance = AzureSpeechPlugin() + + // 初始化ASR通道 + let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) + registrar.addMethodCallDelegate(instance, channel: asrChannel) + instance.asrChannel = asrChannel + + // 设置ASR通道处理器 - 使用实例方法 + asrChannel.setMethodCallHandler { [weak instance] (call, result) in + instance?.handleAsrMethodCall(call, result: result) + } + + // 初始化TTS通道 + let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) + registrar.addMethodCallDelegate(instance, channel: ttsChannel) + instance.ttsChannel = ttsChannel + + // 设置TTS通道处理器 - 使用实例方法 + ttsChannel.setMethodCallHandler { [weak instance] (call, result) in + instance?.handleTtsMethodCall(call, result: result) + } + + // 初始化ASR事件通道 + let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) + asrEventChannel.setStreamHandler(instance) + instance.asrEventChannel = asrEventChannel + + // 初始化TTS事件通道 + let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger()) + ttsEventChannel.setStreamHandler(instance) + instance.ttsEventChannel = ttsEventChannel } - - // 初始化ASR事件通道 - let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) - asrEventChannel.setStreamHandler(instance) - instance.asrEventChannel = asrEventChannel - - // 初始化TTS事件通道 - let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger()) - ttsEventChannel.setStreamHandler(instance) - instance.ttsEventChannel = ttsEventChannel -} // 发送ASR事件方法 internal func sendAsrEvent(_ event: [String: Any]) { if asrEventSink == nil { @@ -104,7 +104,7 @@ import os.log } } - + // // 处理Flutter方法调用 // public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { // switch call.method { @@ -120,7 +120,7 @@ import os.log // MARK: - ASR 方法处理 - private func handleAsrMethodCall(_ call: FlutterMethodCall, result: @escaping FlutterResult) { + private func handleAsrMethodCall(_ call: FlutterMethodCall, result: @escaping FlutterResult) { switch call.method { case "initialize": guard let args = call.arguments as? [String: Any], @@ -171,7 +171,7 @@ import os.log audioSourceType: audioSourceType ) result(success) - case "recognizeCallback": + case "recognizeCallback": // 确保事件通道已准备好 guard asrEventSink != nil else { result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil)) @@ -185,7 +185,6 @@ import os.log ) result(success) case "stopContinuousRecognition": - print("stopContinuousRecognition") let success = azureAsrHelper.stopContinuousRecognition() currentAsrCallback = nil result(success) @@ -194,7 +193,7 @@ import os.log result(azureAsrHelper.isContinuousRecognitionActive()) case "dispose": - print("dispose") + print("dispose") azureAsrHelper.dispose() currentAsrCallback = nil result(true) @@ -205,81 +204,91 @@ import os.log result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil)) return } - + // 安全解包版本 -guard let audioStream = azureAsrHelper.audioStream else { - os_log("音频流未初始化", type: .error) - return -} -audioStream.saveAudioDataTo(data: audioBytes.data) + guard let audioStream = azureAsrHelper.audioStream else { + os_log("音频流未初始化", type: .error) + return + } + audioStream.saveAudioDataTo(data: audioBytes.data) - - result(true) - case "renameFile": - guard let args = call.arguments as? [String: Any], - let filePath = args["filePath"] as? String, - let newName = args["newName"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) - return - } - do { - print("音频文件名称为: \(filePath)") - try azureAsrHelper.renameFile(filePath: filePath, newName: newName) result(true) - } catch { - result(FlutterError(code: "RENAMEFILE_ERROR", message: error.localizedDescription, details: nil)) - } - - case "moveFile": - guard let args = call.arguments as? [String: Any], - let sourcePath = args["sourcePath"] as? String, - let destPath = args["destPath"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) - return - } - do { - print("音频文件名称为: \(sourcePath)") - try azureAsrHelper.moveFile(sourcePath: sourcePath, destPath: destPath) + + case "renameFile": + guard let args = call.arguments as? [String: Any], + let filePath = args["filePath"] as? String, + let newName = args["newName"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) + return + } + do { + print("音频文件名称为: \(filePath)") + try azureAsrHelper.renameFile(filePath: filePath, newName: newName) + result(true) + } catch { + result(FlutterError(code: "RENAMEFILE_ERROR", message: error.localizedDescription, details: nil)) + } + + case "moveFile": + guard let args = call.arguments as? [String: Any], + let sourcePath = args["sourcePath"] as? String, + let destPath = args["destPath"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) + return + } + do { + print("音频文件名称为: \(sourcePath)") + try azureAsrHelper.moveFile(sourcePath: sourcePath, destPath: destPath) + result(true) + } catch { + result(FlutterError(code: "MOVEFILE_ERROR", message: error.localizedDescription, details: nil)) + } + + case "enableRecord": + guard let args = call.arguments as? [String: Any], + let filePath = args["filePath"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENTS", message: "filePath 参数不能为空", details: nil)) + return + } + do { + print("音频文件名称为: \(filePath)") + try azureAsrHelper.enableRecord(filePath: filePath) + result(true) + } catch { + result(FlutterError(code: "ENABLERECORD_ERROR", message: error.localizedDescription, details: nil)) + } + + case "pauseRecord": + azureAsrHelper.pauseRecord() result(true) - } catch { - result(FlutterError(code: "MOVEFILE_ERROR", message: error.localizedDescription, details: nil)) - } - - case "enableRecord": - guard let args = call.arguments as? [String: Any], - let filePath = args["filePath"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "filePath 参数不能为空", details: nil)) - return - } - do { - print("音频文件名称为: \(filePath)") - try azureAsrHelper.enableRecord(filePath: filePath) + + case "stopRecord": + guard let args = call.arguments as? [String: Any], + let isSave = args["isSave"] as? Bool else { + result(FlutterError(code: "INVALID_ARGUMENTS", message: "isSave 参数不能为空", details: nil)) + return + } + do { + try azureAsrHelper.stopRecord(isSave: isSave) + result(true) + } catch { + result(FlutterError(code: "STOPRECORD_ERROR", message: error.localizedDescription, details: nil)) + } + case "disableBluetoothAudio": + print("disableBluetoothAudio") + azureAsrHelper.disableBluetoothAudio(); + azureTtsHelper.setAudioOutputDevice() result(true) - } catch { - result(FlutterError(code: "ENABLERECORD_ERROR", message: error.localizedDescription, details: nil)) - } - - case "pauseRecord": - azureAsrHelper.pauseRecord() - result(true) - - case "stopRecord": - guard let args = call.arguments as? [String: Any], - let isSave = args["isSave"] as? Bool else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "isSave 参数不能为空", details: nil)) - return - } - do { - try azureAsrHelper.stopRecord(isSave: isSave) + + + + case "restoreOriginalAudioState": + print("restoreOriginalAudioState") + azureAsrHelper.restoreOriginalAudioState(); + azureTtsHelper.setAudioOutputDevice() result(true) - } catch { - result(FlutterError(code: "STOPRECORD_ERROR", message: error.localizedDescription, details: nil)) - } - -case "restoreOriginalAudioState": - result(true) - + default: result(FlutterMethodNotImplemented) } @@ -347,8 +356,35 @@ case "restoreOriginalAudioState": case "isSpeaking": result(azureTtsHelper.isSpeaking()) - case "release": + case "dispose": azureTtsHelper.dispose() + result(true) + case "setAudioOutputDevice": + guard let args = call.arguments as? [String: Any], + let type = args["type"] as? Int else { + print("setAudioOutputDevice:type=nil") + return + } + print("setAudioOutputDevice:type=\(type)") + if (type == 0) { + // 默认(如果有耳机选耳机,否则使用系统扬声器) + azureAsrHelper.restoreOriginalAudioState(); + azureTtsHelper.setAudioOutputDevice() + + + } else if (type == 1) { + // 强制使用声器 + azureAsrHelper.disableBluetoothAudio(); + azureTtsHelper.setAudioOutputDevice() + + } else if (type == 2) { + // 强制使用耳机 + azureAsrHelper.restoreOriginalAudioState(); + azureTtsHelper.setAudioOutputDevice() + + } + + result(true) default: diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index 5a390d707..b7c8d7128 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -1,28 +1,11 @@ import Foundation import AVFoundation import MicrosoftCognitiveServicesSpeech +import os.log // 自定义语音处理组件,提供音频流处理等功能 import speech -import os.log - -/** - * Azure TTS Helper - * - * 基于微软Azure语音服务的TTS实现 - * 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis - * - * 特性: - * - 对外接口保持同步,内部异步处理 - * - 使用专用队列保证合成顺序 - * - 避免阻塞主线程和其他后台线程 - * - 事件回调由调用方处理线程切换 - * - * 使用示例: - * let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理 - */ - public class AzureTtsHelper: NSObject, ITtsService { +public class AzureTtsHelper: NSObject, ITtsService { private let tag = "AzureTtsHelper" - // 日志对象 private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") private static let DEFAULT_LANGUAGE = "zh-CN" @@ -43,59 +26,51 @@ import os.log // 事件监听器列表 private var eventListeners = NSHashTable.weakObjects() + private var audioDataListeners = NSHashTable.weakObjects() // 流式文本处理的缓冲区 private var streamBuffer = "" private var lastSpeakTime: TimeInterval = 0 - // 内部异步处理队列,保证顺序执行 + // 内部异步处理队列 private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) + private var isProcessing = false // 添加处理状态标志 private let synthesisGroup = DispatchGroup() private var pendingTasks: [() -> Void] = [] private let taskLock = NSLock() + // 自定义音频输出流 + private var customAudioOutputStream: SPXPushAudioOutputStream? + private var useInternalPlayer = true + // 记录最后播放的文本 + private var lastSpokenText: String? /** * 初始化语音合成服务 - * - * @param appId 服务应用ID - * @param token 服务访问令牌/订阅密钥 - * @param resource 服务资源ID/区域(可选) - * @param language 语言代码,如"zh-CN" - * @return 是否初始化成功 */ public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { do { // 创建语音配置 - if ttsResource.isEmpty { - speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia") - } else { - speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) - } - - // 设置语言 + speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) speechConfig?.speechSynthesisLanguage = language currentLanguage = language - // 设置音频输出格式 - 使用16k、16位的PCM格式 - speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", - byName: "SpeechServiceConnection_SynthOutputFormat") + // 设置音频输出格式 - 与Android一致 + speechConfig?.setPropertyTo("riff-16khz-16bit-mono-pcm", + byName: "SpeechServiceConnection_SynthOutputFormat") - // 创建合成器,使用默认音频输出(扬声器) - synthesizer = try SPXSpeechSynthesizer(speechConfig!) - - // 设置事件监听 - setupEventListeners() + // 设置低延迟属性 + speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_InitialSilenceTimeoutMs") + speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_EndSilenceTimeoutMs") - // 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny - if language.lowercased().starts(with: "zh") { - _ = setVoice("zh-CN-XiaoxiaoNeural") - } else { - _ = setVoice("en-US-JennyNeural") - } + // 创建合成器 + recreateSynthesizer() // 设置初始化完成 isInitialized = true + // 预热TTS引擎 + warmupSynthesizer() + return true } catch { os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) @@ -103,24 +78,25 @@ import os.log } } + /** + * 设置自定义音频输出流 + */ + public func setCustomAudioOutputStream(_ outputStream: SPXPushAudioOutputStream?) { + customAudioOutputStream = outputStream + if isInitialized { + recreateSynthesizer() + } + } + /** * 设置语音角色 - * - * @param voiceName 语音角色名称(不同服务的语音角色命名可能不同) - * @return 是否设置成功 */ public func setVoice(_ voiceName: String) -> Bool { if !isInitialized { return false } do { currentVoice = voiceName - - // 更新语音名称 - if let config = speechConfig { - config.speechSynthesisVoiceName = voiceName - } - - // 重新创建合成器 + speechConfig?.speechSynthesisVoiceName = voiceName recreateSynthesizer() return true } catch { @@ -135,173 +111,88 @@ import os.log /** * 单次合成并播放 - * - * @param text 要合成的文本 - * @return 是否成功开始合成 */ public func speakOnce(_ text: String) -> Bool { if !isInitialized { os_log("语音合成未初始化", log: log, type: .error) + notifyEvent(eventType: .error, params: [ + "errorCode": "NOT_INITIALIZED", + "errorMessage": "TTS引擎未初始化" + ]) return false } - // 立即返回成功,内部异步处理 + // 清理文本 + let cleanedText = cleanTextForTTS(text) + if cleanedText.isEmpty { + return false + } + + // 异步处理 enqueueSynthesisTask { - self.performSynthesis(text: text) + self.performSynthesis(text: cleanedText) } return true } /** - * 将合成任务加入队列,保证顺序执行 + * 流式合成文本 */ - private func enqueueSynthesisTask(_ task: @escaping () -> Void) { - taskLock.lock() - defer { taskLock.unlock() } - - pendingTasks.append(task) - - // 如果当前没有任务在执行,开始处理队列 - if pendingTasks.count == 1 { - processNextTask() + public func speakStream(_ text: String) -> Bool { + if !isInitialized { + notifyEvent(eventType: .error, params: [ + "errorCode": "NOT_INITIALIZED", + "errorMessage": "TTS引擎未初始化" + ]) + return false } - } - - /** - * 处理队列中的下一个任务 - */ - private func processNextTask() { - synthesisQueue.async { - self.synthesisGroup.enter() - - self.taskLock.lock() - guard !self.pendingTasks.isEmpty else { - self.taskLock.unlock() - self.synthesisGroup.leave() - return - } - let task = self.pendingTasks.removeFirst() - self.taskLock.unlock() - - // 执行任务 - task() - - self.synthesisGroup.leave() - - // 处理下一个任务 - self.taskLock.lock() - if !self.pendingTasks.isEmpty { - self.taskLock.unlock() - self.processNextTask() - } else { - self.taskLock.unlock() - } + + if text.isEmpty { + return true } - } - - /** - * 实际执行合成的方法 - */ - private func performSynthesis(text: String) { - // 重置状态 - speaking = true - // 生成SSML - let ssml = generateSsml(text) + // 清理文本并添加到缓冲区 + let cleanedText = cleanTextForTTS(text) + streamBuffer.append(cleanedText) - do { - os_log("开始合成: %{public}@", log: log, type: .debug, ssml) - // 使用同步方法,在后台队列中执行 - let result = try synthesizer?.startSpeakingSsml(ssml) - - // 检查结果 - if let result = result { - os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) - } - } catch { - os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "SYNTHESIS_FAILED", - "errorMessage": error.localizedDescription - ]) - speaking = false + // 防抖逻辑 (150ms) + let currentTime = Date().timeIntervalSince1970 + if currentTime - lastSpeakTime < 0.15 { + return true } - } - - /** - * 流式合成文本 - * - * @param text 要合成的文本片段 - * @return 是否成功处理 - */ - public func speakStream(_ text: String) -> Bool { - if !isInitialized || text.isEmpty { - if !isInitialized { - notifyEvent(eventType: .error, params: [ - "errorCode": "NOT_INITIALIZED", - "errorMessage": "TTS引擎未初始化" - ]) + lastSpeakTime = currentTime + + let currentText = streamBuffer + + // 标点符号列表 (与Android一致) + let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] + + // 从后往前查找最后一个标点符号 + var lastPunctuationIndex = -1 + for (i, char) in currentText.enumerated().reversed() { + if punctuationMarks.contains(char) { + lastPunctuationIndex = i + break } - return false } - do { - // 添加新文本到缓冲区 - streamBuffer.append(text) - - // 增加500ms防抖逻辑 - let currentTime = Date().timeIntervalSince1970 - if currentTime - lastSpeakTime < 0.6 { - return true - } - lastSpeakTime = currentTime - - let currentText = streamBuffer + // 播放到找到的标点符号 + if lastPunctuationIndex >= 0 { + let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)) + let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) + streamBuffer = String(currentText[startIndex...]) - // 定义标点符号列表 - let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] - - // 查找最后一个标点符号的位置 - var lastPunctuationIndex = -1 - for (i, char) in currentText.enumerated().reversed() { - if punctuationMarks.contains(char) { - lastPunctuationIndex = i - break - } - } - - // 如果找到标点符号,则播放到该标点符号 - if lastPunctuationIndex >= 0 { - // 提取要播放的文本(包含标点符号) - let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines) - - // 剩余的文本保存在缓冲区中 - let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) - streamBuffer = String(currentText[startIndex...]) - - // 只有非空文本才播放 - if !textToSpeak.isEmpty { + if !textToSpeak.isEmpty { return speakOnce(textToSpeak) - } } - - // 如果没有找到标点符号,则等待更多文本 - return true - } catch { - os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "STREAM_FAILED", - "errorMessage": "流式语音合成失败: \(error.localizedDescription)" - ]) - return false } + + return true } /** - * 刷新并播放流式文本缓冲区中的剩余内容 - * - * @return 是否成功处理 + * 刷新并播放流式文本 */ public func flushStream() -> Bool { if !isInitialized { @@ -312,115 +203,71 @@ import os.log return false } - do { - // 获取缓冲区中剩余的文本 - let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines) - - // 清空缓冲区 - streamBuffer = "" - - // 如果缓冲区为空,直接返回成功 - if remainingText.isEmpty { - return true - } - - // 播放剩余文本 - return speakOnce(remainingText) - } catch { - os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "FLUSH_FAILED", - "errorMessage": "刷新流式文本失败: \(error.localizedDescription)" - ]) - return false + let remainingText = streamBuffer + streamBuffer = "" + + if remainingText.isEmpty { + return true } + + return speakOnce(remainingText) } /** - * 停止语音合成和播放 - * - * @return 是否成功停止 + * 停止语音合成 */ public func stop() -> Bool { - // 立即更新状态 speaking = false - - // 清除流式缓冲区中的待播放内容 streamBuffer = "" - // 清空待处理的任务队列 taskLock.lock() pendingTasks.removeAll() taskLock.unlock() - // 在后台队列停止合成器,避免阻塞主线程 synthesisQueue.async { - if let synthesizer = self.synthesizer { - do { - try synthesizer.stopSpeaking() - os_log("语音合成已停止", log: self.log, type: .info) - - // 直接通知停止完成 - self.notifyEvent(eventType: .synthesisCanceled) - } catch { - os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) - self.notifyEvent(eventType: .error, params: [ - "errorCode": "STOP_FAILED", - "errorMessage": error.localizedDescription - ]) - } + do { + try self.synthesizer?.stopSpeaking() + self.notifyEvent(eventType: .synthesisCanceled) + } catch { + os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) + self.notifyEvent(eventType: .error, params: [ + "errorCode": "STOP_FAILED", + "errorMessage": error.localizedDescription + ]) } } return true } - /** * 释放资源 - * 在不再需要服务时调用,释放底层资源 */ public func dispose() { - // 停止播放 _ = stop() - - // 等待所有任务完成 synthesisGroup.wait() - // 清理资源 - synthesisQueue.async { - // 释放合成器 - self.synthesizer = nil - - // 释放配置 - self.speechConfig = nil - - // 清空流缓冲区 - self.streamBuffer = "" - - // 重置状态 - self.isInitialized = false - self.speaking = false + synthesisQueue.sync { + synthesizer = nil + speechConfig = nil + customAudioOutputStream = nil + isInitialized = false + speaking = false - // 清空任务队列 - self.taskLock.lock() - self.pendingTasks.removeAll() - self.taskLock.unlock() + taskLock.lock() + pendingTasks.removeAll() + taskLock.unlock() } } /** - * 添加TTS事件监听器 - * - * @param listener 事件监听器 + * 添加事件监听器 */ public func addListener(_ listener: TtsEventListener) { eventListeners.add(listener as AnyObject) } /** - * 移除TTS事件监听器 - * - * @param listener 要移除的事件监听器 + * 移除事件监听器 */ public func removeListener(_ listener: TtsEventListener) { eventListeners.remove(listener as AnyObject) @@ -428,100 +275,179 @@ import os.log /** * 添加音频数据监听器 - * 由于不再支持自定义音频流,此方法实际上不再有效 - * - * @param listener 音频数据监听器 */ public func addAudioDataListener(_ listener: AudioDataListener) { - os_log("警告:不支持音频数据监听器功能", log: log, type: .info) + audioDataListeners.add(listener as AnyObject) } /** * 移除音频数据监听器 - * 由于不再支持自定义音频流,此方法实际上不再有效 - * - * @param listener 要移除的音频数据监听器 */ public func removeAudioDataListener(_ listener: AudioDataListener) { - // 不做任何操作 + audioDataListeners.remove(listener as AnyObject) } /** - * 当前是否正在播放/合成 + * 设置是否使用内部播放器 */ - public func isSpeaking() -> Bool { - return speaking + public func setUseInternalPlayer(_ useInternalPlayer: Bool) { + self.useInternalPlayer = useInternalPlayer + if isInitialized { + recreateSynthesizer() + } } - - // MARK: - 辅助方法 - /** - * 设置事件监听器 + * 设置音频输出设备 */ - private func setupEventListeners() { - guard let synthesizer = synthesizer else { return } + public func setAudioOutputDevice() -> Bool { + // 先暂停当前播放 + let wasSpeaking = speaking + if wasSpeaking { + _ = stop() + } - // 添加合成开始事件处理器 - synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in - guard let self = self else { return } - self.notifyEvent(eventType: .synthesisStarted) + // 重新创建合成器以应用新设备 + if isInitialized { + recreateSynthesizer() } - // 添加合成中事件处理器 - synthesizer.addSynthesizingEventHandler { [weak self] _, _ in - // 可以在这里处理合成中的事件,目前没有特别操作 + // 如果之前正在播放,恢复播放 + if wasSpeaking, let lastText = lastSpokenText { + speakOnce(lastText) } - // 添加合成完成事件处理器 - synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in - guard let self = self else { return } - self.speaking = false - self.notifyEvent(eventType: .synthesisCompleted) + return true + } + // /** + // * 设置音频输出设备 + // */ + // public func setAudioOutputDevice(_ device: AudioOutputDevice) -> Bool { + // do { + // let session = AVAudioSession.sharedInstance() + // try session.setCategory(.playAndRecord, options: [.defaultToSpeaker, .allowBluetooth]) + // + // switch device { + // case .default: + // try session.overrideOutputAudioPort(.none) + // case .speaker: + // try session.overrideOutputAudioPort(.speaker) + // case .headphones: + // try session.overrideOutputAudioPort(.none) + // } + // + // try session.setActive(true) + // return true + // } catch { + // os_log("设置音频输出设备失败: %{public}@", log: log, type: .error, error.localizedDescription) + // return false + // } + // } + // + // MARK: - 私有方法 + + /** + * 加入合成任务队列 + */ + private func enqueueSynthesisTask(_ task: @escaping () -> Void) { + taskLock.lock() + pendingTasks.append(task) + taskLock.unlock() + + // 确保任务被处理 + processTasksIfNeeded() + } + /** + * 处理任务队列 + */ + private func processTasksIfNeeded() { + taskLock.lock() + // 如果已经在处理中或没有任务,则直接返回 + guard !isProcessing && !pendingTasks.isEmpty else { + taskLock.unlock() + return } - // 添加合成取消事件处理器 - synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in - guard let self = self else { return } + isProcessing = true + let task = pendingTasks.removeFirst() + taskLock.unlock() + + synthesisQueue.async { + task() - self.speaking = false + // 任务完成后,检查是否有更多任务 + self.taskLock.lock() + self.isProcessing = false - var params: [String: Any] = [:] - do { - let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) - if cancellationDetails.reason == SPXCancellationReason.error { - params["reason"] = String(describing: cancellationDetails.reason.rawValue) - params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" - } - } catch { - params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)" + // 如果还有任务,递归处理 + if !self.pendingTasks.isEmpty { + self.taskLock.unlock() + self.processTasksIfNeeded() + } else { + self.taskLock.unlock() } - os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误")) - - self.notifyEvent(eventType: .synthesisCanceled, params: params) } } /** - * 触发事件通知 + * 处理下一个任务 */ - private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { - let event = TtsEvent(type: eventType, params: params) + private func processNextTask() { + synthesisQueue.async { + self.taskLock.lock() + guard !self.pendingTasks.isEmpty else { + self.taskLock.unlock() + return + } + let task = self.pendingTasks.removeFirst() + self.taskLock.unlock() + + task() + self.processNextTask() + } + } + + /** + * 执行语音合成 + */ + private func performSynthesis(text: String) { + speaking = true + lastSpokenText = text + // 生成SSML + let ssml = generateOptimizedSsml(text) - // 直接通知事件,由调用方处理线程切换 - for case let listener as TtsEventListener in self.eventListeners.allObjects { - listener.onEvent(event) + do { + os_log("开始合成: %{public}@", log: log, type: .debug, ssml) + let result = try synthesizer?.startSpeakingSsml(ssml) + + if let result = result { + os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) + } + } catch { + speaking = false + os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) + notifyEvent(eventType: .error, params: [ + "errorCode": "SYNTHESIS_FAILED", + "errorMessage": error.localizedDescription + ]) } } + /** * 重新创建合成器 */ private func recreateSynthesizer() { do { - // 使用默认音频输出配置创建合成器 - synthesizer = try SPXSpeechSynthesizer(speechConfig!) + // 创建音频配置 + let audioConfig: SPXAudioConfiguration? + if !useInternalPlayer || customAudioOutputStream != nil { + audioConfig = try SPXAudioConfiguration(streamOutput: customAudioOutputStream ?? SPXPushAudioOutputStream()) + } else { + audioConfig = nil // 使用默认扬声器 + } - // 设置事件监听 + // 创建合成器 + synthesizer = try SPXSpeechSynthesizer(speechConfig!) setupEventListeners() } catch { os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) @@ -533,72 +459,139 @@ import os.log } /** - * 设置语音参数 + * 设置事件监听器 */ - private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { - if !isInitialized { return false } + private func setupEventListeners() { + synthesizer?.addSynthesisStartedEventHandler { [weak self] _, _ in + self?.notifyEvent(eventType: .synthesisStarted) + self?.notifyEvent(eventType: .playbackStarted) + } - do { - currentRate = formatPercentage(rate) - currentPitch = formatPercentage(pitch) - currentVolume = "\(min(max(volume, 0), 100))%" - return true - } catch { - os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "PARAMS_SET_FAILED", - "errorMessage": "设置语音参数失败: \(error.localizedDescription)" - ]) - return false + synthesizer?.addSynthesizingEventHandler { [weak self] _, event in + if let audioData = event.result.audioData { + self?.notifyAudioData(audioData) + } + } + + synthesizer?.addSynthesisCompletedEventHandler { [weak self] _, _ in + guard let self = self else { return } + self.speaking = false + self.notifyEvent(eventType: .synthesisCompleted) + self.notifyEvent(eventType: .playbackCompleted) + } + + synthesizer?.addSynthesisCanceledEventHandler { [weak self] _, event in + guard let self = self else { return } + self.speaking = false + + var params: [String: Any] = [:] + if let details = try? SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: event.result) { + params["errorCode"] = details.errorCode + params["errorDetails"] = details.errorDetails + } + + self.notifyEvent(eventType: .synthesisCanceled, params: params) } } /** - * 格式化百分比值 + * 通知事件 */ - private func formatPercentage(_ value: Int) -> String { - return value >= 0 ? "+\(value)%" : "\(value)%" + private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { + let event = TtsEvent(type: eventType, params: params) + for case let listener as TtsEventListener in eventListeners.allObjects { + listener.onEvent(event) + } } /** - * 生成SSML + * 通知音频数据 */ - private func generateSsml(_ rawText: String) -> String { - // 1. 定义要静音的符号和表情符号列表 - let symbolsToMute = [ - "#", "*", - "😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" - ] - - // 2. 转义 XML 保留字符 - var escapedText = rawText + private func notifyAudioData(_ data: Data) { + for case let listener as AudioDataListener in audioDataListeners.allObjects { + listener.onAudioData(data) + } + } + + /** + * 预热TTS引擎 + */ + private func warmupSynthesizer() { + let warmupSsml = """ + + + + . + + + + """ + + synthesisQueue.async { + _ = try? self.synthesizer?.startSpeakingSsml(warmupSsml) + } + } + + /** + * 清理TTS文本 + */ + private func cleanTextForTTS(_ text: String) -> String { + var cleaned = text + + // 移除URL + if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") + } + + // 移除emoji + if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") + } + + // 合并空格 + if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ") + } + + return cleaned.trimmingCharacters(in: .whitespacesAndNewlines) + } + + /** + * 生成优化的SSML + */ + private func generateOptimizedSsml(_ rawText: String) -> String { + // 转义XML保留字符 + let escapedText = rawText .replacingOccurrences(of: "&", with: "&") .replacingOccurrences(of: "<", with: "<") .replacingOccurrences(of: ">", with: ">") - // 3. 静音处理特殊符号和表情符号 - // 使用空白替换法,直接将符号替换为空字符串 - var processedText = escapedText - for symbol in symbolsToMute { - processedText = processedText.replacingOccurrences(of: symbol, with: "") - } - - // 4. 构造简化的SSML文档,减少嵌套层级 - let ssml = """ - + // 简化SSML结构 + return """ + - - - \(processedText) - - + + \(escapedText) + """ - - return ssml + } + + /** + * 设置语音参数 + */ + private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { + currentRate = rate >= 0 ? "+\(rate)%" : "\(rate)%" + currentPitch = pitch >= 0 ? "+\(pitch)%" : "\(pitch)%" + currentVolume = "\(min(max(volume, 0), 100))%" + return true + } + + /** + * 是否正在播放 + */ + public func isSpeaking() -> Bool { + return speaking } }