From 12711b0cf80e1c0ac91b6393abf1c4b1798e998a Mon Sep 17 00:00:00 2001 From: liwei1dao Date: Fri, 23 Jan 2026 21:15:47 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E4=B8=BA=E8=AF=AD=E9=9F=B3=E6=9C=8D?= =?UTF-8?q?=E5=8A=A1=E6=B7=BB=E5=8A=A0=E6=A8=A1=E5=BC=8F=E5=8F=82=E6=95=B0?= =?UTF-8?q?=E6=94=AF=E6=8C=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 在 ASR 和 TTS 服务接口中添加模式参数,支持 normal、phone_call、push_to_talk 等模式 - 修改翻译控制器,根据当前翻译模式设置对应的 TTS 和 ASR 模式 - 重构 iOS 音频会话配置逻辑,优化不同模式下的音频路由和蓝牙设备处理 - 统一 TTS 模式设置方法名,将 setPhoneCallMode 重命名为 setTtsMode --- lib/data/services/asr_service.dart | 1 + .../speech_impl/azure_asr_service.dart | 2 + .../speech_impl/azure_tts_service.dart | 14 ++ .../speech_impl/voice_clone_tts_service.dart | 12 + .../speech_impl/volcano_asr_api_service.dart | 1 + .../speech_impl/volcano_asr_service.dart | 1 + .../speech_impl/volcano_tts_service.dart | 17 ++ lib/data/services/tts_service.dart | 3 + .../controllers/translation_controller.dart | 26 +- .../agent_service/AgentServiceImpl.swift | 12 +- .../azure_speech/AzureSpeechPlugin.swift | 15 +- .../Sources/azure_speech/AzureTtsHelper.swift | 229 +++++++++--------- .../Sources/tools/MicrophoneCapture.swift | 23 +- .../Sources/tools/SimpleAudioReceiver.swift | 42 ++-- 14 files changed, 230 insertions(+), 168 deletions(-) diff --git a/lib/data/services/asr_service.dart b/lib/data/services/asr_service.dart index 8bb7892c6..2b3610f68 100644 --- a/lib/data/services/asr_service.dart +++ b/lib/data/services/asr_service.dart @@ -22,6 +22,7 @@ abstract class AsrService { Future startContinuousRecognition( bool audioSourceType, { bool isRemoveFirstPunctuation = true, + String mode = "normal", }); /// 停止连续语音识别 diff --git a/lib/data/services/speech_impl/azure_asr_service.dart b/lib/data/services/speech_impl/azure_asr_service.dart index a196a5ddc..bdbc6100e 100644 --- a/lib/data/services/speech_impl/azure_asr_service.dart +++ b/lib/data/services/speech_impl/azure_asr_service.dart @@ -207,6 +207,7 @@ class AzureAsrService extends GetxService implements AsrService { Future startContinuousRecognition( bool audioSourceType, { bool isRemoveFirstPunctuation = true, + String mode = "normal", }) async { // 防抖判断:短时间内重复调用直接拦截 final now = DateTime.now(); @@ -229,6 +230,7 @@ class AzureAsrService extends GetxService implements AsrService { await _channel.invokeMethod('startContinuousRecognition', { 'audioSourceType': audioSourceType, 'isRemoveFirstPunctuation': isRemoveFirstPunctuation, + 'mode': mode, }); if (!result) { diff --git a/lib/data/services/speech_impl/azure_tts_service.dart b/lib/data/services/speech_impl/azure_tts_service.dart index cebcb9a10..200002fb4 100644 --- a/lib/data/services/speech_impl/azure_tts_service.dart +++ b/lib/data/services/speech_impl/azure_tts_service.dart @@ -182,6 +182,20 @@ class AzureTtsService extends GetxService implements TtsService { } } + @override + Future setTtsMode(String mod) async { + if (!_isInitialized) await initialize(); + try { + final result = await _channel.invokeMethod('setTtsMode', { + 'mod': mod, + }); + return result; + } catch (e) { + Logger.error('设置TTS模式失败: ${e.toString()}'); + return false; + } + } + @override Future setVoiceFlocking(String speakerProfileId) async { final result = await _channel.invokeMethod('setVoiceFlocking', { diff --git a/lib/data/services/speech_impl/voice_clone_tts_service.dart b/lib/data/services/speech_impl/voice_clone_tts_service.dart index 15ea6803b..97542c02b 100644 --- a/lib/data/services/speech_impl/voice_clone_tts_service.dart +++ b/lib/data/services/speech_impl/voice_clone_tts_service.dart @@ -94,6 +94,18 @@ class VoiceCloneTtsService extends GetxService implements TtsService { } } + @override + Future setTtsMode(String mod) async { + if (!_isInitialized) await initialize(); + try { + Logger.info('设置TTS模式: $mod'); + return true; + } catch (e) { + Logger.error('设置TTS模式失败: ${e.toString()}'); + return false; + } + } + @override Future setVoiceFlocking(String speakerProfileId) async { return true; diff --git a/lib/data/services/speech_impl/volcano_asr_api_service.dart b/lib/data/services/speech_impl/volcano_asr_api_service.dart index 3e5711b38..d105d97ec 100644 --- a/lib/data/services/speech_impl/volcano_asr_api_service.dart +++ b/lib/data/services/speech_impl/volcano_asr_api_service.dart @@ -865,6 +865,7 @@ class VolcanoAsrApiService implements AsrService { Future startContinuousRecognition( bool audioSourceType, { bool isRemoveFirstPunctuation = true, + String mode = "normal", }) { // TODO: implement startContinuousRecognition throw UnimplementedError(); diff --git a/lib/data/services/speech_impl/volcano_asr_service.dart b/lib/data/services/speech_impl/volcano_asr_service.dart index ec36a0385..d1f59648e 100644 --- a/lib/data/services/speech_impl/volcano_asr_service.dart +++ b/lib/data/services/speech_impl/volcano_asr_service.dart @@ -400,6 +400,7 @@ class VolcanoAsrService extends GetxService implements AsrService { Future startContinuousRecognition( bool audioSourceType, { bool isRemoveFirstPunctuation = true, + String mode = "normal", }) { // TODO: implement startContinuousRecognition throw UnimplementedError(); diff --git a/lib/data/services/speech_impl/volcano_tts_service.dart b/lib/data/services/speech_impl/volcano_tts_service.dart index 2bb04ade0..c0dfb8935 100644 --- a/lib/data/services/speech_impl/volcano_tts_service.dart +++ b/lib/data/services/speech_impl/volcano_tts_service.dart @@ -185,6 +185,23 @@ class VolcanoTtsService extends GetxService implements TtsService { } } + @override + Future setTtsMode(String mod) async { + if (!await _ensureInitialized()) return false; + + try { + final result = await _channel.invokeMethod('setTtsMode', { + 'mod': mod, + }) ?? + false; + + return result; + } catch (e) { + debugPrint('设置TTS模式失败:$e'); + return false; + } + } + @override Future setVoiceFlocking(String speakerProfileId) async { return true; diff --git a/lib/data/services/tts_service.dart b/lib/data/services/tts_service.dart index b4bf72625..c5b24ad18 100644 --- a/lib/data/services/tts_service.dart +++ b/lib/data/services/tts_service.dart @@ -19,6 +19,9 @@ abstract class TtsService { /// 设置语音 Future setVoice(String voiceName); + /// 设置TTS模式 + Future setTtsMode(String mod); + /// 设置语音复刻 Future setVoiceFlocking(String speakerProfileId); diff --git a/lib/modules/translation/controllers/translation_controller.dart b/lib/modules/translation/controllers/translation_controller.dart index 994d1c109..c1d57845b 100644 --- a/lib/modules/translation/controllers/translation_controller.dart +++ b/lib/modules/translation/controllers/translation_controller.dart @@ -344,6 +344,15 @@ class TranslationController extends GetxController with WidgetsBindingObserver { } catch (_) {} } + //设置播报的模式 + if (mode == 'simultaneous') { + await _ttsService.setTtsMode('phone_call'); + } else if (mode == 'faceToFace') { + await _ttsService.setTtsMode('push_to_talk'); + } else { + await _ttsService.setTtsMode('normal'); + } + // 检查悬浮窗支持情况 if (isFloatingWindowEnabled.value && !_isFloatingWindowSupportedMode()) { // 当前模式不支持悬浮窗,自动禁用 @@ -437,6 +446,7 @@ class TranslationController extends GetxController with WidgetsBindingObserver { await _asrService.startContinuousRecognition( _audioSourceType, isRemoveFirstPunctuation: false, + mode: 'push_to_talk', ); //开启识别 _startAsrActiveTracking(); //开启识别活动跟踪 if (isRecording.value) { @@ -946,7 +956,13 @@ class TranslationController extends GetxController with WidgetsBindingObserver { !bleManager.isCodecActive) { return; } - await _startAsrService(); + if (currentMode.value == 'simultaneous') { + await _startAsrService('push_to_talk'); + } else if (currentMode.value == 'faceToFace') { + await _startAsrService('phone_call'); + } else { + await _startAsrService('normal'); + } await _finalizeRecognitionStart(); } } catch (e) { @@ -1045,8 +1061,12 @@ class TranslationController extends GetxController with WidgetsBindingObserver { } /// 启动语音识别 - Future _startAsrService() async { - await _asrService.startContinuousRecognition(_audioSourceType); + Future _startAsrService(String mode) async { + await _asrService.startContinuousRecognition( + _audioSourceType, + isRemoveFirstPunctuation: false, + mode: mode, + ); } /// 完成识别启动 diff --git a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift index b0a7ee5f0..61f5e4d8a 100644 --- a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift +++ b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift @@ -526,7 +526,7 @@ class AgentServiceImpl: NSObject { // 设置当前识别模式 currentRecognitionMode = mode - os_log("设置语音识别模式: %{public}@", log: logger, type: .info, mode) + os_log("liwei--------------- 设置语音识别模式: %{public}@", log: logger, type: .info, mode) let audioSourceType: AzureAsrHelper.AudioSourceType = useBle ? .external : .microphone @@ -724,7 +724,7 @@ class AgentServiceImpl: NSObject { return } _ = azureTtsHelper?.setSpeechParams(rate: ttsRatePercent, pitch: 0, volume: 100) - _ = azureTtsHelper?.setPhoneCallMode(mod: currentRecognitionMode) + _ = azureTtsHelper?.setTtsMode(mod: currentRecognitionMode) let userMessage = chatApiService.createUserMessage(content: text) processWithChatApiServiceInternal(sessionid: sessionid,userMessage: userMessage, displayText: text, speakResponse: speakResponse) } @@ -1810,7 +1810,7 @@ class ChatApiStreamCallback: StreamCallback { responseBuilder += token if speakResponse && reply && broadcast{ - _ = agentService.azureTtsHelper?.setPhoneCallMode(mod:agentService.currentRecognitionMode) +// _ = agentService.azureTtsHelper?.setTtsMode(mod:agentService.currentRecognitionMode) try agentService.azureTtsHelper?.speakStream(sessionid:sessionid,token) // 在开始流式TTS时立即停止气泡音 } @@ -1836,7 +1836,7 @@ class ChatApiStreamCallback: StreamCallback { } if speakResponse && reply && broadcast && sessionid == agentService.currsessionId{ - _ = agentService.azureTtsHelper?.setPhoneCallMode(mod: agentService.currentRecognitionMode) + // _ = agentService.azureTtsHelper?.setTtsMode(mod: agentService.currentRecognitionMode) agentService.azureTtsHelper?.flushStream(sessionid:sessionid) } @@ -1888,7 +1888,7 @@ class ChatApiStreamCallback: StreamCallback { "message": message ]) if(code == 2001){ - _ = agentService.azureTtsHelper?.setPhoneCallMode(mod: agentService.currentRecognitionMode) + // _ = agentService.azureTtsHelper?.setTtsMode(mod: agentService.currentRecognitionMode) agentService.azureTtsHelper?.speakStream(sessionid: sessionid,agentService.insufficientIntegralText) } agentService.isAiStreaming = false @@ -1932,7 +1932,7 @@ class ChatApiStreamCallback: StreamCallback { if (iscallingTool) { os_log("收到函数调用: callingToolText:%{public}@", log: agentService.logger, type: .info, agentService.callingToolText) - _ = agentService.azureTtsHelper?.setPhoneCallMode(mod: agentService.currentRecognitionMode) + // _ = agentService.azureTtsHelper?.setTtsMode(mod: agentService.currentRecognitionMode) agentService.azureTtsHelper?.speakStream(sessionid: sessionid,agentService.callingToolText) iscallingTool = false } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index c74abc624..ce82256b1 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -382,6 +382,7 @@ private func sendAstEvent(_ event: [String: Any]) { } let removeFirstPunctuation = args["isRemoveFirstPunctuation"] as? Bool ?? true + let mode = args["mode"] as? String ?? "normal" // 根据参数确定音频源类型 let audioSourceType = isExternalActive ? @@ -392,6 +393,7 @@ private func sendAstEvent(_ event: [String: Any]) { // 启动连续识别 let success = azureAsrHelper.startContinuousRecognition( + mod: mode, audioSourceType: audioSourceType, isRemoveFirstPunctuation: removeFirstPunctuation ) @@ -631,19 +633,14 @@ private func sendAstEvent(_ event: [String: Any]) { let success = azureTtsHelper.setVoice(voiceName) result(success) - - case "setSpeechParams": - guard let args = call.arguments as? [String: Any] else { + case "setTtsMode": + guard let args = call.arguments as? [String: Any], + let mod = args["mod"] as? String else { result(FlutterError(code: "INVALID_ARGUMENTS", message: "参数不能为空", details: nil)) return } - - let rate = args["rate"] as? Int ?? 0 - let pitch = args["pitch"] as? Int ?? 0 - let volume = args["volume"] as? Int ?? 100 - let success = azureTtsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume) + let success = azureTtsHelper.setTtsMode(mod: mod) result(success) - case "speakText": guard let args = call.arguments as? [String: Any], let sessionid = args["sessionid"] as? String, diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index 6734d418f..68ae36cd2 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -415,14 +415,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { } - public func setPhoneCallMode(mod: String) -> Bool { - // os_log("[liwei] AzureTtsHelper: setPhoneCallMode: %{public}@", log: log, type: .debug, mod) + public func setTtsMode(mod: String) -> Bool { + os_log("[liwei] AzureTtsHelper: setTtsMode: %{public}@", log: log, type: .debug, mod) self.previousRecognitionMode = self.currentRecognitionMode self.currentRecognitionMode = mod self.isHeadphones = detectHeadphonesOutput() - if isInitialized { - safeConfigureAudioSessionForTTS() - } + // if isInitialized { + // safeConfigureAudioSessionForTTS() + // } return true } /** @@ -438,6 +438,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { out.portType == .bluetoothA2DP || out.portType == .bluetoothHFP || out.portType == .bluetoothLE || + out.portType == .headsetMic || out.portType == .headphones } } @@ -1376,6 +1377,18 @@ private var audioSessionObserversAdded = false private var routeChangeDebounceWorkItem: DispatchWorkItem? private var isApplyingAudioSessionConfig = false +private func isTtsBusy() -> Bool { + if speaking { return true } + if isPlaying { return true } + if !playbackQueue.isEmpty { return true } + if pendingTextCount > 0 { return true } + if !pendingTasks.isEmpty { return true } + if activeStreamSynthesisCount > 0 { return true } + if !pcmPendingData.isEmpty { return true } + if scheduledBufferCount > 0 { return true } + return false +} + /** * 确保音频会话队列已设置 specific key(用于避免同队列 sync 死锁) * - Parameters: 无 @@ -1388,6 +1401,12 @@ private func ensureAudioSessionQueueSpecificKeySet() { audioSessionQueue.setSpecific(key: audioSessionQueueKey, value: 1) } +/** + * 同步应用 TTS 音频会话配置(用于需要立即生效的播放/引擎初始化路径) + * - Parameters: 无 + * - Returns: 是否成功激活音频会话 + * - Throws: 无(内部捕获 AVAudioSession 相关异常并记录日志) + */ /** * 同步应用 TTS 音频会话配置(用于需要立即生效的播放/引擎初始化路径) * - Parameters: 无 @@ -1397,7 +1416,6 @@ private func ensureAudioSessionQueueSpecificKeySet() { private func applyAudioSessionForTTS() -> Bool { var success = true ensureAudioSessionQueueSpecificKeySet() - let applyBody = { if self.isApplyingAudioSessionConfig { return } self.isApplyingAudioSessionConfig = true @@ -1405,69 +1423,62 @@ private func applyAudioSessionForTTS() -> Bool { let audioSession = AVAudioSession.sharedInstance() let outputs: [AVAudioSessionPortDescription] = audioSession.currentRoute.outputs - let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP }) let hasWiredHeadphones = outputs.contains(where: { $0.portType == .headphones || $0.portType == .headsetMic }) - let hasBluetoothA2DP = outputs.contains(where: { $0.portType == .bluetoothA2DP }) - let hasHeadphones = hasWiredHeadphones || hasBluetoothA2DP + let hasBluetoothA2DP = outputs.contains(where: { $0.portType == .bluetoothA2DP || $0.portType == .bluetoothLE }) + let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP }) + let hasHeadphones = hasWiredHeadphones || hasBluetoothA2DP || hasBluetoothHFP let otherAudioPlaying = audioSession.isOtherAudioPlaying let isRecording = (self.micCapture?.isCapturing == true) + self.isHeadphones = hasHeadphones let desiredCategory: AVAudioSession.Category let desiredMode: AVAudioSession.Mode var desiredOptions: AVAudioSession.CategoryOptions = [] - let isHeadphonesConnected = outputs.contains(where: { $0.portType == .headphones || $0.portType == .headsetMic || $0.portType == .bluetoothA2DP }) - os_log("[liwei] AzureTtsHelper: applyAudioSessionForTTS: %{public}@", log: self.log, type: .debug, self.currentRecognitionMode) - if self.currentRecognitionMode == "phone_call" || self.currentRecognitionMode == "push_to_talk" { + + if self.currentRecognitionMode == "phone_call" { desiredCategory = .playAndRecord - if isHeadphonesConnected { - desiredMode = .voiceChat - if #available(iOS 18.0, *) { - desiredOptions = [.allowBluetooth, .duckOthers] - } else { - desiredOptions = [.allowBluetooth, .allowBluetoothA2DP] + if hasHeadphones { + desiredMode = .default + desiredOptions = [.allowBluetoothA2DP] + if let inputs = audioSession.availableInputs, let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { + try? audioSession.setPreferredInput(builtInMic) } } else { - desiredMode = .videoChat + desiredMode = .voiceChat desiredOptions = [.allowBluetooth, .defaultToSpeaker] } - } else if self.isTelephonyActive() { + if otherAudioPlaying { + desiredOptions.insert(.duckOthers) + } + } else if self.currentRecognitionMode == "push_to_talk" || self.currentRecognitionMode == "normal" || self.currentRecognitionMode == "ble_wakeup" { + if isRecording { + self.micCapture?.stopCapture() + Thread.sleep(forTimeInterval: 0.05) + } desiredCategory = .playback - desiredMode = .default - desiredOptions.insert(.duckOthers) + desiredMode = .spokenAudio + if otherAudioPlaying { + desiredOptions.insert(.duckOthers) + } } else if isRecording { desiredCategory = .playAndRecord - if self.isHeadphones { - desiredMode = .default - desiredOptions.insert(.allowBluetooth) + desiredMode = .default + if hasHeadphones { desiredOptions.insert(.allowBluetoothA2DP) - if otherAudioPlaying { - desiredOptions.insert(.duckOthers) + if let inputs = audioSession.availableInputs, let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { + try? audioSession.setPreferredInput(builtInMic) } } else { - desiredMode = .videoChat desiredOptions.insert(.allowBluetooth) - desiredOptions.insert(.mixWithOthers) desiredOptions.insert(.defaultToSpeaker) } - } else if hasBluetoothHFP { - desiredCategory = .playback - desiredMode = .voicePrompt - desiredOptions = [.interruptSpokenAudioAndMixWithOthers, .allowBluetooth] - } else if hasHeadphones { - desiredCategory = .playback - desiredMode = .default if otherAudioPlaying { desiredOptions.insert(.duckOthers) } } else { desiredCategory = .playback - desiredMode = .default - if #available(iOS 18.0, *) { - desiredOptions = [] - } else { - desiredOptions = [.defaultToSpeaker] - } + desiredMode = .spokenAudio if otherAudioPlaying { desiredOptions.insert(.duckOthers) } @@ -1483,7 +1494,7 @@ private func applyAudioSessionForTTS() -> Bool { _ = try? audioSession.setPreferredIOBufferDuration(0.02) } try audioSession.setActive(true) - if self.currentRecognitionMode == "phone_call", !self.isHeadphones { + if self.currentRecognitionMode == "phone_call", !hasHeadphones { _ = try? audioSession.overrideOutputAudioPort(.speaker) } break @@ -1551,43 +1562,20 @@ private func isTelephonyActive() -> Bool { return false } -/** - * 安全配置音频会话用于 TTS - * - 播放优先走媒体声道(A2DP),避免切换到通话声道(HFP) - * - 外部音乐播放时:不启用 mix,默认会打断外部音频;同时可 duck 确保可听见 - * - 录音进行中:保持 playAndRecord + A2DP 输出 + 内置麦输入,避免切到 HFP - * - 通话中:保持保守策略,避免触发系统断言或破坏通话链路 - * - Parameters: 无 - * - Returns: 无 - * - Throws: 无(内部捕获 AVAudioSession 相关异常并记录日志) - */ - - /// 为 TTS 安全配置音频会话: - /// - 使用异步队列,避免在同一串行队列上 sync 导致死锁 - /// - 优先使用 .playback + .spokenAudio + .allowBluetoothA2DP(媒体声道),避免进入 HFP - /// - 外部音频播放时启用 spoken audio 策略,确保能在 A2DP 上播放 - /// - 仅当类别或模式发生变化时才 setCategory,降低 AVAudioSessionRouteChangeReason.categoryChange 的触发概率 -private func safeConfigureAudioSessionForTTS() { - os_log("[liwei] AzureTtsHelper: safeConfigureAudioSessionForTTS - recognitionMode: %{public}@", log: log, type: .debug, currentRecognitionMode) - // 使用异步,避免串行队列重入自锁 - audioSessionQueue.async { [weak self] in - guard let self = self else { return } - - let audioSession = AVAudioSession.sharedInstance() - - // 显式指定 outputs 的类型,确保闭包参数类型可推断 - let outputs: [AVAudioSessionPortDescription] = audioSession.currentRoute.outputs - // 使用 contains(where:) 明确谓词闭包。 - // 保持使用 audioSessionQueue.async,防止在串行队列上自锁。 - // 仅在类别或模式发生变化时调用 setCategory,减少 categoryChange 的回调风暴。 - let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP }) - _ = hasBluetoothHFP - _ = audioSession - - _ = self.applyAudioSessionForTTS() - } + /// 为 TTS 安全配置音频会话: + /// - 使用异步队列,避免在同一串行队列上 sync 导致死锁 + /// - 优先使用 .playback + .spokenAudio + .allowBluetoothA2DP(媒体声道),避免进入 HFP + /// - 外部音频播放时启用 spoken audio 策略,确保能在 A2DP 上播放 + /// - 仅当类别或模式发生变化时才 setCategory,降低 AVAudioSessionRouteChangeReason.categoryChange 的触发概率 + private func safeConfigureAudioSessionForTTS() { + if !isTtsBusy() { return } + audioSessionQueue.async { [weak self] in + guard let self = self else { return } + if !self.isTtsBusy() { return } + _ = self.applyAudioSessionForTTS() } + } /** * 在 TTS 完全空闲后释放音频会话(用于恢复外部音乐) @@ -1617,8 +1605,6 @@ private func deactivateAudioSessionAfterTTSIfIdle() { self.playerNode = nil self.pcmFormat = nil } - let session = AVAudioSession.sharedInstance() - _ = try? session.setActive(false, options: .notifyOthersOnDeactivation) } if DispatchQueue.getSpecific(key: audioSessionQueueKey) != nil { work() @@ -1645,8 +1631,10 @@ private func deactivateAudioSessionAfterTTSIfIdle() { os_log("音频中断开始(可能为通话/Siri),停止TTS", log: log, type: .info) case .ended: audioInterrupted = false - os_log("音频中断结束,重新配置会话", log: log, type: .info) - safeConfigureAudioSessionForTTS() + if isTtsBusy() { + os_log("音频中断结束,重新配置会话", log: log, type: .info) + safeConfigureAudioSessionForTTS() + } @unknown default: break } @@ -1660,42 +1648,55 @@ private func deactivateAudioSessionAfterTTSIfIdle() { * - 配置过程中(isApplyingAudioSessionConfig = true)直接忽略 */ @objc private func handleRouteChange(_ notification: Notification) { - guard let info = notification.userInfo, - let reasonValue = info[AVAudioSessionRouteChangeReasonKey] as? UInt, - let reason = AVAudioSession.RouteChangeReason(rawValue: reasonValue) else { return } - - os_log("音频路由变化: %{public}@", log: log, type: .info, String(describing: reason)) - - // 配置中触发的路由变化,直接忽略,避免重入导致死循环 - if isApplyingAudioSessionConfig { - os_log("配置进行中,忽略本次路由变化", log: log, type: .debug) - return - } - - // 忽略会造成循环的原因:categoryChange(你的日志 reason=3),以及 routeConfigurationChange 等 - switch reason { - case .categoryChange, .routeConfigurationChange, .wakeFromSleep, .noSuitableRouteForCategory: - os_log("忽略路由变化原因: %{public}@", log: log, type: .debug, String(describing: reason)) - return - default: - break - } - - // 通话期间避免重配,避免与电话路由冲突 - if isTelephonyActive() { - os_log("通话期间检测到路由变化,跳过重配", log: log, type: .info) - return - } + // guard let info = notification.userInfo, + // let reasonValue = info[AVAudioSessionRouteChangeReasonKey] as? UInt, + // let reason = AVAudioSession.RouteChangeReason(rawValue: reasonValue) else { return } + + // if !isTtsBusy() { return } + // os_log("音频路由变化: %{public}@", log: log, type: .info, String(describing: reason)) + + // // 配置中触发的路由变化,直接忽略,避免重入导致死循环 + // if isApplyingAudioSessionConfig { + // os_log("配置进行中,忽略本次路由变化", log: log, type: .debug) + // return + // } + + // // 忽略会造成循环的原因:categoryChange(你的日志 reason=3),以及 routeConfigurationChange 等 + // switch reason { + // case .categoryChange, .routeConfigurationChange, .wakeFromSleep, .noSuitableRouteForCategory: + // os_log("忽略路由变化原因: %{public}@", log: log, type: .debug, String(describing: reason)) + // return + // default: + // break + // } + + // // 通话期间避免重配,避免与电话路由冲突 + // if isTelephonyActive() { + // os_log("通话期间检测到路由变化,跳过重配", log: log, type: .info) + // return + // } + + // let isActivelyPlayingNow = + // speaking || + // isPlaying || + // (audioPlayer?.isPlaying == true) || + // (playerNode?.isPlaying == true) + // if isActivelyPlayingNow { + // routeChangeDebounceWorkItem?.cancel() + // _ = stop() + // os_log("路由变化时正在播报,已停止以避免音源冲突", log: log, type: .info) + // return + // } // 去抖处理:300ms 内只执行最后一次重配 // 忽略 categoryChange 等 // 去抖:在 audioSessionQueue 上延迟调用 - routeChangeDebounceWorkItem?.cancel() - let work = DispatchWorkItem { [weak self] in - self?.safeConfigureAudioSessionForTTS() - } - routeChangeDebounceWorkItem = work - audioSessionQueue.asyncAfter(deadline: .now() + 0.3, execute: work) + // routeChangeDebounceWorkItem?.cancel() + // let work = DispatchWorkItem { [weak self] in + // self?.safeConfigureAudioSessionForTTS() + // } + // routeChangeDebounceWorkItem = work + // audioSessionQueue.asyncAfter(deadline: .now() + 0.3, execute: work) } /** diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift index b379819de..b346d4924 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift @@ -92,8 +92,7 @@ public class MicrophoneCapture: NSObject { } do { - try audioSession.setPreferredSampleRate(sampleRate) - try audioSession.setPreferredIOBufferDuration(0.005) // 5ms缓冲 + try audioSession.setPreferredIOBufferDuration(0.02) print("音频会话配置成功") } catch { print("音频会话配置失败: \(error.localizedDescription)") @@ -306,15 +305,18 @@ public class MicrophoneCapture: NSObject { sampleRate = audioSession.sampleRate } - // iOS 13+ 启用语音处理(可能在部分路由/模式下失败,失败不应阻断录音启动) if #available(iOS 13.0, *) { + let session = self.audioSession ?? AVAudioSession.sharedInstance() + let outputs = session.currentRoute.outputs + let hasBluetoothA2DPOutput = outputs.contains(where: { $0.portType == .bluetoothA2DP || $0.portType == .bluetoothLE }) + let shouldEnableVoiceProcessing = (session.category == .playAndRecord) && (session.mode == .voiceChat || session.mode == .videoChat) && !hasBluetoothA2DPOutput do { - try audioInputNode.setVoiceProcessingEnabled(true) + try audioInputNode.setVoiceProcessingEnabled(shouldEnableVoiceProcessing) } catch { if let nsError = error as NSError? { - print("启用语音处理失败: domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)") + print("语音处理设置失败(enable=\(shouldEnableVoiceProcessing)): domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)") } else { - print("启用语音处理失败: \(error.localizedDescription)") + print("语音处理设置失败(enable=\(shouldEnableVoiceProcessing)): \(error.localizedDescription)") } } } @@ -514,14 +516,7 @@ public class MicrophoneCapture: NSObject { // 修复:使用安全的清理方法 cleanupAudioEngine() - // 安全地处理音频会话 - guard let audioSession = audioSession else { return } - do { - try audioSession.setActive(false, options: .notifyOthersOnDeactivation) - } catch { - print("音频会话配置失败: \(error.localizedDescription)") - } - + // 安全地处理音频会话:这里不主动 deactivate,避免与后续播报会话切换打架导致卡顿/失败 } /** diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift index a2488a30c..3522a1630 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift @@ -166,25 +166,20 @@ public class SimpleAudioReceiver: NSObject { if audioSourceType == .microphone || isHybridExternalMode { // --- 麦克风输入逻辑 (包括混合模式) --- - desiredCategory = .playAndRecord - if currentRecognitionMode == "phone_call" { - desiredMode = .voiceChat if isHeadphones { - // 修正:增加 A2DP 选项以支持高质量蓝牙音频播放 - desiredOptions = [.allowBluetooth, .allowBluetoothA2DP] + desiredCategory = .playAndRecord + desiredMode = .default + desiredOptions = [.allowBluetoothA2DP] } else { - desiredOptions = [.allowBluetooth, .defaultToSpeaker] - } - } else { // normal, push_to_talk, ble_wakeup 等 - if isHeadphones { - // 修正:使用 .voiceChat 模式以避免 .videoChat 强制外放 + desiredCategory = .playAndRecord desiredMode = .voiceChat - desiredOptions = [.allowBluetooth, .allowBluetoothA2DP, .mixWithOthers] - } else { - desiredMode = .videoChat - desiredOptions = [.mixWithOthers, .defaultToSpeaker] + desiredOptions = [.allowBluetooth, .defaultToSpeaker] } + } else { // normal, push_to_talk, ble_wakeup 等:仅录音,不需要同时播报 + desiredCategory = .record + desiredMode = .default + desiredOptions = [] } // 如果在真实通话中,强制使用更适合的录音设置 @@ -197,8 +192,8 @@ public class SimpleAudioReceiver: NSObject { } else if audioSourceType == .external { // --- 纯外部源 (仅播放) 逻辑 --- desiredCategory = .playback - desiredMode = .videoChat - desiredOptions = isHeadphones ? [.allowBluetoothA2DP, .mixWithOthers] : [.mixWithOthers, .defaultToSpeaker] + desiredMode = .spokenAudio + desiredOptions = isHeadphones ? [.allowBluetoothA2DP, .mixWithOthers] : [.mixWithOthers] } else { // 兜底,理论上不会执行 desiredCategory = .playAndRecord @@ -217,10 +212,12 @@ public class SimpleAudioReceiver: NSObject { // --- 路由设置 --- if audioSourceType == .microphone || isHybridExternalMode { // 为需要麦克风的场景设置输入输出 - if isHeadphones { - try audioSession.overrideOutputAudioPort(.none) // 允许耳机/蓝牙输出 - } else { - try audioSession.overrideOutputAudioPort(.speaker) // 强制外放 + if desiredCategory == .playAndRecord { + if isHeadphones { + try audioSession.overrideOutputAudioPort(.none) // 允许耳机/蓝牙输出 + } else { + try audioSession.overrideOutputAudioPort(.speaker) // 强制外放 + } } // 设置首选输入设备 @@ -248,7 +245,7 @@ public class SimpleAudioReceiver: NSObject { } } } else { // 纯外部源 - try audioSession.overrideOutputAudioPort(.none) + // .overrideOutputAudioPort 仅对 .playAndRecord 有效,.playback 会触发 -50 } } @@ -280,7 +277,8 @@ public class SimpleAudioReceiver: NSObject { out.portType == .bluetoothA2DP || out.portType == .bluetoothHFP || out.portType == .bluetoothLE || - out.portType == .headphones + out.portType == .headphones || + out.portType == .headsetMic } }