diff --git a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift index e7bb7a9af..08f7f4b7f 100644 --- a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift +++ b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift @@ -342,12 +342,12 @@ class AgentServiceImpl: NSObject { sendError("初始化语音识别服务失败", code: "ASR_INIT_ERROR") return false } - guard let ttsSuccess = azureTtsHelper?.initialize( ttsAppId: "", ttsAppToken: azureSpeechKey, ttsResource: azureSpeechRegion, - language: primaryLanguage + language: primaryLanguage, + isExtaudioSource: BleService.shared.isConnected() ), ttsSuccess else { sendError("初始化语音合成服务失败", code: "TTS_INIT_ERROR") return false @@ -619,7 +619,7 @@ class AgentServiceImpl: NSObject { return true } - func speakText(sessionid sessionid:String,_ text: String) -> Bool { + func speakText(sessionid: String, _ text: String) -> Bool { if !isInitialized { sendError("服务未初始化", code: "NOT_INITIALIZED") return false @@ -630,6 +630,7 @@ class AgentServiceImpl: NSObject { } _ = azureTtsHelper?.setSpeechParams(rate: ttsRatePercent, pitch: 0, volume: 100) + _ = azureTtsHelper?.setAudioSourceType(isExtaudioSource: BleService.shared.isConnected()) return azureTtsHelper?.speakOnce(sessionid:sessionid,text) ?? false } @@ -1788,6 +1789,7 @@ class ChatApiStreamCallback: StreamCallback { responseBuilder += token if speakResponse && reply && broadcast{ + _ = agentService.azureTtsHelper?.setAudioSourceType(isExtaudioSource: BleService.shared.isConnected()) try agentService.azureTtsHelper?.speakStream(sessionid:sessionid,token) // 在开始流式TTS时立即停止气泡音 } @@ -1813,6 +1815,7 @@ class ChatApiStreamCallback: StreamCallback { } if speakResponse && reply && broadcast && sessionid == agentService.currsessionId{ + _ = agentService.azureTtsHelper?.setAudioSourceType(isExtaudioSource: BleService.shared.isConnected()) agentService.azureTtsHelper?.flushStream(sessionid:sessionid) } @@ -1864,6 +1867,7 @@ class ChatApiStreamCallback: StreamCallback { "message": message ]) if(code == 2001){ + _ = agentService.azureTtsHelper?.setAudioSourceType(isExtaudioSource: BleService.shared.isConnected()) agentService.azureTtsHelper?.speakStream(sessionid: sessionid,agentService.insufficientIntegralText) } agentService.isAiStreaming = false @@ -1907,6 +1911,7 @@ class ChatApiStreamCallback: StreamCallback { if (iscallingTool) { os_log("收到函数调用: callingToolText:%{public}@", log: agentService.logger, type: .info, agentService.callingToolText) + _ = agentService.azureTtsHelper?.setAudioSourceType(isExtaudioSource: BleService.shared.isConnected()) agentService.azureTtsHelper?.speakStream(sessionid: sessionid,agentService.callingToolText) iscallingTool = false } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index 664d5ea9b..b59031000 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -48,6 +48,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // 自定义音频输出流 private var customAudioOutputStream: SPXPushAudioOutputStream? + private var internalAudioOutputStream: SPXPushAudioOutputStream? private var useInternalPlayer = false private var micCapture: MicrophoneCapture! // 记录最后播放的文本 @@ -67,16 +68,51 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { private var pcmFormat: AVAudioFormat? private var pcmPendingData = Data() private var hasStrippedWavHeader = false + private var wavHeaderProbeBuffer = Data() private var hasNotifiedPlaybackStartedForStream = false private var activeStreamSynthesisCount = 0 private var scheduledBufferCount = 0 - private let streamChunkBytes = 1600 + private let streamChunkBytes = 3200 private var suppressStreamPlayback = false + private var pushStreamChunkCount = 0 + private var pushStreamCallbackCount = 0 + private var synthEventAudioCount = 0 + private var isExtaudioSource = false + + /** * 初始化语音合成服务 + * - Parameters: + * - ttsAppId: 服务应用ID(Azure 场景未使用,可为空) + * - ttsAppToken: Azure 语音服务订阅密钥 + * - ttsResource: Azure 语音服务区域 + * - language: 语言代码,如 "zh-CN" + * - Returns: 是否初始化成功 + * - Throws: 无(内部捕获 SDK 异常并记录日志) */ public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { + return initialize( + ttsAppId: ttsAppId, + ttsAppToken: ttsAppToken, + ttsResource: ttsResource, + language: language, + isExtaudioSource: false + ) + } + + /** + * 初始化语音合成服务(支持外部音频源场景) + * - Parameters: + * - ttsAppId: 服务应用ID(Azure 场景未使用,可为空) + * - ttsAppToken: Azure 语音服务订阅密钥 + * - ttsResource: Azure 语音服务区域 + * - language: 语言代码,如 "zh-CN" + * - isExtaudioSource: 是否为外部音频源(例如 BLE 输入) + * - Returns: 是否初始化成功 + * - Throws: 无(内部捕获 SDK 异常并记录日志) + */ + public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String, isExtaudioSource: Bool) -> Bool { do { micCapture = MicrophoneCapture.shared // 创建语音配置 @@ -94,7 +130,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // 创建合成器 recreateSynthesizer() - + self.isExtaudioSource = isExtaudioSource // 设置初始化完成 isInitialized = true @@ -147,9 +183,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { * @param sessionid 会话ID * @return 是否成功启动 */ - public func startspeak(sessionid sessionid: String) -> Bool { + public func startspeak(sessionid: String) -> Bool { self.sessionid = sessionid print("startspeak sessionid \(sessionid)") + // os_log("调用链: startspeak 设置session=%{public}@", log: log, type: .info, sessionid) // 重置当前会话的文本计数 self.currentSessionTextCount = 0 self.pendingTextCount = 0 @@ -163,7 +200,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { /** * 单次合成并播放 */ - public func speakOnce(sessionid sessionid: String, _ text: String) -> Bool { + public func speakOnce(sessionid: String, _ text: String) -> Bool { if !isInitialized { os_log("语音合成未初始化", log: log, type: .error) notifyEvent(eventType: .error, params: [ @@ -173,20 +210,23 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { return false } - if (self.sessionid != sessionid){ - return false + let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid + if !sessionid.isEmpty { + self.sessionid = sessionid } + // os_log("调用链: speakOnce 接收 session=%{public}@ effective=%{public}@ speaking=%{public}@ pendingTasks=%{public}d", log: log, type: .info, sessionid, effectiveSessionId, speaking.description, pendingTasks.count) // 清理文本 let cleanedText = cleanTextForTTS(text) if cleanedText.isEmpty { + // os_log("调用链: speakOnce 文本清理后为空 session=%{public}@", log: log, type: .info, effectiveSessionId) return false } // 异步处理 - os_log("调用链: speakOnce 入队合成 session=%{public}@ 文本长度=%{public}d", log: log, type: .info, sessionid, cleanedText.count) + // os_log("调用链: speakOnce 入队合成 session=%{public}@ 文本长度=%{public}d", log: log, type: .info, effectiveSessionId, cleanedText.count) enqueueSynthesisTask { - self.performSynthesis(sessionid: sessionid, text: cleanedText) + self.performSynthesis(sessionid: effectiveSessionId, text: cleanedText) } return true @@ -195,7 +235,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { /** * 流式合成文本 */ - public func speakStream(sessionid sessionid: String, _ text: String) -> Bool { + public func speakStream(sessionid: String, _ text: String) -> Bool { if !isInitialized { notifyEvent(eventType: .error, params: [ "errorCode": "NOT_INITIALIZED", @@ -203,12 +243,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { ]) return false } - if (self.sessionid != sessionid){ - return false + let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid + if !sessionid.isEmpty { + self.sessionid = sessionid } - self.sessionid = sessionid + // os_log("调用链: speakStream 接收 session=%{public}@ effective=%{public}@ 输入长度=%{public}d bufferLen=%{public}d", log: log, type: .info, sessionid, effectiveSessionId, text.count, streamBuffer.count) if text.isEmpty { + // os_log("调用链: speakStream 空输入直接返回 session=%{public}@", log: log, type: .info, effectiveSessionId) return true } @@ -218,11 +260,17 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // 清理文本并添加到缓冲区 let cleanedText = cleanTextForTTS(text) + if cleanedText.isEmpty { + // os_log("调用链: speakStream 清理后为空 session=%{public}@ rawLen=%{public}d", log: log, type: .info, effectiveSessionId, text.count) + return true + } streamBuffer.append(cleanedText) + // os_log("调用链: speakStream 入缓冲 session=%{public}@ cleanedLen=%{public}d bufferLen=%{public}d", log: log, type: .info, effectiveSessionId, cleanedText.count, streamBuffer.count) // 防抖逻辑 (150ms) let currentTime = Date().timeIntervalSince1970 if currentTime - lastSpeakTime < 0.15 { + // os_log("调用链: speakStream 防抖跳过 session=%{public}@ deltaMs=%{public}d", log: log, type: .info, effectiveSessionId, Int((currentTime - lastSpeakTime) * 1000.0)) return true } lastSpeakTime = currentTime @@ -248,17 +296,28 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { streamBuffer = String(currentText[startIndex...]) if !textToSpeak.isEmpty { - return speakOnce(sessionid: sessionid, textToSpeak) + // os_log("调用链: speakStream 触发合成 session=%{public}@ speakLen=%{public}d remainLen=%{public}d", log: log, type: .info, effectiveSessionId, textToSpeak.count, streamBuffer.count) + return speakOnce(sessionid: effectiveSessionId, textToSpeak) } } + let maxCharsBeforeForcedSpeak = 120 + if currentText.count >= maxCharsBeforeForcedSpeak { + let textToSpeak = String(currentText.prefix(maxCharsBeforeForcedSpeak)) + let startIndex = currentText.index(currentText.startIndex, offsetBy: maxCharsBeforeForcedSpeak) + streamBuffer = String(currentText[startIndex...]) + // os_log("调用链: speakStream 强制分段合成 session=%{public}@ speakLen=%{public}d remainLen=%{public}d", log: log, type: .info, effectiveSessionId, textToSpeak.count, streamBuffer.count) + return speakOnce(sessionid: effectiveSessionId, textToSpeak) + } + + // os_log("调用链: speakStream 未触发合成 session=%{public}@ bufferLen=%{public}d", log: log, type: .debug, effectiveSessionId, streamBuffer.count) return true } /** * 刷新并播放流式文本 */ - public func flushStream(sessionid sessionid: String) -> Bool { + public func flushStream(sessionid: String) -> Bool { if !isInitialized { notifyEvent(eventType: .error, params: [ "errorCode": "NOT_INITIALIZED", @@ -266,6 +325,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { ]) return false } + // os_log("调用链: flushStream session=%{public}@ bufferLen=%{public}d", log: log, type: .info, sessionid, streamBuffer.count) let remainingText = streamBuffer streamBuffer = "" @@ -327,9 +387,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { synthesisQueue.async { self.recreateSynthesizer() } - - deactivateAudioSessionAfterTTSIfIdle() + deactivateAudioSessionAfterTTSIfIdle() + return true } @@ -372,6 +432,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { } } + public func setAudioSourceType(isExtaudioSource: Bool) -> Bool { + self.isExtaudioSource = isExtaudioSource + if isInitialized { + safeConfigureAudioSessionForTTS() + } + return true + } + /** * 设置音频输出设备 */ @@ -411,7 +479,8 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { taskLock.lock() pendingTasks.append(task) taskLock.unlock() - + // os_log("调用链: enqueueSynthesisTask 入队 queued=%{public}d processing=%{public}@", log: log, type: .debug, queued, isProcessing.description) + // 确保任务被处理 processTasksIfNeeded() } @@ -430,6 +499,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { isProcessing = true let task = pendingTasks.removeFirst() taskLock.unlock() + // os_log("调用链: processTasksIfNeeded 开始执行 remaining=%{public}d", log: log, type: .info, remaining) synthesisQueue.async { task() @@ -444,6 +514,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { self.processTasksIfNeeded() } else { self.taskLock.unlock() + // os_log("调用链: processTasksIfNeeded 队列清空", log: self.log, type: .info) } } } @@ -471,10 +542,12 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { */ private func performSynthesis(sessionid:String,text: String) { if (self.sessionid != sessionid) { + // os_log("调用链: performSynthesis 丢弃(会话不匹配) current=%{public}@ task=%{public}@", log: log, type: .info, self.sessionid, sessionid) return } speaking = true // lastSpokenText = text + // os_log("调用链: performSynthesis 准备合成 session=%{public}@ pushStream=%{public}@ suppress=%{public}@ textLen=%{public}d", log: log, type: .info, sessionid, isUsingPushStreamCapture.description, suppressStreamPlayback.description, text.count) if !applyAudioSessionForTTS() { speaking = false notifyEvent(eventType: .error, params: [ @@ -488,13 +561,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { do { if (self.sessionid != sessionid) { + // os_log("调用链: performSynthesis 中止(会话变化) current=%{public}@ task=%{public}@", log: log, type: .info, self.sessionid, sessionid) return } - os_log("开始合成: %{public}@", log: log, type: .debug, ssml) + // os_log("调用链: performSynthesis startSpeakingSsml session=%{public}@ ssmlLen=%{public}d", log: log, type: .info, sessionid, ssml.count) let result = try synthesizer?.startSpeakingSsml(ssml) if let result = result { - os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) + // os_log("调用链: performSynthesis 返回 reason=%{public}@ session=%{public}@", log: log, type: .debug, String(describing: result.reason), sessionid) } } catch { speaking = false @@ -514,9 +588,11 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { */ private func recreateSynthesizer() { do { + // os_log("调用链: recreateSynthesizer 开始 useInternalPlayer=%{public}@ customStream=%{public}@ initialized=%{public}@", log: log, type: .info, useInternalPlayer.description, (customAudioOutputStream != nil).description, isInitialized.description) // 强制释放旧的合成器 synthesizer = nil isUsingPushStreamCapture = false + internalAudioOutputStream = nil // 创建音频配置 let audioConfig: SPXAudioConfiguration? @@ -535,13 +611,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { return UInt(data.count) }, closeHandler: { [weak self] in guard let self = self else { return } - os_log("调用链: 输出流关闭 累计字节=%{public}d", log: self.log, type: .info, totalBytes) + // os_log("调用链: 输出流关闭 累计字节=%{public}d", log: self.log, type: .info, totalBytes) totalBytes = 0 }) // 如果创建失败,抛出以进入 catch guard let nonNilStream = created else { throw NSError(domain: "AzureTtsHelper", code: -1, userInfo: [NSLocalizedDescriptionKey: "创建推送输出流失败"]) } + internalAudioOutputStream = nonNilStream stream = nonNilStream } audioConfig = try SPXAudioConfiguration(streamOutput: stream) @@ -556,6 +633,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { synthesizer = try SPXSpeechSynthesizer(speechConfig!) } setupEventListeners() + // os_log("调用链: recreateSynthesizer 完成 pushStream=%{public}@", log: log, type: .info, isUsingPushStreamCapture.description) } catch { os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) notifyEvent(eventType: .error, params: [ @@ -578,16 +656,27 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { self.activeStreamSynthesisCount += 1 self.prepareStreamingQueueForNewSynthesisIfNeeded() } + self.pushStreamChunkCount = 0 + self.pushStreamCallbackCount = 0 + self.synthEventAudioCount = 0 } - os_log("调用链: 合成开始 session=%{public}@", log: self.log, type: .info, self.sessionid) + // os_log("调用链: 合成开始 session=%{public}@ pushStream=%{public}@ active=%{public}d", log: self.log, type: .info, self.sessionid, self.isUsingPushStreamCapture.description, self.activeStreamSynthesisCount) self.notifyEvent(eventType: .synthesisStarted) } synthesizer?.addSynthesizingEventHandler { [weak self] _, event in if let audioData = event.result.audioData { guard let self = self else { return } - os_log("调用链: 合成中 接收音频片段=%{public}dB session=%{public}@", log: self.log, type: .debug, audioData.count, self.sessionid) - if !self.isUsingPushStreamCapture { + // os_log("调用链: 合成中 接收音频片段=%{public}dB session=%{public}@", log: self.log, type: .debug, audioData.count, self.sessionid) + if self.isUsingPushStreamCapture { + if self.pushStreamCallbackCount == 0, !audioData.isEmpty { + self.synthEventAudioCount += 1 + if self.synthEventAudioCount == 1 { + // os_log("调用链: 推送流无回调,使用合成事件音频回退 session=%{public}@", log: self.log, type: .info, self.sessionid) + } + self.handleSynthesizedAudioChunk(audioData, source: "synthEvent") + } + } else { self.currentSynthesisBuffer.append(audioData) self.notifyAudioData(audioData) } @@ -602,7 +691,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { } // 将当前合成的音频加入播放队列 if !self.currentSynthesisBuffer.isEmpty { - os_log("调用链: 合成完成 入队播放 数据长度=%{public}dB session=%{public}@", log: self.log, type: .info, self.currentSynthesisBuffer.count, self.sessionid) + // os_log("调用链: 合成完成 入队播放 数据长度=%{public}dB session=%{public}@", log: self.log, type: .info, self.currentSynthesisBuffer.count, self.sessionid) self.enqueuePlaybackItem(self.currentSynthesisBuffer) self.currentSynthesisBuffer = Data() } @@ -613,6 +702,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { self.checkStreamPlaybackCompletedIfNeeded() } } + // os_log("调用链: 合成完成 session=%{public}@ pushStream=%{public}@ pendingPcm=%{public}d scheduled=%{public}d", log: self.log, type: .info, self.sessionid, self.isUsingPushStreamCapture.description, self.pcmPendingData.count, self.scheduledBufferCount) // 减少待处理文本计数 self.pendingTextCount = max(0, self.pendingTextCount - 1) // 通知合成完成 @@ -667,16 +757,50 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { * - Throws: 无 */ private func handleSynthesizedAudioChunk(_ data: Data) { - if suppressStreamPlayback { return } - if sessionid.isEmpty { return } + handleSynthesizedAudioChunk(data, source: "pushStream") + } + + private func handleSynthesizedAudioChunk(_ data: Data, source: String) { + if suppressStreamPlayback { + // os_log("调用链: 推送音频丢弃(suppress) bytes=%{public}d", log: log, type: .debug, data.count) + return + } + if sessionid.isEmpty, !speaking { + // os_log("调用链: 推送音频丢弃(无会话且未speaking) bytes=%{public}d", log: log, type: .debug, data.count) + return + } notifyAudioData(data) - guard isUsingPushStreamCapture else { return } + guard isUsingPushStreamCapture else { + // os_log("调用链: 推送音频忽略(未启用pushStream) bytes=%{public}d", log: log, type: .debug, data.count) + return + } + if source == "pushStream" { + pushStreamCallbackCount += 1 + } else if source == "synthEvent" { + if pushStreamCallbackCount > 0 { return } + } + pushStreamChunkCount += 1 + if pushStreamChunkCount == 1 { + // os_log("调用链: 推送音频首包 source=%{public}@ session=%{public}@ bytes=%{public}d speaking=%{public}@", log: log, type: .info, source, sessionid, data.count, speaking.description) + } audioPlaybackQueue.async { - if self.sessionid.isEmpty { return } + if self.sessionid.isEmpty, !self.speaking { + // os_log("调用链: 推送音频异步丢弃(无会话且未speaking) bytes=%{public}d", log: self.log, type: .debug, data.count) + return + } let pcm = self.stripWavHeaderIfNeeded(data) if !pcm.isEmpty { self.pcmPendingData.append(pcm) + if self.scheduledBufferCount == 0 { + // os_log("调用链: 推送音频入缓冲 pcm=%{public}dB pendingPcm=%{public}d", log: self.log, type: .debug, pcm.count, self.pcmPendingData.count) + } self.scheduleAvailablePcmBuffers() + } else { + if self.pushStreamChunkCount <= 3 { + // os_log("调用链: 推送音频等待WAV头 session=%{public}@ probe=%{public}dB in=%{public}dB", log: self.log, type: .info, self.sessionid, self.wavHeaderProbeBuffer.count, data.count) + } else { + // os_log("调用链: 推送音频等待WAV头 probe=%{public}dB in=%{public}dB", log: self.log, type: .debug, self.wavHeaderProbeBuffer.count, data.count) + } } } } @@ -689,6 +813,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { private func prepareStreamingQueueForNewSynthesisIfNeeded() { if pcmPendingData.isEmpty, scheduledBufferCount == 0, playerNode?.isPlaying != true { hasStrippedWavHeader = false + wavHeaderProbeBuffer.removeAll(keepingCapacity: true) hasNotifiedPlaybackStartedForStream = false } } @@ -700,8 +825,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { */ private func resetStreamingPlaybackState() { audioPlaybackQueue.sync { + // os_log("调用链: resetStreamingPlaybackState 清理 before pendingPcm=%{public}d scheduled=%{public}d active=%{public}d", log: self.log, type: .info, self.pcmPendingData.count, self.scheduledBufferCount, self.activeStreamSynthesisCount) self.pcmPendingData = Data() self.hasStrippedWavHeader = false + self.wavHeaderProbeBuffer = Data() self.hasNotifiedPlaybackStartedForStream = false self.activeStreamSynthesisCount = 0 self.scheduledBufferCount = 0 @@ -710,6 +837,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { self.audioEngine = nil self.playerNode = nil self.pcmFormat = nil + // os_log("调用链: resetStreamingPlaybackState 清理完成", log: self.log, type: .info) } } @@ -720,6 +848,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { */ private func scheduleAvailablePcmBuffers() { guard ensureStreamingEngineIfNeeded() else { return } + if !hasNotifiedPlaybackStartedForStream, pcmPendingData.count >= streamChunkBytes { + // os_log("调用链: scheduleAvailablePcmBuffers 准备调度 pendingPcm=%{public}d chunk=%{public}d active=%{public}d", log: log, type: .info, pcmPendingData.count, streamChunkBytes, activeStreamSynthesisCount) + } while pcmPendingData.count >= streamChunkBytes { let chunk = pcmPendingData.prefix(streamChunkBytes) pcmPendingData.removeFirst(streamChunkBytes) @@ -727,7 +858,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { scheduledBufferCount += 1 if !hasNotifiedPlaybackStartedForStream { hasNotifiedPlaybackStartedForStream = true - os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid) + // os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid) notifyEvent(eventType: .playbackStarted) } playerNode?.scheduleBuffer(buffer, completionHandler: { [weak self] in @@ -747,7 +878,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { scheduledBufferCount += 1 if !hasNotifiedPlaybackStartedForStream { hasNotifiedPlaybackStartedForStream = true - os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid) + // os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid) notifyEvent(eventType: .playbackStarted) } playerNode?.scheduleBuffer(buffer, completionHandler: { [weak self] in @@ -759,7 +890,8 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { }) } } - if playerNode?.isPlaying == false { + if playerNode?.isPlaying == false, scheduledBufferCount > 0 { + // os_log("调用链: playerNode.play scheduled=%{public}d pendingPcm=%{public}d", log: log, type: .debug, scheduledBufferCount, pcmPendingData.count) playerNode?.play() } } @@ -773,7 +905,7 @@ private func checkStreamPlaybackCompletedIfNeeded() { if activeStreamSynthesisCount == 0, pcmPendingData.isEmpty, scheduledBufferCount == 0 { speaking = false hasNotifiedPlaybackStartedForStream = false - os_log("调用链: 流式播放完成 session=%{public}@", log: log, type: .info, sessionid) + // os_log("调用链: 流式播放完成 session=%{public}@", log: log, type: .info, sessionid) notifyEvent(eventType: .playbackCompleted) deactivateAudioSessionAfterTTSIfIdle() } @@ -862,25 +994,36 @@ private func checkStreamPlaybackCompletedIfNeeded() { */ private func stripWavHeaderIfNeeded(_ data: Data) -> Data { if hasStrippedWavHeader { return data } - if data.count >= 12, - String(data: data.subdata(in: 0..<4), encoding: .ascii) == "RIFF", - String(data: data.subdata(in: 8..<12), encoding: .ascii) == "WAVE" { + wavHeaderProbeBuffer.append(data) + if wavHeaderProbeBuffer.count < 12 { return Data() } + if wavHeaderProbeBuffer.count >= 12, + String(data: wavHeaderProbeBuffer.subdata(in: 0..<4), encoding: .ascii) == "RIFF", + String(data: wavHeaderProbeBuffer.subdata(in: 8..<12), encoding: .ascii) == "WAVE" { let marker = Data("data".utf8) - if let range = data.range(of: marker, options: [], in: 0.. 44 { + if wavHeaderProbeBuffer.count > 44 { hasStrippedWavHeader = true - return data.subdata(in: 44.. Bool { let audioSession = AVAudioSession.sharedInstance() let outputs: [AVAudioSessionPortDescription] = audioSession.currentRoute.outputs let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP }) + let hasWiredHeadphones = outputs.contains(where: { $0.portType == .headphones || $0.portType == .headsetMic }) + let hasBluetoothA2DP = outputs.contains(where: { $0.portType == .bluetoothA2DP }) + let hasHeadphones = hasWiredHeadphones || hasBluetoothA2DP let otherAudioPlaying = audioSession.isOtherAudioPlaying + let isRecording = (self.micCapture?.isCapturing == true) let desiredCategory: AVAudioSession.Category let desiredMode: AVAudioSession.Mode @@ -1218,6 +1365,21 @@ private func applyAudioSessionForTTS() -> Bool { desiredCategory = .playback desiredMode = .default desiredOptions.insert(.duckOthers) + } else if isRecording { + desiredCategory = .playAndRecord + desiredMode = self.isExtaudioSource ? .default : .videoChat + desiredOptions.insert(.allowBluetooth) + desiredOptions.insert(.allowBluetoothA2DP) + if self.isExtaudioSource { + if otherAudioPlaying { + desiredOptions.insert(.duckOthers) + } + } else { + desiredOptions.insert(.mixWithOthers) + if !hasHeadphones { + desiredOptions.insert(.defaultToSpeaker) + } + } } else { desiredCategory = .playback desiredMode = .spokenAudio @@ -1233,6 +1395,9 @@ private func applyAudioSessionForTTS() -> Bool { if audioSession.category != desiredCategory || audioSession.mode != desiredMode || audioSession.categoryOptions != desiredOptions { try audioSession.setCategory(desiredCategory, mode: desiredMode, options: desiredOptions) } + if desiredCategory == .playAndRecord { + _ = try? audioSession.setPreferredIOBufferDuration(0.02) + } try audioSession.setActive(true) break } catch { @@ -1244,11 +1409,15 @@ private func applyAudioSessionForTTS() -> Bool { } } catch { do { - try audioSession.setCategory(.playback, mode: .default, options: []) try audioSession.setActive(true) } catch { - success = false - os_log("TTS 音频会话配置失败: %{public}@", log: self.log, type: .error, error.localizedDescription) + do { + try audioSession.setCategory(.playback, mode: .default, options: []) + try audioSession.setActive(true) + } catch { + success = false + os_log("TTS 音频会话配置失败: %{public}@", log: self.log, type: .error, error.localizedDescription) + } } } }