From aa31f526f77eeba44f8a3ad572366ee082f5f9c4 Mon Sep 17 00:00:00 2001 From: liwei1dao Date: Sat, 24 Jan 2026 14:55:09 +0800 Subject: [PATCH] =?UTF-8?q?fix(azure=5Fspeech):=20=E4=BF=AE=E5=A4=8D?= =?UTF-8?q?=E8=AF=AD=E9=9F=B3=E5=90=88=E6=88=90=E4=BC=9A=E8=AF=9D=E7=AE=A1?= =?UTF-8?q?=E7=90=86=E9=80=BB=E8=BE=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修复 Azure TTS 助手中的会话管理问题,确保音频合成和播放仅在正确的会话上下文中执行。移除冗余的会话 ID 处理逻辑,并添加会话快照检查以防止跨会话音频数据干扰。 具体修改包括: - 在合成开始和音频处理事件中添加会话 ID 一致性检查 - 移除 `effectiveSessionId` 计算逻辑,直接使用传入的 sessionid - 在停止合成时清空会话快照 - 注释调试日志以减少控制台输出 --- .../azure_speech/AzureSpeechPlugin.swift | 2 +- .../Sources/azure_speech/AzureTtsHelper.swift | 55 +++++++++++++++---- pubspec.yaml | 2 +- 3 files changed, 45 insertions(+), 14 deletions(-) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index ce82256b1..1a00c6ef8 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -1116,7 +1116,7 @@ extension AzureSpeechPlugin: BleService.Callback { } } - print("分离音频数据 - 左声道: \(leftBuffer.count) bytes, 右声道: \(rightBuffer.count) bytes") + // print("分离音频数据 - 左声道: \(leftBuffer.count) bytes, 右声道: \(rightBuffer.count) bytes") // 安全解包版本 guard let audioStream = azureAsrHelper.audioStream else { diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index cb1be7642..4160e2345 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -18,6 +18,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { private var isInitialized = false private var speaking = false private var sessionid = "" //语音播报任务id + private var synthesisSessionIdSnapshot = "" // 当前配置 private var currentVoice = DEFAULT_VOICE @@ -192,10 +193,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { return false } - let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid - if !sessionid.isEmpty { - self.sessionid = sessionid - } + // let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid + // if !sessionid.isEmpty { + self.sessionid = sessionid + // } // os_log("调用链: speakOnce 接收 session=%{public}@ effective=%{public}@ speaking=%{public}@ pendingTasks=%{public}d", log: log, type: .info, sessionid, effectiveSessionId, speaking.description, pendingTasks.count) // 清理文本 @@ -208,7 +209,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // 异步处理 // os_log("调用链: speakOnce 入队合成 session=%{public}@ 文本长度=%{public}d", log: log, type: .info, effectiveSessionId, cleanedText.count) enqueueSynthesisTask { - self.performSynthesis(sessionid: effectiveSessionId, text: cleanedText) + self.performSynthesis(sessionid: sessionid, text: cleanedText) } return true @@ -225,9 +226,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { ]) return false } - let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid - if !sessionid.isEmpty { - self.sessionid = sessionid + // let effectiveSessionId = sessionid.isEmpty ? self.sessionid : sessionid + // if !sessionid.isEmpty { + if self.sessionid != sessionid { + return false } // os_log("调用链: speakStream 接收 session=%{public}@ effective=%{public}@ 输入长度=%{public}d bufferLen=%{public}d", log: log, type: .info, sessionid, effectiveSessionId, text.count, streamBuffer.count) @@ -279,7 +281,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { if !textToSpeak.isEmpty { // os_log("调用链: speakStream 触发合成 session=%{public}@ speakLen=%{public}d remainLen=%{public}d", log: log, type: .info, effectiveSessionId, textToSpeak.count, streamBuffer.count) - return speakOnce(sessionid: effectiveSessionId, textToSpeak) + return speakOnce(sessionid: sessionid, textToSpeak) } } @@ -289,7 +291,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { let startIndex = currentText.index(currentText.startIndex, offsetBy: maxCharsBeforeForcedSpeak) streamBuffer = String(currentText[startIndex...]) // os_log("调用链: speakStream 强制分段合成 session=%{public}@ speakLen=%{public}d remainLen=%{public}d", log: log, type: .info, effectiveSessionId, textToSpeak.count, streamBuffer.count) - return speakOnce(sessionid: effectiveSessionId, textToSpeak) + return speakOnce(sessionid: sessionid, textToSpeak) } // os_log("调用链: speakStream 未触发合成 session=%{public}@ bufferLen=%{public}d", log: log, type: .debug, effectiveSessionId, streamBuffer.count) @@ -307,6 +309,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { ]) return false } + if self.sessionid != sessionid { + return false + } // os_log("调用链: flushStream session=%{public}@ bufferLen=%{public}d", log: log, type: .info, sessionid, streamBuffer.count) let remainingText = streamBuffer @@ -326,8 +331,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { * - Throws: 无(内部捕获 SDK/音频会话异常并转为 error 事件) */ public func stop() -> Bool { + os_log("liwei------------- 停止语音合成 session=%{public}@", log: self.log, type: .info, sessionid) speaking = false sessionid = "" + synthesisSessionIdSnapshot = "" streamBuffer = "" lastSpokenText = nil @@ -356,6 +363,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { taskLock.unlock() do { + try synthesizer?.stopSpeaking() self.notifyEvent(eventType: .synthesisCanceled) } catch { @@ -563,9 +571,10 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { do { if (self.sessionid != sessionid) { - // os_log("调用链: performSynthesis 中止(会话变化) current=%{public}@ task=%{public}@", log: log, type: .info, self.sessionid, sessionid) + os_log("liwei----------- 调用链: performSynthesis 中止(会话变化) current=%{public}@ task=%{public}@", log: log, type: .info, self.sessionid, sessionid) return } + synthesisSessionIdSnapshot = sessionid // os_log("调用链: performSynthesis startSpeakingSsml session=%{public}@ ssmlLen=%{public}d", log: log, type: .info, sessionid, ssml.count) let result = try synthesizer?.startSpeakingSsml(ssml) @@ -576,7 +585,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { speaking = false // 减少待处理文本计数 pendingTextCount = max(0, pendingTextCount - text.count) - os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) + os_log("liwei-------------- 语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) notifyEvent(eventType: .error, params: [ "errorCode": "SYNTHESIS_FAILED", "errorMessage": error.localizedDescription @@ -653,6 +662,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { guard let self = self else { return } self.currentSynthesisBuffer = Data() if self.suppressStreamPlayback { return } + self.synthesisSessionIdSnapshot = self.sessionid if self.isUsingPushStreamCapture { self.audioPlaybackQueue.async { self.activeStreamSynthesisCount += 1 @@ -669,6 +679,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { synthesizer?.addSynthesizingEventHandler { [weak self] _, event in if let audioData = event.result.audioData { guard let self = self else { return } + if self.synthesisSessionIdSnapshot != self.sessionid { + return + } // os_log("调用链: 合成中 接收音频片段=%{public}dB session=%{public}@", log: self.log, type: .debug, audioData.count, self.sessionid) if self.isUsingPushStreamCapture { if self.pushStreamCallbackCount == 0, !audioData.isEmpty { @@ -691,6 +704,18 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { if !self.isUsingPushStreamCapture { self.speaking = false } + if self.synthesisSessionIdSnapshot != self.sessionid { + self.currentSynthesisBuffer = Data() + if self.isUsingPushStreamCapture { + self.audioPlaybackQueue.async { + self.activeStreamSynthesisCount = max(0, self.activeStreamSynthesisCount - 1) + self.checkStreamPlaybackCompletedIfNeeded() + } + } + self.pendingTextCount = max(0, self.pendingTextCount - 1) + self.notifyEvent(eventType: .synthesisCompleted) + return + } // 将当前合成的音频加入播放队列 if !self.currentSynthesisBuffer.isEmpty { // os_log("调用链: 合成完成 入队播放 数据长度=%{public}dB session=%{public}@", log: self.log, type: .info, self.currentSynthesisBuffer.count, self.sessionid) @@ -771,6 +796,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // os_log("调用链: 推送音频丢弃(无会话且未speaking) bytes=%{public}d", log: log, type: .debug, data.count) return } + if synthesisSessionIdSnapshot != sessionid { + return + } notifyAudioData(data) guard isUsingPushStreamCapture else { // os_log("调用链: 推送音频忽略(未启用pushStream) bytes=%{public}d", log: log, type: .debug, data.count) @@ -790,6 +818,9 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { // os_log("调用链: 推送音频异步丢弃(无会话且未speaking) bytes=%{public}d", log: self.log, type: .debug, data.count) return } + if self.synthesisSessionIdSnapshot != self.sessionid { + return + } let pcm = self.stripWavHeaderIfNeeded(data) if !pcm.isEmpty { self.pcmPendingData.append(pcm) diff --git a/pubspec.yaml b/pubspec.yaml index e7d6eb828..32e746e95 100644 --- a/pubspec.yaml +++ b/pubspec.yaml @@ -1,7 +1,7 @@ name: voitrans description: "Voitrans - AI Voice Assistant." publish_to: "none" -version: 1.0.21+71 +version: 1.0.22+71 environment: sdk: ">=3.3.0 <4.0.0"