From 2d8272f36b8521a6e7b3d98973a860668ed77ddd Mon Sep 17 00:00:00 2001 From: fdp <1286779656@qq.com> Date: Fri, 9 Jan 2026 12:24:16 +0800 Subject: [PATCH] =?UTF-8?q?=E6=8F=90=E5=89=8D=E5=90=A7=E5=90=88=E6=88=90?= =?UTF-8?q?=E6=95=B0=E6=8D=AE=E5=8F=91=E9=80=81?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../azure_speech/AzureAsrToAsr.kt | 54 +++++++++++++++---- .../azure_speech/AzureSpeechPlugin.kt | 8 +-- .../azure_speech/AzureSpeechPlugin.swift | 8 +-- .../IntegratedSpeechTranslationService.swift | 54 ++++++++++++++----- 4 files changed, 95 insertions(+), 29 deletions(-) diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt index b8e5e3056..070b042ed 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt @@ -80,6 +80,36 @@ class IntegratedSpeechTranslationService( // 语音映射缓存 private val voiceCache = ConcurrentHashMap() + + + // TTS合成中的音频缓存 + private var ttsdata: ByteArray = ByteArray(0) + + /** + * 处理合成音频分片(按1280字节对齐) + * 输入的音频数据会追加到缓存;仅当累计长度达到`multiple`的整数倍时, + * 将该整数倍长度的部分通过`onSynthesisAudioGenerated`发送;余数保留在缓存等待下次。 + */ + private fun processSynthesisAudioChunk( + text: String, + incoming: ByteArray, + multiple: Int = 1280 + ) { + if (incoming.isNotEmpty()) { + ttsdata = ttsdata + incoming + } + + val sendLen = (ttsdata.size / multiple) * multiple + if (sendLen > 0) { + val chunk = ttsdata.copyOfRange(0, sendLen) + Log.d(TAG, "按${multiple}字节对齐发送音频: ${chunk.size} 字节,缓存剩余: ${ttsdata.size - sendLen}") + eventCallback?.onSynthesisAudioGenerated(text, chunk) + ttsdata = ttsdata.copyOfRange(sendLen, ttsdata.size) + } else { + Log.d(TAG, "暂未够${multiple}字节,当前缓存: ${ttsdata.size}") + } + } + /** * 服务配置类 */ @@ -399,6 +429,7 @@ class IntegratedSpeechTranslationService( // 合成开始事件 SynthesisStarted.addEventListener { _, _ -> Log.d(TAG, "合成开始事件") + ttsdata = ByteArray(0) serviceState.isSynthesizing.set(true) eventCallback?.onStateChanged("Synthesis", true) } @@ -406,9 +437,11 @@ class IntegratedSpeechTranslationService( // 合成进行中事件 Synthesizing.addEventListener { _, event -> Log.d(TAG, "合成进行中事件") - // 计算进度(简化版) - val progress = 0.5f // 实际应用中可以根据音频数据计算 + // 计算进度(简化版) + val progress = 0.5f // 实际应用中可以根据音频数据计算 eventCallback?.onSynthesisProgress("", progress) + val incoming = event.result.audioData ?: ByteArray(0) + processSynthesisAudioChunk("", incoming, 1280) } // 合成完成事件 @@ -421,6 +454,7 @@ class IntegratedSpeechTranslationService( // 合成取消事件 SynthesisCanceled.addEventListener { _, event -> Log.d(TAG, "合成取消事件") + ttsdata = ByteArray(0) serviceState.isSynthesizing.set(false) val reason = event.result.reason eventCallback?.onError("Synthesis", "语音合成取消: $reason") @@ -605,6 +639,7 @@ class IntegratedSpeechTranslationService( try { Log.d(TAG, "开始语音合成: $text") serviceState.isSynthesizing.set(true) + eventCallback?.onSynthesisStarted(text) val ssml = generateOptimizedSsml(text) @@ -618,12 +653,13 @@ class IntegratedSpeechTranslationService( // 获取音频数据 val audioData = result.audioData - if (audioData != null && audioData.isNotEmpty()) { - // 通过回调返回音频数据 - eventCallback?.onSynthesisAudioGenerated(text, audioData) - Log.d(TAG, "音频数据大小: ${audioData.size} 字节") - } - + // if (audioData != null && audioData.isNotEmpty()) { + // // 通过回调返回音频数据 + // eventCallback?.onSynthesisAudioGenerated(text, audioData) + // Log.d(TAG, "音频数据大小: ${audioData.size} 字节") + // } + eventCallback?.onSynthesisAudioGenerated(text, ttsdata) + ttsdata = ByteArray(0) return audioData } else { Log.e(TAG, "语音合成失败: ${result?.reason}") @@ -1733,4 +1769,4 @@ class VolcanoTranslationServiceImpl : Log.e(TAG, "清理资源失败", e) } } -} \ No newline at end of file +} diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt index abc20de31..c5846713c 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -1027,10 +1027,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { override fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) { FileLogger.d(tag, "语音合成音频生成,文本=${text},音频数据大小=${audioData.size}") // Azure TTS返回的是RIFF WAV格式,需要去除文件头 - val pcmData = extractPcmFromWav(audioData) - if (pcmData.isNotEmpty()) { - FileLogger.d(tag, "提取PCM数据,大小=${pcmData.size}字节") - BleService.writeExternalAudioData(pcmData) + //val pcmData = extractPcmFromWav(audioData) + if (audioData.isNotEmpty()) { + FileLogger.d(tag, "提取PCM数据,大小=${audioData.size}字节") + BleService.writeExternalAudioData(audioData) } else { FileLogger.e(tag, "提取PCM数据失败,音频数据可能不是WAV格式") diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index a3c9cc9b3..0c065cedf 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -1039,10 +1039,10 @@ private class AstEventCallback: IntegratedSpeechTranslationService.ServiceEventC os_log("语音合成音频生成,文本=%@,音频数据大小=%d", log: plugin.log, type: .info, text, audioData.count) // Azure TTS返回的是RIFF WAV格式,需要去除文件头提取PCM数据 - let pcmData = plugin.extractPcmFromWav(audioData) - if !pcmData.isEmpty { - os_log("提取PCM数据,大小=%d字节", log: plugin.log, type: .info, pcmData.count) - BleService.shared.writeExternalAudioData(data: pcmData) + // let pcmData = plugin.extractPcmFromWav(audioData) + if !audioData.isEmpty { + os_log("提取PCM数据,大小=%d字节", log: plugin.log, type: .info, audioData.count) + BleService.shared.writeExternalAudioData(data: audioData) } else { os_log("提取PCM数据失败,音频数据可能不是WAV格式", log: plugin.log, type: .error) } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift index 97db886f4..a404cf89a 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift @@ -66,6 +66,9 @@ import os.log // 处理队列 private let processingQueue = DispatchQueue(label: "com.azure.speech.processing", qos: .userInitiated) + // 合成中的音频缓存 + private var ttsBuffer = Data() + // MARK: - Configuration Structures /** @@ -478,6 +481,7 @@ import os.log // 合成开始事件 synthesizer?.addSynthesisStartedEventHandler { [weak self] (synthesizer, event) in guard let self = self else { return } + self.ttsBuffer = Data() os_log("合成开始事件", log: self.log, type: .info) self.serviceState.isSynthesizing = true @@ -487,7 +491,7 @@ import os.log } } - // 合成进行中事件 - 修改为返回音频数据 + // 合成进行中事件 - 增量发送音频分片(按1280字节对齐) synthesizer?.addSynthesizingEventHandler { [weak self] (synthesizer, event) in guard let self = self else { return } // 安全解包audioData @@ -495,8 +499,8 @@ import os.log os_log("合成进行中事件,音频数据为空", log: self.log, type: .debug) return } - os_log("合成进行中事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) - + self.processSynthesisAudioChunk(text: "", incoming: audioData, multiple: 1280) + os_log("合成进行中事件,累计缓存长度: %d bytes", log: self.log, type: .debug, self.ttsBuffer.count) } // 合成完成事件 @@ -505,16 +509,14 @@ import os.log os_log("合成完成事件", log: self.log, type: .info) // 安全解包audioData - guard let audioData = event.result.audioData else { - os_log("合成完成事件,音频数据为空", log: self.log, type: .debug) - return - } - os_log("合成完成事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) + + // 发送缓存中的剩余数据(可能非1280对齐的余量) + let remaining = self.ttsBuffer + self.ttsBuffer = Data() DispatchQueue.main.async { - self.serviceState.isSynthesizing = false - // 立即返回当前生成的音频数据 - self.eventCallback?.onSynthesisAudioGenerated(text: "", audioData: audioData) - self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) + self.serviceState.isSynthesizing = false + self.eventCallback?.onSynthesisAudioGenerated(text: "", audioData: remaining) + self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) } } @@ -525,6 +527,8 @@ import os.log os_log("合成取消事件", log: self.log, type: .info) self.serviceState.isSynthesizing = false + // 取消时清空缓存,避免脏数据 + self.ttsBuffer = Data() // 修复:移除可选链,因为event.result不是可选类型 let reason = event.result.reason @@ -542,6 +546,32 @@ import os.log } } + /** + * 处理合成音频分片(按multiple字节对齐,默认1280) + * - 参数 text: 正在合成的文本 + * - 参数 incoming: 新到达的音频数据块 + * - 参数 multiple: 对齐字节倍数(例如16000Hz/16bit/mono下常用1280) + * 将incoming追加到缓存,仅当累计长度达到multiple的整数倍时, + * 发送对齐部分;余数继续保留在缓存,等待后续数据到来或合成完成时一次性发送。 + */ + private func processSynthesisAudioChunk(text: String, incoming: Data?, multiple: Int = 1280) { + guard let incoming = incoming, !incoming.isEmpty else { return } + // 追加到缓存 + ttsBuffer.append(incoming) + // 计算可发送的对齐长度 + let sendLen = (ttsBuffer.count / multiple) * multiple + if sendLen > 0 { + let chunk = ttsBuffer.prefix(sendLen) + // 发送对齐部分 + let toSend = Data(chunk) + DispatchQueue.main.async { [weak self] in + self?.eventCallback?.onSynthesisAudioGenerated(text: text, audioData: toSend) + } + // 保留余数 + ttsBuffer = Data(ttsBuffer.suffix(ttsBuffer.count - sendLen)) + } + } + // MARK: - Public Methods /**