From 585ef850b5919390dce1c43fbb3567a32e20ae2f Mon Sep 17 00:00:00 2001 From: liwei1dao Date: Sat, 11 Apr 2026 23:32:26 +0800 Subject: [PATCH] =?UTF-8?q?=E4=B8=8A=E4=BC=A0iOS=20=E7=BF=BB=E8=AF=91?= =?UTF-8?q?=E4=BC=98=E5=8C=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../azure_speech/AliyunBailianE2EHelper.swift | 13 +- .../azure_speech/AzureSpeechPlugin.swift | 293 +++++++---------- .../IntegratedSpeechTranslationService.swift | 311 ++++++++++++++---- .../MicrosoftTranslationServiceImpl.swift | 139 ++++++++ .../VolcanoTranslationServiceImpl.swift | 42 ++- 5 files changed, 555 insertions(+), 243 deletions(-) create mode 100644 local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift index 924d96b1d..41ecacc17 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift @@ -65,6 +65,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { os_log("initialize: wsUrl=%{public}@", log: log, type: .info, config.wsUrl) urlSession?.invalidateAndCancel() urlSession = nil + webSocket = nil + isStarted = false conf = config callback = cb @@ -117,7 +119,7 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { webSocket = session.webSocketTask(with: request) webSocket?.resume() - isStarted = true + // isStarted 在 didOpenWithProtocol 中设置,确保 session.update 先于音频数据发送 return true } @@ -371,8 +373,11 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { * WebSocket 打开回调 */ func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didOpenWithProtocol protocol: String?) { + // 忽略旧会话的回调 + guard session === urlSession else { return } os_log("WebSocket didOpen", log: log, type: .info) sendSessionUpdate() + isStarted = true // 在 session.update 发送后才允许推送音频,与 Android 行为对齐 callback?.onSessionStarted(sessionId: sessionId) receiveLoop() } @@ -385,6 +390,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { guard let self = self else { return } switch result { case .failure(let error): + // 忽略旧连接的错误 + guard self.isStarted else { return } os_log("WebSocket receive error: %{public}@", log: self.log, type: .error, error.localizedDescription) self.callback?.onSessionError(sessionId: self.sessionId, code: 1011, message: error.localizedDescription) case .success(let message): @@ -407,6 +414,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { * WebSocket 关闭回调 */ func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didCloseWith closeCode: URLSessionWebSocketTask.CloseCode, reason: Data?) { + // 忽略旧会话的回调 + guard session === urlSession else { return } let reasonStr = String(data: reason ?? Data(), encoding: .utf8) ?? "" os_log("WebSocket didClose code=%{public}d reason=%{public}@", log: log, type: .info, closeCode.rawValue, reasonStr) isStarted = false @@ -417,6 +426,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate { * 任务完成回调(错误处理) */ func urlSession(_ session: URLSession, task: URLSessionTask, didCompleteWithError error: Error?) { + // 忽略旧会话的回调 + guard session === urlSession else { return } if let e = error { os_log("WebSocket task error: %{public}@", log: log, type: .error, e.localizedDescription) callback?.onSessionError(sessionId: sessionId, code: 1012, message: e.localizedDescription) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index cd1e2fc17..3d53d594e 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -919,6 +919,8 @@ private func sendAudioDataEvent(_ event: [String: Any]) { let callbackA = AstEventCallback(plugin: self, serviceId: "A") let callbackB = AstEventCallback(plugin: self, serviceId: "B") astEventCallback = callbackA + azureAstHelperA.serviceId = "A" + azureAstHelperB.serviceId = "B" let azureConfigI = AzureConfiguration(subscriptionKey: subscriptionKeyI, region: regionI) let translationConfigI = TranslationConfiguration(region: azureTranslationRegion, subscriptionKey: azureTranslationKey, location: azureTranslationRegion) @@ -1033,191 +1035,152 @@ private class AstEventCallback: IntegratedSpeechTranslationService.ServiceEventC self.plugin = plugin self.serviceId = serviceId } - - /** - * 服务初始化完成回调 - */ + public func onServiceInitialized() { -// plugin?.sendAsrEvent([ -// "type": "serviceInitialized" -// ]) + plugin?.sendAstEvent([ + "type": "serviceInitialized", + "serviceId": serviceId + ]) } - - /** - * 识别中回调 - * @param text 正在识别的文本 - * @param language 语言 - * @param confidence 置信度 - */ - public func onRecognizing(text: String, language: String, confidence: Float) { -// plugin?.sendAsrEvent([ -// "type": "recognizing", -// "text": text, -// "language": language, -// "confidence": confidence -// ]) + + public func onRecognizing(utteranceId: String, text: String, language: String, confidence: Float) { + plugin?.sendAstEvent([ + "type": "recognizing", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text, + "language": language, + "confidence": confidence + ]) } - - /** - * 识别完成回调 - * @param text 识别的文本 - * @param language 语言 - * @param confidence 置信度 - */ - public func onRecognized(text: String, language: String, confidence: Float) { - print( "识别到文本: \(text), 语言: \(language), 置信度: \(confidence)") - plugin?.sendAsrEvent([ - "type": "result1", - "text": text, - "language": language, - ]) + + public func onRecognized(utteranceId: String, text: String, language: String, confidence: Float) { + print("识别到文本: \(text), 语言: \(language), 置信度: \(confidence)") + plugin?.sendAstEvent([ + "type": "recognized", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text, + "language": language, + "confidence": confidence + ]) } - - /** - * 翻译完成回调 - * @param originalText 原始文本 - * @param translatedText 翻译文本 - * @param targetLanguage 目标语言 - */ - public func onTranslated(originalText: String, translatedText: String, targetLanguage: String) { -// plugin?.sendAsrEvent([ -// "type": "translated", -// "originalText": originalText, -// "translatedText": translatedText, -// "targetLanguage": targetLanguage -// ]) + + public func onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) { + plugin?.sendAstEvent([ + "type": "interimTranslated", + "serviceId": serviceId, + "utteranceId": utteranceId, + "originalText": originalText, + "translatedText": translatedText, + "targetLanguage": targetLanguage + ]) } - - /** - * 翻译开始回调 - * @param text 要翻译的文本 - */ - public func onTranslationStarted(text: String) { -// plugin?.sendAsrEvent([ -// "type": "translationStarted", -// "text": text -// ]) + + public func onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) { + plugin?.sendAstEvent([ + "type": "translated", + "serviceId": serviceId, + "utteranceId": utteranceId, + "originalText": originalText, + "translatedText": translatedText, + "targetLanguage": targetLanguage + ]) } - - /** - * 翻译失败回调 - * @param text 翻译失败的文本 - * @param error 错误信息 - */ - public func onTranslationFailed(text: String, error: String) { -// plugin?.sendAsrEvent([ -// "type": "translationFailed", -// "text": text, -// "error": error -// ]) + + public func onTranslationStarted(utteranceId: String, text: String) { + plugin?.sendAstEvent([ + "type": "translationStarted", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text + ]) } - - /** - * 语音合成开始回调 - * @param text 要合成的文本 - */ - func onSynthesisStarted(text: String) { -// plugin?.sendAsrEvent([ -// "type": "synthesisStarted", -// "text": text -// ]) + + public func onTranslationFailed(utteranceId: String, text: String, error: String) { + plugin?.sendAstEvent([ + "type": "translationFailed", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text, + "error": error + ]) } - - /** - * 语音合成完成回调 - * @param text 合成的文本 - */ - public func onSynthesisCompleted(text: String) { -// plugin?.sendAsrEvent([ -// "type": "synthesisCompleted", -// "text": text -// ]) + + func onSynthesisStarted(utteranceId: String, text: String) { + plugin?.sendAstEvent([ + "type": "synthesisStarted", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text + ]) } - - /** - * 语音合成失败回调 - * @param text 合成失败的文本 - * @param error 错误信息 - */ - public func onSynthesisFailed(text: String, error: String) { -// plugin?.sendAsrEvent([ -// "type": "synthesisFailed", -// "text": text, -// "error": error -// ]) + + public func onSynthesisCompleted(utteranceId: String, text: String) { + plugin?.sendAstEvent([ + "type": "synthesisCompleted", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text + ]) } - - /** - * 语音合成进度回调 - * @param text 正在合成的文本 - * @param progress 进度(0.0-1.0) - */ - public func onSynthesisProgress(text: String, progress: Float) { -// plugin?.sendAsrEvent([ -// "type": "synthesisProgress", -// "text": text, -// "progress": progress -// ]) + + public func onSynthesisFailed(utteranceId: String, text: String, error: String) { + plugin?.sendAstEvent([ + "type": "synthesisFailed", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text, + "error": error + ]) } - - /** - * 识别开始回调 - */ + + public func onSynthesisProgress(utteranceId: String, text: String, progress: Float) { + plugin?.sendAstEvent([ + "type": "synthesisProgress", + "serviceId": serviceId, + "utteranceId": utteranceId, + "text": text, + "progress": progress + ]) + } + func onRecognitionStarted() { -// plugin?.sendAsrEvent([ -// "type": "recognitionStarted" -// ]) + plugin?.sendAstEvent([ + "type": "recognitionStarted", + "serviceId": serviceId + ]) } - - /** - * 识别停止回调 - */ + public func onRecognitionStopped() { -// plugin?.sendAsrEvent([ -// "type": "recognitionStopped" -// ]) + plugin?.sendAstEvent([ + "type": "recognitionStopped", + "serviceId": serviceId + ]) } - - /** - * 状态变化回调 - * @param component 组件名称 - * @param isActive 是否激活 - */ + public func onStateChanged(component: String, isActive: Bool) { -// plugin?.sendAsrEvent([ -// "type": "stateChanged", -// "component": component, -// "isActive": isActive -// ]) + plugin?.sendAstEvent([ + "type": "stateChanged", + "serviceId": serviceId, + "component": component, + "isActive": isActive + ]) } - - /** - * 错误回调 - * @param component 组件名称 - * @param error 错误信息 - */ + public func onError(component: String, error: String) { -// plugin?.sendAsrEvent([ -// "type": "error", -// "component": component, -// "error": error -// ]) + plugin?.sendAstEvent([ + "type": "error", + "serviceId": serviceId, + "component": component, + "error": error + ]) } - - /** - * 语音合成音频数据生成回调 - * @param text 合成的文本 - * @param audioData 音频数据 - */ - public func onSynthesisAudioGenerated(text: String, audioData: Data) { - guard let plugin = plugin else { return } - - // 只有 A 通道(己方翻译后的语音)推送到耳机,B 通道不推 - guard serviceId == "A" else { return } - let pcmData = plugin.extractPcmFromWav(audioData) - if !pcmData.isEmpty { - os_log("[AzureAST-%{public}@] TTS音频推送到耳机,大小=%d字节", log: plugin.ctLog, type: .info, serviceId, pcmData.count) + public func onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: Data) { + let pcmData = audioData + print("[\(serviceId)] 合成音频生成,PCM大小=\(pcmData.count),合成ID=\(utteranceId)") + if pcmData.count > 0 { BleService.shared.writeExternalAudioData(data: pcmData) } } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift index f554fa9cb..5a3c8b0ba 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift @@ -16,7 +16,10 @@ import os.log public enum AudioSourceType { /** 使用设备麦克风 */ case microphone - + /** 使用系统音频 */ + case systemAudio + /** 使用系统音频+麦克风音频 */ + case systemAudioPlusMicrophone /** 使用外部提供的音频数据 */ case external } @@ -34,6 +37,8 @@ import os.log // MARK: - Properties private let log = OSLog(subsystem: "com.azure.speech", category: "IntegratedSpeechService") + // 实例标识,用于区分 A/B,影响 utteranceId 生成 + public var serviceId: String = "A" // Azure服务组件 private var speechConfig: SPXSpeechConfiguration? @@ -56,6 +61,23 @@ import os.log // 状态管理 private let serviceState = ServiceState() + + // ===== Utterance 关联ID管理 ===== + private var utteranceSeq: Int64 = 0 + private var currentUtteranceId: String? + private var isInUtterance: Bool = false + private var recognizingTranslationTask: Task? + private var currentSynthesisId: String? + private var currentSynthesisText: String? + + /** + * 生成新的 utteranceId + */ + private func nextUtteranceId() -> String { + utteranceSeq += 1 + let ts = Int64(Date().timeIntervalSince1970 * 1000) + return "utt-\(serviceId)-\(ts)-\(utteranceSeq)" + } // 事件回调 private weak var eventCallback: ServiceEventCallback? @@ -75,8 +97,6 @@ import os.log var sourceLanguage: String = "zh-CN" var targetLanguage: String = "en-US" var currentVoice: String = "en-US-AriaNeural" - var translationSourceLanguage: String = "" - var translationTargetLanguage: String = "" var speechRate: String = "0%" var speechPitch: String = "0%" var speechVolume: String = "100%" @@ -86,13 +106,14 @@ import os.log var translationTimeout: TimeInterval = 10.0 // 新增:控制是否播放合成的音频 var enableAudioPlayback: Bool = false + // 新增:可选翻译源/目标语言(优先级高于识别语言对) + var translationSourceLanguage: String = "" + var translationTargetLanguage: String = "" public init( sourceLanguage: String = "zh-CN", targetLanguage: String = "en-US", currentVoice: String = "en-US-AriaNeural", - translationSourceLanguage: String = "", - translationTargetLanguage: String = "", speechRate: String = "0%", speechPitch: String = "0%", speechVolume: String = "100%", @@ -100,13 +121,13 @@ import os.log enableAutoLanguageDetection: Bool = false, maxRetryAttempts: Int = 3, translationTimeout: TimeInterval = 10.0, - enableAudioPlayback: Bool = false + enableAudioPlayback: Bool = false, + translationSourceLanguage: String = "", + translationTargetLanguage: String = "" ) { self.sourceLanguage = sourceLanguage self.targetLanguage = targetLanguage self.currentVoice = currentVoice - self.translationSourceLanguage = translationSourceLanguage - self.translationTargetLanguage = translationTargetLanguage self.speechRate = speechRate self.speechPitch = speechPitch self.speechVolume = speechVolume @@ -115,6 +136,8 @@ import os.log self.maxRetryAttempts = maxRetryAttempts self.translationTimeout = translationTimeout self.enableAudioPlayback = enableAudioPlayback + self.translationSourceLanguage = translationSourceLanguage + self.translationTargetLanguage = translationTargetLanguage } } @@ -156,21 +179,21 @@ import os.log */ public protocol ServiceEventCallback: AnyObject { func onServiceInitialized() - func onRecognizing(text: String, language: String, confidence: Float) - func onRecognized(text: String, language: String, confidence: Float) - func onTranslated(originalText: String, translatedText: String, targetLanguage: String) - func onTranslationStarted(text: String) - func onTranslationFailed(text: String, error: String) - func onSynthesisStarted(text: String) - func onSynthesisCompleted(text: String) - func onSynthesisFailed(text: String, error: String) - func onSynthesisProgress(text: String, progress: Float) + func onRecognizing(utteranceId: String, text: String, language: String, confidence: Float) + func onRecognized(utteranceId: String, text: String, language: String, confidence: Float) + func onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) + func onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) + func onTranslationStarted(utteranceId: String, text: String) + func onTranslationFailed(utteranceId: String, text: String, error: String) + func onSynthesisStarted(utteranceId: String, text: String) + func onSynthesisCompleted(utteranceId: String, text: String) + func onSynthesisFailed(utteranceId: String, text: String, error: String) + func onSynthesisProgress(utteranceId: String, text: String, progress: Float) func onRecognitionStarted() func onRecognitionStopped() func onStateChanged(component: String, isActive: Bool) func onError(component: String, error: String) - // 新增:返回合成的音频数据 - func onSynthesisAudioGenerated(text: String, audioData: Data) + func onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: Data) } /** @@ -295,7 +318,7 @@ import os.log config.speechRecognitionLanguage = serviceConfig.sourceLanguage config.speechSynthesisVoiceName = getVoiceForLanguage(serviceConfig.targetLanguage) - config.setSpeechSynthesisOutputFormat(.riff16Khz16BitMonoPcm) + config.setSpeechSynthesisOutputFormat(.raw16Khz16BitMonoPcm) // 自动语言检测 if serviceConfig.enableAutoLanguageDetection { @@ -316,15 +339,13 @@ import os.log */ private func initializeTranslationService(translationConfig: TranslationConfiguration) async -> Bool { do { - translationService = VolcanoTranslationServiceImpl() + // 切换为微软翻译服务实现 + translationService = MicrosoftTranslationServiceImpl() let success = await translationService?.initialize(config: translationConfig.toConfigMap()) ?? false - if success { - os_log("翻译服务初始化完成", log: log, type: .info) + os_log("翻译服务初始化完成(Microsoft)", log: log, type: .info) } - return success - } catch { os_log("翻译服务初始化失败: %@", log: log, type: .error, error.localizedDescription) return false @@ -337,6 +358,8 @@ import os.log private func initializeAudioProcessor() { audioProcessor = SimpleAudioReceiver() audioProcessor?.initAudioRecord() + // 显式设置推流格式为 16k/16bit/单声道,避免适配层默认推断 + audioProcessor?.setAudioConfig(sampleRate: Int(Self.SAMPLE_RATE), channels: Int(Self.CHANNELS)) // 检查音频配置是否已存在 if audioConfig == nil, let pushStream = audioProcessor?.pushAudioStream { @@ -365,14 +388,28 @@ import os.log guard let text = result.text, !text.isEmpty else { return } let confidence = self.extractConfidence(from: result) os_log("识别中事件: %@", log: self.log, type: .debug, text) - + // 开始新的话段时生成并记录 utteranceId + if !self.isInUtterance { + self.currentUtteranceId = self.nextUtteranceId() + self.isInUtterance = true + } + let uttId = self.currentUtteranceId ?? self.nextUtteranceId() DispatchQueue.main.async { self.eventCallback?.onRecognizing( + utteranceId: uttId, text: text, language: self.serviceConfig.sourceLanguage, confidence: confidence ) } + // 当服务处于运行状态时,为“识别中”的内容启动仅翻译流程(不合成) + if self.serviceState.isRecognizing { + self.recognizingTranslationTask?.cancel() + self.recognizingTranslationTask = Task { [weak self] in + guard let self = self else { return } + await self.processTranslationOnly(utteranceId: uttId, text: text) + } + } } // 识别完成事件 @@ -384,9 +421,10 @@ import os.log case .recognizedSpeech: if let text = result.text, !text.isEmpty { let confidence = self.extractConfidence(from: result) - + let uttId = self.currentUtteranceId ?? self.nextUtteranceId() DispatchQueue.main.async { self.eventCallback?.onRecognized( + utteranceId: uttId, text: text, language: self.serviceConfig.sourceLanguage, confidence: confidence @@ -397,8 +435,13 @@ import os.log // 检查服务状态,只有在识别仍然活跃时才触发翻译流程 if self.serviceState.isRecognizing { + if self.isInUtterance { + self.currentUtteranceId = self.nextUtteranceId() + self.isInUtterance = false + } Task { - await self.processTranslationAndSynthesis(text: text) + await self.cancelRecognizingTranslationTaskSafely() + await self.processTranslationAndSynthesis(utteranceId: uttId, text: text) } } else { os_log("服务已停止,跳过翻译流程", log: self.log, type: .info) @@ -490,6 +533,9 @@ import os.log DispatchQueue.main.async { self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true) + let uttId = self.currentSynthesisId ?? "" + let text = self.currentSynthesisText ?? "" + self.eventCallback?.onSynthesisStarted(utteranceId: uttId, text: text) } } @@ -502,6 +548,14 @@ import os.log return } os_log("合成进行中事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) + + let uttId = self.currentSynthesisId ?? "" + let text = self.currentSynthesisText ?? "" + DispatchQueue.main.async { + let audioCopy = Data(audioData) + self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy) + self.eventCallback?.onSynthesisProgress(utteranceId: uttId, text: text, progress: 0.5) + } } @@ -515,12 +569,18 @@ import os.log os_log("合成完成事件,音频数据为空", log: self.log, type: .debug) return } - os_log("合成完成事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) + os_log("合成完成事件,事件数据长度: %d bytes, 缓冲数据长度: %d bytes", log: self.log, type: .debug, audioData.count, 0) + let audioCopy = Data(audioData) DispatchQueue.main.async { - self.serviceState.isSynthesizing = false + self.serviceState.isSynthesizing = false // 立即返回当前生成的音频数据 - self.eventCallback?.onSynthesisAudioGenerated(text: "", audioData: audioData) - self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) + let uttId = self.currentSynthesisId ?? "" + let text = self.currentSynthesisText ?? "" + //self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy) + self.eventCallback?.onSynthesisCompleted(utteranceId: uttId, text: text) + self.currentSynthesisId = nil + self.currentSynthesisText = nil + self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) } } @@ -536,7 +596,11 @@ import os.log let reason = event.result.reason DispatchQueue.main.async { - self.eventCallback?.onError(component: "Synthesis", error: "语音合成取消: \(reason)") + let uttId = self.currentSynthesisId ?? "" + let text = self.currentSynthesisText ?? "" + self.eventCallback?.onSynthesisFailed(utteranceId: uttId, text: text, error: "语音合成取消: \(reason)") + self.currentSynthesisId = nil + self.currentSynthesisText = nil self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) } } @@ -650,7 +714,14 @@ import os.log /** * 处理翻译和合成流程 */ - private func processTranslationAndSynthesis(text: String) async { + /** + * 处理最终翻译并进行语音合成 + * - Parameters: + * - utteranceId: 该句话的唯一ID + * - text: 识别完成的最终文本 + */ + /// 处理最终翻译并进行语音合成(优先使用配置的翻译语言对) + private func processTranslationAndSynthesis(utteranceId: String, text: String) async { // 添加服务状态检查 guard serviceState.isRecognizing else { os_log("服务已停止,取消翻译流程", log: log, type: .info) @@ -667,19 +738,23 @@ import os.log serviceState.isTranslating = true await MainActor.run { - eventCallback?.onTranslationStarted(text: text) + eventCallback?.onTranslationStarted(utteranceId: utteranceId, text: text) eventCallback?.onStateChanged(component: "Translation", isActive: true) } os_log("翻译开始: %@", log: log, type: .info, text) do { + let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage + let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage + os_log("翻译语言对: %@ -> %@", log: log, type: .debug, srcLang, tgtLang) + let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) { - // 修复:在闭包中显式使用self + os_log("开始翻译文本: %@", log: self.log, type: .debug, text) return await self.translationService?.translateText( text: text, - sourceLanguage: self.serviceConfig.sourceLanguage, - targetLanguage: self.serviceConfig.targetLanguage + sourceLanguage: srcLang, + targetLanguage: tgtLang ) } @@ -694,14 +769,15 @@ import os.log await MainActor.run { eventCallback?.onTranslated( + utteranceId: utteranceId, originalText: text, translatedText: translatedText, - targetLanguage: serviceConfig.targetLanguage + targetLanguage: tgtLang ) } // 进行语音合成 - if synthesizeTextStreaming(translatedText) { + if synthesizeTextStreaming(utteranceId, translatedText) { print("语音合成已启动") } } else { @@ -709,7 +785,7 @@ import os.log os_log("翻译失败: %@", log: log, type: .error, errorMessage) await MainActor.run { - eventCallback?.onTranslationFailed(text: text, error: errorMessage) + eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: errorMessage) } } @@ -718,12 +794,68 @@ import os.log await MainActor.run { eventCallback?.onStateChanged(component: "Translation", isActive: false) - eventCallback?.onTranslationFailed(text: text, error: "翻译超时或异常: \(error.localizedDescription)") + eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时或异常: \(error.localizedDescription)") } os_log("翻译异常: %@", log: log, type: .error, error.localizedDescription) } } + + /** + * 处理识别中临时翻译(不进行合成) + * - Parameters: + * - utteranceId: 当前话段的唯一ID + * - text: 识别中的临时文本 + */ + /// 处理识别中临时翻译(优先使用配置的翻译语言对) + private func processTranslationOnly(utteranceId: String, text: String) async { + guard serviceState.isRecognizing else { return } + do { + let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage + let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage + let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) { + return await self.translationService?.translateText( + text: text, + sourceLanguage: srcLang, + targetLanguage: tgtLang + ) + } + if let result = translationResult, result.success, let translatedText = result.translatedText, !translatedText.isEmpty { + await MainActor.run { + self.eventCallback?.onInterimTranslated( + utteranceId: utteranceId, + originalText: text, + translatedText: translatedText, + targetLanguage: tgtLang + ) + } + } else if let err = translationResult?.error { + await MainActor.run { + self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: err) + } + } else { + await MainActor.run { + self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译服务返回空结果") + } + } + } catch is TimeoutError { + await MainActor.run { + self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时") + } + } catch { + await MainActor.run { + self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译异常: \(error.localizedDescription)") + } + } + } + + /** + * 安全取消识别中翻译任务 + */ + private func cancelRecognizingTranslationTaskSafely() async { + recognizingTranslationTask?.cancel() + recognizingTranslationTask = nil + } /** * 语音合成 @@ -736,24 +868,39 @@ import os.log /// 流式语音合成 /// - Parameter text: 要合成的文本 /// - Returns: 合成是否成功启动 - private func synthesizeTextStreaming(_ text: String) -> Bool { + /** + * 流式语音合成,携带utteranceId + * - Parameters: + * - utteranceId: 当前合成对应的话段ID + * - text: 要合成的文本 + */ + private func synthesizeTextStreaming(_ utteranceId: String, _ text: String) -> Bool { + let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines) + guard !trimmed.isEmpty else { + os_log("合成文本为空或仅空白,跳过合成", log: log, type: .info) + return false + } guard let synthesizer = synthesizer else { os_log("语音合成器未初始化", log: log, type: .error) return false } do { - let ssml = buildSSML(text: text) + let sanitized = cleanTextForTTS(trimmed) + let ssml = generateOptimizedSsml(sanitized) os_log("开始流式合成语音: %@", log: log, type: .info, text) serviceState.isSynthesizing = true + currentSynthesisId = utteranceId + currentSynthesisText = text DispatchQueue.main.async { - self.eventCallback?.onSynthesisStarted(text: text) + self.eventCallback?.onSynthesisStarted(utteranceId: utteranceId, text: text) self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true) } - try synthesizer.speakSsml(ssml) - + // 使用异步启动,提升并发兼容性 + try synthesizer.startSpeakingSsml(ssml) + return true @@ -764,7 +911,7 @@ import os.log os_log("%@", log: log, type: .error, errorMessage) DispatchQueue.main.async { - self.eventCallback?.onSynthesisFailed(text: text, error: errorMessage) + self.eventCallback?.onSynthesisFailed(utteranceId: utteranceId, text: text, error: errorMessage) self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) } @@ -774,36 +921,78 @@ import os.log /// 保留原有的同步合成方法作为备用 private func synthesizeText(_ text: String) -> Data? { - // 现在调用流式合成方法 - let success = synthesizeTextStreaming(text) - return success ? Data() : nil // 返回空数据表示已启动,实际数据通过回调返回 + let success = synthesizeTextStreaming("", text) + return success ? Data() : nil } - /** - * 构建SSML + + + /** + * 清理TTS文本 */ - private func buildSSML(text: String) -> String { - let voice = getVoiceForLanguage(serviceConfig.targetLanguage) + private func cleanTextForTTS(_ text: String) -> String { + var cleaned = text + + // 移除URL + if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") + } + + // 移除emoji + if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") + } + + // 移除标点符号 + if let regex = try? NSRegularExpression(pattern: "[;:#;:*\\n]+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") + } + // 合并空格 + if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) { + cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ") + } + + return cleaned.trimmingCharacters(in: .whitespacesAndNewlines) + } + + /** + * 生成优化的SSML + */ + private func generateOptimizedSsml(_ rawText: String) -> String { + // 转义XML保留字符 + let escapedText = rawText + .replacingOccurrences(of: "&", with: "&") + .replacingOccurrences(of: "<", with: "<") + .replacingOccurrences(of: ">", with: ">") + print("serviceConfig.targetLanguage: \(serviceConfig.targetLanguage)") + let voice = getVoiceForLanguage(serviceConfig.targetLanguage) + // 简化SSML结构 return """ - - - \(text) - - + + + \(escapedText) + + """ } + /** * 根据语言获取对应的语音 */ private func getVoiceForLanguage(_ language: String) -> String { + // 优先使用 Dart 传入的 currentVoice(已包含性别信息) + if !serviceConfig.currentVoice.isEmpty { + return serviceConfig.currentVoice + } + if let cachedVoice = voiceCache[language] { return cachedVoice } - + let voice: String switch language { // 中文相关 diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift new file mode 100644 index 000000000..b5c9aefc0 --- /dev/null +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift @@ -0,0 +1,139 @@ +import Foundation +import os.log + +/** + * 微软翻译服务实现(iOS) + * 实现 IntegratedSpeechTranslationService.TranslationServiceInterface + */ +public class MicrosoftTranslationServiceImpl: IntegratedSpeechTranslationService.TranslationServiceInterface { + private let log = OSLog(subsystem: "com.azure.speech", category: "MicrosoftTranslationService") + private let defaultEndpoint = "https://api.cognitive.microsofttranslator.com" + private let apiPath = "/translate" + private let apiVersion = "3.0" + + private var isInitialized = false + private var subscriptionKey: String = "" + private var region: String = "" + private var endpoint: String = "" + private var maxRetryAttempts: Int = 3 + private var timeout: TimeInterval = 10.0 + + private var _urlSession: URLSession? + + private var urlSession: URLSession { + if let s = _urlSession { return s } + let cfg = URLSessionConfiguration.default + cfg.timeoutIntervalForRequest = timeout + cfg.timeoutIntervalForResource = 30.0 + let s = URLSession(configuration: cfg) + _urlSession = s + return s + } + + /** + * 初始化翻译服务 + */ + public func initialize(config: [String : String]) async -> Bool { + subscriptionKey = (config["subscriptionKey"] ?? config["accessKey"] ?? "").trimmingCharacters(in: .whitespaces) + region = (config["region"] ?? config["location"] ?? "").trimmingCharacters(in: .whitespaces) + endpoint = (config["endpoint"] ?? defaultEndpoint).trimmingCharacters(in: .whitespaces) + os_log("微软翻译初始化参数: endpoint=%@ region=%@ keyLen=%d", log: log, type: .info, endpoint, region, subscriptionKey.count) + if let mr = config["maxRetryAttempts"], let v = Int(mr) { maxRetryAttempts = v } + if let to = config["timeout"], let v = Double(to) { timeout = v } + + // 重置会话以应用新配置或清除旧状态 + if _urlSession != nil { + _urlSession?.invalidateAndCancel() + _urlSession = nil + } + + isInitialized = !subscriptionKey.isEmpty + os_log("微软翻译服务初始化: %{public}@", log: log, type: .info, isInitialized ? "成功" : "失败") + return isInitialized + } + + /** + * 文本翻译 + */ + public func translateText(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult { + guard isInitialized else { + return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "翻译服务未初始化") + } + os_log("调用微软翻译: from=%@ to=%@ textLen=%d", log: log, type: .info, sourceLanguage, targetLanguage, text.count) + return await performTranslationWithRetry(text: text, sourceLanguage: sourceLanguage, targetLanguage: targetLanguage) + } + + /** + * 释放资源 + */ + public func dispose() { + _urlSession?.invalidateAndCancel() + _urlSession = nil + isInitialized = false + } + + /** + * 带重试的翻译执行 + */ + private func performTranslationWithRetry(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult { + var lastError = "未知错误" + for attempt in 1...maxRetryAttempts { + os_log("微软翻译尝试 %d/%d", log: log, type: .debug, attempt, maxRetryAttempts) + let res = await performTranslation(text: text, sourceLanguage: sourceLanguage, targetLanguage: targetLanguage) + if res.success { + os_log("微软翻译成功(尝试 %d)", log: log, type: .info, attempt) + return res + } + lastError = res.error ?? lastError + if attempt < maxRetryAttempts { try? await Task.sleep(nanoseconds: UInt64(attempt) * 500_000_000) } + } + os_log("微软翻译失败,错误: %@", log: log, type: .error, lastError) + return IntegratedSpeechTranslationService.TranslationResult(success: false, error: lastError) + } + + /** + * 实际翻译执行 + */ + private func performTranslation(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult { + do { + var comps = URLComponents(string: endpoint + apiPath)! + comps.queryItems = [ + URLQueryItem(name: "api-version", value: apiVersion), + URLQueryItem(name: "from", value: sourceLanguage), + URLQueryItem(name: "to", value: targetLanguage) + ] + guard let url = comps.url else { throw URLError(.badURL) } + var req = URLRequest(url: url) + req.httpMethod = "POST" + let bodyArr: [[String: Any]] = [["text": text]] + req.httpBody = try JSONSerialization.data(withJSONObject: bodyArr) + req.setValue("application/json", forHTTPHeaderField: "Content-Type") + req.setValue(subscriptionKey, forHTTPHeaderField: "Ocp-Apim-Subscription-Key") + req.setValue(region, forHTTPHeaderField: "Ocp-Apim-Subscription-Region") + req.setValue(UUID().uuidString, forHTTPHeaderField: "X-ClientTraceId") + + os_log("请求微软翻译: endpoint=%@ from=%@ to=%@", log: log, type: .debug, endpoint, sourceLanguage, targetLanguage) + os_log("请求体长度: %d bytes", log: log, type: .debug, (req.httpBody ?? Data()).count) + + let (data, resp) = try await urlSession.data(for: req) + guard let http = resp as? HTTPURLResponse, http.statusCode == 200 else { + let code = (resp as? HTTPURLResponse)?.statusCode ?? -1 + os_log("微软翻译HTTP错误: %d", log: log, type: .error, code) + return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "HTTP错误: \(code)") + } + os_log("微软翻译HTTP成功,响应大小: %d bytes", log: log, type: .debug, data.count) + guard let arr = try JSONSerialization.jsonObject(with: data) as? [[String: Any]], let item = arr.first, + let translations = item["translations"] as? [[String: Any]], let first = translations.first, + let tx = first["text"] as? String else { + os_log("微软翻译解析失败,返回结构不符合预期", log: log, type: .error) + return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "翻译结果为空") + } + let preview = tx.prefix(64) + os_log("微软翻译解析成功,译文预览: %@", log: log, type: .info, String(preview)) + return IntegratedSpeechTranslationService.TranslationResult(success: true, translatedText: tx, confidence: 0.9) + } catch { + os_log("微软翻译请求异常: %@", log: log, type: .error, error.localizedDescription) + return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "请求异常: \(error.localizedDescription)") + } + } +} diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift index 05f7c9006..568123231 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift @@ -267,13 +267,13 @@ public class VolcanoTranslationServiceImpl: IntegratedSpeechTranslationService.T } let jsonData = try JSONSerialization.data(withJSONObject: finalRequestBody) - + // 构建查询参数 let queryParams: [String: String] = [ "Action": action, "Version": version ] - + // 生成签名 let headers = try generateSignature( method: "POST", @@ -414,33 +414,33 @@ public class VolcanoTranslationServiceImpl: IntegratedSpeechTranslationService.T requestBody: [String: Any], queryParams: [String: String] ) throws -> [String: String] { - + let now = Date() let dateFormatter = DateFormatter() dateFormatter.dateFormat = "yyyyMMdd'T'HHmmss'Z'" dateFormatter.timeZone = TimeZone(abbreviation: "UTC") let timestamp = dateFormatter.string(from: now) - + let shortDateFormatter = DateFormatter() shortDateFormatter.dateFormat = "yyyyMMdd" shortDateFormatter.timeZone = TimeZone(abbreviation: "UTC") let shortDate = shortDateFormatter.string(from: now) - + // 构建规范请求 let httpRequestMethod = method let canonicalURI = endpoint - + // 构建规范查询字符串 let sortedQueryParams = queryParams.sorted { $0.key < $1.key } let canonicalQueryString = sortedQueryParams .map { "\($0.key)=\($0.value.addingPercentEncoding(withAllowedCharacters: .urlQueryAllowed) ?? $0.value)" } .joined(separator: "&") - + // 构建规范头部 let host = URL(string: baseURL)!.host! let canonicalHeaders = "host:\(host)\nx-date:\(timestamp)\n" let signedHeaders = "host;x-date" - + // 计算请求体哈希 let requestBodyData = try JSONSerialization.data(withJSONObject: requestBody) let hashedRequestPayload = sha256(data: requestBodyData) @@ -539,20 +539,30 @@ enum TranslationError: Error { */ extension TranslationConfiguration { func toConfigMap() -> [String: String] { - var config: [String: String] = [ - "accessKey": accessKey, - "secretKey": secretKey, - "region": region - ] - + var config: [String: String] = [:] + if !subscriptionKey.isEmpty { + config["subscriptionKey"] = subscriptionKey + } else if !accessKey.isEmpty { + config["accessKey"] = accessKey + } + if !secretKey.isEmpty { + config["secretKey"] = secretKey + } + if !region.isEmpty { + config["region"] = region + } + if !location.isEmpty { + config["location"] = location + } + if !endpoint.isEmpty { + config["endpoint"] = endpoint + } if let maxRetry = maxRetryAttempts { config["maxRetryAttempts"] = String(maxRetry) } - if let timeoutValue = timeout { config["timeout"] = String(timeoutValue) } - return config } }