Browse Source

上传iOS 翻译优化

weicu
liwei1dao 6 months ago
parent
commit
585ef850b5
  1. 13
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift
  2. 293
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift
  3. 311
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift
  4. 139
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift
  5. 42
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift

13
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift

@ -65,6 +65,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
os_log("initialize: wsUrl=%{public}@", log: log, type: .info, config.wsUrl)
urlSession?.invalidateAndCancel()
urlSession = nil
webSocket = nil
isStarted = false
conf = config
callback = cb
@ -117,7 +119,7 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
webSocket = session.webSocketTask(with: request)
webSocket?.resume()
isStarted = true
// isStarted 在 didOpenWithProtocol 中设置,确保 session.update 先于音频数据发送
return true
}
@ -371,8 +373,11 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
* WebSocket 打开回调
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didOpenWithProtocol protocol: String?) {
// 忽略旧会话的回调
guard session === urlSession else { return }
os_log("WebSocket didOpen", log: log, type: .info)
sendSessionUpdate()
isStarted = true // 在 session.update 发送后才允许推送音频,与 Android 行为对齐
callback?.onSessionStarted(sessionId: sessionId)
receiveLoop()
}
@ -385,6 +390,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
guard let self = self else { return }
switch result {
case .failure(let error):
// 忽略旧连接的错误
guard self.isStarted else { return }
os_log("WebSocket receive error: %{public}@", log: self.log, type: .error, error.localizedDescription)
self.callback?.onSessionError(sessionId: self.sessionId, code: 1011, message: error.localizedDescription)
case .success(let message):
@ -407,6 +414,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
* WebSocket 关闭回调
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didCloseWith closeCode: URLSessionWebSocketTask.CloseCode, reason: Data?) {
// 忽略旧会话的回调
guard session === urlSession else { return }
let reasonStr = String(data: reason ?? Data(), encoding: .utf8) ?? ""
os_log("WebSocket didClose code=%{public}d reason=%{public}@", log: log, type: .info, closeCode.rawValue, reasonStr)
isStarted = false
@ -417,6 +426,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
* 任务完成回调(错误处理)
*/
func urlSession(_ session: URLSession, task: URLSessionTask, didCompleteWithError error: Error?) {
// 忽略旧会话的回调
guard session === urlSession else { return }
if let e = error {
os_log("WebSocket task error: %{public}@", log: log, type: .error, e.localizedDescription)
callback?.onSessionError(sessionId: sessionId, code: 1012, message: e.localizedDescription)

293
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift

@ -919,6 +919,8 @@ private func sendAudioDataEvent(_ event: [String: Any]) {
let callbackA = AstEventCallback(plugin: self, serviceId: "A")
let callbackB = AstEventCallback(plugin: self, serviceId: "B")
astEventCallback = callbackA
azureAstHelperA.serviceId = "A"
azureAstHelperB.serviceId = "B"
let azureConfigI = AzureConfiguration(subscriptionKey: subscriptionKeyI, region: regionI)
let translationConfigI = TranslationConfiguration(region: azureTranslationRegion, subscriptionKey: azureTranslationKey, location: azureTranslationRegion)
@ -1033,191 +1035,152 @@ private class AstEventCallback: IntegratedSpeechTranslationService.ServiceEventC
self.plugin = plugin
self.serviceId = serviceId
}
/**
* 服务初始化完成回调
*/
public func onServiceInitialized() {
// plugin?.sendAsrEvent([
// "type": "serviceInitialized"
// ])
plugin?.sendAstEvent([
"type": "serviceInitialized",
"serviceId": serviceId
])
}
/**
* 识别中回调
* @param text 正在识别的文本
* @param language 语言
* @param confidence 置信度
*/
public func onRecognizing(text: String, language: String, confidence: Float) {
// plugin?.sendAsrEvent([
// "type": "recognizing",
// "text": text,
// "language": language,
// "confidence": confidence
// ])
public func onRecognizing(utteranceId: String, text: String, language: String, confidence: Float) {
plugin?.sendAstEvent([
"type": "recognizing",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text,
"language": language,
"confidence": confidence
])
}
/**
* 识别完成回调
* @param text 识别的文本
* @param language 语言
* @param confidence 置信度
*/
public func onRecognized(text: String, language: String, confidence: Float) {
print( "识别到文本: \(text), 语言: \(language), 置信度: \(confidence)")
plugin?.sendAsrEvent([
"type": "result1",
"text": text,
"language": language,
])
public func onRecognized(utteranceId: String, text: String, language: String, confidence: Float) {
print("识别到文本: \(text), 语言: \(language), 置信度: \(confidence)")
plugin?.sendAstEvent([
"type": "recognized",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text,
"language": language,
"confidence": confidence
])
}
/**
* 翻译完成回调
* @param originalText 原始文本
* @param translatedText 翻译文本
* @param targetLanguage 目标语言
*/
public func onTranslated(originalText: String, translatedText: String, targetLanguage: String) {
// plugin?.sendAsrEvent([
// "type": "translated",
// "originalText": originalText,
// "translatedText": translatedText,
// "targetLanguage": targetLanguage
// ])
public func onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) {
plugin?.sendAstEvent([
"type": "interimTranslated",
"serviceId": serviceId,
"utteranceId": utteranceId,
"originalText": originalText,
"translatedText": translatedText,
"targetLanguage": targetLanguage
])
}
/**
* 翻译开始回调
* @param text 要翻译的文本
*/
public func onTranslationStarted(text: String) {
// plugin?.sendAsrEvent([
// "type": "translationStarted",
// "text": text
// ])
public func onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) {
plugin?.sendAstEvent([
"type": "translated",
"serviceId": serviceId,
"utteranceId": utteranceId,
"originalText": originalText,
"translatedText": translatedText,
"targetLanguage": targetLanguage
])
}
/**
* 翻译失败回调
* @param text 翻译失败的文本
* @param error 错误信息
*/
public func onTranslationFailed(text: String, error: String) {
// plugin?.sendAsrEvent([
// "type": "translationFailed",
// "text": text,
// "error": error
// ])
public func onTranslationStarted(utteranceId: String, text: String) {
plugin?.sendAstEvent([
"type": "translationStarted",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text
])
}
/**
* 语音合成开始回调
* @param text 要合成的文本
*/
func onSynthesisStarted(text: String) {
// plugin?.sendAsrEvent([
// "type": "synthesisStarted",
// "text": text
// ])
public func onTranslationFailed(utteranceId: String, text: String, error: String) {
plugin?.sendAstEvent([
"type": "translationFailed",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text,
"error": error
])
}
/**
* 语音合成完成回调
* @param text 合成的文本
*/
public func onSynthesisCompleted(text: String) {
// plugin?.sendAsrEvent([
// "type": "synthesisCompleted",
// "text": text
// ])
func onSynthesisStarted(utteranceId: String, text: String) {
plugin?.sendAstEvent([
"type": "synthesisStarted",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text
])
}
/**
* 语音合成失败回调
* @param text 合成失败的文本
* @param error 错误信息
*/
public func onSynthesisFailed(text: String, error: String) {
// plugin?.sendAsrEvent([
// "type": "synthesisFailed",
// "text": text,
// "error": error
// ])
public func onSynthesisCompleted(utteranceId: String, text: String) {
plugin?.sendAstEvent([
"type": "synthesisCompleted",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text
])
}
/**
* 语音合成进度回调
* @param text 正在合成的文本
* @param progress 进度(0.0-1.0)
*/
public func onSynthesisProgress(text: String, progress: Float) {
// plugin?.sendAsrEvent([
// "type": "synthesisProgress",
// "text": text,
// "progress": progress
// ])
public func onSynthesisFailed(utteranceId: String, text: String, error: String) {
plugin?.sendAstEvent([
"type": "synthesisFailed",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text,
"error": error
])
}
/**
* 识别开始回调
*/
public func onSynthesisProgress(utteranceId: String, text: String, progress: Float) {
plugin?.sendAstEvent([
"type": "synthesisProgress",
"serviceId": serviceId,
"utteranceId": utteranceId,
"text": text,
"progress": progress
])
}
func onRecognitionStarted() {
// plugin?.sendAsrEvent([
// "type": "recognitionStarted"
// ])
plugin?.sendAstEvent([
"type": "recognitionStarted",
"serviceId": serviceId
])
}
/**
* 识别停止回调
*/
public func onRecognitionStopped() {
// plugin?.sendAsrEvent([
// "type": "recognitionStopped"
// ])
plugin?.sendAstEvent([
"type": "recognitionStopped",
"serviceId": serviceId
])
}
/**
* 状态变化回调
* @param component 组件名称
* @param isActive 是否激活
*/
public func onStateChanged(component: String, isActive: Bool) {
// plugin?.sendAsrEvent([
// "type": "stateChanged",
// "component": component,
// "isActive": isActive
// ])
plugin?.sendAstEvent([
"type": "stateChanged",
"serviceId": serviceId,
"component": component,
"isActive": isActive
])
}
/**
* 错误回调
* @param component 组件名称
* @param error 错误信息
*/
public func onError(component: String, error: String) {
// plugin?.sendAsrEvent([
// "type": "error",
// "component": component,
// "error": error
// ])
plugin?.sendAstEvent([
"type": "error",
"serviceId": serviceId,
"component": component,
"error": error
])
}
/**
* 语音合成音频数据生成回调
* @param text 合成的文本
* @param audioData 音频数据
*/
public func onSynthesisAudioGenerated(text: String, audioData: Data) {
guard let plugin = plugin else { return }
// 只有 A 通道(己方翻译后的语音)推送到耳机,B 通道不推
guard serviceId == "A" else { return }
let pcmData = plugin.extractPcmFromWav(audioData)
if !pcmData.isEmpty {
os_log("[AzureAST-%{public}@] TTS音频推送到耳机,大小=%d字节", log: plugin.ctLog, type: .info, serviceId, pcmData.count)
public func onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: Data) {
let pcmData = audioData
print("[\(serviceId)] 合成音频生成,PCM大小=\(pcmData.count),合成ID=\(utteranceId)")
if pcmData.count > 0 {
BleService.shared.writeExternalAudioData(data: pcmData)
}
}

311
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift

@ -16,7 +16,10 @@ import os.log
public enum AudioSourceType {
/** 使用设备麦克风 */
case microphone
/** 使用系统音频 */
case systemAudio
/** 使用系统音频+麦克风音频 */
case systemAudioPlusMicrophone
/** 使用外部提供的音频数据 */
case external
}
@ -34,6 +37,8 @@ import os.log
// MARK: - Properties
private let log = OSLog(subsystem: "com.azure.speech", category: "IntegratedSpeechService")
// 实例标识,用于区分 A/B,影响 utteranceId 生成
public var serviceId: String = "A"
// Azure服务组件
private var speechConfig: SPXSpeechConfiguration?
@ -56,6 +61,23 @@ import os.log
// 状态管理
private let serviceState = ServiceState()
// ===== Utterance 关联ID管理 =====
private var utteranceSeq: Int64 = 0
private var currentUtteranceId: String?
private var isInUtterance: Bool = false
private var recognizingTranslationTask: Task<Void, Never>?
private var currentSynthesisId: String?
private var currentSynthesisText: String?
/**
* 生成新的 utteranceId
*/
private func nextUtteranceId() -> String {
utteranceSeq += 1
let ts = Int64(Date().timeIntervalSince1970 * 1000)
return "utt-\(serviceId)-\(ts)-\(utteranceSeq)"
}
// 事件回调
private weak var eventCallback: ServiceEventCallback?
@ -75,8 +97,6 @@ import os.log
var sourceLanguage: String = "zh-CN"
var targetLanguage: String = "en-US"
var currentVoice: String = "en-US-AriaNeural"
var translationSourceLanguage: String = ""
var translationTargetLanguage: String = ""
var speechRate: String = "0%"
var speechPitch: String = "0%"
var speechVolume: String = "100%"
@ -86,13 +106,14 @@ import os.log
var translationTimeout: TimeInterval = 10.0
// 新增:控制是否播放合成的音频
var enableAudioPlayback: Bool = false
// 新增:可选翻译源/目标语言(优先级高于识别语言对)
var translationSourceLanguage: String = ""
var translationTargetLanguage: String = ""
public init(
sourceLanguage: String = "zh-CN",
targetLanguage: String = "en-US",
currentVoice: String = "en-US-AriaNeural",
translationSourceLanguage: String = "",
translationTargetLanguage: String = "",
speechRate: String = "0%",
speechPitch: String = "0%",
speechVolume: String = "100%",
@ -100,13 +121,13 @@ import os.log
enableAutoLanguageDetection: Bool = false,
maxRetryAttempts: Int = 3,
translationTimeout: TimeInterval = 10.0,
enableAudioPlayback: Bool = false
enableAudioPlayback: Bool = false,
translationSourceLanguage: String = "",
translationTargetLanguage: String = ""
) {
self.sourceLanguage = sourceLanguage
self.targetLanguage = targetLanguage
self.currentVoice = currentVoice
self.translationSourceLanguage = translationSourceLanguage
self.translationTargetLanguage = translationTargetLanguage
self.speechRate = speechRate
self.speechPitch = speechPitch
self.speechVolume = speechVolume
@ -115,6 +136,8 @@ import os.log
self.maxRetryAttempts = maxRetryAttempts
self.translationTimeout = translationTimeout
self.enableAudioPlayback = enableAudioPlayback
self.translationSourceLanguage = translationSourceLanguage
self.translationTargetLanguage = translationTargetLanguage
}
}
@ -156,21 +179,21 @@ import os.log
*/
public protocol ServiceEventCallback: AnyObject {
func onServiceInitialized()
func onRecognizing(text: String, language: String, confidence: Float)
func onRecognized(text: String, language: String, confidence: Float)
func onTranslated(originalText: String, translatedText: String, targetLanguage: String)
func onTranslationStarted(text: String)
func onTranslationFailed(text: String, error: String)
func onSynthesisStarted(text: String)
func onSynthesisCompleted(text: String)
func onSynthesisFailed(text: String, error: String)
func onSynthesisProgress(text: String, progress: Float)
func onRecognizing(utteranceId: String, text: String, language: String, confidence: Float)
func onRecognized(utteranceId: String, text: String, language: String, confidence: Float)
func onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String)
func onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String)
func onTranslationStarted(utteranceId: String, text: String)
func onTranslationFailed(utteranceId: String, text: String, error: String)
func onSynthesisStarted(utteranceId: String, text: String)
func onSynthesisCompleted(utteranceId: String, text: String)
func onSynthesisFailed(utteranceId: String, text: String, error: String)
func onSynthesisProgress(utteranceId: String, text: String, progress: Float)
func onRecognitionStarted()
func onRecognitionStopped()
func onStateChanged(component: String, isActive: Bool)
func onError(component: String, error: String)
// 新增:返回合成的音频数据
func onSynthesisAudioGenerated(text: String, audioData: Data)
func onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: Data)
}
/**
@ -295,7 +318,7 @@ import os.log
config.speechRecognitionLanguage = serviceConfig.sourceLanguage
config.speechSynthesisVoiceName = getVoiceForLanguage(serviceConfig.targetLanguage)
config.setSpeechSynthesisOutputFormat(.riff16Khz16BitMonoPcm)
config.setSpeechSynthesisOutputFormat(.raw16Khz16BitMonoPcm)
// 自动语言检测
if serviceConfig.enableAutoLanguageDetection {
@ -316,15 +339,13 @@ import os.log
*/
private func initializeTranslationService(translationConfig: TranslationConfiguration) async -> Bool {
do {
translationService = VolcanoTranslationServiceImpl()
// 切换为微软翻译服务实现
translationService = MicrosoftTranslationServiceImpl()
let success = await translationService?.initialize(config: translationConfig.toConfigMap()) ?? false
if success {
os_log("翻译服务初始化完成", log: log, type: .info)
os_log("翻译服务初始化完成(Microsoft)", log: log, type: .info)
}
return success
} catch {
os_log("翻译服务初始化失败: %@", log: log, type: .error, error.localizedDescription)
return false
@ -337,6 +358,8 @@ import os.log
private func initializeAudioProcessor() {
audioProcessor = SimpleAudioReceiver()
audioProcessor?.initAudioRecord()
// 显式设置推流格式为 16k/16bit/单声道,避免适配层默认推断
audioProcessor?.setAudioConfig(sampleRate: Int(Self.SAMPLE_RATE), channels: Int(Self.CHANNELS))
// 检查音频配置是否已存在
if audioConfig == nil, let pushStream = audioProcessor?.pushAudioStream {
@ -365,14 +388,28 @@ import os.log
guard let text = result.text, !text.isEmpty else { return }
let confidence = self.extractConfidence(from: result)
os_log("识别中事件: %@", log: self.log, type: .debug, text)
// 开始新的话段时生成并记录 utteranceId
if !self.isInUtterance {
self.currentUtteranceId = self.nextUtteranceId()
self.isInUtterance = true
}
let uttId = self.currentUtteranceId ?? self.nextUtteranceId()
DispatchQueue.main.async {
self.eventCallback?.onRecognizing(
utteranceId: uttId,
text: text,
language: self.serviceConfig.sourceLanguage,
confidence: confidence
)
}
// 当服务处于运行状态时,为“识别中”的内容启动仅翻译流程(不合成)
if self.serviceState.isRecognizing {
self.recognizingTranslationTask?.cancel()
self.recognizingTranslationTask = Task { [weak self] in
guard let self = self else { return }
await self.processTranslationOnly(utteranceId: uttId, text: text)
}
}
}
// 识别完成事件
@ -384,9 +421,10 @@ import os.log
case .recognizedSpeech:
if let text = result.text, !text.isEmpty {
let confidence = self.extractConfidence(from: result)
let uttId = self.currentUtteranceId ?? self.nextUtteranceId()
DispatchQueue.main.async {
self.eventCallback?.onRecognized(
utteranceId: uttId,
text: text,
language: self.serviceConfig.sourceLanguage,
confidence: confidence
@ -397,8 +435,13 @@ import os.log
// 检查服务状态,只有在识别仍然活跃时才触发翻译流程
if self.serviceState.isRecognizing {
if self.isInUtterance {
self.currentUtteranceId = self.nextUtteranceId()
self.isInUtterance = false
}
Task {
await self.processTranslationAndSynthesis(text: text)
await self.cancelRecognizingTranslationTaskSafely()
await self.processTranslationAndSynthesis(utteranceId: uttId, text: text)
}
} else {
os_log("服务已停止,跳过翻译流程", log: self.log, type: .info)
@ -490,6 +533,9 @@ import os.log
DispatchQueue.main.async {
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true)
let uttId = self.currentSynthesisId ?? ""
let text = self.currentSynthesisText ?? ""
self.eventCallback?.onSynthesisStarted(utteranceId: uttId, text: text)
}
}
@ -502,6 +548,14 @@ import os.log
return
}
os_log("合成进行中事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count)
let uttId = self.currentSynthesisId ?? ""
let text = self.currentSynthesisText ?? ""
DispatchQueue.main.async {
let audioCopy = Data(audioData)
self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy)
self.eventCallback?.onSynthesisProgress(utteranceId: uttId, text: text, progress: 0.5)
}
}
@ -515,12 +569,18 @@ import os.log
os_log("合成完成事件,音频数据为空", log: self.log, type: .debug)
return
}
os_log("合成完成事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count)
os_log("合成完成事件,事件数据长度: %d bytes, 缓冲数据长度: %d bytes", log: self.log, type: .debug, audioData.count, 0)
let audioCopy = Data(audioData)
DispatchQueue.main.async {
self.serviceState.isSynthesizing = false
self.serviceState.isSynthesizing = false
// 立即返回当前生成的音频数据
self.eventCallback?.onSynthesisAudioGenerated(text: "", audioData: audioData)
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false)
let uttId = self.currentSynthesisId ?? ""
let text = self.currentSynthesisText ?? ""
//self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy)
self.eventCallback?.onSynthesisCompleted(utteranceId: uttId, text: text)
self.currentSynthesisId = nil
self.currentSynthesisText = nil
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false)
}
}
@ -536,7 +596,11 @@ import os.log
let reason = event.result.reason
DispatchQueue.main.async {
self.eventCallback?.onError(component: "Synthesis", error: "语音合成取消: \(reason)")
let uttId = self.currentSynthesisId ?? ""
let text = self.currentSynthesisText ?? ""
self.eventCallback?.onSynthesisFailed(utteranceId: uttId, text: text, error: "语音合成取消: \(reason)")
self.currentSynthesisId = nil
self.currentSynthesisText = nil
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false)
}
}
@ -650,7 +714,14 @@ import os.log
/**
* 处理翻译和合成流程
*/
private func processTranslationAndSynthesis(text: String) async {
/**
* 处理最终翻译并进行语音合成
* - Parameters:
* - utteranceId: 该句话的唯一ID
* - text: 识别完成的最终文本
*/
/// 处理最终翻译并进行语音合成(优先使用配置的翻译语言对)
private func processTranslationAndSynthesis(utteranceId: String, text: String) async {
// 添加服务状态检查
guard serviceState.isRecognizing else {
os_log("服务已停止,取消翻译流程", log: log, type: .info)
@ -667,19 +738,23 @@ import os.log
serviceState.isTranslating = true
await MainActor.run {
eventCallback?.onTranslationStarted(text: text)
eventCallback?.onTranslationStarted(utteranceId: utteranceId, text: text)
eventCallback?.onStateChanged(component: "Translation", isActive: true)
}
os_log("翻译开始: %@", log: log, type: .info, text)
do {
let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage
let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage
os_log("翻译语言对: %@ -> %@", log: log, type: .debug, srcLang, tgtLang)
let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) {
// 修复:在闭包中显式使用self
os_log("开始翻译文本: %@", log: self.log, type: .debug, text)
return await self.translationService?.translateText(
text: text,
sourceLanguage: self.serviceConfig.sourceLanguage,
targetLanguage: self.serviceConfig.targetLanguage
sourceLanguage: srcLang,
targetLanguage: tgtLang
)
}
@ -694,14 +769,15 @@ import os.log
await MainActor.run {
eventCallback?.onTranslated(
utteranceId: utteranceId,
originalText: text,
translatedText: translatedText,
targetLanguage: serviceConfig.targetLanguage
targetLanguage: tgtLang
)
}
// 进行语音合成
if synthesizeTextStreaming(translatedText) {
if synthesizeTextStreaming(utteranceId, translatedText) {
print("语音合成已启动")
}
} else {
@ -709,7 +785,7 @@ import os.log
os_log("翻译失败: %@", log: log, type: .error, errorMessage)
await MainActor.run {
eventCallback?.onTranslationFailed(text: text, error: errorMessage)
eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: errorMessage)
}
}
@ -718,12 +794,68 @@ import os.log
await MainActor.run {
eventCallback?.onStateChanged(component: "Translation", isActive: false)
eventCallback?.onTranslationFailed(text: text, error: "翻译超时或异常: \(error.localizedDescription)")
eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时或异常: \(error.localizedDescription)")
}
os_log("翻译异常: %@", log: log, type: .error, error.localizedDescription)
}
}
/**
* 处理识别中临时翻译(不进行合成)
* - Parameters:
* - utteranceId: 当前话段的唯一ID
* - text: 识别中的临时文本
*/
/// 处理识别中临时翻译(优先使用配置的翻译语言对)
private func processTranslationOnly(utteranceId: String, text: String) async {
guard serviceState.isRecognizing else { return }
do {
let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage
let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage
let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) {
return await self.translationService?.translateText(
text: text,
sourceLanguage: srcLang,
targetLanguage: tgtLang
)
}
if let result = translationResult, result.success, let translatedText = result.translatedText, !translatedText.isEmpty {
await MainActor.run {
self.eventCallback?.onInterimTranslated(
utteranceId: utteranceId,
originalText: text,
translatedText: translatedText,
targetLanguage: tgtLang
)
}
} else if let err = translationResult?.error {
await MainActor.run {
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: err)
}
} else {
await MainActor.run {
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译服务返回空结果")
}
}
} catch is TimeoutError {
await MainActor.run {
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时")
}
} catch {
await MainActor.run {
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译异常: \(error.localizedDescription)")
}
}
}
/**
* 安全取消识别中翻译任务
*/
private func cancelRecognizingTranslationTaskSafely() async {
recognizingTranslationTask?.cancel()
recognizingTranslationTask = nil
}
/**
* 语音合成
@ -736,24 +868,39 @@ import os.log
/// 流式语音合成
/// - Parameter text: 要合成的文本
/// - Returns: 合成是否成功启动
private func synthesizeTextStreaming(_ text: String) -> Bool {
/**
* 流式语音合成,携带utteranceId
* - Parameters:
* - utteranceId: 当前合成对应的话段ID
* - text: 要合成的文本
*/
private func synthesizeTextStreaming(_ utteranceId: String, _ text: String) -> Bool {
let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines)
guard !trimmed.isEmpty else {
os_log("合成文本为空或仅空白,跳过合成", log: log, type: .info)
return false
}
guard let synthesizer = synthesizer else {
os_log("语音合成器未初始化", log: log, type: .error)
return false
}
do {
let ssml = buildSSML(text: text)
let sanitized = cleanTextForTTS(trimmed)
let ssml = generateOptimizedSsml(sanitized)
os_log("开始流式合成语音: %@", log: log, type: .info, text)
serviceState.isSynthesizing = true
currentSynthesisId = utteranceId
currentSynthesisText = text
DispatchQueue.main.async {
self.eventCallback?.onSynthesisStarted(text: text)
self.eventCallback?.onSynthesisStarted(utteranceId: utteranceId, text: text)
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true)
}
try synthesizer.speakSsml(ssml)
// 使用异步启动,提升并发兼容性
try synthesizer.startSpeakingSsml(ssml)
return true
@ -764,7 +911,7 @@ import os.log
os_log("%@", log: log, type: .error, errorMessage)
DispatchQueue.main.async {
self.eventCallback?.onSynthesisFailed(text: text, error: errorMessage)
self.eventCallback?.onSynthesisFailed(utteranceId: utteranceId, text: text, error: errorMessage)
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false)
}
@ -774,36 +921,78 @@ import os.log
/// 保留原有的同步合成方法作为备用
private func synthesizeText(_ text: String) -> Data? {
// 现在调用流式合成方法
let success = synthesizeTextStreaming(text)
return success ? Data() : nil // 返回空数据表示已启动,实际数据通过回调返回
let success = synthesizeTextStreaming("", text)
return success ? Data() : nil
}
/**
* 构建SSML
/**
* 清理TTS文本
*/
private func buildSSML(text: String) -> String {
let voice = getVoiceForLanguage(serviceConfig.targetLanguage)
private func cleanTextForTTS(_ text: String) -> String {
var cleaned = text
// 移除URL
if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "")
}
// 移除emoji
if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "")
}
// 移除标点符号
if let regex = try? NSRegularExpression(pattern: "[;:#;:*\\n]+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "")
}
// 合并空格
if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ")
}
return cleaned.trimmingCharacters(in: .whitespacesAndNewlines)
}
/**
* 生成优化的SSML
*/
private func generateOptimizedSsml(_ rawText: String) -> String {
// 转义XML保留字符
let escapedText = rawText
.replacingOccurrences(of: "&", with: "&amp;")
.replacingOccurrences(of: "<", with: "&lt;")
.replacingOccurrences(of: ">", with: "&gt;")
print("serviceConfig.targetLanguage: \(serviceConfig.targetLanguage)")
let voice = getVoiceForLanguage(serviceConfig.targetLanguage)
// 简化SSML结构
return """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(serviceConfig.targetLanguage)">
<voice name="\(voice)">
<prosody rate="\(serviceConfig.speechRate)" pitch="\(serviceConfig.speechPitch)" volume="\(serviceConfig.speechVolume)">
\(text)
</prosody>
</voice>
<voice name="\(voice)">
<prosody rate="\(serviceConfig.speechRate)" pitch="\(serviceConfig.speechPitch)" volume="\(serviceConfig.speechVolume)">
\(escapedText)
</prosody>
</voice>
</speak>
"""
}
/**
* 根据语言获取对应的语音
*/
private func getVoiceForLanguage(_ language: String) -> String {
// 优先使用 Dart 传入的 currentVoice(已包含性别信息)
if !serviceConfig.currentVoice.isEmpty {
return serviceConfig.currentVoice
}
if let cachedVoice = voiceCache[language] {
return cachedVoice
}
let voice: String
switch language {
// 中文相关

139
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/MicrosoftTranslationServiceImpl.swift

@ -0,0 +1,139 @@
import Foundation
import os.log
/**
* 微软翻译服务实现(iOS)
* 实现 IntegratedSpeechTranslationService.TranslationServiceInterface
*/
public class MicrosoftTranslationServiceImpl: IntegratedSpeechTranslationService.TranslationServiceInterface {
private let log = OSLog(subsystem: "com.azure.speech", category: "MicrosoftTranslationService")
private let defaultEndpoint = "https://api.cognitive.microsofttranslator.com"
private let apiPath = "/translate"
private let apiVersion = "3.0"
private var isInitialized = false
private var subscriptionKey: String = ""
private var region: String = ""
private var endpoint: String = ""
private var maxRetryAttempts: Int = 3
private var timeout: TimeInterval = 10.0
private var _urlSession: URLSession?
private var urlSession: URLSession {
if let s = _urlSession { return s }
let cfg = URLSessionConfiguration.default
cfg.timeoutIntervalForRequest = timeout
cfg.timeoutIntervalForResource = 30.0
let s = URLSession(configuration: cfg)
_urlSession = s
return s
}
/**
* 初始化翻译服务
*/
public func initialize(config: [String : String]) async -> Bool {
subscriptionKey = (config["subscriptionKey"] ?? config["accessKey"] ?? "").trimmingCharacters(in: .whitespaces)
region = (config["region"] ?? config["location"] ?? "").trimmingCharacters(in: .whitespaces)
endpoint = (config["endpoint"] ?? defaultEndpoint).trimmingCharacters(in: .whitespaces)
os_log("微软翻译初始化参数: endpoint=%@ region=%@ keyLen=%d", log: log, type: .info, endpoint, region, subscriptionKey.count)
if let mr = config["maxRetryAttempts"], let v = Int(mr) { maxRetryAttempts = v }
if let to = config["timeout"], let v = Double(to) { timeout = v }
// 重置会话以应用新配置或清除旧状态
if _urlSession != nil {
_urlSession?.invalidateAndCancel()
_urlSession = nil
}
isInitialized = !subscriptionKey.isEmpty
os_log("微软翻译服务初始化: %{public}@", log: log, type: .info, isInitialized ? "成功" : "失败")
return isInitialized
}
/**
* 文本翻译
*/
public func translateText(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult {
guard isInitialized else {
return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "翻译服务未初始化")
}
os_log("调用微软翻译: from=%@ to=%@ textLen=%d", log: log, type: .info, sourceLanguage, targetLanguage, text.count)
return await performTranslationWithRetry(text: text, sourceLanguage: sourceLanguage, targetLanguage: targetLanguage)
}
/**
* 释放资源
*/
public func dispose() {
_urlSession?.invalidateAndCancel()
_urlSession = nil
isInitialized = false
}
/**
* 带重试的翻译执行
*/
private func performTranslationWithRetry(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult {
var lastError = "未知错误"
for attempt in 1...maxRetryAttempts {
os_log("微软翻译尝试 %d/%d", log: log, type: .debug, attempt, maxRetryAttempts)
let res = await performTranslation(text: text, sourceLanguage: sourceLanguage, targetLanguage: targetLanguage)
if res.success {
os_log("微软翻译成功(尝试 %d)", log: log, type: .info, attempt)
return res
}
lastError = res.error ?? lastError
if attempt < maxRetryAttempts { try? await Task.sleep(nanoseconds: UInt64(attempt) * 500_000_000) }
}
os_log("微软翻译失败,错误: %@", log: log, type: .error, lastError)
return IntegratedSpeechTranslationService.TranslationResult(success: false, error: lastError)
}
/**
* 实际翻译执行
*/
private func performTranslation(text: String, sourceLanguage: String, targetLanguage: String) async -> IntegratedSpeechTranslationService.TranslationResult {
do {
var comps = URLComponents(string: endpoint + apiPath)!
comps.queryItems = [
URLQueryItem(name: "api-version", value: apiVersion),
URLQueryItem(name: "from", value: sourceLanguage),
URLQueryItem(name: "to", value: targetLanguage)
]
guard let url = comps.url else { throw URLError(.badURL) }
var req = URLRequest(url: url)
req.httpMethod = "POST"
let bodyArr: [[String: Any]] = [["text": text]]
req.httpBody = try JSONSerialization.data(withJSONObject: bodyArr)
req.setValue("application/json", forHTTPHeaderField: "Content-Type")
req.setValue(subscriptionKey, forHTTPHeaderField: "Ocp-Apim-Subscription-Key")
req.setValue(region, forHTTPHeaderField: "Ocp-Apim-Subscription-Region")
req.setValue(UUID().uuidString, forHTTPHeaderField: "X-ClientTraceId")
os_log("请求微软翻译: endpoint=%@ from=%@ to=%@", log: log, type: .debug, endpoint, sourceLanguage, targetLanguage)
os_log("请求体长度: %d bytes", log: log, type: .debug, (req.httpBody ?? Data()).count)
let (data, resp) = try await urlSession.data(for: req)
guard let http = resp as? HTTPURLResponse, http.statusCode == 200 else {
let code = (resp as? HTTPURLResponse)?.statusCode ?? -1
os_log("微软翻译HTTP错误: %d", log: log, type: .error, code)
return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "HTTP错误: \(code)")
}
os_log("微软翻译HTTP成功,响应大小: %d bytes", log: log, type: .debug, data.count)
guard let arr = try JSONSerialization.jsonObject(with: data) as? [[String: Any]], let item = arr.first,
let translations = item["translations"] as? [[String: Any]], let first = translations.first,
let tx = first["text"] as? String else {
os_log("微软翻译解析失败,返回结构不符合预期", log: log, type: .error)
return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "翻译结果为空")
}
let preview = tx.prefix(64)
os_log("微软翻译解析成功,译文预览: %@", log: log, type: .info, String(preview))
return IntegratedSpeechTranslationService.TranslationResult(success: true, translatedText: tx, confidence: 0.9)
} catch {
os_log("微软翻译请求异常: %@", log: log, type: .error, error.localizedDescription)
return IntegratedSpeechTranslationService.TranslationResult(success: false, error: "请求异常: \(error.localizedDescription)")
}
}
}

42
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/VolcanoTranslationServiceImpl.swift

@ -267,13 +267,13 @@ public class VolcanoTranslationServiceImpl: IntegratedSpeechTranslationService.T
}
let jsonData = try JSONSerialization.data(withJSONObject: finalRequestBody)
// 构建查询参数
let queryParams: [String: String] = [
"Action": action,
"Version": version
]
// 生成签名
let headers = try generateSignature(
method: "POST",
@ -414,33 +414,33 @@ public class VolcanoTranslationServiceImpl: IntegratedSpeechTranslationService.T
requestBody: [String: Any],
queryParams: [String: String]
) throws -> [String: String] {
let now = Date()
let dateFormatter = DateFormatter()
dateFormatter.dateFormat = "yyyyMMdd'T'HHmmss'Z'"
dateFormatter.timeZone = TimeZone(abbreviation: "UTC")
let timestamp = dateFormatter.string(from: now)
let shortDateFormatter = DateFormatter()
shortDateFormatter.dateFormat = "yyyyMMdd"
shortDateFormatter.timeZone = TimeZone(abbreviation: "UTC")
let shortDate = shortDateFormatter.string(from: now)
// 构建规范请求
let httpRequestMethod = method
let canonicalURI = endpoint
// 构建规范查询字符串
let sortedQueryParams = queryParams.sorted { $0.key < $1.key }
let canonicalQueryString = sortedQueryParams
.map { "\($0.key)=\($0.value.addingPercentEncoding(withAllowedCharacters: .urlQueryAllowed) ?? $0.value)" }
.joined(separator: "&")
// 构建规范头部
let host = URL(string: baseURL)!.host!
let canonicalHeaders = "host:\(host)\nx-date:\(timestamp)\n"
let signedHeaders = "host;x-date"
// 计算请求体哈希
let requestBodyData = try JSONSerialization.data(withJSONObject: requestBody)
let hashedRequestPayload = sha256(data: requestBodyData)
@ -539,20 +539,30 @@ enum TranslationError: Error {
*/
extension TranslationConfiguration {
func toConfigMap() -> [String: String] {
var config: [String: String] = [
"accessKey": accessKey,
"secretKey": secretKey,
"region": region
]
var config: [String: String] = [:]
if !subscriptionKey.isEmpty {
config["subscriptionKey"] = subscriptionKey
} else if !accessKey.isEmpty {
config["accessKey"] = accessKey
}
if !secretKey.isEmpty {
config["secretKey"] = secretKey
}
if !region.isEmpty {
config["region"] = region
}
if !location.isEmpty {
config["location"] = location
}
if !endpoint.isEmpty {
config["endpoint"] = endpoint
}
if let maxRetry = maxRetryAttempts {
config["maxRetryAttempts"] = String(maxRetry)
}
if let timeoutValue = timeout {
config["timeout"] = String(timeoutValue)
}
return config
}
}

Loading…
Cancel
Save