|
|
|
@ -16,7 +16,10 @@ import os.log |
|
|
|
public enum AudioSourceType { |
|
|
|
/** 使用设备麦克风 */ |
|
|
|
case microphone |
|
|
|
|
|
|
|
/** 使用系统音频 */ |
|
|
|
case systemAudio |
|
|
|
/** 使用系统音频+麦克风音频 */ |
|
|
|
case systemAudioPlusMicrophone |
|
|
|
/** 使用外部提供的音频数据 */ |
|
|
|
case external |
|
|
|
} |
|
|
|
@ -34,6 +37,8 @@ import os.log |
|
|
|
|
|
|
|
// MARK: - Properties |
|
|
|
private let log = OSLog(subsystem: "com.azure.speech", category: "IntegratedSpeechService") |
|
|
|
// 实例标识,用于区分 A/B,影响 utteranceId 生成 |
|
|
|
public var serviceId: String = "A" |
|
|
|
|
|
|
|
// Azure服务组件 |
|
|
|
private var speechConfig: SPXSpeechConfiguration? |
|
|
|
@ -56,6 +61,23 @@ import os.log |
|
|
|
|
|
|
|
// 状态管理 |
|
|
|
private let serviceState = ServiceState() |
|
|
|
|
|
|
|
// ===== Utterance 关联ID管理 ===== |
|
|
|
private var utteranceSeq: Int64 = 0 |
|
|
|
private var currentUtteranceId: String? |
|
|
|
private var isInUtterance: Bool = false |
|
|
|
private var recognizingTranslationTask: Task<Void, Never>? |
|
|
|
private var currentSynthesisId: String? |
|
|
|
private var currentSynthesisText: String? |
|
|
|
|
|
|
|
/** |
|
|
|
* 生成新的 utteranceId |
|
|
|
*/ |
|
|
|
private func nextUtteranceId() -> String { |
|
|
|
utteranceSeq += 1 |
|
|
|
let ts = Int64(Date().timeIntervalSince1970 * 1000) |
|
|
|
return "utt-\(serviceId)-\(ts)-\(utteranceSeq)" |
|
|
|
} |
|
|
|
|
|
|
|
// 事件回调 |
|
|
|
private weak var eventCallback: ServiceEventCallback? |
|
|
|
@ -75,8 +97,6 @@ import os.log |
|
|
|
var sourceLanguage: String = "zh-CN" |
|
|
|
var targetLanguage: String = "en-US" |
|
|
|
var currentVoice: String = "en-US-AriaNeural" |
|
|
|
var translationSourceLanguage: String = "" |
|
|
|
var translationTargetLanguage: String = "" |
|
|
|
var speechRate: String = "0%" |
|
|
|
var speechPitch: String = "0%" |
|
|
|
var speechVolume: String = "100%" |
|
|
|
@ -86,13 +106,14 @@ import os.log |
|
|
|
var translationTimeout: TimeInterval = 10.0 |
|
|
|
// 新增:控制是否播放合成的音频 |
|
|
|
var enableAudioPlayback: Bool = false |
|
|
|
// 新增:可选翻译源/目标语言(优先级高于识别语言对) |
|
|
|
var translationSourceLanguage: String = "" |
|
|
|
var translationTargetLanguage: String = "" |
|
|
|
|
|
|
|
public init( |
|
|
|
sourceLanguage: String = "zh-CN", |
|
|
|
targetLanguage: String = "en-US", |
|
|
|
currentVoice: String = "en-US-AriaNeural", |
|
|
|
translationSourceLanguage: String = "", |
|
|
|
translationTargetLanguage: String = "", |
|
|
|
speechRate: String = "0%", |
|
|
|
speechPitch: String = "0%", |
|
|
|
speechVolume: String = "100%", |
|
|
|
@ -100,13 +121,13 @@ import os.log |
|
|
|
enableAutoLanguageDetection: Bool = false, |
|
|
|
maxRetryAttempts: Int = 3, |
|
|
|
translationTimeout: TimeInterval = 10.0, |
|
|
|
enableAudioPlayback: Bool = false |
|
|
|
enableAudioPlayback: Bool = false, |
|
|
|
translationSourceLanguage: String = "", |
|
|
|
translationTargetLanguage: String = "" |
|
|
|
) { |
|
|
|
self.sourceLanguage = sourceLanguage |
|
|
|
self.targetLanguage = targetLanguage |
|
|
|
self.currentVoice = currentVoice |
|
|
|
self.translationSourceLanguage = translationSourceLanguage |
|
|
|
self.translationTargetLanguage = translationTargetLanguage |
|
|
|
self.speechRate = speechRate |
|
|
|
self.speechPitch = speechPitch |
|
|
|
self.speechVolume = speechVolume |
|
|
|
@ -115,6 +136,8 @@ import os.log |
|
|
|
self.maxRetryAttempts = maxRetryAttempts |
|
|
|
self.translationTimeout = translationTimeout |
|
|
|
self.enableAudioPlayback = enableAudioPlayback |
|
|
|
self.translationSourceLanguage = translationSourceLanguage |
|
|
|
self.translationTargetLanguage = translationTargetLanguage |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
@ -156,21 +179,21 @@ import os.log |
|
|
|
*/ |
|
|
|
public protocol ServiceEventCallback: AnyObject { |
|
|
|
func onServiceInitialized() |
|
|
|
func onRecognizing(text: String, language: String, confidence: Float) |
|
|
|
func onRecognized(text: String, language: String, confidence: Float) |
|
|
|
func onTranslated(originalText: String, translatedText: String, targetLanguage: String) |
|
|
|
func onTranslationStarted(text: String) |
|
|
|
func onTranslationFailed(text: String, error: String) |
|
|
|
func onSynthesisStarted(text: String) |
|
|
|
func onSynthesisCompleted(text: String) |
|
|
|
func onSynthesisFailed(text: String, error: String) |
|
|
|
func onSynthesisProgress(text: String, progress: Float) |
|
|
|
func onRecognizing(utteranceId: String, text: String, language: String, confidence: Float) |
|
|
|
func onRecognized(utteranceId: String, text: String, language: String, confidence: Float) |
|
|
|
func onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) |
|
|
|
func onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) |
|
|
|
func onTranslationStarted(utteranceId: String, text: String) |
|
|
|
func onTranslationFailed(utteranceId: String, text: String, error: String) |
|
|
|
func onSynthesisStarted(utteranceId: String, text: String) |
|
|
|
func onSynthesisCompleted(utteranceId: String, text: String) |
|
|
|
func onSynthesisFailed(utteranceId: String, text: String, error: String) |
|
|
|
func onSynthesisProgress(utteranceId: String, text: String, progress: Float) |
|
|
|
func onRecognitionStarted() |
|
|
|
func onRecognitionStopped() |
|
|
|
func onStateChanged(component: String, isActive: Bool) |
|
|
|
func onError(component: String, error: String) |
|
|
|
// 新增:返回合成的音频数据 |
|
|
|
func onSynthesisAudioGenerated(text: String, audioData: Data) |
|
|
|
func onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: Data) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
@ -295,7 +318,7 @@ import os.log |
|
|
|
|
|
|
|
config.speechRecognitionLanguage = serviceConfig.sourceLanguage |
|
|
|
config.speechSynthesisVoiceName = getVoiceForLanguage(serviceConfig.targetLanguage) |
|
|
|
config.setSpeechSynthesisOutputFormat(.riff16Khz16BitMonoPcm) |
|
|
|
config.setSpeechSynthesisOutputFormat(.raw16Khz16BitMonoPcm) |
|
|
|
|
|
|
|
// 自动语言检测 |
|
|
|
if serviceConfig.enableAutoLanguageDetection { |
|
|
|
@ -316,15 +339,13 @@ import os.log |
|
|
|
*/ |
|
|
|
private func initializeTranslationService(translationConfig: TranslationConfiguration) async -> Bool { |
|
|
|
do { |
|
|
|
translationService = VolcanoTranslationServiceImpl() |
|
|
|
// 切换为微软翻译服务实现 |
|
|
|
translationService = MicrosoftTranslationServiceImpl() |
|
|
|
let success = await translationService?.initialize(config: translationConfig.toConfigMap()) ?? false |
|
|
|
|
|
|
|
if success { |
|
|
|
os_log("翻译服务初始化完成", log: log, type: .info) |
|
|
|
os_log("翻译服务初始化完成(Microsoft)", log: log, type: .info) |
|
|
|
} |
|
|
|
|
|
|
|
return success |
|
|
|
|
|
|
|
} catch { |
|
|
|
os_log("翻译服务初始化失败: %@", log: log, type: .error, error.localizedDescription) |
|
|
|
return false |
|
|
|
@ -337,6 +358,8 @@ import os.log |
|
|
|
private func initializeAudioProcessor() { |
|
|
|
audioProcessor = SimpleAudioReceiver() |
|
|
|
audioProcessor?.initAudioRecord() |
|
|
|
// 显式设置推流格式为 16k/16bit/单声道,避免适配层默认推断 |
|
|
|
audioProcessor?.setAudioConfig(sampleRate: Int(Self.SAMPLE_RATE), channels: Int(Self.CHANNELS)) |
|
|
|
|
|
|
|
// 检查音频配置是否已存在 |
|
|
|
if audioConfig == nil, let pushStream = audioProcessor?.pushAudioStream { |
|
|
|
@ -365,14 +388,28 @@ import os.log |
|
|
|
guard let text = result.text, !text.isEmpty else { return } |
|
|
|
let confidence = self.extractConfidence(from: result) |
|
|
|
os_log("识别中事件: %@", log: self.log, type: .debug, text) |
|
|
|
|
|
|
|
// 开始新的话段时生成并记录 utteranceId |
|
|
|
if !self.isInUtterance { |
|
|
|
self.currentUtteranceId = self.nextUtteranceId() |
|
|
|
self.isInUtterance = true |
|
|
|
} |
|
|
|
let uttId = self.currentUtteranceId ?? self.nextUtteranceId() |
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onRecognizing( |
|
|
|
utteranceId: uttId, |
|
|
|
text: text, |
|
|
|
language: self.serviceConfig.sourceLanguage, |
|
|
|
confidence: confidence |
|
|
|
) |
|
|
|
} |
|
|
|
// 当服务处于运行状态时,为“识别中”的内容启动仅翻译流程(不合成) |
|
|
|
if self.serviceState.isRecognizing { |
|
|
|
self.recognizingTranslationTask?.cancel() |
|
|
|
self.recognizingTranslationTask = Task { [weak self] in |
|
|
|
guard let self = self else { return } |
|
|
|
await self.processTranslationOnly(utteranceId: uttId, text: text) |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 识别完成事件 |
|
|
|
@ -384,9 +421,10 @@ import os.log |
|
|
|
case .recognizedSpeech: |
|
|
|
if let text = result.text, !text.isEmpty { |
|
|
|
let confidence = self.extractConfidence(from: result) |
|
|
|
|
|
|
|
let uttId = self.currentUtteranceId ?? self.nextUtteranceId() |
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onRecognized( |
|
|
|
utteranceId: uttId, |
|
|
|
text: text, |
|
|
|
language: self.serviceConfig.sourceLanguage, |
|
|
|
confidence: confidence |
|
|
|
@ -397,8 +435,13 @@ import os.log |
|
|
|
|
|
|
|
// 检查服务状态,只有在识别仍然活跃时才触发翻译流程 |
|
|
|
if self.serviceState.isRecognizing { |
|
|
|
if self.isInUtterance { |
|
|
|
self.currentUtteranceId = self.nextUtteranceId() |
|
|
|
self.isInUtterance = false |
|
|
|
} |
|
|
|
Task { |
|
|
|
await self.processTranslationAndSynthesis(text: text) |
|
|
|
await self.cancelRecognizingTranslationTaskSafely() |
|
|
|
await self.processTranslationAndSynthesis(utteranceId: uttId, text: text) |
|
|
|
} |
|
|
|
} else { |
|
|
|
os_log("服务已停止,跳过翻译流程", log: self.log, type: .info) |
|
|
|
@ -490,6 +533,9 @@ import os.log |
|
|
|
|
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true) |
|
|
|
let uttId = self.currentSynthesisId ?? "" |
|
|
|
let text = self.currentSynthesisText ?? "" |
|
|
|
self.eventCallback?.onSynthesisStarted(utteranceId: uttId, text: text) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
@ -502,6 +548,14 @@ import os.log |
|
|
|
return |
|
|
|
} |
|
|
|
os_log("合成进行中事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) |
|
|
|
|
|
|
|
let uttId = self.currentSynthesisId ?? "" |
|
|
|
let text = self.currentSynthesisText ?? "" |
|
|
|
DispatchQueue.main.async { |
|
|
|
let audioCopy = Data(audioData) |
|
|
|
self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy) |
|
|
|
self.eventCallback?.onSynthesisProgress(utteranceId: uttId, text: text, progress: 0.5) |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
|
|
|
|
@ -515,12 +569,18 @@ import os.log |
|
|
|
os_log("合成完成事件,音频数据为空", log: self.log, type: .debug) |
|
|
|
return |
|
|
|
} |
|
|
|
os_log("合成完成事件,音频数据长度: %d bytes", log: self.log, type: .debug, audioData.count) |
|
|
|
os_log("合成完成事件,事件数据长度: %d bytes, 缓冲数据长度: %d bytes", log: self.log, type: .debug, audioData.count, 0) |
|
|
|
let audioCopy = Data(audioData) |
|
|
|
DispatchQueue.main.async { |
|
|
|
self.serviceState.isSynthesizing = false |
|
|
|
self.serviceState.isSynthesizing = false |
|
|
|
// 立即返回当前生成的音频数据 |
|
|
|
self.eventCallback?.onSynthesisAudioGenerated(text: "", audioData: audioData) |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) |
|
|
|
let uttId = self.currentSynthesisId ?? "" |
|
|
|
let text = self.currentSynthesisText ?? "" |
|
|
|
//self.eventCallback?.onSynthesisAudioGenerated(utteranceId: uttId, text: text, audioData: audioCopy) |
|
|
|
self.eventCallback?.onSynthesisCompleted(utteranceId: uttId, text: text) |
|
|
|
self.currentSynthesisId = nil |
|
|
|
self.currentSynthesisText = nil |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
@ -536,7 +596,11 @@ import os.log |
|
|
|
let reason = event.result.reason |
|
|
|
|
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onError(component: "Synthesis", error: "语音合成取消: \(reason)") |
|
|
|
let uttId = self.currentSynthesisId ?? "" |
|
|
|
let text = self.currentSynthesisText ?? "" |
|
|
|
self.eventCallback?.onSynthesisFailed(utteranceId: uttId, text: text, error: "语音合成取消: \(reason)") |
|
|
|
self.currentSynthesisId = nil |
|
|
|
self.currentSynthesisText = nil |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) |
|
|
|
} |
|
|
|
} |
|
|
|
@ -650,7 +714,14 @@ import os.log |
|
|
|
/** |
|
|
|
* 处理翻译和合成流程 |
|
|
|
*/ |
|
|
|
private func processTranslationAndSynthesis(text: String) async { |
|
|
|
/** |
|
|
|
* 处理最终翻译并进行语音合成 |
|
|
|
* - Parameters: |
|
|
|
* - utteranceId: 该句话的唯一ID |
|
|
|
* - text: 识别完成的最终文本 |
|
|
|
*/ |
|
|
|
/// 处理最终翻译并进行语音合成(优先使用配置的翻译语言对) |
|
|
|
private func processTranslationAndSynthesis(utteranceId: String, text: String) async { |
|
|
|
// 添加服务状态检查 |
|
|
|
guard serviceState.isRecognizing else { |
|
|
|
os_log("服务已停止,取消翻译流程", log: log, type: .info) |
|
|
|
@ -667,19 +738,23 @@ import os.log |
|
|
|
serviceState.isTranslating = true |
|
|
|
|
|
|
|
await MainActor.run { |
|
|
|
eventCallback?.onTranslationStarted(text: text) |
|
|
|
eventCallback?.onTranslationStarted(utteranceId: utteranceId, text: text) |
|
|
|
eventCallback?.onStateChanged(component: "Translation", isActive: true) |
|
|
|
} |
|
|
|
|
|
|
|
os_log("翻译开始: %@", log: log, type: .info, text) |
|
|
|
|
|
|
|
do { |
|
|
|
let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage |
|
|
|
let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage |
|
|
|
os_log("翻译语言对: %@ -> %@", log: log, type: .debug, srcLang, tgtLang) |
|
|
|
|
|
|
|
let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) { |
|
|
|
// 修复:在闭包中显式使用self |
|
|
|
os_log("开始翻译文本: %@", log: self.log, type: .debug, text) |
|
|
|
return await self.translationService?.translateText( |
|
|
|
text: text, |
|
|
|
sourceLanguage: self.serviceConfig.sourceLanguage, |
|
|
|
targetLanguage: self.serviceConfig.targetLanguage |
|
|
|
sourceLanguage: srcLang, |
|
|
|
targetLanguage: tgtLang |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
@ -694,14 +769,15 @@ import os.log |
|
|
|
|
|
|
|
await MainActor.run { |
|
|
|
eventCallback?.onTranslated( |
|
|
|
utteranceId: utteranceId, |
|
|
|
originalText: text, |
|
|
|
translatedText: translatedText, |
|
|
|
targetLanguage: serviceConfig.targetLanguage |
|
|
|
targetLanguage: tgtLang |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
// 进行语音合成 |
|
|
|
if synthesizeTextStreaming(translatedText) { |
|
|
|
if synthesizeTextStreaming(utteranceId, translatedText) { |
|
|
|
print("语音合成已启动") |
|
|
|
} |
|
|
|
} else { |
|
|
|
@ -709,7 +785,7 @@ import os.log |
|
|
|
os_log("翻译失败: %@", log: log, type: .error, errorMessage) |
|
|
|
|
|
|
|
await MainActor.run { |
|
|
|
eventCallback?.onTranslationFailed(text: text, error: errorMessage) |
|
|
|
eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: errorMessage) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
@ -718,12 +794,68 @@ import os.log |
|
|
|
|
|
|
|
await MainActor.run { |
|
|
|
eventCallback?.onStateChanged(component: "Translation", isActive: false) |
|
|
|
eventCallback?.onTranslationFailed(text: text, error: "翻译超时或异常: \(error.localizedDescription)") |
|
|
|
eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时或异常: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
|
|
|
|
os_log("翻译异常: %@", log: log, type: .error, error.localizedDescription) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 处理识别中临时翻译(不进行合成) |
|
|
|
* - Parameters: |
|
|
|
* - utteranceId: 当前话段的唯一ID |
|
|
|
* - text: 识别中的临时文本 |
|
|
|
*/ |
|
|
|
/// 处理识别中临时翻译(优先使用配置的翻译语言对) |
|
|
|
private func processTranslationOnly(utteranceId: String, text: String) async { |
|
|
|
guard serviceState.isRecognizing else { return } |
|
|
|
do { |
|
|
|
let srcLang = self.serviceConfig.translationSourceLanguage.isEmpty ? self.serviceConfig.sourceLanguage : self.serviceConfig.translationSourceLanguage |
|
|
|
let tgtLang = self.serviceConfig.translationTargetLanguage.isEmpty ? self.serviceConfig.targetLanguage : self.serviceConfig.translationTargetLanguage |
|
|
|
let translationResult = try await withTimeout(self.serviceConfig.translationTimeout) { |
|
|
|
return await self.translationService?.translateText( |
|
|
|
text: text, |
|
|
|
sourceLanguage: srcLang, |
|
|
|
targetLanguage: tgtLang |
|
|
|
) |
|
|
|
} |
|
|
|
if let result = translationResult, result.success, let translatedText = result.translatedText, !translatedText.isEmpty { |
|
|
|
await MainActor.run { |
|
|
|
self.eventCallback?.onInterimTranslated( |
|
|
|
utteranceId: utteranceId, |
|
|
|
originalText: text, |
|
|
|
translatedText: translatedText, |
|
|
|
targetLanguage: tgtLang |
|
|
|
) |
|
|
|
} |
|
|
|
} else if let err = translationResult?.error { |
|
|
|
await MainActor.run { |
|
|
|
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: err) |
|
|
|
} |
|
|
|
} else { |
|
|
|
await MainActor.run { |
|
|
|
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译服务返回空结果") |
|
|
|
} |
|
|
|
} |
|
|
|
} catch is TimeoutError { |
|
|
|
await MainActor.run { |
|
|
|
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译超时") |
|
|
|
} |
|
|
|
} catch { |
|
|
|
await MainActor.run { |
|
|
|
self.eventCallback?.onTranslationFailed(utteranceId: utteranceId, text: text, error: "翻译异常: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 安全取消识别中翻译任务 |
|
|
|
*/ |
|
|
|
private func cancelRecognizingTranslationTaskSafely() async { |
|
|
|
recognizingTranslationTask?.cancel() |
|
|
|
recognizingTranslationTask = nil |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 语音合成 |
|
|
|
@ -736,24 +868,39 @@ import os.log |
|
|
|
/// 流式语音合成 |
|
|
|
/// - Parameter text: 要合成的文本 |
|
|
|
/// - Returns: 合成是否成功启动 |
|
|
|
private func synthesizeTextStreaming(_ text: String) -> Bool { |
|
|
|
/** |
|
|
|
* 流式语音合成,携带utteranceId |
|
|
|
* - Parameters: |
|
|
|
* - utteranceId: 当前合成对应的话段ID |
|
|
|
* - text: 要合成的文本 |
|
|
|
*/ |
|
|
|
private func synthesizeTextStreaming(_ utteranceId: String, _ text: String) -> Bool { |
|
|
|
let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines) |
|
|
|
guard !trimmed.isEmpty else { |
|
|
|
os_log("合成文本为空或仅空白,跳过合成", log: log, type: .info) |
|
|
|
return false |
|
|
|
} |
|
|
|
guard let synthesizer = synthesizer else { |
|
|
|
os_log("语音合成器未初始化", log: log, type: .error) |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
do { |
|
|
|
let ssml = buildSSML(text: text) |
|
|
|
let sanitized = cleanTextForTTS(trimmed) |
|
|
|
let ssml = generateOptimizedSsml(sanitized) |
|
|
|
os_log("开始流式合成语音: %@", log: log, type: .info, text) |
|
|
|
|
|
|
|
serviceState.isSynthesizing = true |
|
|
|
currentSynthesisId = utteranceId |
|
|
|
currentSynthesisText = text |
|
|
|
|
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onSynthesisStarted(text: text) |
|
|
|
self.eventCallback?.onSynthesisStarted(utteranceId: utteranceId, text: text) |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: true) |
|
|
|
} |
|
|
|
try synthesizer.speakSsml(ssml) |
|
|
|
|
|
|
|
// 使用异步启动,提升并发兼容性 |
|
|
|
try synthesizer.startSpeakingSsml(ssml) |
|
|
|
|
|
|
|
|
|
|
|
return true |
|
|
|
|
|
|
|
@ -764,7 +911,7 @@ import os.log |
|
|
|
os_log("%@", log: log, type: .error, errorMessage) |
|
|
|
|
|
|
|
DispatchQueue.main.async { |
|
|
|
self.eventCallback?.onSynthesisFailed(text: text, error: errorMessage) |
|
|
|
self.eventCallback?.onSynthesisFailed(utteranceId: utteranceId, text: text, error: errorMessage) |
|
|
|
self.eventCallback?.onStateChanged(component: "Synthesis", isActive: false) |
|
|
|
} |
|
|
|
|
|
|
|
@ -774,36 +921,78 @@ import os.log |
|
|
|
|
|
|
|
/// 保留原有的同步合成方法作为备用 |
|
|
|
private func synthesizeText(_ text: String) -> Data? { |
|
|
|
// 现在调用流式合成方法 |
|
|
|
let success = synthesizeTextStreaming(text) |
|
|
|
return success ? Data() : nil // 返回空数据表示已启动,实际数据通过回调返回 |
|
|
|
let success = synthesizeTextStreaming("", text) |
|
|
|
return success ? Data() : nil |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 构建SSML |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 清理TTS文本 |
|
|
|
*/ |
|
|
|
private func buildSSML(text: String) -> String { |
|
|
|
let voice = getVoiceForLanguage(serviceConfig.targetLanguage) |
|
|
|
private func cleanTextForTTS(_ text: String) -> String { |
|
|
|
var cleaned = text |
|
|
|
|
|
|
|
// 移除URL |
|
|
|
if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 移除emoji |
|
|
|
if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 移除标点符号 |
|
|
|
if let regex = try? NSRegularExpression(pattern: "[;:#;:*\\n]+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 合并空格 |
|
|
|
if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ") |
|
|
|
} |
|
|
|
|
|
|
|
return cleaned.trimmingCharacters(in: .whitespacesAndNewlines) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 生成优化的SSML |
|
|
|
*/ |
|
|
|
private func generateOptimizedSsml(_ rawText: String) -> String { |
|
|
|
// 转义XML保留字符 |
|
|
|
let escapedText = rawText |
|
|
|
.replacingOccurrences(of: "&", with: "&") |
|
|
|
.replacingOccurrences(of: "<", with: "<") |
|
|
|
.replacingOccurrences(of: ">", with: ">") |
|
|
|
print("serviceConfig.targetLanguage: \(serviceConfig.targetLanguage)") |
|
|
|
let voice = getVoiceForLanguage(serviceConfig.targetLanguage) |
|
|
|
// 简化SSML结构 |
|
|
|
return """ |
|
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(serviceConfig.targetLanguage)"> |
|
|
|
<voice name="\(voice)"> |
|
|
|
<prosody rate="\(serviceConfig.speechRate)" pitch="\(serviceConfig.speechPitch)" volume="\(serviceConfig.speechVolume)"> |
|
|
|
\(text) |
|
|
|
</prosody> |
|
|
|
</voice> |
|
|
|
<voice name="\(voice)"> |
|
|
|
<prosody rate="\(serviceConfig.speechRate)" pitch="\(serviceConfig.speechPitch)" volume="\(serviceConfig.speechVolume)"> |
|
|
|
\(escapedText) |
|
|
|
</prosody> |
|
|
|
</voice> |
|
|
|
</speak> |
|
|
|
""" |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 根据语言获取对应的语音 |
|
|
|
*/ |
|
|
|
private func getVoiceForLanguage(_ language: String) -> String { |
|
|
|
// 优先使用 Dart 传入的 currentVoice(已包含性别信息) |
|
|
|
if !serviceConfig.currentVoice.isEmpty { |
|
|
|
return serviceConfig.currentVoice |
|
|
|
} |
|
|
|
|
|
|
|
if let cachedVoice = voiceCache[language] { |
|
|
|
return cachedVoice |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
let voice: String |
|
|
|
switch language { |
|
|
|
// 中文相关 |
|
|
|
|