|
|
|
@ -1,28 +1,11 @@ |
|
|
|
import Foundation |
|
|
|
import AVFoundation |
|
|
|
import MicrosoftCognitiveServicesSpeech |
|
|
|
import os.log |
|
|
|
// 自定义语音处理组件,提供音频流处理等功能 |
|
|
|
import speech |
|
|
|
import os.log |
|
|
|
|
|
|
|
/** |
|
|
|
* Azure TTS Helper |
|
|
|
* |
|
|
|
* 基于微软Azure语音服务的TTS实现 |
|
|
|
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis |
|
|
|
* |
|
|
|
* 特性: |
|
|
|
* - 对外接口保持同步,内部异步处理 |
|
|
|
* - 使用专用队列保证合成顺序 |
|
|
|
* - 避免阻塞主线程和其他后台线程 |
|
|
|
* - 事件回调由调用方处理线程切换 |
|
|
|
* |
|
|
|
* 使用示例: |
|
|
|
* let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理 |
|
|
|
*/ |
|
|
|
public class AzureTtsHelper: NSObject, ITtsService { |
|
|
|
public class AzureTtsHelper: NSObject, ITtsService { |
|
|
|
private let tag = "AzureTtsHelper" |
|
|
|
// 日志对象 |
|
|
|
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") |
|
|
|
|
|
|
|
private static let DEFAULT_LANGUAGE = "zh-CN" |
|
|
|
@ -43,59 +26,51 @@ import os.log |
|
|
|
|
|
|
|
// 事件监听器列表 |
|
|
|
private var eventListeners = NSHashTable<AnyObject>.weakObjects() |
|
|
|
private var audioDataListeners = NSHashTable<AnyObject>.weakObjects() |
|
|
|
|
|
|
|
// 流式文本处理的缓冲区 |
|
|
|
private var streamBuffer = "" |
|
|
|
private var lastSpeakTime: TimeInterval = 0 |
|
|
|
|
|
|
|
// 内部异步处理队列,保证顺序执行 |
|
|
|
// 内部异步处理队列 |
|
|
|
private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) |
|
|
|
private var isProcessing = false // 添加处理状态标志 |
|
|
|
private let synthesisGroup = DispatchGroup() |
|
|
|
private var pendingTasks: [() -> Void] = [] |
|
|
|
private let taskLock = NSLock() |
|
|
|
|
|
|
|
// 自定义音频输出流 |
|
|
|
private var customAudioOutputStream: SPXPushAudioOutputStream? |
|
|
|
private var useInternalPlayer = true |
|
|
|
// 记录最后播放的文本 |
|
|
|
private var lastSpokenText: String? |
|
|
|
/** |
|
|
|
* 初始化语音合成服务 |
|
|
|
* |
|
|
|
* @param appId 服务应用ID |
|
|
|
* @param token 服务访问令牌/订阅密钥 |
|
|
|
* @param resource 服务资源ID/区域(可选) |
|
|
|
* @param language 语言代码,如"zh-CN" |
|
|
|
* @return 是否初始化成功 |
|
|
|
*/ |
|
|
|
public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { |
|
|
|
do { |
|
|
|
// 创建语音配置 |
|
|
|
if ttsResource.isEmpty { |
|
|
|
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia") |
|
|
|
} else { |
|
|
|
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) |
|
|
|
} |
|
|
|
|
|
|
|
// 设置语言 |
|
|
|
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) |
|
|
|
speechConfig?.speechSynthesisLanguage = language |
|
|
|
currentLanguage = language |
|
|
|
|
|
|
|
// 设置音频输出格式 - 使用16k、16位的PCM格式 |
|
|
|
speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", |
|
|
|
byName: "SpeechServiceConnection_SynthOutputFormat") |
|
|
|
// 设置音频输出格式 - 与Android一致 |
|
|
|
speechConfig?.setPropertyTo("riff-16khz-16bit-mono-pcm", |
|
|
|
byName: "SpeechServiceConnection_SynthOutputFormat") |
|
|
|
|
|
|
|
// 创建合成器,使用默认音频输出(扬声器) |
|
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
|
|
// 设置低延迟属性 |
|
|
|
speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_InitialSilenceTimeoutMs") |
|
|
|
speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_EndSilenceTimeoutMs") |
|
|
|
|
|
|
|
// 设置事件监听 |
|
|
|
setupEventListeners() |
|
|
|
|
|
|
|
// 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny |
|
|
|
if language.lowercased().starts(with: "zh") { |
|
|
|
_ = setVoice("zh-CN-XiaoxiaoNeural") |
|
|
|
} else { |
|
|
|
_ = setVoice("en-US-JennyNeural") |
|
|
|
} |
|
|
|
// 创建合成器 |
|
|
|
recreateSynthesizer() |
|
|
|
|
|
|
|
// 设置初始化完成 |
|
|
|
isInitialized = true |
|
|
|
|
|
|
|
// 预热TTS引擎 |
|
|
|
warmupSynthesizer() |
|
|
|
|
|
|
|
return true |
|
|
|
} catch { |
|
|
|
os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
@ -103,24 +78,25 @@ import os.log |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置自定义音频输出流 |
|
|
|
*/ |
|
|
|
public func setCustomAudioOutputStream(_ outputStream: SPXPushAudioOutputStream?) { |
|
|
|
customAudioOutputStream = outputStream |
|
|
|
if isInitialized { |
|
|
|
recreateSynthesizer() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置语音角色 |
|
|
|
* |
|
|
|
* @param voiceName 语音角色名称(不同服务的语音角色命名可能不同) |
|
|
|
* @return 是否设置成功 |
|
|
|
*/ |
|
|
|
public func setVoice(_ voiceName: String) -> Bool { |
|
|
|
if !isInitialized { return false } |
|
|
|
|
|
|
|
do { |
|
|
|
currentVoice = voiceName |
|
|
|
|
|
|
|
// 更新语音名称 |
|
|
|
if let config = speechConfig { |
|
|
|
config.speechSynthesisVoiceName = voiceName |
|
|
|
} |
|
|
|
|
|
|
|
// 重新创建合成器 |
|
|
|
speechConfig?.speechSynthesisVoiceName = voiceName |
|
|
|
recreateSynthesizer() |
|
|
|
return true |
|
|
|
} catch { |
|
|
|
@ -135,173 +111,88 @@ import os.log |
|
|
|
|
|
|
|
/** |
|
|
|
* 单次合成并播放 |
|
|
|
* |
|
|
|
* @param text 要合成的文本 |
|
|
|
* @return 是否成功开始合成 |
|
|
|
*/ |
|
|
|
public func speakOnce(_ text: String) -> Bool { |
|
|
|
if !isInitialized { |
|
|
|
os_log("语音合成未初始化", log: log, type: .error) |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "NOT_INITIALIZED", |
|
|
|
"errorMessage": "TTS引擎未初始化" |
|
|
|
]) |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
// 立即返回成功,内部异步处理 |
|
|
|
enqueueSynthesisTask { |
|
|
|
self.performSynthesis(text: text) |
|
|
|
// 清理文本 |
|
|
|
let cleanedText = cleanTextForTTS(text) |
|
|
|
if cleanedText.isEmpty { |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 将合成任务加入队列,保证顺序执行 |
|
|
|
*/ |
|
|
|
private func enqueueSynthesisTask(_ task: @escaping () -> Void) { |
|
|
|
taskLock.lock() |
|
|
|
defer { taskLock.unlock() } |
|
|
|
|
|
|
|
pendingTasks.append(task) |
|
|
|
|
|
|
|
// 如果当前没有任务在执行,开始处理队列 |
|
|
|
if pendingTasks.count == 1 { |
|
|
|
processNextTask() |
|
|
|
// 异步处理 |
|
|
|
enqueueSynthesisTask { |
|
|
|
self.performSynthesis(text: cleanedText) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 处理队列中的下一个任务 |
|
|
|
*/ |
|
|
|
private func processNextTask() { |
|
|
|
synthesisQueue.async { |
|
|
|
self.synthesisGroup.enter() |
|
|
|
|
|
|
|
self.taskLock.lock() |
|
|
|
guard !self.pendingTasks.isEmpty else { |
|
|
|
self.taskLock.unlock() |
|
|
|
self.synthesisGroup.leave() |
|
|
|
return |
|
|
|
} |
|
|
|
let task = self.pendingTasks.removeFirst() |
|
|
|
self.taskLock.unlock() |
|
|
|
|
|
|
|
// 执行任务 |
|
|
|
task() |
|
|
|
|
|
|
|
self.synthesisGroup.leave() |
|
|
|
|
|
|
|
// 处理下一个任务 |
|
|
|
self.taskLock.lock() |
|
|
|
if !self.pendingTasks.isEmpty { |
|
|
|
self.taskLock.unlock() |
|
|
|
self.processNextTask() |
|
|
|
} else { |
|
|
|
self.taskLock.unlock() |
|
|
|
} |
|
|
|
} |
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 实际执行合成的方法 |
|
|
|
* 流式合成文本 |
|
|
|
*/ |
|
|
|
private func performSynthesis(text: String) { |
|
|
|
// 重置状态 |
|
|
|
speaking = true |
|
|
|
|
|
|
|
// 生成SSML |
|
|
|
let ssml = generateSsml(text) |
|
|
|
|
|
|
|
do { |
|
|
|
os_log("开始合成: %{public}@", log: log, type: .debug, ssml) |
|
|
|
// 使用同步方法,在后台队列中执行 |
|
|
|
let result = try synthesizer?.startSpeakingSsml(ssml) |
|
|
|
|
|
|
|
// 检查结果 |
|
|
|
if let result = result { |
|
|
|
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) |
|
|
|
} |
|
|
|
} catch { |
|
|
|
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
public func speakStream(_ text: String) -> Bool { |
|
|
|
if !isInitialized { |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "SYNTHESIS_FAILED", |
|
|
|
"errorMessage": error.localizedDescription |
|
|
|
"errorCode": "NOT_INITIALIZED", |
|
|
|
"errorMessage": "TTS引擎未初始化" |
|
|
|
]) |
|
|
|
speaking = false |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 流式合成文本 |
|
|
|
* |
|
|
|
* @param text 要合成的文本片段 |
|
|
|
* @return 是否成功处理 |
|
|
|
*/ |
|
|
|
public func speakStream(_ text: String) -> Bool { |
|
|
|
if !isInitialized || text.isEmpty { |
|
|
|
if !isInitialized { |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "NOT_INITIALIZED", |
|
|
|
"errorMessage": "TTS引擎未初始化" |
|
|
|
]) |
|
|
|
} |
|
|
|
return false |
|
|
|
if text.isEmpty { |
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
do { |
|
|
|
// 添加新文本到缓冲区 |
|
|
|
streamBuffer.append(text) |
|
|
|
// 清理文本并添加到缓冲区 |
|
|
|
let cleanedText = cleanTextForTTS(text) |
|
|
|
streamBuffer.append(cleanedText) |
|
|
|
|
|
|
|
// 增加500ms防抖逻辑 |
|
|
|
let currentTime = Date().timeIntervalSince1970 |
|
|
|
if currentTime - lastSpeakTime < 0.6 { |
|
|
|
return true |
|
|
|
} |
|
|
|
lastSpeakTime = currentTime |
|
|
|
// 防抖逻辑 (150ms) |
|
|
|
let currentTime = Date().timeIntervalSince1970 |
|
|
|
if currentTime - lastSpeakTime < 0.15 { |
|
|
|
return true |
|
|
|
} |
|
|
|
lastSpeakTime = currentTime |
|
|
|
|
|
|
|
let currentText = streamBuffer |
|
|
|
let currentText = streamBuffer |
|
|
|
|
|
|
|
// 定义标点符号列表 |
|
|
|
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] |
|
|
|
// 标点符号列表 (与Android一致) |
|
|
|
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] |
|
|
|
|
|
|
|
// 查找最后一个标点符号的位置 |
|
|
|
var lastPunctuationIndex = -1 |
|
|
|
for (i, char) in currentText.enumerated().reversed() { |
|
|
|
if punctuationMarks.contains(char) { |
|
|
|
lastPunctuationIndex = i |
|
|
|
break |
|
|
|
} |
|
|
|
// 从后往前查找最后一个标点符号 |
|
|
|
var lastPunctuationIndex = -1 |
|
|
|
for (i, char) in currentText.enumerated().reversed() { |
|
|
|
if punctuationMarks.contains(char) { |
|
|
|
lastPunctuationIndex = i |
|
|
|
break |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 如果找到标点符号,则播放到该标点符号 |
|
|
|
if lastPunctuationIndex >= 0 { |
|
|
|
// 提取要播放的文本(包含标点符号) |
|
|
|
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines) |
|
|
|
|
|
|
|
// 剩余的文本保存在缓冲区中 |
|
|
|
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) |
|
|
|
streamBuffer = String(currentText[startIndex...]) |
|
|
|
// 播放到找到的标点符号 |
|
|
|
if lastPunctuationIndex >= 0 { |
|
|
|
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)) |
|
|
|
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) |
|
|
|
streamBuffer = String(currentText[startIndex...]) |
|
|
|
|
|
|
|
// 只有非空文本才播放 |
|
|
|
if !textToSpeak.isEmpty { |
|
|
|
if !textToSpeak.isEmpty { |
|
|
|
return speakOnce(textToSpeak) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 如果没有找到标点符号,则等待更多文本 |
|
|
|
return true |
|
|
|
} catch { |
|
|
|
os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "STREAM_FAILED", |
|
|
|
"errorMessage": "流式语音合成失败: \(error.localizedDescription)" |
|
|
|
]) |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 刷新并播放流式文本缓冲区中的剩余内容 |
|
|
|
* |
|
|
|
* @return 是否成功处理 |
|
|
|
* 刷新并播放流式文本 |
|
|
|
*/ |
|
|
|
public func flushStream() -> Bool { |
|
|
|
if !isInitialized { |
|
|
|
@ -312,115 +203,71 @@ import os.log |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
do { |
|
|
|
// 获取缓冲区中剩余的文本 |
|
|
|
let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines) |
|
|
|
|
|
|
|
// 清空缓冲区 |
|
|
|
streamBuffer = "" |
|
|
|
|
|
|
|
// 如果缓冲区为空,直接返回成功 |
|
|
|
if remainingText.isEmpty { |
|
|
|
return true |
|
|
|
} |
|
|
|
let remainingText = streamBuffer |
|
|
|
streamBuffer = "" |
|
|
|
|
|
|
|
// 播放剩余文本 |
|
|
|
return speakOnce(remainingText) |
|
|
|
} catch { |
|
|
|
os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "FLUSH_FAILED", |
|
|
|
"errorMessage": "刷新流式文本失败: \(error.localizedDescription)" |
|
|
|
]) |
|
|
|
return false |
|
|
|
if remainingText.isEmpty { |
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
return speakOnce(remainingText) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 停止语音合成和播放 |
|
|
|
* |
|
|
|
* @return 是否成功停止 |
|
|
|
* 停止语音合成 |
|
|
|
*/ |
|
|
|
public func stop() -> Bool { |
|
|
|
// 立即更新状态 |
|
|
|
speaking = false |
|
|
|
|
|
|
|
// 清除流式缓冲区中的待播放内容 |
|
|
|
streamBuffer = "" |
|
|
|
|
|
|
|
// 清空待处理的任务队列 |
|
|
|
taskLock.lock() |
|
|
|
pendingTasks.removeAll() |
|
|
|
taskLock.unlock() |
|
|
|
|
|
|
|
// 在后台队列停止合成器,避免阻塞主线程 |
|
|
|
synthesisQueue.async { |
|
|
|
if let synthesizer = self.synthesizer { |
|
|
|
do { |
|
|
|
try synthesizer.stopSpeaking() |
|
|
|
os_log("语音合成已停止", log: self.log, type: .info) |
|
|
|
|
|
|
|
// 直接通知停止完成 |
|
|
|
self.notifyEvent(eventType: .synthesisCanceled) |
|
|
|
} catch { |
|
|
|
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) |
|
|
|
self.notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "STOP_FAILED", |
|
|
|
"errorMessage": error.localizedDescription |
|
|
|
]) |
|
|
|
} |
|
|
|
do { |
|
|
|
try self.synthesizer?.stopSpeaking() |
|
|
|
self.notifyEvent(eventType: .synthesisCanceled) |
|
|
|
} catch { |
|
|
|
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) |
|
|
|
self.notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "STOP_FAILED", |
|
|
|
"errorMessage": error.localizedDescription |
|
|
|
]) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 释放资源 |
|
|
|
* 在不再需要服务时调用,释放底层资源 |
|
|
|
*/ |
|
|
|
public func dispose() { |
|
|
|
// 停止播放 |
|
|
|
_ = stop() |
|
|
|
|
|
|
|
// 等待所有任务完成 |
|
|
|
synthesisGroup.wait() |
|
|
|
|
|
|
|
// 清理资源 |
|
|
|
synthesisQueue.async { |
|
|
|
// 释放合成器 |
|
|
|
self.synthesizer = nil |
|
|
|
|
|
|
|
// 释放配置 |
|
|
|
self.speechConfig = nil |
|
|
|
|
|
|
|
// 清空流缓冲区 |
|
|
|
self.streamBuffer = "" |
|
|
|
|
|
|
|
// 重置状态 |
|
|
|
self.isInitialized = false |
|
|
|
self.speaking = false |
|
|
|
synthesisQueue.sync { |
|
|
|
synthesizer = nil |
|
|
|
speechConfig = nil |
|
|
|
customAudioOutputStream = nil |
|
|
|
isInitialized = false |
|
|
|
speaking = false |
|
|
|
|
|
|
|
// 清空任务队列 |
|
|
|
self.taskLock.lock() |
|
|
|
self.pendingTasks.removeAll() |
|
|
|
self.taskLock.unlock() |
|
|
|
taskLock.lock() |
|
|
|
pendingTasks.removeAll() |
|
|
|
taskLock.unlock() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 添加TTS事件监听器 |
|
|
|
* |
|
|
|
* @param listener 事件监听器 |
|
|
|
* 添加事件监听器 |
|
|
|
*/ |
|
|
|
public func addListener(_ listener: TtsEventListener) { |
|
|
|
eventListeners.add(listener as AnyObject) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 移除TTS事件监听器 |
|
|
|
* |
|
|
|
* @param listener 要移除的事件监听器 |
|
|
|
* 移除事件监听器 |
|
|
|
*/ |
|
|
|
public func removeListener(_ listener: TtsEventListener) { |
|
|
|
eventListeners.remove(listener as AnyObject) |
|
|
|
@ -428,100 +275,179 @@ import os.log |
|
|
|
|
|
|
|
/** |
|
|
|
* 添加音频数据监听器 |
|
|
|
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|
|
|
* |
|
|
|
* @param listener 音频数据监听器 |
|
|
|
*/ |
|
|
|
public func addAudioDataListener(_ listener: AudioDataListener) { |
|
|
|
os_log("警告:不支持音频数据监听器功能", log: log, type: .info) |
|
|
|
audioDataListeners.add(listener as AnyObject) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 移除音频数据监听器 |
|
|
|
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|
|
|
* |
|
|
|
* @param listener 要移除的音频数据监听器 |
|
|
|
*/ |
|
|
|
public func removeAudioDataListener(_ listener: AudioDataListener) { |
|
|
|
// 不做任何操作 |
|
|
|
audioDataListeners.remove(listener as AnyObject) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 当前是否正在播放/合成 |
|
|
|
* 设置是否使用内部播放器 |
|
|
|
*/ |
|
|
|
public func isSpeaking() -> Bool { |
|
|
|
return speaking |
|
|
|
public func setUseInternalPlayer(_ useInternalPlayer: Bool) { |
|
|
|
self.useInternalPlayer = useInternalPlayer |
|
|
|
if isInitialized { |
|
|
|
recreateSynthesizer() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// MARK: - 辅助方法 |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置事件监听器 |
|
|
|
* 设置音频输出设备 |
|
|
|
*/ |
|
|
|
private func setupEventListeners() { |
|
|
|
guard let synthesizer = synthesizer else { return } |
|
|
|
public func setAudioOutputDevice() -> Bool { |
|
|
|
// 先暂停当前播放 |
|
|
|
let wasSpeaking = speaking |
|
|
|
if wasSpeaking { |
|
|
|
_ = stop() |
|
|
|
} |
|
|
|
|
|
|
|
// 添加合成开始事件处理器 |
|
|
|
synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in |
|
|
|
guard let self = self else { return } |
|
|
|
self.notifyEvent(eventType: .synthesisStarted) |
|
|
|
// 重新创建合成器以应用新设备 |
|
|
|
if isInitialized { |
|
|
|
recreateSynthesizer() |
|
|
|
} |
|
|
|
|
|
|
|
// 添加合成中事件处理器 |
|
|
|
synthesizer.addSynthesizingEventHandler { [weak self] _, _ in |
|
|
|
// 可以在这里处理合成中的事件,目前没有特别操作 |
|
|
|
// 如果之前正在播放,恢复播放 |
|
|
|
if wasSpeaking, let lastText = lastSpokenText { |
|
|
|
speakOnce(lastText) |
|
|
|
} |
|
|
|
|
|
|
|
// 添加合成完成事件处理器 |
|
|
|
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in |
|
|
|
guard let self = self else { return } |
|
|
|
self.speaking = false |
|
|
|
self.notifyEvent(eventType: .synthesisCompleted) |
|
|
|
return true |
|
|
|
} |
|
|
|
// /** |
|
|
|
// * 设置音频输出设备 |
|
|
|
// */ |
|
|
|
// public func setAudioOutputDevice(_ device: AudioOutputDevice) -> Bool { |
|
|
|
// do { |
|
|
|
// let session = AVAudioSession.sharedInstance() |
|
|
|
// try session.setCategory(.playAndRecord, options: [.defaultToSpeaker, .allowBluetooth]) |
|
|
|
// |
|
|
|
// switch device { |
|
|
|
// case .default: |
|
|
|
// try session.overrideOutputAudioPort(.none) |
|
|
|
// case .speaker: |
|
|
|
// try session.overrideOutputAudioPort(.speaker) |
|
|
|
// case .headphones: |
|
|
|
// try session.overrideOutputAudioPort(.none) |
|
|
|
// } |
|
|
|
// |
|
|
|
// try session.setActive(true) |
|
|
|
// return true |
|
|
|
// } catch { |
|
|
|
// os_log("设置音频输出设备失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
// return false |
|
|
|
// } |
|
|
|
// } |
|
|
|
// |
|
|
|
// MARK: - 私有方法 |
|
|
|
|
|
|
|
/** |
|
|
|
* 加入合成任务队列 |
|
|
|
*/ |
|
|
|
private func enqueueSynthesisTask(_ task: @escaping () -> Void) { |
|
|
|
taskLock.lock() |
|
|
|
pendingTasks.append(task) |
|
|
|
taskLock.unlock() |
|
|
|
|
|
|
|
// 确保任务被处理 |
|
|
|
processTasksIfNeeded() |
|
|
|
} |
|
|
|
/** |
|
|
|
* 处理任务队列 |
|
|
|
*/ |
|
|
|
private func processTasksIfNeeded() { |
|
|
|
taskLock.lock() |
|
|
|
// 如果已经在处理中或没有任务,则直接返回 |
|
|
|
guard !isProcessing && !pendingTasks.isEmpty else { |
|
|
|
taskLock.unlock() |
|
|
|
return |
|
|
|
} |
|
|
|
|
|
|
|
// 添加合成取消事件处理器 |
|
|
|
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in |
|
|
|
guard let self = self else { return } |
|
|
|
isProcessing = true |
|
|
|
let task = pendingTasks.removeFirst() |
|
|
|
taskLock.unlock() |
|
|
|
|
|
|
|
self.speaking = false |
|
|
|
synthesisQueue.async { |
|
|
|
task() |
|
|
|
|
|
|
|
var params: [String: Any] = [:] |
|
|
|
do { |
|
|
|
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) |
|
|
|
if cancellationDetails.reason == SPXCancellationReason.error { |
|
|
|
params["reason"] = String(describing: cancellationDetails.reason.rawValue) |
|
|
|
params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" |
|
|
|
} |
|
|
|
} catch { |
|
|
|
params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)" |
|
|
|
// 任务完成后,检查是否有更多任务 |
|
|
|
self.taskLock.lock() |
|
|
|
self.isProcessing = false |
|
|
|
|
|
|
|
// 如果还有任务,递归处理 |
|
|
|
if !self.pendingTasks.isEmpty { |
|
|
|
self.taskLock.unlock() |
|
|
|
self.processTasksIfNeeded() |
|
|
|
} else { |
|
|
|
self.taskLock.unlock() |
|
|
|
} |
|
|
|
os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误")) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
self.notifyEvent(eventType: .synthesisCanceled, params: params) |
|
|
|
/** |
|
|
|
* 处理下一个任务 |
|
|
|
*/ |
|
|
|
private func processNextTask() { |
|
|
|
synthesisQueue.async { |
|
|
|
self.taskLock.lock() |
|
|
|
guard !self.pendingTasks.isEmpty else { |
|
|
|
self.taskLock.unlock() |
|
|
|
return |
|
|
|
} |
|
|
|
let task = self.pendingTasks.removeFirst() |
|
|
|
self.taskLock.unlock() |
|
|
|
|
|
|
|
task() |
|
|
|
self.processNextTask() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 触发事件通知 |
|
|
|
* 执行语音合成 |
|
|
|
*/ |
|
|
|
private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { |
|
|
|
let event = TtsEvent(type: eventType, params: params) |
|
|
|
private func performSynthesis(text: String) { |
|
|
|
speaking = true |
|
|
|
lastSpokenText = text |
|
|
|
// 生成SSML |
|
|
|
let ssml = generateOptimizedSsml(text) |
|
|
|
|
|
|
|
// 直接通知事件,由调用方处理线程切换 |
|
|
|
for case let listener as TtsEventListener in self.eventListeners.allObjects { |
|
|
|
listener.onEvent(event) |
|
|
|
do { |
|
|
|
os_log("开始合成: %{public}@", log: log, type: .debug, ssml) |
|
|
|
let result = try synthesizer?.startSpeakingSsml(ssml) |
|
|
|
|
|
|
|
if let result = result { |
|
|
|
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) |
|
|
|
} |
|
|
|
} catch { |
|
|
|
speaking = false |
|
|
|
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "SYNTHESIS_FAILED", |
|
|
|
"errorMessage": error.localizedDescription |
|
|
|
]) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 重新创建合成器 |
|
|
|
*/ |
|
|
|
private func recreateSynthesizer() { |
|
|
|
do { |
|
|
|
// 使用默认音频输出配置创建合成器 |
|
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
|
|
// 创建音频配置 |
|
|
|
let audioConfig: SPXAudioConfiguration? |
|
|
|
if !useInternalPlayer || customAudioOutputStream != nil { |
|
|
|
audioConfig = try SPXAudioConfiguration(streamOutput: customAudioOutputStream ?? SPXPushAudioOutputStream()) |
|
|
|
} else { |
|
|
|
audioConfig = nil // 使用默认扬声器 |
|
|
|
} |
|
|
|
|
|
|
|
// 设置事件监听 |
|
|
|
// 创建合成器 |
|
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
|
|
setupEventListeners() |
|
|
|
} catch { |
|
|
|
os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
@ -533,72 +459,139 @@ import os.log |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置语音参数 |
|
|
|
* 设置事件监听器 |
|
|
|
*/ |
|
|
|
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|
|
|
if !isInitialized { return false } |
|
|
|
private func setupEventListeners() { |
|
|
|
synthesizer?.addSynthesisStartedEventHandler { [weak self] _, _ in |
|
|
|
self?.notifyEvent(eventType: .synthesisStarted) |
|
|
|
self?.notifyEvent(eventType: .playbackStarted) |
|
|
|
} |
|
|
|
|
|
|
|
do { |
|
|
|
currentRate = formatPercentage(rate) |
|
|
|
currentPitch = formatPercentage(pitch) |
|
|
|
currentVolume = "\(min(max(volume, 0), 100))%" |
|
|
|
return true |
|
|
|
} catch { |
|
|
|
os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
notifyEvent(eventType: .error, params: [ |
|
|
|
"errorCode": "PARAMS_SET_FAILED", |
|
|
|
"errorMessage": "设置语音参数失败: \(error.localizedDescription)" |
|
|
|
]) |
|
|
|
return false |
|
|
|
synthesizer?.addSynthesizingEventHandler { [weak self] _, event in |
|
|
|
if let audioData = event.result.audioData { |
|
|
|
self?.notifyAudioData(audioData) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
synthesizer?.addSynthesisCompletedEventHandler { [weak self] _, _ in |
|
|
|
guard let self = self else { return } |
|
|
|
self.speaking = false |
|
|
|
self.notifyEvent(eventType: .synthesisCompleted) |
|
|
|
self.notifyEvent(eventType: .playbackCompleted) |
|
|
|
} |
|
|
|
|
|
|
|
synthesizer?.addSynthesisCanceledEventHandler { [weak self] _, event in |
|
|
|
guard let self = self else { return } |
|
|
|
self.speaking = false |
|
|
|
|
|
|
|
var params: [String: Any] = [:] |
|
|
|
if let details = try? SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: event.result) { |
|
|
|
params["errorCode"] = details.errorCode |
|
|
|
params["errorDetails"] = details.errorDetails |
|
|
|
} |
|
|
|
|
|
|
|
self.notifyEvent(eventType: .synthesisCanceled, params: params) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 通知事件 |
|
|
|
*/ |
|
|
|
private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { |
|
|
|
let event = TtsEvent(type: eventType, params: params) |
|
|
|
for case let listener as TtsEventListener in eventListeners.allObjects { |
|
|
|
listener.onEvent(event) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 通知音频数据 |
|
|
|
*/ |
|
|
|
private func notifyAudioData(_ data: Data) { |
|
|
|
for case let listener as AudioDataListener in audioDataListeners.allObjects { |
|
|
|
listener.onAudioData(data) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 预热TTS引擎 |
|
|
|
*/ |
|
|
|
private func warmupSynthesizer() { |
|
|
|
let warmupSsml = """ |
|
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(currentLanguage)"> |
|
|
|
<voice name="\(currentVoice)"> |
|
|
|
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="0%"> |
|
|
|
. |
|
|
|
</prosody> |
|
|
|
</voice> |
|
|
|
</speak> |
|
|
|
""" |
|
|
|
|
|
|
|
synthesisQueue.async { |
|
|
|
_ = try? self.synthesizer?.startSpeakingSsml(warmupSsml) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 格式化百分比值 |
|
|
|
* 清理TTS文本 |
|
|
|
*/ |
|
|
|
private func formatPercentage(_ value: Int) -> String { |
|
|
|
return value >= 0 ? "+\(value)%" : "\(value)%" |
|
|
|
private func cleanTextForTTS(_ text: String) -> String { |
|
|
|
var cleaned = text |
|
|
|
|
|
|
|
// 移除URL |
|
|
|
if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 移除emoji |
|
|
|
if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 合并空格 |
|
|
|
if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) { |
|
|
|
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ") |
|
|
|
} |
|
|
|
|
|
|
|
return cleaned.trimmingCharacters(in: .whitespacesAndNewlines) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 生成SSML |
|
|
|
* 生成优化的SSML |
|
|
|
*/ |
|
|
|
private func generateSsml(_ rawText: String) -> String { |
|
|
|
// 1. 定义要静音的符号和表情符号列表 |
|
|
|
let symbolsToMute = [ |
|
|
|
"#", "*", |
|
|
|
"😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" |
|
|
|
] |
|
|
|
|
|
|
|
// 2. 转义 XML 保留字符 |
|
|
|
var escapedText = rawText |
|
|
|
private func generateOptimizedSsml(_ rawText: String) -> String { |
|
|
|
// 转义XML保留字符 |
|
|
|
let escapedText = rawText |
|
|
|
.replacingOccurrences(of: "&", with: "&") |
|
|
|
.replacingOccurrences(of: "<", with: "<") |
|
|
|
.replacingOccurrences(of: ">", with: ">") |
|
|
|
|
|
|
|
// 3. 静音处理特殊符号和表情符号 |
|
|
|
// 使用空白替换法,直接将符号替换为空字符串 |
|
|
|
var processedText = escapedText |
|
|
|
for symbol in symbolsToMute { |
|
|
|
processedText = processedText.replacingOccurrences(of: symbol, with: "") |
|
|
|
} |
|
|
|
|
|
|
|
// 4. 构造简化的SSML文档,减少嵌套层级 |
|
|
|
let ssml = """ |
|
|
|
<speak version="1.0" |
|
|
|
xmlns="http://www.w3.org/2001/10/synthesis" |
|
|
|
xmlns:mstts="https://www.w3.org/2001/mstts" |
|
|
|
xml:lang="zh-CN"> |
|
|
|
// 简化SSML结构 |
|
|
|
return """ |
|
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(currentLanguage)"> |
|
|
|
<voice name="\(currentVoice)"> |
|
|
|
<mstts:express-as style="cheerful"> |
|
|
|
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)"> |
|
|
|
<say-as interpret-as="text">\(processedText)</say-as> |
|
|
|
</prosody> |
|
|
|
</mstts:express-as> |
|
|
|
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)"> |
|
|
|
\(escapedText) |
|
|
|
</prosody> |
|
|
|
</voice> |
|
|
|
</speak> |
|
|
|
""" |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置语音参数 |
|
|
|
*/ |
|
|
|
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|
|
|
currentRate = rate >= 0 ? "+\(rate)%" : "\(rate)%" |
|
|
|
currentPitch = pitch >= 0 ? "+\(pitch)%" : "\(pitch)%" |
|
|
|
currentVolume = "\(min(max(volume, 0), 100))%" |
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
return ssml |
|
|
|
/** |
|
|
|
* 是否正在播放 |
|
|
|
*/ |
|
|
|
public func isSpeaking() -> Bool { |
|
|
|
return speaking |
|
|
|
} |
|
|
|
} |
|
|
|
|