You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
446 lines
14 KiB
446 lines
14 KiB
import Foundation
|
|
import AVFoundation
|
|
import MicrosoftCognitiveServicesSpeech
|
|
|
|
/// Azure 语音合成辅助类
|
|
class AzureTtsHelper: NSObject {
|
|
// 常量定义
|
|
private let tag = "AzureTtsHelper"
|
|
private let DEFAULT_LANGUAGE = "zh-CN"
|
|
private let DEFAULT_VOICE = "zh-CN-XiaoxiaoNeural"
|
|
|
|
// 核心组件
|
|
private var synthesizer: SPXSpeechSynthesizer?
|
|
private var speechConfig: SPXSpeechConfiguration?
|
|
private var audioConfig: SPXAudioConfiguration?
|
|
private var initialized = false
|
|
private var speaking = false
|
|
|
|
// 当前设置
|
|
private var currentVoice = "zh-CN-XiaoxiaoNeural"
|
|
private var currentRate = "0%"
|
|
private var currentPitch = "0%"
|
|
private var currentVolume = "100%"
|
|
|
|
// 音频会话管理
|
|
private let audioSession = AVAudioSession.sharedInstance()
|
|
|
|
// 自定义音频输出流
|
|
private var customAudioOutputStream: SPXPushAudioOutputStream?
|
|
|
|
// 流式文本处理的缓冲区
|
|
private var streamBuffer = ""
|
|
private var lastSpeakTime: TimeInterval = 0
|
|
|
|
// 应用上下文
|
|
private let context: Any
|
|
|
|
// 全局回调
|
|
private var ttsCallback: TtsCallback?
|
|
|
|
/// TTS回调接口
|
|
protocol TtsCallback {
|
|
/// 合成开始
|
|
func onSynthesisStarted()
|
|
|
|
/// 合成中
|
|
func onSynthesizing()
|
|
|
|
/// 合成完成
|
|
func onSynthesisCompleted()
|
|
|
|
/// 合成取消
|
|
func onSynthesisCanceled()
|
|
}
|
|
|
|
/// 初始化
|
|
/// - Parameter context: 应用上下文
|
|
init(_ context: Any) {
|
|
self.context = context
|
|
super.init()
|
|
}
|
|
|
|
/// 初始化 TTS 引擎
|
|
/// - Parameters:
|
|
/// - subscriptionKey: Azure 语音服务订阅密钥
|
|
/// - region: Azure 语音服务区域
|
|
/// - callback: 可选,全局TTS回调接口
|
|
/// - Returns: 是否初始化成功
|
|
func initialize(
|
|
subscriptionKey: String,
|
|
region: String,
|
|
callback: TtsCallback? = nil
|
|
) -> Bool {
|
|
print("\(tag): 初始化 Azure 语音服务")
|
|
|
|
// 检查配置是否为空
|
|
if subscriptionKey.isEmpty || region.isEmpty {
|
|
print("\(tag): 错误: Azure 配置信息不完整")
|
|
return false
|
|
}
|
|
|
|
// 释放之前的资源
|
|
dispose()
|
|
|
|
// 保存全局回调
|
|
ttsCallback = callback
|
|
|
|
do {
|
|
// 创建语音配置
|
|
speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region)
|
|
|
|
// 设置语音合成输出格式为高质量音频
|
|
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm)
|
|
|
|
// 设置默认语音
|
|
speechConfig?.setSpeechSynthesisVoiceName(currentVoice)
|
|
|
|
// 创建音频配置
|
|
if let customOutput = customAudioOutputStream {
|
|
// 使用自定义音频输出流
|
|
audioConfig = try SPXAudioConfiguration(streamOutput: customOutput)
|
|
} else {
|
|
// 使用默认扬声器
|
|
audioConfig = SPXAudioConfiguration()
|
|
}
|
|
|
|
// 创建语音合成器
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
|
|
|
|
// 设置事件监听
|
|
setupEventListeners()
|
|
|
|
initialized = true
|
|
print("\(tag): TTS 引擎初始化成功")
|
|
return true
|
|
} catch {
|
|
print("\(tag): TTS 引擎初始化失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 设置事件监听器
|
|
private func setupEventListeners() {
|
|
guard let synthesizer = synthesizer else { return }
|
|
|
|
// 合成开始事件
|
|
synthesizer.addSynthesisStarted { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
self.ttsCallback?.onSynthesisStarted()
|
|
}
|
|
|
|
// 合成中事件
|
|
synthesizer.addSynthesizing { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
self.ttsCallback?.onSynthesizing()
|
|
}
|
|
|
|
// 合成完成事件
|
|
synthesizer.addSynthesisCompleted { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
self.speaking = false
|
|
self.ttsCallback?.onSynthesisCompleted()
|
|
}
|
|
|
|
// 合成取消事件
|
|
synthesizer.addSynthesisCanceled { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
print("\(self.tag): 语音合成取消")
|
|
self.speaking = false
|
|
self.ttsCallback?.onSynthesisCanceled()
|
|
}
|
|
}
|
|
|
|
/// 设置TTS回调
|
|
/// - Parameter callback: TTS回调接口
|
|
func setTtsCallback(_ callback: TtsCallback?) {
|
|
ttsCallback = callback
|
|
}
|
|
|
|
/// 设置自定义音频输出流
|
|
/// - Parameter outputStream: 自定义音频输出流,如果为nil则使用默认音频输出
|
|
/// - Returns: 是否设置成功
|
|
func setCustomAudioOutputStream(_ outputStream: SPXPushAudioOutputStream?) -> Bool {
|
|
do {
|
|
// 保存引用
|
|
customAudioOutputStream = outputStream
|
|
|
|
// 如果已初始化,需要重新创建合成器以应用新的音频输出流
|
|
if initialized {
|
|
recreateSynthesizer()
|
|
}
|
|
|
|
return true
|
|
} catch {
|
|
print("\(tag): 设置自定义音频输出流失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 重新创建合成器
|
|
private func recreateSynthesizer() {
|
|
do {
|
|
// 关闭现有合成器
|
|
synthesizer = nil
|
|
|
|
// 创建音频配置
|
|
if let customOutput = customAudioOutputStream {
|
|
// 使用自定义音频输出流
|
|
audioConfig = try SPXAudioConfiguration(streamOutput: customOutput)
|
|
} else {
|
|
// 使用默认扬声器
|
|
audioConfig = SPXAudioConfiguration()
|
|
}
|
|
|
|
// 使用新的音频配置创建合成器
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
|
|
|
|
// 重新设置事件监听
|
|
setupEventListeners()
|
|
} catch {
|
|
print("\(tag): 重新创建合成器失败: \(error.localizedDescription)")
|
|
}
|
|
}
|
|
|
|
/// 设置语音
|
|
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural"
|
|
/// - Returns: 是否设置成功
|
|
func setVoice(voiceName: String) -> Bool {
|
|
if !initialized {
|
|
print("\(tag): TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
if voiceName == currentVoice {
|
|
print("\(tag): 已设置语音: \(voiceName)")
|
|
return true
|
|
}
|
|
|
|
do {
|
|
currentVoice = voiceName
|
|
speechConfig?.setSpeechSynthesisVoiceName(voiceName)
|
|
|
|
// 重新创建合成器
|
|
recreateSynthesizer()
|
|
|
|
print("\(tag): 已设置语音: \(voiceName)")
|
|
return true
|
|
} catch {
|
|
print("\(tag): 设置语音失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 设置语音合成参数
|
|
/// - Parameters:
|
|
/// - rate: 语速,范围 -100 到 100,默认为 0
|
|
/// - pitch: 音调,范围 -100 到 100,默认为 0
|
|
/// - volume: 音量,范围 0 到 100,默认为 100
|
|
/// - Returns: 是否设置成功
|
|
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
|
|
if !initialized {
|
|
print("\(tag): TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
currentRate = formatPercentage(rate)
|
|
currentPitch = formatPercentage(pitch)
|
|
currentVolume = "\(min(max(volume, 0), 100))%"
|
|
|
|
print("\(tag): 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)")
|
|
return true
|
|
} catch {
|
|
print("\(tag): 设置语音参数失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 格式化百分比值
|
|
private func formatPercentage(_ value: Int) -> String {
|
|
return value >= 0 ? "+\(value)%" : "\(value)%"
|
|
}
|
|
|
|
/// 生成 SSML 文本
|
|
/// - Parameter text: 要转换的文本
|
|
/// - Returns: SSML 格式的文本
|
|
private func generateSsml(text: String) -> String {
|
|
return """
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN">
|
|
<voice name="\(currentVoice)">
|
|
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)">
|
|
\(text)
|
|
</prosody>
|
|
</voice>
|
|
</speak>
|
|
"""
|
|
}
|
|
|
|
/// 合成文本为语音并播放
|
|
/// - Parameter text: 要合成的文本
|
|
/// - Returns: 是否成功开始合成
|
|
func speakText(_ text: String) -> Bool {
|
|
if !initialized {
|
|
print("\(tag): TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
print("\(tag): 开始合成文本: \(text)")
|
|
|
|
// 生成 SSML
|
|
let ssml = generateSsml(text: text)
|
|
|
|
// 标记为正在播放
|
|
speaking = true
|
|
|
|
// 异步合成语音
|
|
try synthesizer?.speakSsmlAsync(ssml)
|
|
|
|
return true
|
|
} catch {
|
|
print("\(tag): 语音合成异常: \(error.localizedDescription)")
|
|
speaking = false
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 处理流式文本并在遇到标点符号时播放
|
|
/// - Parameter text: 收到的文本流片段
|
|
/// - Returns: 是否成功处理
|
|
func speakStream(_ text: String) -> Bool {
|
|
if !initialized || text.isEmpty {
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 添加新文本到缓冲区
|
|
streamBuffer.append(text)
|
|
|
|
// 增加500ms防抖逻辑
|
|
let currentTime = Date().timeIntervalSince1970
|
|
if currentTime - lastSpeakTime < 0.5 {
|
|
return true
|
|
}
|
|
lastSpeakTime = currentTime
|
|
|
|
let currentText = streamBuffer
|
|
|
|
// 定义标点符号列表
|
|
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":"]
|
|
|
|
// 查找最后一个标点符号的位置
|
|
var lastPunctuationIndex = -1
|
|
for i in currentText.indices.reversed() {
|
|
if punctuationMarks.contains(currentText[i]) {
|
|
lastPunctuationIndex = currentText.distance(from: currentText.startIndex, to: i)
|
|
break
|
|
}
|
|
}
|
|
|
|
// 如果找到标点符号,则播放到该标点符号
|
|
if lastPunctuationIndex >= 0 {
|
|
// 提取要播放的文本(包含标点符号)
|
|
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1))
|
|
|
|
// 剩余的文本保存在缓冲区中
|
|
streamBuffer = String(currentText.dropFirst(lastPunctuationIndex + 1))
|
|
|
|
// 播放提取的文本
|
|
return speakText(textToSpeak)
|
|
}
|
|
|
|
// 如果没有找到标点符号,则等待更多文本
|
|
return true
|
|
} catch {
|
|
print("\(tag): 流式语音合成失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 播放剩余的流式文本
|
|
/// - Returns: 是否成功播放剩余文本
|
|
func flushStream() -> Bool {
|
|
if !initialized {
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 获取缓冲区中剩余的文本
|
|
let remainingText = streamBuffer
|
|
|
|
// 清空缓冲区
|
|
streamBuffer = ""
|
|
|
|
// 如果缓冲区为空,直接返回成功
|
|
if remainingText.isEmpty {
|
|
return true
|
|
}
|
|
|
|
// 播放剩余文本
|
|
return speakText(remainingText)
|
|
} catch {
|
|
print("\(tag): 刷新流式文本失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 停止当前语音合成
|
|
/// - Returns: 是否停止成功
|
|
func stopSpeaking() -> Bool {
|
|
if !initialized {
|
|
print("\(tag): TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 清空流缓冲区
|
|
streamBuffer = ""
|
|
|
|
try synthesizer?.stopSpeakingAsync()
|
|
speaking = false
|
|
print("\(tag): 已停止语音合成")
|
|
return true
|
|
} catch {
|
|
print("\(tag): 停止语音合成失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 释放资源
|
|
func dispose() {
|
|
do {
|
|
stopSpeaking()
|
|
|
|
// 恢复音频会话
|
|
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
|
|
|
|
// 关闭自定义音频输出流
|
|
customAudioOutputStream = nil
|
|
|
|
// 清空流缓冲区
|
|
streamBuffer = ""
|
|
|
|
// 清理资源
|
|
synthesizer = nil
|
|
speechConfig = nil
|
|
audioConfig = nil
|
|
ttsCallback = nil
|
|
|
|
initialized = false
|
|
speaking = false
|
|
print("\(tag): TTS 引擎已释放")
|
|
} catch {
|
|
print("\(tag): 释放 TTS 引擎失败: \(error.localizedDescription)")
|
|
// 确保重置状态
|
|
initialized = false
|
|
speaking = false
|
|
}
|
|
}
|
|
|
|
/// 检查当前是否正在播放语音
|
|
/// - Returns: 是否正在播放语音
|
|
func isSpeaking() -> Bool {
|
|
return speaking
|
|
}
|
|
}
|