You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.

330 lines
12 KiB

import Foundation
import AVFoundation
import MicrosoftCognitiveServicesSpeech
/// Azure 语音合成辅助类
class AzureTtsHelper: NSObject {
private var synthesizer: SPXSpeechSynthesizer?
private var speechConfig: SPXSpeechConfig?
private var audioConfig: SPXAudioConfig?
private var initialized = false
private var speaking = false
// 音频输出类型
enum AudioOutputType {
case speaker // 扬声器
case earpiece // 听筒
case auto // 自动选择
}
// 当前设置
private var currentVoiceName = "zh-CN-XiaoxiaoNeural"
private var currentSpeechRate = 0
private var currentPitch = 0
private var currentVolume = 100
private var currentAudioOutputType: AudioOutputType = .auto
// 音频会话管理
private let audioSession = AVAudioSession.sharedInstance()
/// 初始化 TTS 引擎
///
/// - Parameters:
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
/// - serviceRegion: Azure 语音服务区域
/// - language: 可选,默认语言,默认为 "zh-CN"
/// - Returns: 是否初始化成功
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool {
print("[AzureTtsHelper] 初始化 Azure 语音服务")
// 检查配置是否为空
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureTtsHelper] 错误: Azure 配置信息不完整")
return false
}
// 释放之前的资源
dispose()
do {
// 创建语音配置
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置语音合成输出格式为高质量音频
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm)
// 设置默认语言
speechConfig?.setSpeechSynthesisLanguage(language)
// 设置默认语音
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName)
// 创建音频配置 - 使用默认扬声器
audioConfig = SPXAudioConfig.default()
// 创建语音合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
initialized = true
// 设置默认音频输出类型为自动
setAudioOutputType(outputType: .auto)
print("[AzureTtsHelper] TTS 引擎初始化成功")
return true
} catch {
print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)")
return false
}
}
/// 设置音频输出设备类型
///
/// - Parameter outputType: 音频输出设备类型
/// - Returns: 是否设置成功
func setAudioOutputType(outputType: AudioOutputType) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
do {
currentAudioOutputType = outputType
switch outputType {
case .speaker:
// 使用扬声器
try audioSession.setCategory(.playback, mode: .default)
try audioSession.overrideOutputAudioPort(.speaker)
print("[AzureTtsHelper] 已设置音频输出设备为扬声器")
case .earpiece:
// 使用听筒
try audioSession.setCategory(.playback, mode: .voiceChat)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
case .auto:
// 检查是否有耳机连接
let outputs = audioSession.currentRoute.outputs
let hasHeadphones = outputs.contains { output in
return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP
}
if hasHeadphones {
// 有耳机,使用耳机
try audioSession.setCategory(.playback, mode: .default)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为耳机")
} else {
// 无耳机,使用听筒
try audioSession.setCategory(.playback, mode: .voiceChat)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
}
}
try audioSession.setActive(true)
return true
} catch {
print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)")
return false
}
}
/// 设置语音
///
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural"
/// - Returns: 是否设置成功
func setVoice(voiceName: String) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
if voiceName == currentVoiceName {
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
return true
}
do {
currentVoiceName = voiceName
speechConfig?.setSpeechSynthesisVoiceName(voiceName)
// 重新创建合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
return true
} catch {
print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)")
return false
}
}
/// 设置语音合成参数
///
/// - Parameters:
/// - rate: 语速,范围 -100 到 100,默认为 0
/// - pitch: 音调,范围 -100 到 100,默认为 0
/// - volume: 音量,范围 0 到 100,默认为 100
/// - Returns: 是否设置成功
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
currentSpeechRate = rate
currentPitch = pitch
currentVolume = volume
print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)")
return true
}
/// 合成文本为语音并播放
///
/// - Parameters:
/// - text: 要合成的文本
/// - completion: 完成回调,返回是否成功和可能的错误信息
func speakText(text: String, completion: @escaping (Bool, String?) -> Void) {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
completion(false, "TTS 引擎尚未初始化")
return
}
do {
print("[AzureTtsHelper] 开始合成文本: \(text)")
// 生成 SSML
let ssml = generateSsml(text: text)
// 使用 SSML 合成语音
speakSsml(ssml: ssml, completion: completion)
} catch {
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
completion(false, "语音合成异常: \(error.localizedDescription)")
}
}
/// 生成 SSML 文本
///
/// - Parameter text: 要转换的文本
/// - Returns: SSML 格式的文本
private func generateSsml(text: String) -> String {
// 计算 SSML 参数
let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%")
let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%"
let volumeParam = "\(min(max(currentVolume, 0), 100))%"
return """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN">
<voice name="\(currentVoiceName)">
<prosody rate="\(rateParam)" pitch="\(pitchParam)" volume="\(volumeParam)">
\(text)
</prosody>
</voice>
</speak>
"""
}
/// 合成 SSML 为语音并播放
///
/// - Parameters:
/// - ssml: SSML 格式的文本
/// - completion: 完成回调,返回是否成功和可能的错误信息
private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
completion(false, "TTS 引擎尚未初始化")
return
}
do {
print("[AzureTtsHelper] 开始合成 SSML")
// 标记为正在播放
speaking = true
// 激活音频会话
try audioSession.setActive(true)
// 异步合成语音
let result = try synthesizer!.speakSsml(ssml)
switch result.reason {
case .synthesizingAudioCompleted:
print("[AzureTtsHelper] 语音合成完成")
speaking = false
completion(true, "语音合成完成")
case .canceled:
if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) {
print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
speaking = false
completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
} else {
print("[AzureTtsHelper] 语音合成取消")
speaking = false
completion(false, "语音合成取消")
}
default:
print("[AzureTtsHelper] 语音合成失败: \(result.reason)")
speaking = false
completion(false, "语音合成失败: \(result.reason)")
}
} catch {
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
speaking = false
completion(false, "语音合成异常: \(error.localizedDescription)")
}
}
/// 停止当前语音合成
///
/// - Returns: 是否停止成功
func stopSpeaking() -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
do {
try synthesizer?.stopSpeaking()
speaking = false
print("[AzureTtsHelper] 已停止语音合成")
return true
} catch {
print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)")
return false
}
}
/// 释放资源
func dispose() {
do {
stopSpeaking()
// 恢复音频会话
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
synthesizer = nil
speechConfig = nil
audioConfig = nil
initialized = false
speaking = false
print("[AzureTtsHelper] TTS 引擎已释放")
} catch {
print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)")
}
}
/// 检查当前是否正在播放语音
///
/// - Returns: 是否正在播放语音
func isSpeaking() -> Bool {
return speaking
}
}