You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
330 lines
12 KiB
330 lines
12 KiB
import Foundation
|
|
import AVFoundation
|
|
import MicrosoftCognitiveServicesSpeech
|
|
|
|
/// Azure 语音合成辅助类
|
|
class AzureTtsHelper: NSObject {
|
|
private var synthesizer: SPXSpeechSynthesizer?
|
|
private var speechConfig: SPXSpeechConfig?
|
|
private var audioConfig: SPXAudioConfig?
|
|
private var initialized = false
|
|
private var speaking = false
|
|
|
|
// 音频输出类型
|
|
enum AudioOutputType {
|
|
case speaker // 扬声器
|
|
case earpiece // 听筒
|
|
case auto // 自动选择
|
|
}
|
|
|
|
// 当前设置
|
|
private var currentVoiceName = "zh-CN-XiaoxiaoNeural"
|
|
private var currentSpeechRate = 0
|
|
private var currentPitch = 0
|
|
private var currentVolume = 100
|
|
private var currentAudioOutputType: AudioOutputType = .auto
|
|
|
|
// 音频会话管理
|
|
private let audioSession = AVAudioSession.sharedInstance()
|
|
|
|
/// 初始化 TTS 引擎
|
|
///
|
|
/// - Parameters:
|
|
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
|
|
/// - serviceRegion: Azure 语音服务区域
|
|
/// - language: 可选,默认语言,默认为 "zh-CN"
|
|
/// - Returns: 是否初始化成功
|
|
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool {
|
|
print("[AzureTtsHelper] 初始化 Azure 语音服务")
|
|
|
|
// 检查配置是否为空
|
|
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
|
|
print("[AzureTtsHelper] 错误: Azure 配置信息不完整")
|
|
return false
|
|
}
|
|
|
|
// 释放之前的资源
|
|
dispose()
|
|
|
|
do {
|
|
// 创建语音配置
|
|
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion)
|
|
|
|
// 设置语音合成输出格式为高质量音频
|
|
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm)
|
|
|
|
// 设置默认语言
|
|
speechConfig?.setSpeechSynthesisLanguage(language)
|
|
|
|
// 设置默认语音
|
|
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName)
|
|
|
|
// 创建音频配置 - 使用默认扬声器
|
|
audioConfig = SPXAudioConfig.default()
|
|
|
|
// 创建语音合成器
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
|
|
|
|
initialized = true
|
|
|
|
// 设置默认音频输出类型为自动
|
|
setAudioOutputType(outputType: .auto)
|
|
|
|
print("[AzureTtsHelper] TTS 引擎初始化成功")
|
|
return true
|
|
} catch {
|
|
print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 设置音频输出设备类型
|
|
///
|
|
/// - Parameter outputType: 音频输出设备类型
|
|
/// - Returns: 是否设置成功
|
|
func setAudioOutputType(outputType: AudioOutputType) -> Bool {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
currentAudioOutputType = outputType
|
|
|
|
switch outputType {
|
|
case .speaker:
|
|
// 使用扬声器
|
|
try audioSession.setCategory(.playback, mode: .default)
|
|
try audioSession.overrideOutputAudioPort(.speaker)
|
|
print("[AzureTtsHelper] 已设置音频输出设备为扬声器")
|
|
|
|
case .earpiece:
|
|
// 使用听筒
|
|
try audioSession.setCategory(.playback, mode: .voiceChat)
|
|
try audioSession.overrideOutputAudioPort(.none)
|
|
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
|
|
|
|
case .auto:
|
|
// 检查是否有耳机连接
|
|
let outputs = audioSession.currentRoute.outputs
|
|
let hasHeadphones = outputs.contains { output in
|
|
return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP
|
|
}
|
|
|
|
if hasHeadphones {
|
|
// 有耳机,使用耳机
|
|
try audioSession.setCategory(.playback, mode: .default)
|
|
try audioSession.overrideOutputAudioPort(.none)
|
|
print("[AzureTtsHelper] 已设置音频输出设备为耳机")
|
|
} else {
|
|
// 无耳机,使用听筒
|
|
try audioSession.setCategory(.playback, mode: .voiceChat)
|
|
try audioSession.overrideOutputAudioPort(.none)
|
|
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
|
|
}
|
|
}
|
|
|
|
try audioSession.setActive(true)
|
|
return true
|
|
} catch {
|
|
print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 设置语音
|
|
///
|
|
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural"
|
|
/// - Returns: 是否设置成功
|
|
func setVoice(voiceName: String) -> Bool {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
if voiceName == currentVoiceName {
|
|
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
|
|
return true
|
|
}
|
|
|
|
do {
|
|
currentVoiceName = voiceName
|
|
speechConfig?.setSpeechSynthesisVoiceName(voiceName)
|
|
|
|
// 重新创建合成器
|
|
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
|
|
|
|
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
|
|
return true
|
|
} catch {
|
|
print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 设置语音合成参数
|
|
///
|
|
/// - Parameters:
|
|
/// - rate: 语速,范围 -100 到 100,默认为 0
|
|
/// - pitch: 音调,范围 -100 到 100,默认为 0
|
|
/// - volume: 音量,范围 0 到 100,默认为 100
|
|
/// - Returns: 是否设置成功
|
|
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
currentSpeechRate = rate
|
|
currentPitch = pitch
|
|
currentVolume = volume
|
|
|
|
print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)")
|
|
return true
|
|
}
|
|
|
|
/// 合成文本为语音并播放
|
|
///
|
|
/// - Parameters:
|
|
/// - text: 要合成的文本
|
|
/// - completion: 完成回调,返回是否成功和可能的错误信息
|
|
func speakText(text: String, completion: @escaping (Bool, String?) -> Void) {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
completion(false, "TTS 引擎尚未初始化")
|
|
return
|
|
}
|
|
|
|
do {
|
|
print("[AzureTtsHelper] 开始合成文本: \(text)")
|
|
|
|
// 生成 SSML
|
|
let ssml = generateSsml(text: text)
|
|
|
|
// 使用 SSML 合成语音
|
|
speakSsml(ssml: ssml, completion: completion)
|
|
} catch {
|
|
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
|
|
completion(false, "语音合成异常: \(error.localizedDescription)")
|
|
}
|
|
}
|
|
|
|
/// 生成 SSML 文本
|
|
///
|
|
/// - Parameter text: 要转换的文本
|
|
/// - Returns: SSML 格式的文本
|
|
private func generateSsml(text: String) -> String {
|
|
// 计算 SSML 参数
|
|
let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%")
|
|
let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%"
|
|
let volumeParam = "\(min(max(currentVolume, 0), 100))%"
|
|
|
|
return """
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN">
|
|
<voice name="\(currentVoiceName)">
|
|
<prosody rate="\(rateParam)" pitch="\(pitchParam)" volume="\(volumeParam)">
|
|
\(text)
|
|
</prosody>
|
|
</voice>
|
|
</speak>
|
|
"""
|
|
}
|
|
|
|
/// 合成 SSML 为语音并播放
|
|
///
|
|
/// - Parameters:
|
|
/// - ssml: SSML 格式的文本
|
|
/// - completion: 完成回调,返回是否成功和可能的错误信息
|
|
private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
completion(false, "TTS 引擎尚未初始化")
|
|
return
|
|
}
|
|
|
|
do {
|
|
print("[AzureTtsHelper] 开始合成 SSML")
|
|
|
|
// 标记为正在播放
|
|
speaking = true
|
|
|
|
// 激活音频会话
|
|
try audioSession.setActive(true)
|
|
|
|
// 异步合成语音
|
|
let result = try synthesizer!.speakSsml(ssml)
|
|
|
|
switch result.reason {
|
|
case .synthesizingAudioCompleted:
|
|
print("[AzureTtsHelper] 语音合成完成")
|
|
speaking = false
|
|
completion(true, "语音合成完成")
|
|
case .canceled:
|
|
if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) {
|
|
print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
|
|
speaking = false
|
|
completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
|
|
} else {
|
|
print("[AzureTtsHelper] 语音合成取消")
|
|
speaking = false
|
|
completion(false, "语音合成取消")
|
|
}
|
|
default:
|
|
print("[AzureTtsHelper] 语音合成失败: \(result.reason)")
|
|
speaking = false
|
|
completion(false, "语音合成失败: \(result.reason)")
|
|
}
|
|
} catch {
|
|
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
|
|
speaking = false
|
|
completion(false, "语音合成异常: \(error.localizedDescription)")
|
|
}
|
|
}
|
|
|
|
/// 停止当前语音合成
|
|
///
|
|
/// - Returns: 是否停止成功
|
|
func stopSpeaking() -> Bool {
|
|
if !initialized {
|
|
print("[AzureTtsHelper] TTS 引擎尚未初始化")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
try synthesizer?.stopSpeaking()
|
|
speaking = false
|
|
print("[AzureTtsHelper] 已停止语音合成")
|
|
return true
|
|
} catch {
|
|
print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 释放资源
|
|
func dispose() {
|
|
do {
|
|
stopSpeaking()
|
|
|
|
// 恢复音频会话
|
|
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
|
|
|
|
synthesizer = nil
|
|
speechConfig = nil
|
|
audioConfig = nil
|
|
|
|
initialized = false
|
|
speaking = false
|
|
print("[AzureTtsHelper] TTS 引擎已释放")
|
|
} catch {
|
|
print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)")
|
|
}
|
|
}
|
|
|
|
/// 检查当前是否正在播放语音
|
|
///
|
|
/// - Returns: 是否正在播放语音
|
|
func isSpeaking() -> Bool {
|
|
return speaking
|
|
}
|
|
}
|