import Foundation import AVFoundation import MicrosoftCognitiveServicesSpeech /// Azure 语音合成辅助类 class AzureTtsHelper: NSObject { private var synthesizer: SPXSpeechSynthesizer? private var speechConfig: SPXSpeechConfig? private var audioConfig: SPXAudioConfig? private var initialized = false private var speaking = false // 音频输出类型 enum AudioOutputType { case speaker // 扬声器 case earpiece // 听筒 case auto // 自动选择 } // 当前设置 private var currentVoiceName = "zh-CN-XiaoxiaoNeural" private var currentSpeechRate = 0 private var currentPitch = 0 private var currentVolume = 100 private var currentAudioOutputType: AudioOutputType = .auto // 音频会话管理 private let audioSession = AVAudioSession.sharedInstance() /// 初始化 TTS 引擎 /// /// - Parameters: /// - speechSubscriptionKey: Azure 语音服务订阅密钥 /// - serviceRegion: Azure 语音服务区域 /// - language: 可选,默认语言,默认为 "zh-CN" /// - Returns: 是否初始化成功 func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { print("[AzureTtsHelper] 初始化 Azure 语音服务") // 检查配置是否为空 if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { print("[AzureTtsHelper] 错误: Azure 配置信息不完整") return false } // 释放之前的资源 dispose() do { // 创建语音配置 speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) // 设置语音合成输出格式为高质量音频 speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm) // 设置默认语言 speechConfig?.setSpeechSynthesisLanguage(language) // 设置默认语音 speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) // 创建音频配置 - 使用默认扬声器 audioConfig = SPXAudioConfig.default() // 创建语音合成器 synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) initialized = true // 设置默认音频输出类型为自动 setAudioOutputType(outputType: .auto) print("[AzureTtsHelper] TTS 引擎初始化成功") return true } catch { print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)") return false } } /// 设置音频输出设备类型 /// /// - Parameter outputType: 音频输出设备类型 /// - Returns: 是否设置成功 func setAudioOutputType(outputType: AudioOutputType) -> Bool { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") return false } do { currentAudioOutputType = outputType switch outputType { case .speaker: // 使用扬声器 try audioSession.setCategory(.playback, mode: .default) try audioSession.overrideOutputAudioPort(.speaker) print("[AzureTtsHelper] 已设置音频输出设备为扬声器") case .earpiece: // 使用听筒 try audioSession.setCategory(.playback, mode: .voiceChat) try audioSession.overrideOutputAudioPort(.none) print("[AzureTtsHelper] 已设置音频输出设备为听筒") case .auto: // 检查是否有耳机连接 let outputs = audioSession.currentRoute.outputs let hasHeadphones = outputs.contains { output in return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP } if hasHeadphones { // 有耳机,使用耳机 try audioSession.setCategory(.playback, mode: .default) try audioSession.overrideOutputAudioPort(.none) print("[AzureTtsHelper] 已设置音频输出设备为耳机") } else { // 无耳机,使用听筒 try audioSession.setCategory(.playback, mode: .voiceChat) try audioSession.overrideOutputAudioPort(.none) print("[AzureTtsHelper] 已设置音频输出设备为听筒") } } try audioSession.setActive(true) return true } catch { print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)") return false } } /// 设置语音 /// /// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural" /// - Returns: 是否设置成功 func setVoice(voiceName: String) -> Bool { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") return false } if voiceName == currentVoiceName { print("[AzureTtsHelper] 已设置语音: \(voiceName)") return true } do { currentVoiceName = voiceName speechConfig?.setSpeechSynthesisVoiceName(voiceName) // 重新创建合成器 synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) print("[AzureTtsHelper] 已设置语音: \(voiceName)") return true } catch { print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)") return false } } /// 设置语音合成参数 /// /// - Parameters: /// - rate: 语速,范围 -100 到 100,默认为 0 /// - pitch: 音调,范围 -100 到 100,默认为 0 /// - volume: 音量,范围 0 到 100,默认为 100 /// - Returns: 是否设置成功 func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") return false } currentSpeechRate = rate currentPitch = pitch currentVolume = volume print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)") return true } /// 合成文本为语音并播放 /// /// - Parameters: /// - text: 要合成的文本 /// - completion: 完成回调,返回是否成功和可能的错误信息 func speakText(text: String, completion: @escaping (Bool, String?) -> Void) { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") completion(false, "TTS 引擎尚未初始化") return } do { print("[AzureTtsHelper] 开始合成文本: \(text)") // 生成 SSML let ssml = generateSsml(text: text) // 使用 SSML 合成语音 speakSsml(ssml: ssml, completion: completion) } catch { print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") completion(false, "语音合成异常: \(error.localizedDescription)") } } /// 生成 SSML 文本 /// /// - Parameter text: 要转换的文本 /// - Returns: SSML 格式的文本 private func generateSsml(text: String) -> String { // 计算 SSML 参数 let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%") let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%" let volumeParam = "\(min(max(currentVolume, 0), 100))%" return """ \(text) """ } /// 合成 SSML 为语音并播放 /// /// - Parameters: /// - ssml: SSML 格式的文本 /// - completion: 完成回调,返回是否成功和可能的错误信息 private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") completion(false, "TTS 引擎尚未初始化") return } do { print("[AzureTtsHelper] 开始合成 SSML") // 标记为正在播放 speaking = true // 激活音频会话 try audioSession.setActive(true) // 异步合成语音 let result = try synthesizer!.speakSsml(ssml) switch result.reason { case .synthesizingAudioCompleted: print("[AzureTtsHelper] 语音合成完成") speaking = false completion(true, "语音合成完成") case .canceled: if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) { print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") speaking = false completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") } else { print("[AzureTtsHelper] 语音合成取消") speaking = false completion(false, "语音合成取消") } default: print("[AzureTtsHelper] 语音合成失败: \(result.reason)") speaking = false completion(false, "语音合成失败: \(result.reason)") } } catch { print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") speaking = false completion(false, "语音合成异常: \(error.localizedDescription)") } } /// 停止当前语音合成 /// /// - Returns: 是否停止成功 func stopSpeaking() -> Bool { if !initialized { print("[AzureTtsHelper] TTS 引擎尚未初始化") return false } do { try synthesizer?.stopSpeaking() speaking = false print("[AzureTtsHelper] 已停止语音合成") return true } catch { print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)") return false } } /// 释放资源 func dispose() { do { stopSpeaking() // 恢复音频会话 try audioSession.setActive(false, options: .notifyOthersOnDeactivation) synthesizer = nil speechConfig = nil audioConfig = nil initialized = false speaking = false print("[AzureTtsHelper] TTS 引擎已释放") } catch { print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)") } } /// 检查当前是否正在播放语音 /// /// - Returns: 是否正在播放语音 func isSpeaking() -> Bool { return speaking } }