import Foundation import MicrosoftCognitiveServicesSpeech import AVFoundation import AudioToolbox /// Azure ASR工具类,负责实现语音识别服务接口 @available(iOS 13.0, *) class AzureAsrHelper: NSObject { // MARK: - 属性 /// 事件处理回调 private var eventHandler: (String, [String: Any]) -> Void /// 语音配置信息 private var speechSubscriptionKey: String = "" private var serviceRegion: String = "" /// 语音识别相关 private var speechConfig: SPXSpeechConfiguration? private var recognizer: SPXSpeechRecognizer? private var audioConfig: SPXAudioConfiguration? private var pushStream: SPXPushAudioInputStream? /// 音频处理相关 private var audioProcessor: CustomAudioProcessor? private var isProcessingAudio = false private var audioProcessingTimer: Timer? /// 状态标志 private var isInitialized = false private var _isContinuousRecognitionActive = false /// 当前语言和支持的语言 private var currentLanguage = "zh-CN" private var supportedLanguages: [String] = ["zh-CN", "en-US"] private var isAutoDetectLanguage = false // MARK: - 初始化 init(eventHandler: @escaping (String, [String: Any]) -> Void) { self.eventHandler = eventHandler super.init() } deinit { dispose() } // MARK: - ASR Service 接口实现 /// 初始化语音识别服务 /// - Parameters: /// - speechSubscriptionKey: Azure 语音服务订阅密钥 /// - serviceRegion: Azure 服务区域 (如 eastasia) /// - supportedLanguages: 支持的语言代码数组 (可选) /// - Returns: 初始化是否成功 func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool { print("[AzureAsrHelper] 初始化 Azure 语音服务") // 检查配置是否为空 if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { print("[AzureAsrHelper] 错误: Azure 配置信息不完整") eventHandler("error", ["message": "Azure 配置信息不完整"]) return false } // 释放之前的资源 dispose() // 记录配置信息 self.speechSubscriptionKey = speechSubscriptionKey self.serviceRegion = serviceRegion // 设置语言 if let languages = supportedLanguages, !languages.isEmpty { self.supportedLanguages = languages } // 根据支持的语言数量决定是否启用自动语言检测 isAutoDetectLanguage = self.supportedLanguages.count >= 2 // 如果只有一种语言,设置为当前语言 if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty { currentLanguage = self.supportedLanguages[0] } // 创建识别器和设置回调 if !createRecognizerAndSetupCallbacks() { return false } print("[AzureAsrHelper] Azure 语音服务初始化成功") isInitialized = true return true } /// 创建识别器并设置回调 private func createRecognizerAndSetupCallbacks() -> Bool { // 释放之前的 recognizer recognizer = nil audioConfig = nil do { // 创建语音配置 speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) // 设置音频输入参数 try setupAudioSession() // 创建自定义推送流,替代默认的麦克风输入 pushStream = try SPXPushAudioInputStream() audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) // 初始化自定义音频处理器 audioProcessor = CustomAudioProcessor() // 设置语言配置 if isAutoDetectLanguage { // 设置自动语言检测 speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode) // 创建自动语言检测配置 let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) // 创建识别器 recognizer = try SPXSpeechRecognizer( speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig, audioConfiguration: audioConfig! ) } else { // 设置指定的识别语言 speechConfig?.speechRecognitionLanguage = currentLanguage // 创建识别器 recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) } // 设置所有回调 setupAllCallbacks() return true } catch { print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) return false } } /// 设置音频会话 private func setupAudioSession() throws { let audioSession = AVAudioSession.sharedInstance() // 使用playAndRecord类别允许同时录音和播放 try audioSession.setCategory(.playAndRecord, mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) // 设置首选的输入和输出 let currentRoute = audioSession.currentRoute // 获取当前是否连接了耳机或外部麦克风 let hasHeadphones = currentRoute.outputs.contains { $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP } // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 if !hasHeadphones { try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 // 启用回音消除和噪声抑制 try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 } else { // 耳机模式,可以使用不同的设置 try audioSession.setMode(.voiceChat) try audioSession.setInputGain(1.0) } // 设置合适的采样率 try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 // 激活音频会话 try audioSession.setActive(true, options: .notifyOthersOnDeactivation) print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") } /// 设置所有回调 private func setupAllCallbacks() { guard let recognizer = recognizer else { return } // 最终识别结果 recognizer.addRecognizedEventHandler { [weak self] _, event in guard let self = self else { return } if event.result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: event.result) print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") self.eventHandler("result", [ "text": event.result.text ?? "", "detectedLanguage": detectedLanguage ]) } } // 识别中事件 recognizer.addRecognizingEventHandler { [weak self] _, event in guard let self = self else { return } if event.result.reason == SPXResultReason.recognizingSpeech { let detectedLanguage = self.getDetectedLanguage(from: event.result) // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") self.eventHandler("recognizing", [ "text": event.result.text ?? "", "detectedLanguage": detectedLanguage ]) } } // 会话事件 recognizer.addSessionStartedEventHandler { [weak self] _, _ in guard let self = self else { return } print("[AzureAsrHelper] 识别会话已开始") self._isContinuousRecognitionActive = true self.eventHandler("sessionStarted", [:]) } recognizer.addSessionStoppedEventHandler { [weak self] _, _ in guard let self = self else { return } print("[AzureAsrHelper] 识别会话已结束") self._isContinuousRecognitionActive = false self.eventHandler("sessionStopped", [:]) } // 取消事件 recognizer.addCanceledEventHandler { [weak self] _, event in guard let self = self else { return } let reason = event.reason.rawValue let errorDetails = event.errorDetails ?? "未知错误" print("[AzureAsrHelper] 识别取消: \(errorDetails)") self.eventHandler("canceled", [ "reason": reason, "errorDetails": errorDetails ]) self._isContinuousRecognitionActive = false } } /// 执行一次性语音识别 /// - Returns: 是否成功启动识别 func recognizeOnce() -> Bool { if !isInitialized { print("[AzureAsrHelper] 错误: 语音服务未初始化") eventHandler("error", ["message": "语音服务未初始化"]) return false } // 如果正在连续识别,先停止 if _isContinuousRecognitionActive { stopContinuousRecognition() } // 确保识别器已创建 if recognizer == nil && !createRecognizerAndSetupCallbacks() { return false } do { // 启动音频处理 startAudioProcessing() // 通知会话开始 eventHandler("sessionStarted", [:]) // 执行识别 try recognizer?.recognizeOnceAsync { [weak self] result in guard let self = self else { return } // 停止音频处理 self.stopAudioProcessing() if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) self.eventHandler("result", [ "text": result.text ?? "", "detectedLanguage": detectedLanguage ]) } else if result.reason == SPXResultReason.noMatch { print("[AzureAsrHelper] 无匹配结果") self.eventHandler("noMatch", [:]) } else if result.reason == SPXResultReason.canceled { do { let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) let errorDetails = details.errorDetails ?? "未知错误" self.eventHandler("error", ["message": "识别取消: \(errorDetails)"]) } catch { print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)") self.eventHandler("error", ["message": "识别取消,无法获取详细原因"]) } } } return true } catch { print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) stopAudioProcessing() return false } } /// 开始连续语音识别 /// - Returns: 是否成功启动识别 func startContinuousRecognition() -> Bool { if !isInitialized { print("[AzureAsrHelper] 错误: 语音服务未初始化") eventHandler("error", ["message": "语音服务未初始化"]) return false } // 如果已经在进行连续识别,先停止 if _isContinuousRecognitionActive { stopContinuousRecognition() } // 确保识别器已创建 if recognizer == nil && !createRecognizerAndSetupCallbacks() { return false } // 重新确保音频设置正确 do { try setupAudioSession() } catch { print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") } do { // 启动音频处理 startAudioProcessing() // 启动连续识别 try recognizer?.startContinuousRecognition() _isContinuousRecognitionActive = true print("[AzureAsrHelper] 连续识别开始") return true } catch { print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) _isContinuousRecognitionActive = false stopAudioProcessing() return false } } /// 停止连续语音识别 /// - Returns: 是否成功停止识别 func stopContinuousRecognition() -> Bool { // 停止音频处理 stopAudioProcessing() if !_isContinuousRecognitionActive || recognizer == nil { return true } do { try recognizer?.stopContinuousRecognition() _isContinuousRecognitionActive = false print("[AzureAsrHelper] 连续识别已停止") return true } catch { print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"]) _isContinuousRecognitionActive = false return false } } /// 检查连续识别是否活跃 /// - Returns: 连续识别是否处于活跃状态 func isContinuousRecognitionActive() -> Bool { return _isContinuousRecognitionActive } /// 释放资源 func dispose() { print("[AzureAsrHelper] 释放资源") // 停止音频处理 stopAudioProcessing() // 停止连续识别 if _isContinuousRecognitionActive { stopContinuousRecognition() } // 释放音频会话 do { try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) } catch { print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") } // 释放资源 recognizer = nil speechConfig = nil audioConfig = nil pushStream = nil audioProcessor = nil // 重置状态 _isContinuousRecognitionActive = false isInitialized = false } /// 从结果中获取检测到的语言 private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { if isAutoDetectLanguage { do { let langResult = try SPXAutoDetectSourceLanguageResult(result) return langResult.language ?? currentLanguage } catch { print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") return currentLanguage } } else { return currentLanguage } } // MARK: - 音频处理 /// 开始音频处理 private func startAudioProcessing() { guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } isProcessingAudio = true // 启动音频处理器 if !audioProcessor.startRecord() { print("[AzureAsrHelper] 错误: 启动音频处理器失败") eventHandler("error", ["message": "启动音频处理器失败"]) return } // 启动音频处理定时器 audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { return } // 读取处理后的音频数据 var bytes = [UInt8](repeating: 0, count: 2560) let bytesRead = processor.read(bytes: &bytes) if bytesRead > 0 { // 推送数据到Azure语音服务 let data = Data(bytes: bytes, count: bytesRead) stream.write(data) // 通知音频数据可用 self.eventHandler("audioData", ["data": bytes]) } } print("[AzureAsrHelper] 音频处理已启动") } /// 停止音频处理 private func stopAudioProcessing() { // 停止定时器 audioProcessingTimer?.invalidate() audioProcessingTimer = nil // 停止音频处理器 audioProcessor?.stopRecord() isProcessingAudio = false print("[AzureAsrHelper] 音频处理已停止") } } // MARK: - 自定义音频处理器 @available(iOS 13.0, *) class CustomAudioProcessor: NSObject { // 音频单元 private var ioUnit: AudioUnit? // 音频格式 private var audioFormat: AudioStreamBasicDescription // 音频缓冲 private var audioBufferList: AudioBufferList private var audioList: [Float] = [] private let audioListQueue = DispatchQueue(label: "audioListQueue") // 回音消除状态 private var isEchoCancellationEnabled = true override init() { // 设置音频格式 - 16kHz, 16位, 单声道 audioFormat = AudioStreamBasicDescription( mSampleRate: 16000.0, mFormatID: kAudioFormatLinearPCM, mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, mBytesPerPacket: 2, mFramesPerPacket: 1, mBytesPerFrame: 2, mChannelsPerFrame: 1, mBitsPerChannel: 16, mReserved: 0 ) // 初始化音频缓冲 audioBufferList = AudioBufferList( mNumberBuffers: 1, mBuffers: AudioBuffer( mNumberChannels: 1, mDataByteSize: 4096, mData: malloc(4096) ) ) super.init() } deinit { stopRecord() free(audioBufferList.mBuffers.mData) } /// 启动音频处理 /// - Returns: 是否成功启动 func startRecord() -> Bool { print("[CustomAudioProcessor] 配置音频单元") // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 var ioUnitDescription = AudioComponentDescription( componentType: kAudioUnitType_Output, componentSubType: kAudioUnitSubType_VoiceProcessingIO, componentManufacturer: kAudioUnitManufacturer_Apple, componentFlags: 0, componentFlagsMask: 0 ) // 查找音频组件 guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { print("[CustomAudioProcessor] 错误: 未找到音频组件") return false } // 创建音频单元实例 if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { ioUnit = nil return false } // 启用输入端口 var enableInput: UInt32 = 1 let kInputBus: AudioUnitElement = 1 let kOutputBus: AudioUnitElement = 0 if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, kAudioUnitScope_Input, kInputBus, &enableInput, UInt32(MemoryLayout.size)), "启用输入端口") { return false } // 禁用输出端口 (我们只需要输入) var enableOutput: UInt32 = 0 if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, kAudioUnitScope_Output, kOutputBus, &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") { return false } // 设置缓冲区分配标志 var flag: UInt32 = 0 if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") { return false } // 设置音频格式 let size = UInt32(MemoryLayout.size) if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { return false } if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { return false } // 启用回音消除 if isEchoCancellationEnabled { var echoCancellation: UInt32 = 1 AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size)) } // 设置输入回调 - 当有新音频数据时调用 var inputCallback = AURenderCallbackStruct( inputProc: CustomAudioProcessor.onAudioDataAvailable, inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) ) if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_SetInputCallback, kAudioUnitScope_Global, kInputBus, &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { return false } // 初始化音频单元 var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") while hasError { Thread.sleep(forTimeInterval: 0.1) hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") } // 启动音频单元 hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") return !hasError } /// 停止音频处理 func stopRecord() { print("[CustomAudioProcessor] 停止音频处理器") if let ioUnit = ioUnit { // 停止音频单元 _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") // 关闭音频单元 _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") self.ioUnit = nil } // 清空音频数据缓冲 audioListQueue.sync { audioList.removeAll() } } /// 音频数据回调 - 当有新的音频数据可用时调用 private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in // 获取实例 let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() // 计算预期数据大小 let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame // 确保缓冲区足够大 if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize } // 渲染音频数据 let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, &processor.audioBufferList), "渲染音频数据") // 将Int16数据转换为浮点数据进行处理 var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) let buffer = processor.audioBufferList.mBuffers let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) for j in 0...size)) { // 归一化到[-1.0, 1.0]范围 audioDataFloat[j] = Float(bufferData[j]) / 32768.0 } // 应用附加处理 (如有需要) // processor.applyAdditionalProcessing(&audioDataFloat) // 保存处理后的数据 if status == noErr { processor.audioListQueue.async { processor.audioList.append(contentsOf: audioDataFloat) } } return status } /// 读取处理后的音频数据 /// - Parameter bytes: 输出字节数组 /// - Returns: 读取的字节数 func read(bytes: inout [UInt8]) -> Int { return audioListQueue.sync { // 如果没有数据,返回0 if audioList.isEmpty { return 0 } // 确保有足够的数据 (至少1280个样本) if audioList.count < 1280 { return 0 } // 读取一帧数据 (1280个样本) let frameLength = 1280 let buffer = Array(audioList.prefix(frameLength)) audioList.removeFirst(frameLength) // 将浮点数据转回Int16格式 var int16Data = buffer.map { Int16($0 * 32767) } // 转换为字节数组 let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) bytes = [UInt8](data) // 每个样本2字节 (16位PCM) return frameLength * 2 } } /// 检查错误并打印日志 /// - Parameters: /// - status: 操作状态 /// - operation: 操作描述 /// - Returns: 是否发生错误 private func checkError(_ status: OSStatus, _ operation: String) -> Bool { if status != noErr { print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") return true } return false } /// 检查OSStatus并返回状态 /// - Parameters: /// - status: 操作状态 /// - operation: 操作描述 /// - Returns: 原始状态 private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { if status != noErr { print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") } return status } }