diff --git a/azure/ios/Classes/AzureAsrHelper.swift b/azure/ios/Classes/AzureAsrHelper.swift index 4bd02368f..58b142f60 100644 --- a/azure/ios/Classes/AzureAsrHelper.swift +++ b/azure/ios/Classes/AzureAsrHelper.swift @@ -1,6 +1,7 @@ import Foundation import MicrosoftCognitiveServicesSpeech import AVFoundation +import AudioToolbox /// Azure ASR工具类,负责实现语音识别服务接口 @available(iOS 13.0, *) @@ -18,6 +19,12 @@ class AzureAsrHelper: NSObject { private var speechConfig: SPXSpeechConfiguration? private var recognizer: SPXSpeechRecognizer? private var audioConfig: SPXAudioConfiguration? + private var pushStream: SPXPushAudioInputStream? + + /// 音频处理相关 + private var audioProcessor: CustomAudioProcessor? + private var isProcessingAudio = false + private var audioProcessingTimer: Timer? /// 状态标志 private var isInitialized = false @@ -100,8 +107,12 @@ class AzureAsrHelper: NSObject { // 设置音频输入参数 try setupAudioSession() - // 直接使用麦克风音频配置 - audioConfig = try SPXAudioConfiguration() + // 创建自定义推送流,替代默认的麦克风输入 + pushStream = try SPXPushAudioInputStream() + audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) + + // 初始化自定义音频处理器 + audioProcessor = CustomAudioProcessor() // 设置语言配置 if isAutoDetectLanguage { @@ -143,7 +154,7 @@ class AzureAsrHelper: NSObject { // 使用playAndRecord类别允许同时录音和播放 try audioSession.setCategory(.playAndRecord, mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 - options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay]) + options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) // 设置首选的输入和输出 let currentRoute = audioSession.currentRoute @@ -262,6 +273,9 @@ class AzureAsrHelper: NSObject { } do { + // 启动音频处理 + startAudioProcessing() + // 通知会话开始 eventHandler("sessionStarted", [:]) @@ -269,6 +283,9 @@ class AzureAsrHelper: NSObject { try recognizer?.recognizeOnceAsync { [weak self] result in guard let self = self else { return } + // 停止音频处理 + self.stopAudioProcessing() + if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) self.eventHandler("result", [ @@ -294,6 +311,7 @@ class AzureAsrHelper: NSObject { } catch { print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) + stopAudioProcessing() return false } } @@ -325,6 +343,9 @@ class AzureAsrHelper: NSObject { } do { + // 启动音频处理 + startAudioProcessing() + // 启动连续识别 try recognizer?.startContinuousRecognition() _isContinuousRecognitionActive = true @@ -335,6 +356,7 @@ class AzureAsrHelper: NSObject { print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) _isContinuousRecognitionActive = false + stopAudioProcessing() return false } } @@ -342,6 +364,9 @@ class AzureAsrHelper: NSObject { /// 停止连续语音识别 /// - Returns: 是否成功停止识别 func stopContinuousRecognition() -> Bool { + // 停止音频处理 + stopAudioProcessing() + if !_isContinuousRecognitionActive || recognizer == nil { return true } @@ -369,6 +394,9 @@ class AzureAsrHelper: NSObject { func dispose() { print("[AzureAsrHelper] 释放资源") + // 停止音频处理 + stopAudioProcessing() + // 停止连续识别 if _isContinuousRecognitionActive { stopContinuousRecognition() @@ -385,6 +413,8 @@ class AzureAsrHelper: NSObject { recognizer = nil speechConfig = nil audioConfig = nil + pushStream = nil + audioProcessor = nil // 重置状态 _isContinuousRecognitionActive = false @@ -405,4 +435,323 @@ class AzureAsrHelper: NSObject { return currentLanguage } } + + // MARK: - 音频处理 + + /// 开始音频处理 + private func startAudioProcessing() { + guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } + + isProcessingAudio = true + + // 启动音频处理器 + if !audioProcessor.startRecord() { + print("[AzureAsrHelper] 错误: 启动音频处理器失败") + eventHandler("error", ["message": "启动音频处理器失败"]) + return + } + + // 启动音频处理定时器 + audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in + guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { + return + } + + // 读取处理后的音频数据 + var bytes = [UInt8](repeating: 0, count: 2560) + let bytesRead = processor.read(bytes: &bytes) + + if bytesRead > 0 { + // 推送数据到Azure语音服务 + let data = Data(bytes: bytes, count: bytesRead) + stream.write(data) + + // 通知音频数据可用 + self.eventHandler("audioData", ["data": bytes]) + } + } + + print("[AzureAsrHelper] 音频处理已启动") + } + + /// 停止音频处理 + private func stopAudioProcessing() { + // 停止定时器 + audioProcessingTimer?.invalidate() + audioProcessingTimer = nil + + // 停止音频处理器 + audioProcessor?.stopRecord() + + isProcessingAudio = false + print("[AzureAsrHelper] 音频处理已停止") + } +} + +// MARK: - 自定义音频处理器 + +@available(iOS 13.0, *) +class CustomAudioProcessor: NSObject { + // 音频单元 + private var ioUnit: AudioUnit? + + // 音频格式 + private var audioFormat: AudioStreamBasicDescription + + // 音频缓冲 + private var audioBufferList: AudioBufferList + private var audioList: [Float] = [] + private let audioListQueue = DispatchQueue(label: "audioListQueue") + + // 回音消除状态 + private var isEchoCancellationEnabled = true + + override init() { + // 设置音频格式 - 16kHz, 16位, 单声道 + audioFormat = AudioStreamBasicDescription( + mSampleRate: 16000.0, + mFormatID: kAudioFormatLinearPCM, + mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, + mBytesPerPacket: 2, + mFramesPerPacket: 1, + mBytesPerFrame: 2, + mChannelsPerFrame: 1, + mBitsPerChannel: 16, + mReserved: 0 + ) + + // 初始化音频缓冲 + audioBufferList = AudioBufferList( + mNumberBuffers: 1, + mBuffers: AudioBuffer( + mNumberChannels: 1, + mDataByteSize: 4096, + mData: malloc(4096) + ) + ) + + super.init() + } + + deinit { + stopRecord() + free(audioBufferList.mBuffers.mData) + } + + /// 启动音频处理 + /// - Returns: 是否成功启动 + func startRecord() -> Bool { + print("[CustomAudioProcessor] 配置音频单元") + + // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 + var ioUnitDescription = AudioComponentDescription( + componentType: kAudioUnitType_Output, + componentSubType: kAudioUnitSubType_VoiceProcessingIO, + componentManufacturer: kAudioUnitManufacturer_Apple, + componentFlags: 0, + componentFlagsMask: 0 + ) + + // 查找音频组件 + guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { + print("[CustomAudioProcessor] 错误: 未找到音频组件") + return false + } + + // 创建音频单元实例 + if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { + ioUnit = nil + return false + } + + // 启用输入端口 + var enableInput: UInt32 = 1 + let kInputBus: AudioUnitElement = 1 + let kOutputBus: AudioUnitElement = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, + kAudioUnitScope_Input, kInputBus, &enableInput, + UInt32(MemoryLayout.size)), "启用输入端口") { + return false + } + + // 禁用输出端口 (我们只需要输入) + var enableOutput: UInt32 = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, + kAudioUnitScope_Output, kOutputBus, + &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") { + return false + } + + // 设置缓冲区分配标志 + var flag: UInt32 = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, + kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") { + return false + } + + // 设置音频格式 + let size = UInt32(MemoryLayout.size) + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, + kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { + return false + } + + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, + kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { + return false + } + + // 启用回音消除 + if isEchoCancellationEnabled { + var echoCancellation: UInt32 = 1 + AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, + kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size)) + } + + // 设置输入回调 - 当有新音频数据时调用 + var inputCallback = AURenderCallbackStruct( + inputProc: CustomAudioProcessor.onAudioDataAvailable, + inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) + ) + + if checkError(AudioUnitSetProperty(ioUnit!, + kAudioOutputUnitProperty_SetInputCallback, + kAudioUnitScope_Global, kInputBus, + &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { + return false + } + + // 初始化音频单元 + var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") + while hasError { + Thread.sleep(forTimeInterval: 0.1) + hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") + } + + // 启动音频单元 + hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") + + print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") + return !hasError + } + + /// 停止音频处理 + func stopRecord() { + print("[CustomAudioProcessor] 停止音频处理器") + + if let ioUnit = ioUnit { + // 停止音频单元 + _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") + + // 关闭音频单元 + _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") + _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") + + self.ioUnit = nil + } + + // 清空音频数据缓冲 + audioListQueue.sync { + audioList.removeAll() + } + } + + /// 音频数据回调 - 当有新的音频数据可用时调用 + private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in + // 获取实例 + let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() + + // 计算预期数据大小 + let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame + + // 确保缓冲区足够大 + if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { + processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) + processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize + } + + // 渲染音频数据 + let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, + inBusNumber, inNumberFrames, &processor.audioBufferList), + "渲染音频数据") + + // 将Int16数据转换为浮点数据进行处理 + var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) + let buffer = processor.audioBufferList.mBuffers + let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) + + for j in 0...size)) { + // 归一化到[-1.0, 1.0]范围 + audioDataFloat[j] = Float(bufferData[j]) / 32768.0 + } + + // 应用附加处理 (如有需要) + // processor.applyAdditionalProcessing(&audioDataFloat) + + // 保存处理后的数据 + if status == noErr { + processor.audioListQueue.async { + processor.audioList.append(contentsOf: audioDataFloat) + } + } + + return status + } + + /// 读取处理后的音频数据 + /// - Parameter bytes: 输出字节数组 + /// - Returns: 读取的字节数 + func read(bytes: inout [UInt8]) -> Int { + return audioListQueue.sync { + // 如果没有数据,返回0 + if audioList.isEmpty { + return 0 + } + + // 确保有足够的数据 (至少1280个样本) + if audioList.count < 1280 { + return 0 + } + + // 读取一帧数据 (1280个样本) + let frameLength = 1280 + let buffer = Array(audioList.prefix(frameLength)) + audioList.removeFirst(frameLength) + + // 将浮点数据转回Int16格式 + var int16Data = buffer.map { Int16($0 * 32767) } + + // 转换为字节数组 + let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) + bytes = [UInt8](data) + + // 每个样本2字节 (16位PCM) + return frameLength * 2 + } + } + + /// 检查错误并打印日志 + /// - Parameters: + /// - status: 操作状态 + /// - operation: 操作描述 + /// - Returns: 是否发生错误 + private func checkError(_ status: OSStatus, _ operation: String) -> Bool { + if status != noErr { + print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") + return true + } + return false + } + + /// 检查OSStatus并返回状态 + /// - Parameters: + /// - status: 操作状态 + /// - operation: 操作描述 + /// - Returns: 原始状态 + private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { + if status != noErr { + print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") + } + return status + } } \ No newline at end of file diff --git a/ios/Runner/AudioSessionManager.swift b/ios/Runner/AudioSessionManager.swift index b4a78ec17..011c32c45 100644 --- a/ios/Runner/AudioSessionManager.swift +++ b/ios/Runner/AudioSessionManager.swift @@ -190,9 +190,9 @@ import UIKit // NSLog("[AudioSessionManager] 配置音频会话: 类别=\(category), 模式=\(mode)") // 暂时停用当前会话(如果正在活动) - // if isActive { - // try audioSession.setActive(false, options: .notifyOthersOnDeactivation) - // } + if isActive { + try audioSession.setActive(false, options: .notifyOthersOnDeactivation) + } // 设置新的分类、模式和选项 try audioSession.setCategory(category, mode: mode, options: options) @@ -214,9 +214,9 @@ import UIKit isConfigured = true // 检查并设置首选的输入设备(如果需要) - // if category == .playAndRecord || category == .record { - // try configurePreferredInput() - // } + if category == .playAndRecord || category == .record { + try configurePreferredInput() + } return } catch { NSLog("[AudioSessionManager] 配置音频会话失败: \(error.localizedDescription)") @@ -420,35 +420,141 @@ import UIKit - /// 配置用于语音交互(同时支持ASR和TTS)的音频会话 - /// 此函数综合优化语音识别和语音合成,适用于需要双向交互的场景 + /// 为Azure语音识别专门配置的音频会话 + /// 此函数优化Azure语音识别的音频处理配置,提供更强的回音消除能力 + /// - Parameters: + /// - force: 是否强制重新配置 + /// - configureAdditionalSettings: 配置完成后进行额外的音频设置 /// - Returns: 配置是否成功 - @objc func configureForVoiceInteraction(force: Bool = false) -> Bool { + @objc func configureForAzureSpeechRecognition(force: Bool = false, configureAdditionalSettings: Bool = true) -> Bool { + NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话") do { - // 配置音频会话以支持语音交互 - // 使用playAndRecord类别允许同时录音和播放 - // 使用spokenAudio模式优化语音交互 + // 配置音频会话以优化语音识别 try configureSession( - category: .playAndRecord, - mode: .spokenAudio, + category: .playAndRecord, // 允许同时录音和播放 + mode: .voiceChat, // 使用voiceChat模式获得最佳回音消除效果 + options: [ + .allowBluetooth, // 允许蓝牙设备 + .defaultToSpeaker, // 默认使用扬声器 + .mixWithOthers // 允许与其他应用混音 + ], + force: force // 是否强制重新配置 + ) + + // 额外的优化配置 + if configureAdditionalSettings { + // 获取当前是否使用耳机 + let currentRoute = audioSession.currentRoute + let hasHeadphones = currentRoute.outputs.contains { + $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP + } + + // 设置采样率为16kHz(Azure Speech API推荐) + try audioSession.setPreferredSampleRate(16000.0) + + // 设置较小的缓冲区大小以减少延迟 + try audioSession.setPreferredIOBufferDuration(0.01) + + // 根据是否有耳机连接调整输入增益 + if !hasHeadphones { + // 无耳机时降低输入增益以减少回音 + try audioSession.setInputGain(0.8) + NSLog("[AudioSessionManager] 启用扬声器回音消除优化") + } else { + // 使用耳机时可以使用较高增益 + try audioSession.setInputGain(1.0) + NSLog("[AudioSessionManager] 检测到耳机连接,应用耳机模式") + } + } + + NSLog("[AudioSessionManager] 已成功配置Azure语音识别的音频会话") + return true + } catch { + NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话失败: \(error.localizedDescription)") + return false + } + } + + +@objc func configureForVoiceInteraction(force: Bool = false, configureAdditionalSettings: Bool = true) -> Bool { + NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话") + do { + // 配置音频会话以优化语音识别 + try configureSession( + category: .playAndRecord, // 允许同时录音和播放 + mode: .voiceChat, // 使用voiceChat模式获得最佳回音消除效果 options: [ - .allowBluetoothA2DP, // 允许通过蓝牙A2DP连接进行高质量音频输出 - // .allowBluetooth, // 允许通过蓝牙SCO连接进行输入和输出 - // .defaultToSpeaker, // 默认使用扬声器输出 - // .mixWithOthers + .allowBluetooth, // 允许蓝牙设备 + .defaultToSpeaker, // 默认使用扬声器 + .mixWithOthers // 允许与其他应用混音 ], - force: force // 强制重新配置,确保设置生效 + force: force // 是否强制重新配置 ) - // 设置首选输入设备 - // try configurePreferredInput() + // 额外的优化配置 + if configureAdditionalSettings { + // 获取当前是否使用耳机 + let currentRoute = audioSession.currentRoute + let hasHeadphones = currentRoute.outputs.contains { + $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP + } + + // 设置采样率为16kHz(Azure Speech API推荐) + try audioSession.setPreferredSampleRate(16000.0) + + // 设置较小的缓冲区大小以减少延迟 + try audioSession.setPreferredIOBufferDuration(0.01) + + // 根据是否有耳机连接调整输入增益 + if !hasHeadphones { + // 无耳机时降低输入增益以减少回音 + try audioSession.setInputGain(0.8) + NSLog("[AudioSessionManager] 启用扬声器回音消除优化") + } else { + // 使用耳机时可以使用较高增益 + try audioSession.setInputGain(1.0) + NSLog("[AudioSessionManager] 检测到耳机连接,应用耳机模式") + } + } - NSLog("[AudioSessionManager] 已配置语音交互(ASR+TTS)的音频会话") + NSLog("[AudioSessionManager] 已成功配置Azure语音识别的音频会话") return true } catch { - NSLog("[AudioSessionManager] 配置语音交互(ASR+TTS)的音频会话失败: \(error.localizedDescription)") + NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话失败: \(error.localizedDescription)") return false } } + + /// 配置用于语音交互(同时支持ASR和TTS)的音频会话 + /// 此函数综合优化语音识别和语音合成,适用于需要双向交互的场景 + /// - Returns: 配置是否成功 + // @objc func configureForVoiceInteraction(force: Bool = false) -> Bool { + // do { + // // 配置音频会话以支持语音交互 + // // 使用playAndRecord类别允许同时录音和播放 + // // 使用spokenAudio模式优化语音交互 + // try configureSession( + // category: .playAndRecord, + // mode: .voiceChat, + // options: [ + // // .allowBluetoothA2DP, // 允许通过蓝牙A2DP连接进行高质量音频输出 + // .allowBluetooth, // 允许通过蓝牙SCO连接进行输入和输出 + // .defaultToSpeaker, // 默认使用扬声器输出 + // .mixWithOthers + // ], + // force: force // 强制重新配置,确保设置生效 + // ) + + // // 设置首选输入设备 + // try configurePreferredInput() + + // NSLog("[AudioSessionManager] 已配置语音交互(ASR+TTS)的音频会话") + // return true + // } catch { + // NSLog("[AudioSessionManager] 配置语音交互(ASR+TTS)的音频会话失败: \(error.localizedDescription)") + // return false + // } + // } + } \ No newline at end of file diff --git a/ios/Runner/AzureAsrHelper.swift b/ios/Runner/AzureAsrHelper.swift index 652a57528..e43850165 100644 --- a/ios/Runner/AzureAsrHelper.swift +++ b/ios/Runner/AzureAsrHelper.swift @@ -1,6 +1,7 @@ import Foundation import MicrosoftCognitiveServicesSpeech import AVFoundation +import AudioToolbox /// Azure ASR工具类,负责实现语音识别服务接口 @available(iOS 13.0, *) @@ -19,6 +20,12 @@ class AzureAsrHelper: NSObject { private var recognizer: SPXSpeechRecognizer? private var audioConfig: SPXAudioConfiguration? + /// 添加自定义音频处理相关 + private var pushStream: SPXPushAudioInputStream? + private var audioProcessor: CustomAudioProcessor? + private var isProcessingAudio = false + private var audioProcessingTimer: Timer? + /// 状态标志 private var isInitialized = false private var _isContinuousRecognitionActive = false @@ -108,8 +115,12 @@ class AzureAsrHelper: NSObject { // 设置音频输入参数 try setupAudioSession() - // 直接使用麦克风音频配置 - audioConfig = try SPXAudioConfiguration() + // 创建自定义音频流和处理器,替代默认的麦克风输入 + pushStream = try SPXPushAudioInputStream() + audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) + + // 初始化自定义音频处理器 + audioProcessor = CustomAudioProcessor() // 设置语言配置 if isAutoDetectLanguage { @@ -147,11 +158,15 @@ class AzureAsrHelper: NSObject { /// 设置音频会话 private func setupAudioSession() throws { print("[AzureAsrHelper] 开始配置音频会话...") - - audioSessionManager.configureForVoiceInteraction() - + + // 使用AudioSessionManager配置音频会话,使用专门为Azure ASR优化的配置 + let success = audioSessionManager.configureForAzureSpeechRecognition(force: true) + if !success { + print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败") + } } + /// 设置所有回调 private func setupAllCallbacks() { guard let recognizer = recognizer else { return } @@ -244,6 +259,9 @@ class AzureAsrHelper: NSObject { } do { + // 启动音频处理 + startAudioProcessing() + // 通知会话开始 eventHandler?(["type": "sessionStarted"]) @@ -251,6 +269,9 @@ class AzureAsrHelper: NSObject { try recognizer?.recognizeOnceAsync { [weak self] result in guard let self = self else { return } + // 停止音频处理 + self.stopAudioProcessing() + if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) self.eventHandler?(["type": "result", @@ -276,6 +297,7 @@ class AzureAsrHelper: NSObject { } catch { print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") eventHandler?(["type": "error", "message": "识别异常: \(error.localizedDescription)"]) + stopAudioProcessing() return false } } @@ -311,7 +333,9 @@ class AzureAsrHelper: NSObject { return false } } - + + // 启动音频处理 + startAudioProcessing() // 尝试启动连续识别 do { @@ -319,13 +343,15 @@ class AzureAsrHelper: NSObject { try recognizer?.startContinuousRecognition() _isContinuousRecognitionActive = true - print("[AzureAsrHelper] 连续识别已启动") return true } catch { _isContinuousRecognitionActive = false print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") + // 停止音频处理 + stopAudioProcessing() + // 发送错误通知 eventHandler?(["type": "error", "message": "开始连续识别失败: \(error.localizedDescription)"]) @@ -337,6 +363,9 @@ class AzureAsrHelper: NSObject { /// 停止连续语音识别 /// - Returns: 是否成功停止识别 func stopContinuousRecognition() -> Bool { + // 停止音频处理 + stopAudioProcessing() + // 检查是否初始化 if !isInitialized { print("[AzureAsrHelper] 错误: 语音服务未初始化") @@ -362,14 +391,7 @@ class AzureAsrHelper: NSObject { // 先标记为非活跃状态,防止重复调用 _isContinuousRecognitionActive = false - // do { - // try recognizer.stopContinuousRecognition() - - // } catch { - // print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") - // } - - // // 异步执行停止操作,避免阻塞主线程 + // 异步执行停止操作,避免阻塞主线程 DispatchQueue.global(qos: .userInitiated).async { [weak self] in guard let self = self else { return } @@ -405,6 +427,9 @@ class AzureAsrHelper: NSObject { /// 释放资源 func dispose() { + // 停止音频处理 + stopAudioProcessing() + // 尝试停止所有识别操作 if _isContinuousRecognitionActive { do { @@ -422,6 +447,8 @@ class AzureAsrHelper: NSObject { recognizer = nil audioConfig = nil speechConfig = nil + pushStream = nil + audioProcessor = nil // 重置状态 isInitialized = false @@ -448,4 +475,320 @@ class AzureAsrHelper: NSObject { @objc func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) { self.eventHandler = handler } + + // MARK: - 音频处理 + + /// 开始音频处理 + private func startAudioProcessing() { + guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } + + isProcessingAudio = true + + // 启动音频处理器 + if !audioProcessor.startRecord() { + print("[AzureAsrHelper] 错误: 启动音频处理器失败") + eventHandler?(["type": "error", "message": "启动音频处理器失败"]) + return + } + + // 启动音频处理定时器 + audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in + guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { + return + } + + // 读取处理后的音频数据 + var bytes = [UInt8](repeating: 0, count: 2560) + let bytesRead = processor.read(bytes: &bytes) + + if bytesRead > 0 { + // 推送数据到Azure语音服务 + let data = Data(bytes: bytes, count: bytesRead) + stream.write(data) + + // 通知音频数据可用(可选) + // self.eventHandler?(["type": "audioData", "data": bytes]) + } + } + + print("[AzureAsrHelper] 音频处理已启动") + } + + /// 停止音频处理 + private func stopAudioProcessing() { + // 停止定时器 + audioProcessingTimer?.invalidate() + audioProcessingTimer = nil + + // 停止音频处理器 + audioProcessor?.stopRecord() + + isProcessingAudio = false + print("[AzureAsrHelper] 音频处理已停止") + } +} + +// MARK: - 自定义音频处理器 + +@available(iOS 13.0, *) +class CustomAudioProcessor: NSObject { + // 音频单元 + private var ioUnit: AudioUnit? + + // 音频格式 + private var audioFormat: AudioStreamBasicDescription + + // 音频缓冲 + private var audioBufferList: AudioBufferList + private var audioList: [Float] = [] + private let audioListQueue = DispatchQueue(label: "audioListQueue") + + // 回音消除状态 + private var isEchoCancellationEnabled = true + + override init() { + // 设置音频格式 - 16kHz, 16位, 单声道 + audioFormat = AudioStreamBasicDescription( + mSampleRate: 16000.0, + mFormatID: kAudioFormatLinearPCM, + mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, + mBytesPerPacket: 2, + mFramesPerPacket: 1, + mBytesPerFrame: 2, + mChannelsPerFrame: 1, + mBitsPerChannel: 16, + mReserved: 0 + ) + + // 初始化音频缓冲 + audioBufferList = AudioBufferList( + mNumberBuffers: 1, + mBuffers: AudioBuffer( + mNumberChannels: 1, + mDataByteSize: 4096, + mData: malloc(4096) + ) + ) + + super.init() + } + + deinit { + stopRecord() + free(audioBufferList.mBuffers.mData) + } + + /// 启动音频处理 + /// - Returns: 是否成功启动 + func startRecord() -> Bool { + print("[CustomAudioProcessor] 配置音频单元") + + // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 + var ioUnitDescription = AudioComponentDescription( + componentType: kAudioUnitType_Output, + componentSubType: kAudioUnitSubType_VoiceProcessingIO, + componentManufacturer: kAudioUnitManufacturer_Apple, + componentFlags: 0, + componentFlagsMask: 0 + ) + + // 查找音频组件 + guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { + print("[CustomAudioProcessor] 错误: 未找到音频组件") + return false + } + + // 创建音频单元实例 + if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { + ioUnit = nil + return false + } + + // 启用输入端口 + var enableInput: UInt32 = 1 + let kInputBus: AudioUnitElement = 1 + let kOutputBus: AudioUnitElement = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, + kAudioUnitScope_Input, kInputBus, &enableInput, + UInt32(MemoryLayout.size)), "启用输入端口") { + return false + } + + // 禁用输出端口 (我们只需要输入) + var enableOutput: UInt32 = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, + kAudioUnitScope_Output, kOutputBus, + &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") { + return false + } + + // 设置缓冲区分配标志 + var flag: UInt32 = 0 + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, + kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") { + return false + } + + // 设置音频格式 + let size = UInt32(MemoryLayout.size) + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, + kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { + return false + } + + if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, + kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { + return false + } + + // 启用回音消除 - 注意: kAUVoiceIOProperty_BypassVoiceProcessing值为1时表示绕过处理,值为0表示启用处理 + if isEchoCancellationEnabled { + var echoCancellation: UInt32 = 0 // 0表示不绕过,即启用回音消除 + AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, + kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size)) + } + + // 设置输入回调 - 当有新音频数据时调用 + var inputCallback = AURenderCallbackStruct( + inputProc: CustomAudioProcessor.onAudioDataAvailable, + inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) + ) + + if checkError(AudioUnitSetProperty(ioUnit!, + kAudioOutputUnitProperty_SetInputCallback, + kAudioUnitScope_Global, kInputBus, + &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { + return false + } + + // 初始化音频单元 + var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") + while hasError { + Thread.sleep(forTimeInterval: 0.1) + hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") + } + + // 启动音频单元 + hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") + + print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") + return !hasError + } + + /// 停止音频处理 + func stopRecord() { + print("[CustomAudioProcessor] 停止音频处理器") + + if let ioUnit = ioUnit { + // 停止音频单元 + _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") + + // 关闭音频单元 + _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") + _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") + + self.ioUnit = nil + } + + // 清空音频数据缓冲 + audioListQueue.sync { + audioList.removeAll() + } + } + + /// 音频数据回调 - 当有新的音频数据可用时调用 + private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in + // 获取实例 + let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() + + // 计算预期数据大小 + let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame + + // 确保缓冲区足够大 + if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { + processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) + processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize + } + + // 渲染音频数据 + let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, + inBusNumber, inNumberFrames, &processor.audioBufferList), + "渲染音频数据") + + // 将Int16数据转换为浮点数据进行处理 + var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) + let buffer = processor.audioBufferList.mBuffers + let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) + + for j in 0...size)) { + // 归一化到[-1.0, 1.0]范围 + audioDataFloat[j] = Float(bufferData[j]) / 32768.0 + } + + // 保存处理后的数据 + if status == noErr { + processor.audioListQueue.async { + processor.audioList.append(contentsOf: audioDataFloat) + } + } + + return status + } + + /// 读取处理后的音频数据 + /// - Parameter bytes: 输出字节数组 + /// - Returns: 读取的字节数 + func read(bytes: inout [UInt8]) -> Int { + return audioListQueue.sync { + // 如果没有数据,返回0 + if audioList.isEmpty { + return 0 + } + + // 确保有足够的数据 (至少1280个样本) + if audioList.count < 1280 { + return 0 + } + + // 读取一帧数据 (1280个样本) + let frameLength = 1280 + let buffer = Array(audioList.prefix(frameLength)) + audioList.removeFirst(frameLength) + + // 将浮点数据转回Int16格式 + var int16Data = buffer.map { Int16($0 * 32767) } + + // 转换为字节数组 + let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) + bytes = [UInt8](data) + + // 每个样本2字节 (16位PCM) + return frameLength * 2 + } + } + + /// 检查错误并打印日志 + /// - Parameters: + /// - status: 操作状态 + /// - operation: 操作描述 + /// - Returns: 是否发生错误 + private func checkError(_ status: OSStatus, _ operation: String) -> Bool { + if status != noErr { + print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") + return true + } + return false + } + + /// 检查OSStatus并返回状态 + /// - Parameters: + /// - status: 操作状态 + /// - operation: 操作描述 + /// - Returns: 原始状态 + private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { + if status != noErr { + print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") + } + return status + } } \ No newline at end of file diff --git a/ios/Runner/ClassicBluetoothHelper.swift b/ios/Runner/ClassicBluetoothHelper.swift index 2011670c5..2694075e0 100644 --- a/ios/Runner/ClassicBluetoothHelper.swift +++ b/ios/Runner/ClassicBluetoothHelper.swift @@ -80,7 +80,7 @@ import UIKit NotificationCenter.default.removeObserver(self) // 添加新的监听器 audioSessionManager.addRouteChangeListener(self, selector: #selector(handleRouteChange(_:))) - BluetoothMediaButtonHelper.shared.startButtonListening() + // BluetoothMediaButtonHelper.shared.startButtonListening() } // 处理音频路由变化