From d9d87d07072f34d2f5fcd48a420ad425b49b62f1 Mon Sep 17 00:00:00 2001 From: fdp <1286779656@qq.com> Date: Sat, 23 Aug 2025 14:06:47 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BF=AE=E5=A4=8D=E9=BA=A6=E5=85=8B=E9=A3=8E?= =?UTF-8?q?=E6=97=B6=E5=A4=A7=E6=97=B6=E5=B0=8F=E7=9A=84=E9=97=AE=E9=A2=98?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Sources/azure_speech/AzureAsrHelper.swift | 22 +- .../Sources/azure_speech/AzureTtsHelper.swift | 26 +- .../Sources/tools/SimpleAudioReceiver.swift | 453 ++++++++++-------- 3 files changed, 263 insertions(+), 238 deletions(-) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index f5d2cf052..0ee8b5c0a 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -206,12 +206,11 @@ public class AzureAsrHelper: NSObject { audioStream = SimpleAudioReceiver() audioStream?.initAudioRecord() } - // 预创建音频配置 - if let pushStream = audioStream?.pushAudioStream { + if let pushStream = audioStream?.pushAudioStream { audioConfig = SPXAudioConfiguration(streamInput: pushStream) os_log("预初始化音频组件完成", log: log, type: .info) - } + } } @@ -465,12 +464,17 @@ public class AzureAsrHelper: NSObject { * 设置麦克风音频流 */ private func setupMicrophoneStream() { - audioStream = SimpleAudioReceiver() - // 【优化】检查音频配置是否已存在 - if audioConfig == nil, let pushStream = audioStream?.pushAudioStream { - audioConfig = SPXAudioConfiguration(streamInput: pushStream) - os_log("设置麦克风流完成", log: log, type: .info) - } + // 只有在 audioStream 为 nil 时才创建新实例 + if audioStream == nil { + audioStream = SimpleAudioReceiver() + audioStream?.initAudioRecord() // 确保调用初始化 + } + + // 检查音频配置是否已存在 + if audioConfig == nil, let pushStream = audioStream?.pushAudioStream { + audioConfig = SPXAudioConfiguration(streamInput: pushStream) + os_log("设置麦克风流完成", log: log, type: .info) + } } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index 86ac66bb9..225bf71c2 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -319,31 +319,7 @@ public class AzureTtsHelper: NSObject, ITtsService { return true } - // /** - // * 设置音频输出设备 - // */ - // public func setAudioOutputDevice(_ device: AudioOutputDevice) -> Bool { - // do { - // let session = AVAudioSession.sharedInstance() - // try session.setCategory(.playAndRecord, options: [.defaultToSpeaker, .allowBluetooth]) - // - // switch device { - // case .default: - // try session.overrideOutputAudioPort(.none) - // case .speaker: - // try session.overrideOutputAudioPort(.speaker) - // case .headphones: - // try session.overrideOutputAudioPort(.none) - // } - // - // try session.setActive(true) - // return true - // } catch { - // os_log("设置音频输出设备失败: %{public}@", log: log, type: .error, error.localizedDescription) - // return false - // } - // } - // + // MARK: - 私有方法 /** diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift index 0cf7951ae..710f7a1e2 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift @@ -54,11 +54,28 @@ public class SimpleAudioReceiver: NSObject { // 初始化音频引擎 audioEngine = AVAudioEngine() + isRunning = true // 创建新的写线程 writeThread = DispatchQueue(label: "audio.stream.writer") startWriteThread() + // 配置音频引擎 + do { + try startCapture { buffer in + // 处理实时音频数据 + let channelData = buffer.floatChannelData?[0] + let frameCount = buffer.frameLength + print("麦克风启动: \(frameCount)") + print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") + self.writeQueue.put(self.audioBufferToData(buffer)) + // 分析或传输数据... + } + + } catch { + print("麦克风启动失败: \(error)") + + } } /// 获取最佳音频格式 (iOS 通常支持标准采样率) @@ -145,6 +162,7 @@ public class SimpleAudioReceiver: NSObject { print("写入数据长度: \(dataToWrite.count)") self.audioDataCallback!.onAudio(dataToWrite) } + print("写入数据长度: \(dataToWrite.count)") try self.pushAudioStream?.write(dataToWrite) if self.recordfile != nil { // print("写入recordfile数据长度: \(dataToWrite.count)") @@ -173,107 +191,113 @@ public class SimpleAudioReceiver: NSObject { // 放入队列,由写线程写入 writeQueue.put(data) } + + /** + * 开始采集音频数据 + * @param bufferHandler 音频缓冲区处理回调 + * @throws 音频配置或启动错误 + */ + func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { + try setupAudioSession() + + guard let audioEngine = audioEngine else { + throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) + } + + // 确保音频引擎完全停止 + if audioEngine.isRunning { + audioEngine.stop() + // 等待一小段时间确保完全停止 + Thread.sleep(forTimeInterval: 0.1) + } + + // 先安全地移除已存在的 tap + removeTapSafely() + + // 重置音频引擎(在获取格式之前) + audioEngine.reset() + + // 重新获取输入节点和格式(reset后需要重新获取) + let inputNode = audioEngine.inputNode + let inputFormat = inputNode.outputFormat(forBus: 0) + + print("硬件输入格式: 采样率=\(inputFormat.sampleRate)Hz, 声道数=\(inputFormat.channelCount), 格式=\(inputFormat.commonFormat.rawValue)") + + // 验证格式是否有效 + if inputFormat.sampleRate <= 0 || inputFormat.channelCount <= 0 { + // 如果格式无效,使用默认格式 + let defaultFormat = AVAudioFormat(standardFormatWithSampleRate: 48000, channels: 1)! + print("使用默认音频格式: 采样率=\(defaultFormat.sampleRate)Hz, 声道数=\(defaultFormat.channelCount)") + + inputNode.installTap(onBus: 0, + bufferSize: 1024, + format: defaultFormat) { (buffer, time) in + bufferHandler(buffer) + } + } else { + // 使用硬件原生格式 + inputNode.installTap(onBus: 0, + bufferSize: 1024, + format: inputFormat) { (buffer, time) in + bufferHandler(buffer) + } + } + + if #available(iOS 13.0, *) { + try inputNode.setVoiceProcessingEnabled(true) + } + + audioEngine.prepare() + + + print("音频引擎启动成功") + } - // 私有方法:启动麦克风捕获 /** - * 启动麦克风捕获 - * 优化:复用已初始化的音频引擎,减少启动延迟 + * 安全地移除音频tap + * 避免在移除不存在的tap时出错 */ + private func removeTapSafely() { + guard let audioEngine = audioEngine else { return } + + let inputNode = audioEngine.inputNode + + // 尝试移除tap,如果没有tap也不会出错 + do { + inputNode.removeTap(onBus: 0) + print("成功移除音频tap") + } catch { + // 如果没有tap可移除,这是正常的 + print("移除tap时出现错误(可能没有tap存在): \(error)") + } + } private func runMicrophoneCapture() { do { - // 【新增】首先配置音频会话 - try audioSession.setCategory( - .playAndRecord, - mode: .measurement, // 使用 measurement 模式获得最佳录音质量 - options: [.allowBluetooth, .defaultToSpeaker] - ) - - // 【新增】请求麦克风权限(如果尚未授权) - if audioSession.recordPermission != .granted { - audioSession.requestRecordPermission { granted in - if !granted { - print("麦克风权限被拒绝") - } - } - } - - // 【新增】激活音频会话(在配置音频引擎之前) - try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) + print("🎤 开始配置麦克风捕获") - // 检查音频引擎是否已初始化,避免重复创建 if audioEngine == nil { audioEngine = AVAudioEngine() + print("🔧 创建新的音频引擎") } // 如果音频引擎正在运行,先停止 if audioEngine?.isRunning == true { audioEngine?.stop() + print("⏹️ 停止现有音频引擎") } - // 获取音频输入节点(麦克风) - guard let inputNode = audioEngine?.inputNode else { - throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"]) - } - - // 移除之前的音频处理块,避免重复添加 - inputNode.removeTap(onBus: 0) - - // 获取硬件支持的原始音频格式 - let hardwareFormat = inputNode.inputFormat(forBus: 0) - - // iOS 13+ 启用语音处理 - if #available(iOS 13.0, *) { - try inputNode.setVoiceProcessingEnabled(true) - } - - // 检查目标音频格式和转换器是否可用 - guard let targetFormat = audioFormat, // 外部定义的期望音频格式 - let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { - throw NSError(domain: "AudioSetup", code: 2) - } - - // 在输入节点上安装录音回调 - inputNode.installTap(onBus: 0, - bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小 - format: hardwareFormat) { // 使用原始硬件格式 - [weak self] buffer, time in // 弱引用避免循环引用 - - // 确保实例存在且正在写入状态 - guard let self = self, self._isWriting else { return } - - // 创建目标格式的音频缓冲区 - let convertedBuffer = AVAudioPCMBuffer( - pcmFormat: targetFormat, - // 计算转换后的帧容量(考虑采样率差异) - frameCapacity: AVAudioFrameCount( - targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate - ) - )! - - var error: NSError? - // 执行音频格式转换 - let status = converter.convert( - to: convertedBuffer, - error: &error, - withInputFrom: { inNumPackets, outStatus in - outStatus.pointee = .haveData // 标记有数据可用 - return buffer // 返回原始音频数据 - } - ) - - // 转换成功且无错误 - if status == .haveData, error == nil { - // 将音频缓冲区转换为二进制数据 - let data = self.audioBufferToData(convertedBuffer) - // 将数据放入写入队列(后续处理) - self.writeQueue.put(data) - } - } - - // 激活音频会话(允许录音) - try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) + // try startCapture { buffer in + // // 处理实时音频数据 + // let channelData = buffer.floatChannelData?[0] + // let frameCount = buffer.frameLength + // print("麦克风启动: \(frameCount)") + // print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") + // self.writeQueue.put(self.audioBufferToData(buffer)) + // // 分析或传输数据... + // } + try audioEngine?.start() + print("✅ 麦克风捕获配置完成") - try audioEngine?.start() } catch { print("麦克风启动失败: \(error)") // 【新增】添加详细错误处理 @@ -283,38 +307,7 @@ public class SimpleAudioReceiver: NSObject { } } } - private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { - let frameLength = Int(buffer.frameLength) - let channelCount = 1 - let dataLength = frameLength * channelCount * MemoryLayout.size - - // Handle 16-bit integer format - if let int16Data = buffer.int16ChannelData { - return Data( - bytes: int16Data.pointee, - count: dataLength - ) - } - // Handle float format - else if let floatData = buffer.floatChannelData { - var int16Array = [Int16](repeating: 0, count: frameLength) - let floatBuffer = floatData.pointee - - for i in 0.. Data { + guard let floatChannelData = buffer.floatChannelData else { + return Data() } - public func setAudioOutputRoute(_ route: AudioOutputRoute) { + + let frameLength = Int(buffer.frameLength) + let channelCount = Int(buffer.format.channelCount) + let inputSampleRate = buffer.format.sampleRate + let targetSampleRate: Double = 16000 + + // 统一处理:只使用第一个声道(单声道),确保一致性 + let firstChannelData = floatChannelData[0] + + // 如果采样率不是16kHz,进行降采样 + if inputSampleRate != targetSampleRate { + let ratio = inputSampleRate / targetSampleRate + let outputFrameCount = Int(Double(frameLength) / ratio) - do { - // - if self._isWriting { - print("当前正在识别") - try audioEngine?.pause() - // 先停用以避免冲突 - try audioSession.setActive(false) - - } - print("setAudioOutputRoutecurrent,route=\(route)") - switch route { - case .speaker: - print("setAudioOutputRoutecurrent,speaker") - // 使用扬声器时必须用videoChat模式 - try audioSession.setCategory( - .playAndRecord, - mode: .videoChat, - options: [.defaultToSpeaker] - ) - try audioSession.overrideOutputAudioPort(.speaker) + var outputData = Data() + outputData.reserveCapacity(outputFrameCount * 2) // 16位 = 2字节 + + for outputIndex in 0.. \(targetSampleRate)Hz, 帧数: \(frameLength) -> \(outputFrameCount)") + return outputData + } else { + // 采样率已经是16kHz,直接转换格式(只处理第一个声道) + var data = Data() + data.reserveCapacity(frameLength * 2) // 单声道,16位 + + for frame in 0.. { } } -// MARK: - 协议定义 -