From 15deebb95cf47ba19658dc32ca990c830fc56b59 Mon Sep 17 00:00:00 2001 From: fdp <1286779656@qq.com> Date: Sun, 24 Aug 2025 21:53:56 +0800 Subject: [PATCH] =?UTF-8?q?=E8=BF=98=E5=8E=9F=E5=8E=9F=E6=9C=AC=E7=9A=84?= =?UTF-8?q?=E9=9F=B3=E6=BA=90=E7=AE=A1=E6=8E=A7=E9=97=AE=E9=A2=98?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../Sources/azure_speech/AzureAsrHelper.swift | 2 - .../IntegratedSpeechTranslationService.swift | 5 +- .../Sources/tools/SimpleAudioReceiver.swift | 591 +++++++----------- .../realtime/RealtimeAudioManager.swift | 15 +- 4 files changed, 254 insertions(+), 359 deletions(-) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index 4625666f6..f3fa241ce 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -204,7 +204,6 @@ public class AzureAsrHelper: NSObject { private func preInitializeAudioComponents() { if audioStream == nil { audioStream = SimpleAudioReceiver.shared - //audioStream?.initAudioRecord() } // 预创建音频配置 if let pushStream = audioStream?.pushAudioStream { @@ -670,7 +669,6 @@ public class AzureAsrHelper: NSObject { if audioStream == nil { // 创建外部音频拉流对象 audioStream = SimpleAudioReceiver.shared - //audioStream?.initAudioRecord() } if audioStream?.recordfile == nil { diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift index bb117738a..bf1feed33 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift @@ -330,7 +330,6 @@ import os.log */ private func initializeAudioProcessor() { audioProcessor = SimpleAudioReceiver.shared - // audioProcessor?.initAudioRecord() // 检查音频配置是否已存在 if audioConfig == nil, let pushStream = audioProcessor?.pushAudioStream { audioConfig = SPXAudioConfiguration(streamInput: pushStream) @@ -913,8 +912,8 @@ import os.log audioConfig = nil // 释放其他组件 - //audioProcessor?.releaseAudioResources() - //audioProcessor = nil + // audioProcessor?.releaseAudioResources() + // audioProcessor = nil translationService?.dispose() translationService = nil diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift index 439a3cf83..dbe76a6e6 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift @@ -6,15 +6,20 @@ import os.log /** * 简单音频接收器类,用于处理音频录制和流传输 * 对应Android的SimpleAudioReceiver功能 + * 使用单例模式确保全局唯一实例 */ public class SimpleAudioReceiver: NSObject { - /// 单例实例(线程安全) - /// 通过 SimpleAudioReceiver.shared 获取全局唯一实例 + // MARK: - 单例实现 + + /** + * 单例实例 + */ public static let shared = SimpleAudioReceiver() - /// 私有化构造函数,防止外部直接实例化 - /// 使用方式:通过 SimpleAudioReceiver.shared 访问实例 + /** + * 私有初始化方法,防止外部创建实例 + */ private override init() { super.init() initAudioRecord() @@ -22,19 +27,7 @@ public class SimpleAudioReceiver: NSObject { private let tag = "SimpleAudioReceiver" private let log = OSLog(subsystem: "com.azure.speech", category: "SimpleAudioReceiver") - /// 音频格式转换专用队列(避免在音频渲染线程上做重操作) - private let conversionQueue = DispatchQueue(label: "audio.stream.convert", qos: .userInitiated) - - /// 缓存的音频转换器,避免每个缓冲区重复创建 - private var cachedConverter: AVAudioConverter? - /// 统一的目标格式(16kHz/单声道/Float32,用于后续再转 Int16) - private lazy var targetFormat16000Mono: AVAudioFormat = { - return AVAudioFormat(commonFormat: .pcmFormatFloat32, - sampleRate: 16000, - channels: 1, - interleaved: false)! - }() /** * 音频来源类型 */ @@ -48,7 +41,7 @@ public class SimpleAudioReceiver: NSObject { // MARK: - 音频流类 public private(set) var pushAudioStream: SPXPushAudioInputStream? - private var writeQueue = LinkedBlockingQueue() + private let writeQueue = LinkedBlockingQueue() private var audioEngine: AVAudioEngine? private var audioFormat: AVAudioFormat? private let audioSession = AVAudioSession.sharedInstance() @@ -69,10 +62,14 @@ public class SimpleAudioReceiver: NSObject { /** * 初始化音频录制组件 * 包括音频格式、推流、音频引擎等核心组件的初始化 + * 注意:由于是单例模式,此方法可以被多次调用但只会初始化一次 */ public func initAudioRecord() { - // 每次初始化前,先释放上一次的资源,避免重复引擎/线程/tap - releaseAudioResources() + // 防止重复初始化 + guard audioEngine == nil else { + print("音频录制组件已初始化,跳过重复初始化") + return + } print("初始化了") audioFormat = getOptimalAudioFormat() @@ -85,20 +82,6 @@ public class SimpleAudioReceiver: NSObject { // 创建新的写线程 writeThread = DispatchQueue(label: "audio.stream.writer") startWriteThread() - // 配置音频引擎 - do { - try startCapture { buffer in - // 处理实时音频数据 - let channelData = buffer.floatChannelData?[0] - let frameCount = buffer.frameLength - // print("麦克风启动: \(frameCount)") - // print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") - self.writeQueue.put(self.audioBufferToDataOptimized(buffer)) - // 分析或传输数据... - } - } catch { - print("麦克风启动失败: \(error)") - } } /// 获取最佳音频格式 (iOS 通常支持标准采样率) @@ -185,7 +168,6 @@ public class SimpleAudioReceiver: NSObject { print("写入数据长度: \(dataToWrite.count)") self.audioDataCallback!.onAudio(dataToWrite) } - print("写入数据长度: \(dataToWrite.count)") try self.pushAudioStream?.write(dataToWrite) if self.recordfile != nil { // print("写入recordfile数据长度: \(dataToWrite.count)") @@ -214,381 +196,262 @@ public class SimpleAudioReceiver: NSObject { // 放入队列,由写线程写入 writeQueue.put(data) } - -/** - * 开始音频捕获 - * 1) 重置音频引擎并配置音频会话 - * 2) 安装 tap 时不指定格式(format: nil),由系统使用硬件实际格式 - * 3) 在回调内将硬件格式转换为统一的 16kHz 单声道后回调上层 - * @param bufferHandler 音频缓冲区处理回调(已转换为目标格式) - * @throws 音频引擎启动失败时抛出错误 - */ -func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { - guard let audioEngine = audioEngine else { - throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) - } - - // 确保音频引擎完全停止 - if audioEngine.isRunning { - audioEngine.stop() - // 等待一小段时间确保完全停止 - Thread.sleep(forTimeInterval: 0.1) - } - - // 先安全地移除已存在的 tap - removeTapSafely() - - // 重置音频引擎(在获取格式之前) - audioEngine.reset() - - // 配置音频会话 - do { - try setupAudioSession() - } catch { - print("音频会话设置失败: \(error.localizedDescription)") - throw error - } - - // 重新获取输入节点(此处获取的输出格式在未启动时可能为 0Hz,仅用于日志) - let inputNode = audioEngine.inputNode - let preFormat = inputNode.outputFormat(forBus: 0) - print("当前节点(启动前)输出格式: 采样率=\(preFormat.sampleRate)Hz, 声道数=\(preFormat.channelCount), 格式=\(preFormat.commonFormat.rawValue)") - - // 目标格式 - 统一使用16000Hz单声道(Float32),后续再转 Int16 - let targetFormat = self.targetFormat16000Mono - print("使用目标音频格式: 采样率=\(targetFormat.sampleRate)Hz, 声道数=\(targetFormat.channelCount)") - - // 使用硬件原生格式安装 tap(format: nil),回调中异步转换 - inputNode.installTap(onBus: 0, - bufferSize: 1024, - format: nil) { [weak self] (buffer, time) in - guard let self = self else { return } - let srcFormat = buffer.format - - // 将重采样与格式转换放到专用队列,避免阻塞音频渲染线程 - self.conversionQueue.async { - // 如果源格式与目标格式不同,复用/重建转换器 - var outputBuffer: AVAudioPCMBuffer? = buffer - if srcFormat.sampleRate != targetFormat.sampleRate || - srcFormat.channelCount != targetFormat.channelCount || - srcFormat.commonFormat != targetFormat.commonFormat { - - if self.cachedConverter == nil || - self.cachedConverter?.inputFormat != srcFormat || - self.cachedConverter?.outputFormat != targetFormat { - self.cachedConverter = AVAudioConverter(from: srcFormat, to: targetFormat) - } - - if let converter = self.cachedConverter { - let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / srcFormat.sampleRate) - if let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) { - var error: NSError? - let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in - outStatus.pointee = .haveData - return buffer - } - if status == .error { - // 限制日志:仅在错误时打印,避免频繁输出 - print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") - return - } - outputBuffer = convertedBuffer - } else { - print("无法创建转换后的音频缓冲区") - return - } - } else { - print("无法创建音频转换器: 从\(srcFormat.sampleRate)Hz到\(targetFormat.sampleRate)Hz") - return - } - } - - // 回调上层(注意:此时不在实时渲染线程) - if let out = outputBuffer { - bufferHandler(out) - } - } - } - // 启用语音处理(如果支持) - enableVoiceProcessingIfAvailable(inputNode: inputNode) - - // 准备并启动音频引擎 - audioEngine.prepare() - // print("音频引擎启动成功。Tap使用硬件格式: 采样率=\(postFormat.sampleRate)Hz, 声道数=\(postFormat.channelCount)") - //print("音频引擎启动成功,使用格式: 采样率=\(tapFormat.sampleRate)Hz, 声道数=\(tapFormat.channelCount)") -} - - -/** - * 启用语音处理(如果设备支持) - * @param inputNode 输入节点 - */ -private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) { - if #available(iOS 13.0, *) { - do { - try inputNode.setVoiceProcessingEnabled(true) - print("语音处理已启用") - } catch { - print("启用语音处理失败: \(error.localizedDescription)") - // 不抛出错误,因为这不是关键功能 - } - } else { - print("当前iOS版本不支持语音处理") - } -} - /** - * 安全地移除音频tap - * 避免在移除tap时出现崩溃 - */ + // 私有方法:启动麦克风捕获 /** - * 安全地移除音频tap - * 避免在移除tap时出现崩溃 + * 启动麦克风捕获 + * 优化:复用已初始化的音频引擎,减少启动延迟 */ - private func removeTapSafely() { - guard let audioEngine = audioEngine else { return } - - let inputNode = audioEngine.inputNode - - // 检查是否有tap需要移除 - if inputNode.numberOfInputs > 0 { - do { - inputNode.removeTap(onBus: 0) - print("成功移除音频tap") - } catch { - print("移除音频tap时出错: \(error.localizedDescription)") - } - } - } - // 修改 runMicrophoneCapture 方法 private func runMicrophoneCapture() { do { - print("🎤 开始配置麦克风捕获") + // 【新增】首先配置音频会话 + try audioSession.setCategory( + .playAndRecord, + mode: .videoChat, + options: [.defaultToSpeaker,.allowBluetooth,.mixWithOthers] + ) + // 【新增】请求麦克风权限(如果尚未授权) + if audioSession.recordPermission != .granted { + audioSession.requestRecordPermission { granted in + if !granted { + print("麦克风权限被拒绝") + } + } + } + + // 【新增】激活音频会话(在配置音频引擎之前) + try audioSession.setActive(true) + + // 检查音频引擎是否已初始化,避免重复创建 if audioEngine == nil { audioEngine = AVAudioEngine() - print("🔧 创建新的音频引擎") } + // 如果音频引擎正在运行,先停止 if audioEngine?.isRunning == true { audioEngine?.stop() - print("⏹️ 停止现有音频引擎") } - try startCapture { buffer in - // 将音频处理移到专用队列,避免阻塞音频线程 - self.conversionQueue.async { - let audioData = self.audioBufferToDataOptimized(buffer) - self.writeQueue.put(audioData) + // 获取音频输入节点(麦克风) + guard let inputNode = audioEngine?.inputNode else { + throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"]) + } + + // 移除之前的音频处理块,避免重复添加 + inputNode.removeTap(onBus: 0) + + // 获取硬件支持的原始音频格式 + let hardwareFormat = inputNode.inputFormat(forBus: 0) + + // iOS 13+ 启用语音处理 + if #available(iOS 13.0, *) { + try inputNode.setVoiceProcessingEnabled(true) + } + + // 检查目标音频格式和转换器是否可用 + guard let targetFormat = audioFormat, // 外部定义的期望音频格式 + let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { + throw NSError(domain: "AudioSetup", code: 2) + } + + // 在输入节点上安装录音回调 + inputNode.installTap(onBus: 0, + bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小 + format: hardwareFormat) { // 使用原始硬件格式 + [weak self] buffer, time in // 弱引用避免循环引用 + + // 确保实例存在且正在写入状态 + guard let self = self, self._isWriting else { return } + + // 创建目标格式的音频缓冲区 + let convertedBuffer = AVAudioPCMBuffer( + pcmFormat: targetFormat, + // 计算转换后的帧容量(考虑采样率差异) + frameCapacity: AVAudioFrameCount( + targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate + ) + )! + + var error: NSError? + // 执行音频格式转换 + let status = converter.convert( + to: convertedBuffer, + error: &error, + withInputFrom: { inNumPackets, outStatus in + outStatus.pointee = .haveData // 标记有数据可用 + return buffer // 返回原始音频数据 + } + ) + + // 转换成功且无错误 + if status == .haveData, error == nil { + // 将音频缓冲区转换为二进制数据 + let data = self.audioBufferToData(convertedBuffer) + // 将数据放入写入队列(后续处理) + self.writeQueue.put(data) } } - try audioEngine?.start() - print("✅ 麦克风捕获配置完成") + try audioEngine?.start() } catch { print("麦克风启动失败: \(error)") + // 【新增】添加详细错误处理 if let nsError = error as NSError? { print("错误域: \(nsError.domain), 错误代码: \(nsError.code)") print("错误描述: \(nsError.localizedDescription)") } } } - + private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { + let frameLength = Int(buffer.frameLength) + let channelCount = 1 + let dataLength = frameLength * channelCount * MemoryLayout.size + + // Handle 16-bit integer format + if let int16Data = buffer.int16ChannelData { + return Data( + bytes: int16Data.pointee, + count: dataLength + ) + } + // Handle float format + else if let floatData = buffer.floatChannelData { + var int16Array = [Int16](repeating: 0, count: frameLength) + let floatBuffer = floatData.pointee + + for i in 0..() - - // 再次安全地移除 tap(幂等) - removeTapSafely() - - // 重置状态 - audioEngine = nil - } -// MARK: - 协议定义 -/** - * 将音频缓冲区转换为Data格式 - * 统一处理采样率转换和格式转换 - * @param buffer 音频缓冲区 - * @return 转换后的音频数据 - */ -/** - * 优化的音频缓冲区转换方法 - * 减少内存分配,提高转换效率 - */ -// 在类属性中添加 -private var reusableDataBuffer: Data? -private let bufferReuseQueue = DispatchQueue(label: "buffer.reuse", qos: .userInitiated) - -/** - * 带缓冲区复用的音频转换方法 - */ -private func audioBufferToDataOptimized(_ buffer: AVAudioPCMBuffer) -> Data { - guard let floatChannelData = buffer.floatChannelData else { - return Data() } - let frameLength = Int(buffer.frameLength) - let requiredSize = frameLength * 2 - // 复用或创建缓冲区 - if reusableDataBuffer == nil || reusableDataBuffer!.count < requiredSize { - reusableDataBuffer = Data(count: requiredSize) - } - var outputData = reusableDataBuffer! - let firstChannelData = floatChannelData[0] - outputData.withUnsafeMutableBytes { rawBytes in - let int16Ptr = rawBytes.bindMemory(to: Int16.self) - - for frame in 0.. AVAudioPCMBuffer? { - guard let converter = AVAudioConverter(from: buffer.format, to: targetFormat) else { - print("无法创建音频转换器: 从\(buffer.format.sampleRate)Hz到\(targetFormat.sampleRate)Hz") - return nil - } - - // 计算转换后的帧数 - let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / buffer.format.sampleRate) - guard let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) else { - print("无法创建转换后的音频缓冲区") - return nil + public func stopMicrophoneCapture() { + guard _isWriting==true else { return } + + _isWriting=false + audioEngine?.pause() + print("stopMicrophoneCapture") + } - - var error: NSError? - let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in - outStatus.pointee = .haveData - return buffer + public func releaseAudioResources() { + print("释放了") + stopMicrophoneCapture() + audioEngine?.stop() + writeThread?.async { + self.writeQueue.close() + } + writeThread = nil + audioEngine?.inputNode.removeTap(onBus: 0) + isRunning=false + audioEngine = nil } - if status == .error { - print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") - return nil + // 音频路由管理 + public enum AudioOutputRoute { + case speaker + case receiver + case bluetooth } - - return convertedBuffer -} - - -/** - * 配置音频会话 - * 设置录音和播放所需的音频会话参数 - * @throws 音频会话配置失败时抛出错误 - */ -private func setupAudioSession() throws { - // 添加 .mixWithOthers 选项以支持同时播放和录音 - try audioSession.setCategory(.playAndRecord, - mode: .default, - options: [.defaultToSpeaker, .allowBluetooth, .mixWithOthers]) - - // 设置较低的音频延迟以减少卡顿 - try audioSession.setPreferredIOBufferDuration(0.005) // 5ms - - try audioSession.setActive(true) - - print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)") -} - - - /** - * 继续麦克风捕获 - */ - public func resumeRecord() { - guard _isWriting == false else { return } - _isWriting = true + public func setAudioOutputRoute(_ route: AudioOutputRoute) { - // 启动录音引擎 do { - print("resumeRecord") - - // 确保音频引擎存在且已准备好 - guard let audioEngine = audioEngine else { - print("Audio engine not initialized") - return + // + if self._isWriting { + print("当前正在识别") + try audioEngine?.pause() + // 先停用以避免冲突 + try audioSession.setActive(false) + } - - if !audioEngine.isRunning { - try audioEngine.start() + print("setAudioOutputRoutecurrent,route=\(route)") + switch route { + case .speaker: + print("setAudioOutputRoutecurrent,speaker") + // 使用扬声器时必须用videoChat模式 + try audioSession.setCategory( + .playAndRecord, + mode: .videoChat, + options: [.defaultToSpeaker,.allowBluetooth,.mixWithOthers] + ) + try audioSession.overrideOutputAudioPort(.speaker) + + case .receiver: + // 听筒模式使用voiceChat节省资源 + try audioSession.setCategory( + .playAndRecord, + mode: .voiceChat, + options: [.allowBluetooth] + ) + try audioSession.overrideOutputAudioPort(.none) + + case .bluetooth: + // 完整蓝牙设备支持 + try audioSession.setCategory( + .playAndRecord, + mode: .voiceChat, + options: [.allowBluetooth, .allowBluetoothA2DP] + ) + try audioSession.overrideOutputAudioPort(.none) + // 不需要override,系统自动路由 + } + self.currentRoute = route + if self._isWriting { + + // 重新激活 + try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) + + try audioEngine?.start() + // writeThread?.async { + // self.writeQueue.close() + // } } } catch { - os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription) - _isWriting = false + print("路由切换失败: \(error)") } } - // 音频路由管理 - public enum AudioOutputRoute { - case speaker - case receiver - case bluetooth - } - + /** * 音频数据回调接口 @@ -609,6 +472,25 @@ private func setupAudioSession() throws { return _isWriting } + /** + * 重置单例实例(用于测试或特殊情况下的重新初始化) + * 注意:调用此方法会释放所有音频资源并重置状态 + */ + public func resetInstance() { + releaseAudioResources() + + // 重置所有状态 + pushAudioStream = nil + audioEngine = nil + audioFormat = nil + audioSourceType = .microphone + audioDataCallback = nil + isRunning = false + _isWriting = false + writeThread = nil + currentRoute = nil + recordfile = nil + } } // MARK: - iOS版LinkedBlockingQueue实现 private class LinkedBlockingQueue { @@ -683,3 +565,6 @@ private class LinkedBlockingQueue { } } } + +// MARK: - 协议定义 + diff --git a/local_plugins/realtime/ios/realtime/Sources/realtime/RealtimeAudioManager.swift b/local_plugins/realtime/ios/realtime/Sources/realtime/RealtimeAudioManager.swift index 6309d7936..9277b9ea9 100644 --- a/local_plugins/realtime/ios/realtime/Sources/realtime/RealtimeAudioManager.swift +++ b/local_plugins/realtime/ios/realtime/Sources/realtime/RealtimeAudioManager.swift @@ -38,9 +38,22 @@ class RealtimeAudioManager { init() { playbackQueue = DispatchQueue(label: "com.realtime.playback", qos: .userInitiated) + // setupAudioSession() } - + // /// 设置音频会话 + // private func setupAudioSession() { + // do { + // let session = AVAudioSession.sharedInstance() + // try session.setCategory(.playAndRecord, + // mode: .default, + // options: [.defaultToSpeaker, .allowBluetooth]) + // try session.setActive(true) + // os_log("音频会话设置成功", log: log, type: .info) + // } catch { + // os_log("音频会话设置失败: %@", log: log, type: .error, error.localizedDescription) + // } + // } /// 初始化音频管理器 func initialize(config: AudioConfig) -> Bool {