diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index 0ee8b5c0a..4625666f6 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -203,8 +203,8 @@ public class AzureAsrHelper: NSObject { */ private func preInitializeAudioComponents() { if audioStream == nil { - audioStream = SimpleAudioReceiver() - audioStream?.initAudioRecord() + audioStream = SimpleAudioReceiver.shared + //audioStream?.initAudioRecord() } // 预创建音频配置 if let pushStream = audioStream?.pushAudioStream { @@ -346,14 +346,14 @@ public class AzureAsrHelper: NSObject { } // 停止音频处理 - audioStream?.releaseAudioResources() + //audioStream?.releaseAudioResources() // 释放网络监听资源 networkMonitor.dispose() // 释放资源 recognizer = nil speechConfig = nil audioConfig = nil - audioStream = nil + //audioStream = nil // 确保状态被重置 _isContinuousRecognitionActive = false //externalAudioStream = nil @@ -466,8 +466,8 @@ public class AzureAsrHelper: NSObject { private func setupMicrophoneStream() { // 只有在 audioStream 为 nil 时才创建新实例 if audioStream == nil { - audioStream = SimpleAudioReceiver() - audioStream?.initAudioRecord() // 确保调用初始化 + audioStream = SimpleAudioReceiver.shared + //audioStream?.initAudioRecord() // 确保调用初始化 } // 检查音频配置是否已存在 @@ -669,14 +669,14 @@ public class AzureAsrHelper: NSObject { if audioStream == nil { // 创建外部音频拉流对象 - audioStream = SimpleAudioReceiver() - audioStream?.initAudioRecord() + audioStream = SimpleAudioReceiver.shared + //audioStream?.initAudioRecord() } if audioStream?.recordfile == nil { audioStream?.recordfile = RecordFile() } - + // 修复:移除多余的 audioDataCallback 参数 audioStream?.startAudioRecord( diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift index cd0269c34..bb117738a 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/IntegratedSpeechTranslationService.swift @@ -329,8 +329,8 @@ import os.log * 初始化音频处理器 */ private func initializeAudioProcessor() { - audioProcessor = SimpleAudioReceiver() - audioProcessor?.initAudioRecord() + audioProcessor = SimpleAudioReceiver.shared + // audioProcessor?.initAudioRecord() // 检查音频配置是否已存在 if audioConfig == nil, let pushStream = audioProcessor?.pushAudioStream { audioConfig = SPXAudioConfiguration(streamInput: pushStream) @@ -913,8 +913,8 @@ import os.log audioConfig = nil // 释放其他组件 - audioProcessor?.releaseAudioResources() - audioProcessor = nil + //audioProcessor?.releaseAudioResources() + //audioProcessor = nil translationService?.dispose() translationService = nil diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift index 35ad33407..398c0eadb 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift @@ -9,9 +9,32 @@ import os.log */ public class SimpleAudioReceiver: NSObject { + /// 单例实例(线程安全) + /// 通过 SimpleAudioReceiver.shared 获取全局唯一实例 + public static let shared = SimpleAudioReceiver() + + /// 私有化构造函数,防止外部直接实例化 + /// 使用方式:通过 SimpleAudioReceiver.shared 访问实例 + private override init() { + super.init() + initAudioRecord() + } + private let tag = "SimpleAudioReceiver" private let log = OSLog(subsystem: "com.azure.speech", category: "SimpleAudioReceiver") + /// 音频格式转换专用队列(避免在音频渲染线程上做重操作) + private let conversionQueue = DispatchQueue(label: "audio.stream.convert", qos: .userInitiated) + + /// 缓存的音频转换器,避免每个缓冲区重复创建 + private var cachedConverter: AVAudioConverter? + /// 统一的目标格式(16kHz/单声道/Float32,用于后续再转 Int16) + private lazy var targetFormat16000Mono: AVAudioFormat = { + return AVAudioFormat(commonFormat: .pcmFormatFloat32, + sampleRate: 16000, + channels: 1, + interleaved: false)! + }() /** * 音频来源类型 */ @@ -25,7 +48,7 @@ public class SimpleAudioReceiver: NSObject { // MARK: - 音频流类 public private(set) var pushAudioStream: SPXPushAudioInputStream? - private let writeQueue = LinkedBlockingQueue() + private var writeQueue = LinkedBlockingQueue() private var audioEngine: AVAudioEngine? private var audioFormat: AVAudioFormat? private let audioSession = AVAudioSession.sharedInstance() @@ -48,13 +71,15 @@ public class SimpleAudioReceiver: NSObject { * 包括音频格式、推流、音频引擎等核心组件的初始化 */ public func initAudioRecord() { + // 每次初始化前,先释放上一次的资源,避免重复引擎/线程/tap + releaseAudioResources() + print("初始化了") audioFormat = getOptimalAudioFormat() pushAudioStream = SPXPushAudioInputStream() // 初始化音频引擎 audioEngine = AVAudioEngine() - isRunning = true // 创建新的写线程 @@ -66,15 +91,13 @@ public class SimpleAudioReceiver: NSObject { // 处理实时音频数据 let channelData = buffer.floatChannelData?[0] let frameCount = buffer.frameLength - print("麦克风启动: \(frameCount)") - print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") + // print("麦克风启动: \(frameCount)") + // print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") self.writeQueue.put(self.audioBufferToData(buffer)) // 分析或传输数据... } - } catch { print("麦克风启动失败: \(error)") - } } @@ -192,93 +215,128 @@ public class SimpleAudioReceiver: NSObject { writeQueue.put(data) } - /** - * 开始采集音频数据 - * @param bufferHandler 音频缓冲区处理回调 - * @throws 音频引擎启动失败时抛出错误 - */ - func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { - guard let audioEngine = audioEngine else { - throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) - } - - // 确保音频引擎完全停止 - if audioEngine.isRunning { - audioEngine.stop() - // 等待一小段时间确保完全停止 - Thread.sleep(forTimeInterval: 0.1) - } - - // 先安全地移除已存在的 tap - removeTapSafely() - - // 重置音频引擎(在获取格式之前) - audioEngine.reset() +/** + * 开始音频捕获 + * 1) 重置音频引擎并配置音频会话 + * 2) 安装 tap 时不指定格式(format: nil),由系统使用硬件实际格式 + * 3) 在回调内将硬件格式转换为统一的 16kHz 单声道后回调上层 + * @param bufferHandler 音频缓冲区处理回调(已转换为目标格式) + * @throws 音频引擎启动失败时抛出错误 + */ +func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { + guard let audioEngine = audioEngine else { + throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) + } + + // 确保音频引擎完全停止 + if audioEngine.isRunning { + audioEngine.stop() + // 等待一小段时间确保完全停止 + Thread.sleep(forTimeInterval: 0.1) + } + + // 先安全地移除已存在的 tap + removeTapSafely() + + // 重置音频引擎(在获取格式之前) + audioEngine.reset() + + // 配置音频会话 + do { try setupAudioSession() - - // 重新获取输入节点和格式(reset后需要重新获取) - let inputNode = audioEngine.inputNode - let inputFormat = inputNode.outputFormat(forBus: 0) - - print("硬件输入格式: 采样率=\(inputFormat.sampleRate)Hz, 声道数=\(inputFormat.channelCount), 格式=\(inputFormat.commonFormat.rawValue)") - - // 创建目标格式 - 统一使用16000Hz单声道 - guard let targetFormat = AVAudioFormat(commonFormat: .pcmFormatFloat32, - sampleRate: 16000, - channels: 1, - interleaved: false) else { - throw NSError(domain: "SimpleAudioReceiver", code: -2, userInfo: [NSLocalizedDescriptionKey: "无法创建目标音频格式"]) - } - - print("使用目标音频格式: 采样率=\(targetFormat.sampleRate)Hz, 声道数=\(targetFormat.channelCount)") - - // 检查硬件格式是否有效 - let isHardwareFormatValid = inputFormat.sampleRate > 0 && inputFormat.channelCount > 0 - - if isHardwareFormatValid { - print("使用硬件原生格式安装tap") - // 使用硬件原生格式安装tap,然后在回调中进行格式转换 - inputNode.installTap(onBus: 0, - bufferSize: 1024, - format: inputFormat) { (buffer, time) in - // 如果硬件格式与目标格式不匹配,进行转换 - if inputFormat.sampleRate != targetFormat.sampleRate || - inputFormat.channelCount != targetFormat.channelCount { - // 执行格式转换 - if let convertedBuffer = self.convertAudioBuffer(buffer, to: targetFormat) { - bufferHandler(convertedBuffer) + } catch { + print("音频会话设置失败: \(error.localizedDescription)") + throw error + } + + // 重新获取输入节点(此处获取的输出格式在未启动时可能为 0Hz,仅用于日志) + let inputNode = audioEngine.inputNode + let preFormat = inputNode.outputFormat(forBus: 0) + print("当前节点(启动前)输出格式: 采样率=\(preFormat.sampleRate)Hz, 声道数=\(preFormat.channelCount), 格式=\(preFormat.commonFormat.rawValue)") + + // 目标格式 - 统一使用16000Hz单声道(Float32),后续再转 Int16 + let targetFormat = self.targetFormat16000Mono + print("使用目标音频格式: 采样率=\(targetFormat.sampleRate)Hz, 声道数=\(targetFormat.channelCount)") + + // 使用硬件原生格式安装 tap(format: nil),回调中异步转换 + inputNode.installTap(onBus: 0, + bufferSize: 1024, + format: nil) { [weak self] (buffer, time) in + guard let self = self else { return } + let srcFormat = buffer.format + + // 将重采样与格式转换放到专用队列,避免阻塞音频渲染线程 + self.conversionQueue.async { + // 如果源格式与目标格式不同,复用/重建转换器 + var outputBuffer: AVAudioPCMBuffer? = buffer + if srcFormat.sampleRate != targetFormat.sampleRate || + srcFormat.channelCount != targetFormat.channelCount || + srcFormat.commonFormat != targetFormat.commonFormat { + + if self.cachedConverter == nil || + self.cachedConverter?.inputFormat != srcFormat || + self.cachedConverter?.outputFormat != targetFormat { + self.cachedConverter = AVAudioConverter(from: srcFormat, to: targetFormat) + } + + if let converter = self.cachedConverter { + let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / srcFormat.sampleRate) + if let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) { + var error: NSError? + let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in + outStatus.pointee = .haveData + return buffer + } + if status == .error { + // 限制日志:仅在错误时打印,避免频繁输出 + print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") + return + } + outputBuffer = convertedBuffer } else { - print("音频格式转换失败,跳过此缓冲区") + print("无法创建转换后的音频缓冲区") + return } } else { - bufferHandler(buffer) + print("无法创建音频转换器: 从\(srcFormat.sampleRate)Hz到\(targetFormat.sampleRate)Hz") + return } } - } else { - print("硬件音频格式无效,使用目标格式直接安装tap") - // 硬件格式无效时,直接使用目标格式安装tap - inputNode.installTap(onBus: 0, - bufferSize: 1024, - format: targetFormat) { (buffer, time) in - // 直接使用目标格式的缓冲区 - bufferHandler(buffer) - } - } - - if #available(iOS 13.0, *) { - do { - try inputNode.setVoiceProcessingEnabled(true) - } catch { - print("启用语音处理失败: \(error.localizedDescription)") + + // 回调上层(注意:此时不在实时渲染线程) + if let out = outputBuffer { + bufferHandler(out) } } - - audioEngine.prepare() - //try audioEngine.start() - - print("音频引擎启动成功") } + // 启用语音处理(如果支持) + enableVoiceProcessingIfAvailable(inputNode: inputNode) + + // 准备并启动音频引擎 + audioEngine.prepare() + // print("音频引擎启动成功。Tap使用硬件格式: 采样率=\(postFormat.sampleRate)Hz, 声道数=\(postFormat.channelCount)") + //print("音频引擎启动成功,使用格式: 采样率=\(tapFormat.sampleRate)Hz, 声道数=\(tapFormat.channelCount)") +} + + +/** + * 启用语音处理(如果设备支持) + * @param inputNode 输入节点 + */ +private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) { + if #available(iOS 13.0, *) { + do { + try inputNode.setVoiceProcessingEnabled(true) + print("语音处理已启用") + } catch { + print("启用语音处理失败: \(error.localizedDescription)") + // 不抛出错误,因为这不是关键功能 + } + } else { + print("当前iOS版本不支持语音处理") + } +} /** * 安全地移除音频tap * 避免在移除tap时出现崩溃 @@ -321,8 +379,8 @@ public class SimpleAudioReceiver: NSObject { // 处理实时音频数据 let channelData = buffer.floatChannelData?[0] let frameCount = buffer.frameLength - print("麦克风启动: \(frameCount)") - print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") + // print("麦克风启动: \(frameCount)") + // print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") self.writeQueue.put(self.audioBufferToData(buffer)) // 分析或传输数据... } @@ -342,7 +400,7 @@ public class SimpleAudioReceiver: NSObject { private func runExternalCapture() { if audioSourceType == .external { //pushAudioData(data: Data()) - + //stopMicrophoneCapture() audioEngine?.pause() //audioEngine?.inputNode.removeTap(onBus: 0) } @@ -375,23 +433,26 @@ public class SimpleAudioReceiver: NSObject { public func releaseAudioResources() { print("释放了") - // 停止麦克风捕获 + // 停止麦克风捕获(含 _isWriting=false 与移除 tap) stopMicrophoneCapture() // 停止音频引擎 audioEngine?.stop() - // 关闭写入队列 + // 安全关闭旧的写队列以唤醒阻塞的 take(),然后重建一个新的队列 + let oldQueue = self.writeQueue + isRunning = false writeThread?.async { - self.writeQueue.close() + oldQueue.close() } writeThread = nil + // 重建队列,避免旧数据残留以及被 close 影响 + writeQueue = LinkedBlockingQueue() - // 安全地移除 tap + // 再次安全地移除 tap(幂等) removeTapSafely() // 重置状态 - isRunning = false audioEngine = nil } // MARK: - 协议定义 @@ -503,10 +564,10 @@ private func setupAudioSession() throws { try audioSession.setCategory(.playAndRecord, mode: .default, options: [.defaultToSpeaker, .allowBluetooth]) try audioSession.setActive(true) - // 设置首选的音频参数 - 修改为16000Hz以匹配Azure Speech要求 - try audioSession.setPreferredSampleRate(16000) - // 设置输入声道数 - try audioSession.setPreferredInputNumberOfChannels(1) + // // 设置首选的音频参数 - 修改为16000Hz以匹配Azure Speech要求 + // try audioSession.setPreferredSampleRate(16000) + // // 设置输入声道数 + // try audioSession.setPreferredInputNumberOfChannels(1) print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)") }