|
|
|
@ -9,9 +9,32 @@ import os.log |
|
|
|
*/ |
|
|
|
public class SimpleAudioReceiver: NSObject { |
|
|
|
|
|
|
|
/// 单例实例(线程安全) |
|
|
|
/// 通过 SimpleAudioReceiver.shared 获取全局唯一实例 |
|
|
|
public static let shared = SimpleAudioReceiver() |
|
|
|
|
|
|
|
/// 私有化构造函数,防止外部直接实例化 |
|
|
|
/// 使用方式:通过 SimpleAudioReceiver.shared 访问实例 |
|
|
|
private override init() { |
|
|
|
super.init() |
|
|
|
initAudioRecord() |
|
|
|
} |
|
|
|
|
|
|
|
private let tag = "SimpleAudioReceiver" |
|
|
|
private let log = OSLog(subsystem: "com.azure.speech", category: "SimpleAudioReceiver") |
|
|
|
/// 音频格式转换专用队列(避免在音频渲染线程上做重操作) |
|
|
|
private let conversionQueue = DispatchQueue(label: "audio.stream.convert", qos: .userInitiated) |
|
|
|
|
|
|
|
/// 缓存的音频转换器,避免每个缓冲区重复创建 |
|
|
|
private var cachedConverter: AVAudioConverter? |
|
|
|
|
|
|
|
/// 统一的目标格式(16kHz/单声道/Float32,用于后续再转 Int16) |
|
|
|
private lazy var targetFormat16000Mono: AVAudioFormat = { |
|
|
|
return AVAudioFormat(commonFormat: .pcmFormatFloat32, |
|
|
|
sampleRate: 16000, |
|
|
|
channels: 1, |
|
|
|
interleaved: false)! |
|
|
|
}() |
|
|
|
/** |
|
|
|
* 音频来源类型 |
|
|
|
*/ |
|
|
|
@ -25,7 +48,7 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
// MARK: - 音频流类 |
|
|
|
|
|
|
|
public private(set) var pushAudioStream: SPXPushAudioInputStream? |
|
|
|
private let writeQueue = LinkedBlockingQueue<Data>() |
|
|
|
private var writeQueue = LinkedBlockingQueue<Data>() |
|
|
|
private var audioEngine: AVAudioEngine? |
|
|
|
private var audioFormat: AVAudioFormat? |
|
|
|
private let audioSession = AVAudioSession.sharedInstance() |
|
|
|
@ -48,13 +71,15 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
* 包括音频格式、推流、音频引擎等核心组件的初始化 |
|
|
|
*/ |
|
|
|
public func initAudioRecord() { |
|
|
|
// 每次初始化前,先释放上一次的资源,避免重复引擎/线程/tap |
|
|
|
releaseAudioResources() |
|
|
|
|
|
|
|
print("初始化了") |
|
|
|
audioFormat = getOptimalAudioFormat() |
|
|
|
pushAudioStream = SPXPushAudioInputStream() |
|
|
|
|
|
|
|
// 初始化音频引擎 |
|
|
|
audioEngine = AVAudioEngine() |
|
|
|
|
|
|
|
|
|
|
|
isRunning = true |
|
|
|
// 创建新的写线程 |
|
|
|
@ -66,15 +91,13 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
// 处理实时音频数据 |
|
|
|
let channelData = buffer.floatChannelData?[0] |
|
|
|
let frameCount = buffer.frameLength |
|
|
|
print("麦克风启动: \(frameCount)") |
|
|
|
print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") |
|
|
|
// print("麦克风启动: \(frameCount)") |
|
|
|
// print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") |
|
|
|
self.writeQueue.put(self.audioBufferToData(buffer)) |
|
|
|
// 分析或传输数据... |
|
|
|
} |
|
|
|
|
|
|
|
} catch { |
|
|
|
print("麦克风启动失败: \(error)") |
|
|
|
|
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
@ -192,83 +215,149 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
writeQueue.put(data) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 开始采集音频数据 |
|
|
|
* @param bufferHandler 音频缓冲区处理回调 |
|
|
|
* @throws 音频配置或启动错误 |
|
|
|
*/ |
|
|
|
func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { |
|
|
|
|
|
|
|
|
|
|
|
guard let audioEngine = audioEngine else { |
|
|
|
throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) |
|
|
|
} |
|
|
|
|
|
|
|
// 确保音频引擎完全停止 |
|
|
|
if audioEngine.isRunning { |
|
|
|
audioEngine.stop() |
|
|
|
// 等待一小段时间确保完全停止 |
|
|
|
Thread.sleep(forTimeInterval: 0.1) |
|
|
|
} |
|
|
|
|
|
|
|
// 先安全地移除已存在的 tap |
|
|
|
removeTapSafely() |
|
|
|
|
|
|
|
// 重置音频引擎(在获取格式之前) |
|
|
|
audioEngine.reset() |
|
|
|
try setupAudioSession() |
|
|
|
// 重新获取输入节点和格式(reset后需要重新获取) |
|
|
|
let inputNode = audioEngine.inputNode |
|
|
|
let inputFormat = inputNode.outputFormat(forBus: 0) |
|
|
|
|
|
|
|
print("硬件输入格式: 采样率=\(inputFormat.sampleRate)Hz, 声道数=\(inputFormat.channelCount), 格式=\(inputFormat.commonFormat.rawValue)") |
|
|
|
|
|
|
|
// 验证格式是否有效 |
|
|
|
if inputFormat.sampleRate <= 0 || inputFormat.channelCount <= 0 { |
|
|
|
// 如果格式无效,使用默认格式 |
|
|
|
let defaultFormat = AVAudioFormat(standardFormatWithSampleRate: 48000, channels: 1)! |
|
|
|
print("使用默认音频格式: 采样率=\(defaultFormat.sampleRate)Hz, 声道数=\(defaultFormat.channelCount)") |
|
|
|
|
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: 1024, |
|
|
|
format: defaultFormat) { (buffer, time) in |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
} else { |
|
|
|
// 使用硬件原生格式 |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: 1024, |
|
|
|
format: inputFormat) { (buffer, time) in |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
/** |
|
|
|
* 开始音频捕获 |
|
|
|
* 1) 重置音频引擎并配置音频会话 |
|
|
|
* 2) 安装 tap 时不指定格式(format: nil),由系统使用硬件实际格式 |
|
|
|
* 3) 在回调内将硬件格式转换为统一的 16kHz 单声道后回调上层 |
|
|
|
* @param bufferHandler 音频缓冲区处理回调(已转换为目标格式) |
|
|
|
* @throws 音频引擎启动失败时抛出错误 |
|
|
|
*/ |
|
|
|
func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { |
|
|
|
guard let audioEngine = audioEngine else { |
|
|
|
throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) |
|
|
|
} |
|
|
|
|
|
|
|
// 确保音频引擎完全停止 |
|
|
|
if audioEngine.isRunning { |
|
|
|
audioEngine.stop() |
|
|
|
// 等待一小段时间确保完全停止 |
|
|
|
Thread.sleep(forTimeInterval: 0.1) |
|
|
|
} |
|
|
|
|
|
|
|
// 先安全地移除已存在的 tap |
|
|
|
removeTapSafely() |
|
|
|
|
|
|
|
// 重置音频引擎(在获取格式之前) |
|
|
|
audioEngine.reset() |
|
|
|
|
|
|
|
// 配置音频会话 |
|
|
|
do { |
|
|
|
try setupAudioSession() |
|
|
|
} catch { |
|
|
|
print("音频会话设置失败: \(error.localizedDescription)") |
|
|
|
throw error |
|
|
|
} |
|
|
|
|
|
|
|
// 重新获取输入节点(此处获取的输出格式在未启动时可能为 0Hz,仅用于日志) |
|
|
|
let inputNode = audioEngine.inputNode |
|
|
|
let preFormat = inputNode.outputFormat(forBus: 0) |
|
|
|
print("当前节点(启动前)输出格式: 采样率=\(preFormat.sampleRate)Hz, 声道数=\(preFormat.channelCount), 格式=\(preFormat.commonFormat.rawValue)") |
|
|
|
|
|
|
|
// 目标格式 - 统一使用16000Hz单声道(Float32),后续再转 Int16 |
|
|
|
let targetFormat = self.targetFormat16000Mono |
|
|
|
print("使用目标音频格式: 采样率=\(targetFormat.sampleRate)Hz, 声道数=\(targetFormat.channelCount)") |
|
|
|
|
|
|
|
// 使用硬件原生格式安装 tap(format: nil),回调中异步转换 |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: 1024, |
|
|
|
format: nil) { [weak self] (buffer, time) in |
|
|
|
guard let self = self else { return } |
|
|
|
let srcFormat = buffer.format |
|
|
|
|
|
|
|
// 将重采样与格式转换放到专用队列,避免阻塞音频渲染线程 |
|
|
|
self.conversionQueue.async { |
|
|
|
// 如果源格式与目标格式不同,复用/重建转换器 |
|
|
|
var outputBuffer: AVAudioPCMBuffer? = buffer |
|
|
|
if srcFormat.sampleRate != targetFormat.sampleRate || |
|
|
|
srcFormat.channelCount != targetFormat.channelCount || |
|
|
|
srcFormat.commonFormat != targetFormat.commonFormat { |
|
|
|
|
|
|
|
if self.cachedConverter == nil || |
|
|
|
self.cachedConverter?.inputFormat != srcFormat || |
|
|
|
self.cachedConverter?.outputFormat != targetFormat { |
|
|
|
self.cachedConverter = AVAudioConverter(from: srcFormat, to: targetFormat) |
|
|
|
} |
|
|
|
|
|
|
|
if let converter = self.cachedConverter { |
|
|
|
let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / srcFormat.sampleRate) |
|
|
|
if let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) { |
|
|
|
var error: NSError? |
|
|
|
let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in |
|
|
|
outStatus.pointee = .haveData |
|
|
|
return buffer |
|
|
|
} |
|
|
|
if status == .error { |
|
|
|
// 限制日志:仅在错误时打印,避免频繁输出 |
|
|
|
print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") |
|
|
|
return |
|
|
|
} |
|
|
|
outputBuffer = convertedBuffer |
|
|
|
} else { |
|
|
|
print("无法创建转换后的音频缓冲区") |
|
|
|
return |
|
|
|
} |
|
|
|
} else { |
|
|
|
print("无法创建音频转换器: 从\(srcFormat.sampleRate)Hz到\(targetFormat.sampleRate)Hz") |
|
|
|
return |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 回调上层(注意:此时不在实时渲染线程) |
|
|
|
if let out = outputBuffer { |
|
|
|
bufferHandler(out) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
if #available(iOS 13.0, *) { |
|
|
|
} |
|
|
|
|
|
|
|
// 启用语音处理(如果支持) |
|
|
|
enableVoiceProcessingIfAvailable(inputNode: inputNode) |
|
|
|
|
|
|
|
// 准备并启动音频引擎 |
|
|
|
audioEngine.prepare() |
|
|
|
// print("音频引擎启动成功。Tap使用硬件格式: 采样率=\(postFormat.sampleRate)Hz, 声道数=\(postFormat.channelCount)") |
|
|
|
//print("音频引擎启动成功,使用格式: 采样率=\(tapFormat.sampleRate)Hz, 声道数=\(tapFormat.channelCount)") |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 启用语音处理(如果设备支持) |
|
|
|
* @param inputNode 输入节点 |
|
|
|
*/ |
|
|
|
private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) { |
|
|
|
if #available(iOS 13.0, *) { |
|
|
|
do { |
|
|
|
try inputNode.setVoiceProcessingEnabled(true) |
|
|
|
print("语音处理已启用") |
|
|
|
} catch { |
|
|
|
print("启用语音处理失败: \(error.localizedDescription)") |
|
|
|
// 不抛出错误,因为这不是关键功能 |
|
|
|
} |
|
|
|
|
|
|
|
audioEngine.prepare() |
|
|
|
|
|
|
|
|
|
|
|
print("音频引擎启动成功") |
|
|
|
} else { |
|
|
|
print("当前iOS版本不支持语音处理") |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
/** |
|
|
|
* 安全地移除音频tap |
|
|
|
* 避免在移除tap时出现崩溃 |
|
|
|
*/ |
|
|
|
/** |
|
|
|
* 安全地移除音频tap |
|
|
|
* 避免在移除不存在的tap时出错 |
|
|
|
* 避免在移除tap时出现崩溃 |
|
|
|
*/ |
|
|
|
private func removeTapSafely() { |
|
|
|
guard let audioEngine = audioEngine else { return } |
|
|
|
|
|
|
|
let inputNode = audioEngine.inputNode |
|
|
|
|
|
|
|
// 尝试移除tap,如果没有tap也不会出错 |
|
|
|
do { |
|
|
|
inputNode.removeTap(onBus: 0) |
|
|
|
print("成功移除音频tap") |
|
|
|
} catch { |
|
|
|
// 如果没有tap可移除,这是正常的 |
|
|
|
print("移除tap时出现错误(可能没有tap存在): \(error)") |
|
|
|
// 检查是否有tap需要移除 |
|
|
|
if inputNode.numberOfInputs > 0 { |
|
|
|
do { |
|
|
|
inputNode.removeTap(onBus: 0) |
|
|
|
print("成功移除音频tap") |
|
|
|
} catch { |
|
|
|
print("移除音频tap时出错: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
private func runMicrophoneCapture() { |
|
|
|
@ -290,8 +379,8 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
// 处理实时音频数据 |
|
|
|
let channelData = buffer.floatChannelData?[0] |
|
|
|
let frameCount = buffer.frameLength |
|
|
|
print("麦克风启动: \(frameCount)") |
|
|
|
print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") |
|
|
|
// print("麦克风启动: \(frameCount)") |
|
|
|
// print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")") |
|
|
|
self.writeQueue.put(self.audioBufferToData(buffer)) |
|
|
|
// 分析或传输数据... |
|
|
|
} |
|
|
|
@ -311,7 +400,7 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
private func runExternalCapture() { |
|
|
|
if audioSourceType == .external { |
|
|
|
//pushAudioData(data: Data()) |
|
|
|
|
|
|
|
//stopMicrophoneCapture() |
|
|
|
audioEngine?.pause() |
|
|
|
//audioEngine?.inputNode.removeTap(onBus: 0) |
|
|
|
} |
|
|
|
@ -344,23 +433,26 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
public func releaseAudioResources() { |
|
|
|
print("释放了") |
|
|
|
|
|
|
|
// 停止麦克风捕获 |
|
|
|
// 停止麦克风捕获(含 _isWriting=false 与移除 tap) |
|
|
|
stopMicrophoneCapture() |
|
|
|
|
|
|
|
// 停止音频引擎 |
|
|
|
audioEngine?.stop() |
|
|
|
|
|
|
|
// 关闭写入队列 |
|
|
|
// 安全关闭旧的写队列以唤醒阻塞的 take(),然后重建一个新的队列 |
|
|
|
let oldQueue = self.writeQueue |
|
|
|
isRunning = false |
|
|
|
writeThread?.async { |
|
|
|
self.writeQueue.close() |
|
|
|
oldQueue.close() |
|
|
|
} |
|
|
|
writeThread = nil |
|
|
|
// 重建队列,避免旧数据残留以及被 close 影响 |
|
|
|
writeQueue = LinkedBlockingQueue<Data>() |
|
|
|
|
|
|
|
// 安全地移除 tap |
|
|
|
// 再次安全地移除 tap(幂等) |
|
|
|
removeTapSafely() |
|
|
|
|
|
|
|
// 重置状态 |
|
|
|
isRunning = false |
|
|
|
audioEngine = nil |
|
|
|
} |
|
|
|
// MARK: - 协议定义 |
|
|
|
@ -429,6 +521,40 @@ private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 转换音频缓冲区格式 |
|
|
|
* @param buffer 原始音频缓冲区 |
|
|
|
* @param targetFormat 目标音频格式 |
|
|
|
* @return 转换后的音频缓冲区,转换失败时返回nil |
|
|
|
*/ |
|
|
|
private func convertAudioBuffer(_ buffer: AVAudioPCMBuffer, to targetFormat: AVAudioFormat) -> AVAudioPCMBuffer? { |
|
|
|
guard let converter = AVAudioConverter(from: buffer.format, to: targetFormat) else { |
|
|
|
print("无法创建音频转换器: 从\(buffer.format.sampleRate)Hz到\(targetFormat.sampleRate)Hz") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
// 计算转换后的帧数 |
|
|
|
let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / buffer.format.sampleRate) |
|
|
|
guard let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) else { |
|
|
|
print("无法创建转换后的音频缓冲区") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
var error: NSError? |
|
|
|
let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in |
|
|
|
outStatus.pointee = .haveData |
|
|
|
return buffer |
|
|
|
} |
|
|
|
|
|
|
|
if status == .error { |
|
|
|
print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
return convertedBuffer |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 配置音频会话 |
|
|
|
* 设置录音和播放所需的音频会话参数 |
|
|
|
@ -438,10 +564,10 @@ private func setupAudioSession() throws { |
|
|
|
try audioSession.setCategory(.playAndRecord, mode: .default, options: [.defaultToSpeaker, .allowBluetooth]) |
|
|
|
try audioSession.setActive(true) |
|
|
|
|
|
|
|
// 设置首选的音频参数 |
|
|
|
try audioSession.setPreferredSampleRate(48000) |
|
|
|
// 设置输入声道数 |
|
|
|
try audioSession.setPreferredInputNumberOfChannels(1) |
|
|
|
// // 设置首选的音频参数 - 修改为16000Hz以匹配Azure Speech要求 |
|
|
|
// try audioSession.setPreferredSampleRate(16000) |
|
|
|
// // 设置输入声道数 |
|
|
|
// try audioSession.setPreferredInputNumberOfChannels(1) |
|
|
|
|
|
|
|
print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)") |
|
|
|
} |
|
|
|
@ -574,4 +700,3 @@ private class LinkedBlockingQueue<T> { |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|