|
|
|
@ -195,11 +195,9 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
/** |
|
|
|
* 开始采集音频数据 |
|
|
|
* @param bufferHandler 音频缓冲区处理回调 |
|
|
|
* @throws 音频配置或启动错误 |
|
|
|
* @throws 音频引擎启动失败时抛出错误 |
|
|
|
*/ |
|
|
|
func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws { |
|
|
|
|
|
|
|
|
|
|
|
guard let audioEngine = audioEngine else { |
|
|
|
throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"]) |
|
|
|
} |
|
|
|
@ -216,59 +214,92 @@ public class SimpleAudioReceiver: NSObject { |
|
|
|
|
|
|
|
// 重置音频引擎(在获取格式之前) |
|
|
|
audioEngine.reset() |
|
|
|
try setupAudioSession() |
|
|
|
try setupAudioSession() |
|
|
|
|
|
|
|
// 重新获取输入节点和格式(reset后需要重新获取) |
|
|
|
let inputNode = audioEngine.inputNode |
|
|
|
let inputFormat = inputNode.outputFormat(forBus: 0) |
|
|
|
|
|
|
|
print("硬件输入格式: 采样率=\(inputFormat.sampleRate)Hz, 声道数=\(inputFormat.channelCount), 格式=\(inputFormat.commonFormat.rawValue)") |
|
|
|
|
|
|
|
// 验证格式是否有效 |
|
|
|
if inputFormat.sampleRate <= 0 || inputFormat.channelCount <= 0 { |
|
|
|
// 如果格式无效,使用默认格式 |
|
|
|
let defaultFormat = AVAudioFormat(standardFormatWithSampleRate: 48000, channels: 1)! |
|
|
|
print("使用默认音频格式: 采样率=\(defaultFormat.sampleRate)Hz, 声道数=\(defaultFormat.channelCount)") |
|
|
|
// 创建目标格式 - 统一使用16000Hz单声道 |
|
|
|
guard let targetFormat = AVAudioFormat(commonFormat: .pcmFormatFloat32, |
|
|
|
sampleRate: 16000, |
|
|
|
channels: 1, |
|
|
|
interleaved: false) else { |
|
|
|
throw NSError(domain: "SimpleAudioReceiver", code: -2, userInfo: [NSLocalizedDescriptionKey: "无法创建目标音频格式"]) |
|
|
|
} |
|
|
|
|
|
|
|
print("使用目标音频格式: 采样率=\(targetFormat.sampleRate)Hz, 声道数=\(targetFormat.channelCount)") |
|
|
|
|
|
|
|
// 检查硬件格式是否有效 |
|
|
|
let isHardwareFormatValid = inputFormat.sampleRate > 0 && inputFormat.channelCount > 0 |
|
|
|
|
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
if isHardwareFormatValid { |
|
|
|
print("使用硬件原生格式安装tap") |
|
|
|
// 使用硬件原生格式安装tap,然后在回调中进行格式转换 |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: 1024, |
|
|
|
format: defaultFormat) { (buffer, time) in |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
format: inputFormat) { (buffer, time) in |
|
|
|
// 如果硬件格式与目标格式不匹配,进行转换 |
|
|
|
if inputFormat.sampleRate != targetFormat.sampleRate || |
|
|
|
inputFormat.channelCount != targetFormat.channelCount { |
|
|
|
// 执行格式转换 |
|
|
|
if let convertedBuffer = self.convertAudioBuffer(buffer, to: targetFormat) { |
|
|
|
bufferHandler(convertedBuffer) |
|
|
|
} else { |
|
|
|
print("音频格式转换失败,跳过此缓冲区") |
|
|
|
} |
|
|
|
} else { |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
} |
|
|
|
} else { |
|
|
|
// 使用硬件原生格式 |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
print("硬件音频格式无效,使用目标格式直接安装tap") |
|
|
|
// 硬件格式无效时,直接使用目标格式安装tap |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: 1024, |
|
|
|
format: inputFormat) { (buffer, time) in |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
format: targetFormat) { (buffer, time) in |
|
|
|
// 直接使用目标格式的缓冲区 |
|
|
|
bufferHandler(buffer) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
if #available(iOS 13.0, *) { |
|
|
|
try inputNode.setVoiceProcessingEnabled(true) |
|
|
|
do { |
|
|
|
try inputNode.setVoiceProcessingEnabled(true) |
|
|
|
} catch { |
|
|
|
print("启用语音处理失败: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
audioEngine.prepare() |
|
|
|
|
|
|
|
try audioEngine.start() |
|
|
|
|
|
|
|
print("音频引擎启动成功") |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 安全地移除音频tap |
|
|
|
* 避免在移除不存在的tap时出错 |
|
|
|
* 避免在移除tap时出现崩溃 |
|
|
|
*/ |
|
|
|
/** |
|
|
|
* 安全地移除音频tap |
|
|
|
* 避免在移除tap时出现崩溃 |
|
|
|
*/ |
|
|
|
private func removeTapSafely() { |
|
|
|
guard let audioEngine = audioEngine else { return } |
|
|
|
|
|
|
|
let inputNode = audioEngine.inputNode |
|
|
|
|
|
|
|
// 尝试移除tap,如果没有tap也不会出错 |
|
|
|
do { |
|
|
|
inputNode.removeTap(onBus: 0) |
|
|
|
print("成功移除音频tap") |
|
|
|
} catch { |
|
|
|
// 如果没有tap可移除,这是正常的 |
|
|
|
print("移除tap时出现错误(可能没有tap存在): \(error)") |
|
|
|
// 检查是否有tap需要移除 |
|
|
|
if inputNode.numberOfInputs > 0 { |
|
|
|
do { |
|
|
|
inputNode.removeTap(onBus: 0) |
|
|
|
print("成功移除音频tap") |
|
|
|
} catch { |
|
|
|
print("移除音频tap时出错: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
private func runMicrophoneCapture() { |
|
|
|
@ -429,6 +460,40 @@ private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 转换音频缓冲区格式 |
|
|
|
* @param buffer 原始音频缓冲区 |
|
|
|
* @param targetFormat 目标音频格式 |
|
|
|
* @return 转换后的音频缓冲区,转换失败时返回nil |
|
|
|
*/ |
|
|
|
private func convertAudioBuffer(_ buffer: AVAudioPCMBuffer, to targetFormat: AVAudioFormat) -> AVAudioPCMBuffer? { |
|
|
|
guard let converter = AVAudioConverter(from: buffer.format, to: targetFormat) else { |
|
|
|
print("无法创建音频转换器: 从\(buffer.format.sampleRate)Hz到\(targetFormat.sampleRate)Hz") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
// 计算转换后的帧数 |
|
|
|
let capacity = AVAudioFrameCount(Double(buffer.frameLength) * targetFormat.sampleRate / buffer.format.sampleRate) |
|
|
|
guard let convertedBuffer = AVAudioPCMBuffer(pcmFormat: targetFormat, frameCapacity: capacity) else { |
|
|
|
print("无法创建转换后的音频缓冲区") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
var error: NSError? |
|
|
|
let status = converter.convert(to: convertedBuffer, error: &error) { _, outStatus in |
|
|
|
outStatus.pointee = .haveData |
|
|
|
return buffer |
|
|
|
} |
|
|
|
|
|
|
|
if status == .error { |
|
|
|
print("音频转换失败: \(error?.localizedDescription ?? "未知错误")") |
|
|
|
return nil |
|
|
|
} |
|
|
|
|
|
|
|
return convertedBuffer |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 配置音频会话 |
|
|
|
* 设置录音和播放所需的音频会话参数 |
|
|
|
@ -438,10 +503,10 @@ private func setupAudioSession() throws { |
|
|
|
try audioSession.setCategory(.playAndRecord, mode: .default, options: [.defaultToSpeaker, .allowBluetooth]) |
|
|
|
try audioSession.setActive(true) |
|
|
|
|
|
|
|
// 设置首选的音频参数 |
|
|
|
try audioSession.setPreferredSampleRate(48000) |
|
|
|
// 设置输入声道数 |
|
|
|
try audioSession.setPreferredInputNumberOfChannels(1) |
|
|
|
// 设置首选的音频参数 - 修改为16000Hz以匹配Azure Speech要求 |
|
|
|
try audioSession.setPreferredSampleRate(16000) |
|
|
|
// 设置输入声道数 |
|
|
|
try audioSession.setPreferredInputNumberOfChannels(1) |
|
|
|
|
|
|
|
print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)") |
|
|
|
} |
|
|
|
@ -574,4 +639,3 @@ private class LinkedBlockingQueue<T> { |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|