Browse Source

修复麦克风时大时小的问题

weicu
fdp 1 year ago
parent
commit
0582ef4ee8
  1. 22
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift
  2. 26
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift
  3. 453
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

22
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

@ -206,12 +206,11 @@ public class AzureAsrHelper: NSObject {
audioStream = SimpleAudioReceiver() audioStream = SimpleAudioReceiver()
audioStream?.initAudioRecord() audioStream?.initAudioRecord()
} }
// 预创建音频配置 // 预创建音频配置
if let pushStream = audioStream?.pushAudioStream { if let pushStream = audioStream?.pushAudioStream {
audioConfig = SPXAudioConfiguration(streamInput: pushStream) audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("预初始化音频组件完成", log: log, type: .info) os_log("预初始化音频组件完成", log: log, type: .info)
} }
} }
@ -465,12 +464,17 @@ public class AzureAsrHelper: NSObject {
* 设置麦克风音频流 * 设置麦克风音频流
*/ */
private func setupMicrophoneStream() { private func setupMicrophoneStream() {
audioStream = SimpleAudioReceiver() // 只有在 audioStream 为 nil 时才创建新实例
// 【优化】检查音频配置是否已存在 if audioStream == nil {
if audioConfig == nil, let pushStream = audioStream?.pushAudioStream { audioStream = SimpleAudioReceiver()
audioConfig = SPXAudioConfiguration(streamInput: pushStream) audioStream?.initAudioRecord() // 确保调用初始化
os_log("设置麦克风流完成", log: log, type: .info) }
}
// 检查音频配置是否已存在
if audioConfig == nil, let pushStream = audioStream?.pushAudioStream {
audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("设置麦克风流完成", log: log, type: .info)
}
} }

26
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift

@ -319,31 +319,7 @@ public class AzureTtsHelper: NSObject, ITtsService {
return true return true
} }
// /**
// * 设置音频输出设备
// */
// public func setAudioOutputDevice(_ device: AudioOutputDevice) -> Bool {
// do {
// let session = AVAudioSession.sharedInstance()
// try session.setCategory(.playAndRecord, options: [.defaultToSpeaker, .allowBluetooth])
//
// switch device {
// case .default:
// try session.overrideOutputAudioPort(.none)
// case .speaker:
// try session.overrideOutputAudioPort(.speaker)
// case .headphones:
// try session.overrideOutputAudioPort(.none)
// }
//
// try session.setActive(true)
// return true
// } catch {
// os_log("设置音频输出设备失败: %{public}@", log: log, type: .error, error.localizedDescription)
// return false
// }
// }
//
// MARK: - 私有方法 // MARK: - 私有方法
/** /**

453
local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

@ -54,11 +54,28 @@ public class SimpleAudioReceiver: NSObject {
// 初始化音频引擎 // 初始化音频引擎
audioEngine = AVAudioEngine() audioEngine = AVAudioEngine()
isRunning = true isRunning = true
// 创建新的写线程 // 创建新的写线程
writeThread = DispatchQueue(label: "audio.stream.writer") writeThread = DispatchQueue(label: "audio.stream.writer")
startWriteThread() startWriteThread()
// 配置音频引擎
do {
try startCapture { buffer in
// 处理实时音频数据
let channelData = buffer.floatChannelData?[0]
let frameCount = buffer.frameLength
print("麦克风启动: \(frameCount)")
print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")")
self.writeQueue.put(self.audioBufferToData(buffer))
// 分析或传输数据...
}
} catch {
print("麦克风启动失败: \(error)")
}
} }
/// 获取最佳音频格式 (iOS 通常支持标准采样率) /// 获取最佳音频格式 (iOS 通常支持标准采样率)
@ -145,6 +162,7 @@ public class SimpleAudioReceiver: NSObject {
print("写入数据长度: \(dataToWrite.count)") print("写入数据长度: \(dataToWrite.count)")
self.audioDataCallback!.onAudio(dataToWrite) self.audioDataCallback!.onAudio(dataToWrite)
} }
print("写入数据长度: \(dataToWrite.count)")
try self.pushAudioStream?.write(dataToWrite) try self.pushAudioStream?.write(dataToWrite)
if self.recordfile != nil { if self.recordfile != nil {
// print("写入recordfile数据长度: \(dataToWrite.count)") // print("写入recordfile数据长度: \(dataToWrite.count)")
@ -173,107 +191,113 @@ public class SimpleAudioReceiver: NSObject {
// 放入队列,由写线程写入 // 放入队列,由写线程写入
writeQueue.put(data) writeQueue.put(data)
} }
/**
* 开始采集音频数据
* @param bufferHandler 音频缓冲区处理回调
* @throws 音频配置或启动错误
*/
func startCapture(bufferHandler: @escaping (AVAudioPCMBuffer) -> Void) throws {
try setupAudioSession()
guard let audioEngine = audioEngine else {
throw NSError(domain: "SimpleAudioReceiver", code: -1, userInfo: [NSLocalizedDescriptionKey: "Audio engine not initialized"])
}
// 确保音频引擎完全停止
if audioEngine.isRunning {
audioEngine.stop()
// 等待一小段时间确保完全停止
Thread.sleep(forTimeInterval: 0.1)
}
// 先安全地移除已存在的 tap
removeTapSafely()
// 重置音频引擎(在获取格式之前)
audioEngine.reset()
// 重新获取输入节点和格式(reset后需要重新获取)
let inputNode = audioEngine.inputNode
let inputFormat = inputNode.outputFormat(forBus: 0)
print("硬件输入格式: 采样率=\(inputFormat.sampleRate)Hz, 声道数=\(inputFormat.channelCount), 格式=\(inputFormat.commonFormat.rawValue)")
// 验证格式是否有效
if inputFormat.sampleRate <= 0 || inputFormat.channelCount <= 0 {
// 如果格式无效,使用默认格式
let defaultFormat = AVAudioFormat(standardFormatWithSampleRate: 48000, channels: 1)!
print("使用默认音频格式: 采样率=\(defaultFormat.sampleRate)Hz, 声道数=\(defaultFormat.channelCount)")
inputNode.installTap(onBus: 0,
bufferSize: 1024,
format: defaultFormat) { (buffer, time) in
bufferHandler(buffer)
}
} else {
// 使用硬件原生格式
inputNode.installTap(onBus: 0,
bufferSize: 1024,
format: inputFormat) { (buffer, time) in
bufferHandler(buffer)
}
}
if #available(iOS 13.0, *) {
try inputNode.setVoiceProcessingEnabled(true)
}
audioEngine.prepare()
print("音频引擎启动成功")
}
// 私有方法:启动麦克风捕获
/** /**
* 启动麦克风捕获 * 安全地移除音频tap
* 优化:复用已初始化的音频引擎,减少启动延迟 * 避免在移除不存在的tap时出错
*/ */
private func removeTapSafely() {
guard let audioEngine = audioEngine else { return }
let inputNode = audioEngine.inputNode
// 尝试移除tap,如果没有tap也不会出错
do {
inputNode.removeTap(onBus: 0)
print("成功移除音频tap")
} catch {
// 如果没有tap可移除,这是正常的
print("移除tap时出现错误(可能没有tap存在): \(error)")
}
}
private func runMicrophoneCapture() { private func runMicrophoneCapture() {
do { do {
// 【新增】首先配置音频会话 print("🎤 开始配置麦克风捕获")
try audioSession.setCategory(
.playAndRecord,
mode: .measurement, // 使用 measurement 模式获得最佳录音质量
options: [.allowBluetooth, .defaultToSpeaker]
)
// 【新增】请求麦克风权限(如果尚未授权)
if audioSession.recordPermission != .granted {
audioSession.requestRecordPermission { granted in
if !granted {
print("麦克风权限被拒绝")
}
}
}
// 【新增】激活音频会话(在配置音频引擎之前)
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
// 检查音频引擎是否已初始化,避免重复创建
if audioEngine == nil { if audioEngine == nil {
audioEngine = AVAudioEngine() audioEngine = AVAudioEngine()
print("🔧 创建新的音频引擎")
} }
// 如果音频引擎正在运行,先停止 // 如果音频引擎正在运行,先停止
if audioEngine?.isRunning == true { if audioEngine?.isRunning == true {
audioEngine?.stop() audioEngine?.stop()
print("⏹️ 停止现有音频引擎")
} }
// 获取音频输入节点(麦克风) // try startCapture { buffer in
guard let inputNode = audioEngine?.inputNode else { // // 处理实时音频数据
throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"]) // let channelData = buffer.floatChannelData?[0]
} // let frameCount = buffer.frameLength
// print("麦克风启动: \(frameCount)")
// 移除之前的音频处理块,避免重复添加 // print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")")
inputNode.removeTap(onBus: 0) // self.writeQueue.put(self.audioBufferToData(buffer))
// // 分析或传输数据...
// 获取硬件支持的原始音频格式 // }
let hardwareFormat = inputNode.inputFormat(forBus: 0) try audioEngine?.start()
print("✅ 麦克风捕获配置完成")
// iOS 13+ 启用语音处理
if #available(iOS 13.0, *) {
try inputNode.setVoiceProcessingEnabled(true)
}
// 检查目标音频格式和转换器是否可用
guard let targetFormat = audioFormat, // 外部定义的期望音频格式
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2)
}
// 在输入节点上安装录音回调
inputNode.installTap(onBus: 0,
bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小
format: hardwareFormat) { // 使用原始硬件格式
[weak self] buffer, time in // 弱引用避免循环引用
// 确保实例存在且正在写入状态
guard let self = self, self._isWriting else { return }
// 创建目标格式的音频缓冲区
let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
// 计算转换后的帧容量(考虑采样率差异)
frameCapacity: AVAudioFrameCount(
targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate
)
)!
var error: NSError?
// 执行音频格式转换
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
outStatus.pointee = .haveData // 标记有数据可用
return buffer // 返回原始音频数据
}
)
// 转换成功且无错误
if status == .haveData, error == nil {
// 将音频缓冲区转换为二进制数据
let data = self.audioBufferToData(convertedBuffer)
// 将数据放入写入队列(后续处理)
self.writeQueue.put(data)
}
}
// 激活音频会话(允许录音)
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
try audioEngine?.start()
} catch { } catch {
print("麦克风启动失败: \(error)") print("麦克风启动失败: \(error)")
// 【新增】添加详细错误处理 // 【新增】添加详细错误处理
@ -283,38 +307,7 @@ public class SimpleAudioReceiver: NSObject {
} }
} }
} }
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
let frameLength = Int(buffer.frameLength)
let channelCount = 1
let dataLength = frameLength * channelCount * MemoryLayout<Int16>.size
// Handle 16-bit integer format
if let int16Data = buffer.int16ChannelData {
return Data(
bytes: int16Data.pointee,
count: dataLength
)
}
// Handle float format
else if let floatData = buffer.floatChannelData {
var int16Array = [Int16](repeating: 0, count: frameLength)
let floatBuffer = floatData.pointee
for i in 0..<frameLength {
let sample = floatBuffer[i]
let clamped = max(-1.0, min(sample, 1.0))
let scaled = clamped * Float(Int16.max)
int16Array[i] = Int16(scaled)
}
return Data(
bytes: int16Array,
count: dataLength
)
}
return Data() // Fallback for unsupported formats
}
private func runExternalCapture() { private func runExternalCapture() {
if audioSourceType == .external { if audioSourceType == .external {
//pushAudioData(data: Data()) //pushAudioData(data: Data())
@ -323,114 +316,168 @@ public class SimpleAudioReceiver: NSObject {
//audioEngine?.inputNode.removeTap(onBus: 0) //audioEngine?.inputNode.removeTap(onBus: 0)
} }
} }
/** /**
* 继续麦克风捕获 * 停止麦克风捕获
*/ * 安全地停止音频会话和音频引擎
/**
* 继续麦克风捕获
*/ */
public func resumeRecord() {
guard _isWriting==false else { return }
_isWriting = true
// 启动录音引擎
do {
print("resumeRecord")
try audioEngine?.start()
} catch {
os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription)
}
}
public func stopMicrophoneCapture() { public func stopMicrophoneCapture() {
guard _isWriting==true else { return } guard _isWriting == true else { return }
_isWriting=false _isWriting = false
audioEngine?.pause()
print("stopMicrophoneCapture") // 先停止音频引擎
audioEngine?.stop()
// 然后安全地移除 tap
removeTapSafely()
print("stopMicrophoneCapture")
} }
/**
* 释放音频资源
* 完全清理所有音频相关资源
*/
public func releaseAudioResources() { public func releaseAudioResources() {
print("释放了") print("释放了")
// 停止麦克风捕获
stopMicrophoneCapture() stopMicrophoneCapture()
// 停止音频引擎
audioEngine?.stop() audioEngine?.stop()
writeThread?.async {
// 关闭写入队列
writeThread?.async {
self.writeQueue.close() self.writeQueue.close()
} }
writeThread = nil writeThread = nil
audioEngine?.inputNode.removeTap(onBus: 0)
isRunning=false // 安全地移除 tap
removeTapSafely()
// 重置状态
isRunning = false
audioEngine = nil audioEngine = nil
} }
// MARK: - 协议定义
// 音频路由管理 /**
public enum AudioOutputRoute { * 将音频缓冲区转换为Data格式
case speaker * 统一处理采样率转换和格式转换
case receiver * @param buffer 音频缓冲区
case bluetooth * @return 转换后的音频数据
*/
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
guard let floatChannelData = buffer.floatChannelData else {
return Data()
} }
public func setAudioOutputRoute(_ route: AudioOutputRoute) {
let frameLength = Int(buffer.frameLength)
let channelCount = Int(buffer.format.channelCount)
let inputSampleRate = buffer.format.sampleRate
let targetSampleRate: Double = 16000
// 统一处理:只使用第一个声道(单声道),确保一致性
let firstChannelData = floatChannelData[0]
// 如果采样率不是16kHz,进行降采样
if inputSampleRate != targetSampleRate {
let ratio = inputSampleRate / targetSampleRate
let outputFrameCount = Int(Double(frameLength) / ratio)
do { var outputData = Data()
// outputData.reserveCapacity(outputFrameCount * 2) // 16位 = 2字节
if self._isWriting {
print("当前正在识别") for outputIndex in 0..<outputFrameCount {
try audioEngine?.pause() let inputIndex = Int(Double(outputIndex) * ratio)
// 先停用以避免冲突 if inputIndex < frameLength {
try audioSession.setActive(false) let floatSample = firstChannelData[inputIndex]
}
print("setAudioOutputRoutecurrent,route=\(route)")
switch route {
case .speaker:
print("setAudioOutputRoutecurrent,speaker")
// 使用扬声器时必须用videoChat模式
try audioSession.setCategory(
.playAndRecord,
mode: .videoChat,
options: [.defaultToSpeaker]
)
try audioSession.overrideOutputAudioPort(.speaker)
case .receiver: // 将Float32 (-1.0 to 1.0) 转换为Int16 (-32768 to 32767)
// 听筒模式使用voiceChat节省资源 let clampedSample = max(-1.0, min(1.0, floatSample))
try audioSession.setCategory( let int16Sample = Int16(clampedSample * 32767.0)
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth]
)
try audioSession.overrideOutputAudioPort(.none)
case .bluetooth: // 转换为小端字节序
// 完整蓝牙设备支持 withUnsafeBytes(of: int16Sample.littleEndian) { bytes in
try audioSession.setCategory( outputData.append(contentsOf: bytes)
.playAndRecord, }
mode: .voiceChat,
options: [.allowBluetooth, .allowBluetoothA2DP]
)
try audioSession.overrideOutputAudioPort(.none)
// 不需要override,系统自动路由
} }
self.currentRoute = route }
if self._isWriting {
print("音频降采样: \(inputSampleRate)Hz -> \(targetSampleRate)Hz, 帧数: \(frameLength) -> \(outputFrameCount)")
// 重新激活 return outputData
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) } else {
// 采样率已经是16kHz,直接转换格式(只处理第一个声道)
try audioEngine?.start() var data = Data()
// writeThread?.async { data.reserveCapacity(frameLength * 2) // 单声道,16位
// self.writeQueue.close()
// } for frame in 0..<frameLength {
let floatSample = firstChannelData[frame]
let clampedSample = max(-1.0, min(1.0, floatSample))
let int16Sample = Int16(clampedSample * 32767.0)
withUnsafeBytes(of: int16Sample.littleEndian) { bytes in
data.append(contentsOf: bytes)
} }
} catch {
print("路由切换失败: \(error)")
} }
print("音频格式转换: \(inputSampleRate)Hz, 帧数: \(frameLength), 输出大小: \(data.count) 字节")
return data
} }
}
/**
* 设置音频会话配置
* 确保音频会话正确配置用于录音
*/
private func setupAudioSession() throws {
try audioSession.setCategory(.playAndRecord, mode: .default, options: [.defaultToSpeaker, .allowBluetooth])
try audioSession.setActive(true)
// 设置首选的音频参数
try audioSession.setPreferredSampleRate(48000)
try audioSession.setPreferredInputNumberOfChannels(1)
print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)")
}
/**
* 继续麦克风捕获
*/
public func resumeRecord() {
guard _isWriting == false else { return }
_isWriting = true
// 启动录音引擎
do {
print("resumeRecord")
// 确保音频引擎存在且已准备好
guard let audioEngine = audioEngine else {
print("Audio engine not initialized")
return
}
if !audioEngine.isRunning {
try audioEngine.start()
}
} catch {
os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription)
_isWriting = false
}
}
// 音频路由管理
public enum AudioOutputRoute {
case speaker
case receiver
case bluetooth
}
/** /**
* 音频数据回调接口 * 音频数据回调接口
@ -526,5 +573,3 @@ private class LinkedBlockingQueue<T> {
} }
} }
// MARK: - 协议定义

Loading…
Cancel
Save