Browse Source

优化按住说话说话体验

weicu
fdp 1 year ago
parent
commit
024bcaa271
  1. 101
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

101
local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

@ -93,7 +93,7 @@ public class SimpleAudioReceiver: NSObject {
let frameCount = buffer.frameLength
// print("麦克风启动: \(frameCount)")
// print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")")
self.writeQueue.put(self.audioBufferToData(buffer))
self.writeQueue.put(self.audioBufferToDataOptimized(buffer))
// 分析或传输数据...
}
} catch {
@ -360,6 +360,7 @@ private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) {
}
}
}
// 修改 runMicrophoneCapture 方法
private func runMicrophoneCapture() {
do {
print("🎤 开始配置麦克风捕获")
@ -369,27 +370,24 @@ private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) {
print("🔧 创建新的音频引擎")
}
// 如果音频引擎正在运行,先停止
if audioEngine?.isRunning == true {
audioEngine?.stop()
print("⏹️ 停止现有音频引擎")
}
try startCapture { buffer in
// 处理实时音频数据
let channelData = buffer.floatChannelData?[0]
let frameCount = buffer.frameLength
// print("麦克风启动: \(frameCount)")
// print("🎵 音频数据: 帧数=\(frameCount), 声道数据=\(channelData != nil ? "有效" : "无效")")
self.writeQueue.put(self.audioBufferToData(buffer))
// 分析或传输数据...
// 将音频处理移到专用队列,避免阻塞音频线程
self.conversionQueue.async {
let audioData = self.audioBufferToDataOptimized(buffer)
self.writeQueue.put(audioData)
}
}
try audioEngine?.start()
try audioEngine?.start()
print("✅ 麦克风捕获配置完成")
} catch {
print("麦克风启动失败: \(error)")
// 【新增】添加详细错误处理
if let nsError = error as NSError? {
print("错误域: \(nsError.domain), 错误代码: \(nsError.code)")
print("错误描述: \(nsError.localizedDescription)")
@ -462,63 +460,46 @@ private func enableVoiceProcessingIfAvailable(inputNode: AVAudioInputNode) {
* @param buffer 音频缓冲区
* @return 转换后的音频数据
*/
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
/**
* 优化的音频缓冲区转换方法
* 减少内存分配,提高转换效率
*/
// 在类属性中添加
private var reusableDataBuffer: Data?
private let bufferReuseQueue = DispatchQueue(label: "buffer.reuse", qos: .userInitiated)
/**
* 带缓冲区复用的音频转换方法
*/
private func audioBufferToDataOptimized(_ buffer: AVAudioPCMBuffer) -> Data {
guard let floatChannelData = buffer.floatChannelData else {
return Data()
}
let frameLength = Int(buffer.frameLength)
let channelCount = Int(buffer.format.channelCount)
let inputSampleRate = buffer.format.sampleRate
let targetSampleRate: Double = 16000
let requiredSize = frameLength * 2
// 复用或创建缓冲区
if reusableDataBuffer == nil || reusableDataBuffer!.count < requiredSize {
reusableDataBuffer = Data(count: requiredSize)
}
// 统一处理:只使用第一个声道(单声道),确保一致性
var outputData = reusableDataBuffer!
let firstChannelData = floatChannelData[0]
// 如果采样率不是16kHz,进行降采样
if inputSampleRate != targetSampleRate {
let ratio = inputSampleRate / targetSampleRate
let outputFrameCount = Int(Double(frameLength) / ratio)
var outputData = Data()
outputData.reserveCapacity(outputFrameCount * 2) // 16位 = 2字节
for outputIndex in 0..<outputFrameCount {
let inputIndex = Int(Double(outputIndex) * ratio)
if inputIndex < frameLength {
let floatSample = firstChannelData[inputIndex]
// 将Float32 (-1.0 to 1.0) 转换为Int16 (-32768 to 32767)
let clampedSample = max(-1.0, min(1.0, floatSample))
let int16Sample = Int16(clampedSample * 32767.0)
// 转换为小端字节序
withUnsafeBytes(of: int16Sample.littleEndian) { bytes in
outputData.append(contentsOf: bytes)
}
}
}
print("音频降采样: \(inputSampleRate)Hz -> \(targetSampleRate)Hz, 帧数: \(frameLength) -> \(outputFrameCount)")
return outputData
} else {
// 采样率已经是16kHz,直接转换格式(只处理第一个声道)
var data = Data()
data.reserveCapacity(frameLength * 2) // 单声道,16位
outputData.withUnsafeMutableBytes { rawBytes in
let int16Ptr = rawBytes.bindMemory(to: Int16.self)
for frame in 0..<frameLength {
let floatSample = firstChannelData[frame]
let clampedSample = max(-1.0, min(1.0, floatSample))
let int16Sample = Int16(clampedSample * 32767.0)
withUnsafeBytes(of: int16Sample.littleEndian) { bytes in
data.append(contentsOf: bytes)
}
int16Ptr[frame] = int16Sample.littleEndian
}
print("音频格式转换: \(inputSampleRate)Hz, 帧数: \(frameLength), 输出大小: \(data.count) 字节")
return data
}
// 返回实际使用的数据部分
return outputData.prefix(requiredSize)
}
/**
@ -561,13 +542,15 @@ private func convertAudioBuffer(_ buffer: AVAudioPCMBuffer, to targetFormat: AVA
* @throws 音频会话配置失败时抛出错误
*/
private func setupAudioSession() throws {
try audioSession.setCategory(.playAndRecord, mode: .default, options: [.defaultToSpeaker, .allowBluetooth,.mixWithOthers])
// 添加 .mixWithOthers 选项以支持同时播放和录音
try audioSession.setCategory(.playAndRecord,
mode: .default,
options: [.defaultToSpeaker, .allowBluetooth, .mixWithOthers])
// 设置较低的音频延迟以减少卡顿
try audioSession.setPreferredIOBufferDuration(0.005) // 5ms
try audioSession.setActive(true)
// // 设置首选的音频参数 - 修改为16000Hz以匹配Azure Speech要求
// try audioSession.setPreferredSampleRate(16000)
// // 设置输入声道数
// try audioSession.setPreferredInputNumberOfChannels(1)
print("音频会话配置成功: 采样率=\(audioSession.sampleRate)Hz, 输入声道=\(audioSession.inputNumberOfChannels)")
}

Loading…
Cancel
Save