Browse Source

add

newdev_shunjiawei
wolfplus 2 years ago
parent
commit
e5a053254b
  1. 355
      azure/ios/Classes/AzureAsrHelper.swift
  2. 152
      ios/Runner/AudioSessionManager.swift
  3. 373
      ios/Runner/AzureAsrHelper.swift
  4. 2
      ios/Runner/ClassicBluetoothHelper.swift

355
azure/ios/Classes/AzureAsrHelper.swift

@ -1,6 +1,7 @@
import Foundation
import MicrosoftCognitiveServicesSpeech
import AVFoundation
import AudioToolbox
/// Azure ASR工具类,负责实现语音识别服务接口
@available(iOS 13.0, *)
@ -18,6 +19,12 @@ class AzureAsrHelper: NSObject {
private var speechConfig: SPXSpeechConfiguration?
private var recognizer: SPXSpeechRecognizer?
private var audioConfig: SPXAudioConfiguration?
private var pushStream: SPXPushAudioInputStream?
/// 音频处理相关
private var audioProcessor: CustomAudioProcessor?
private var isProcessingAudio = false
private var audioProcessingTimer: Timer?
/// 状态标志
private var isInitialized = false
@ -100,8 +107,12 @@ class AzureAsrHelper: NSObject {
// 设置音频输入参数
try setupAudioSession()
// 直接使用麦克风音频配置
audioConfig = try SPXAudioConfiguration()
// 创建自定义推送流,替代默认的麦克风输入
pushStream = try SPXPushAudioInputStream()
audioConfig = try SPXAudioConfiguration(streamInput: pushStream!)
// 初始化自定义音频处理器
audioProcessor = CustomAudioProcessor()
// 设置语言配置
if isAutoDetectLanguage {
@ -143,7 +154,7 @@ class AzureAsrHelper: NSObject {
// 使用playAndRecord类别允许同时录音和播放
try audioSession.setCategory(.playAndRecord,
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay])
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers])
// 设置首选的输入和输出
let currentRoute = audioSession.currentRoute
@ -262,6 +273,9 @@ class AzureAsrHelper: NSObject {
}
do {
// 启动音频处理
startAudioProcessing()
// 通知会话开始
eventHandler("sessionStarted", [:])
@ -269,6 +283,9 @@ class AzureAsrHelper: NSObject {
try recognizer?.recognizeOnceAsync { [weak self] result in
guard let self = self else { return }
// 停止音频处理
self.stopAudioProcessing()
if result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: result)
self.eventHandler("result", [
@ -294,6 +311,7 @@ class AzureAsrHelper: NSObject {
} catch {
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"])
stopAudioProcessing()
return false
}
}
@ -325,6 +343,9 @@ class AzureAsrHelper: NSObject {
}
do {
// 启动音频处理
startAudioProcessing()
// 启动连续识别
try recognizer?.startContinuousRecognition()
_isContinuousRecognitionActive = true
@ -335,6 +356,7 @@ class AzureAsrHelper: NSObject {
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"])
_isContinuousRecognitionActive = false
stopAudioProcessing()
return false
}
}
@ -342,6 +364,9 @@ class AzureAsrHelper: NSObject {
/// 停止连续语音识别
/// - Returns: 是否成功停止识别
func stopContinuousRecognition() -> Bool {
// 停止音频处理
stopAudioProcessing()
if !_isContinuousRecognitionActive || recognizer == nil {
return true
}
@ -369,6 +394,9 @@ class AzureAsrHelper: NSObject {
func dispose() {
print("[AzureAsrHelper] 释放资源")
// 停止音频处理
stopAudioProcessing()
// 停止连续识别
if _isContinuousRecognitionActive {
stopContinuousRecognition()
@ -385,6 +413,8 @@ class AzureAsrHelper: NSObject {
recognizer = nil
speechConfig = nil
audioConfig = nil
pushStream = nil
audioProcessor = nil
// 重置状态
_isContinuousRecognitionActive = false
@ -405,4 +435,323 @@ class AzureAsrHelper: NSObject {
return currentLanguage
}
}
// MARK: - 音频处理
/// 开始音频处理
private func startAudioProcessing() {
guard !isProcessingAudio, let audioProcessor = audioProcessor else { return }
isProcessingAudio = true
// 启动音频处理器
if !audioProcessor.startRecord() {
print("[AzureAsrHelper] 错误: 启动音频处理器失败")
eventHandler("error", ["message": "启动音频处理器失败"])
return
}
// 启动音频处理定时器
audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in
guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else {
return
}
// 读取处理后的音频数据
var bytes = [UInt8](repeating: 0, count: 2560)
let bytesRead = processor.read(bytes: &bytes)
if bytesRead > 0 {
// 推送数据到Azure语音服务
let data = Data(bytes: bytes, count: bytesRead)
stream.write(data)
// 通知音频数据可用
self.eventHandler("audioData", ["data": bytes])
}
}
print("[AzureAsrHelper] 音频处理已启动")
}
/// 停止音频处理
private func stopAudioProcessing() {
// 停止定时器
audioProcessingTimer?.invalidate()
audioProcessingTimer = nil
// 停止音频处理器
audioProcessor?.stopRecord()
isProcessingAudio = false
print("[AzureAsrHelper] 音频处理已停止")
}
}
// MARK: - 自定义音频处理器
@available(iOS 13.0, *)
class CustomAudioProcessor: NSObject {
// 音频单元
private var ioUnit: AudioUnit?
// 音频格式
private var audioFormat: AudioStreamBasicDescription
// 音频缓冲
private var audioBufferList: AudioBufferList
private var audioList: [Float] = []
private let audioListQueue = DispatchQueue(label: "audioListQueue")
// 回音消除状态
private var isEchoCancellationEnabled = true
override init() {
// 设置音频格式 - 16kHz, 16位, 单声道
audioFormat = AudioStreamBasicDescription(
mSampleRate: 16000.0,
mFormatID: kAudioFormatLinearPCM,
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
mBytesPerPacket: 2,
mFramesPerPacket: 1,
mBytesPerFrame: 2,
mChannelsPerFrame: 1,
mBitsPerChannel: 16,
mReserved: 0
)
// 初始化音频缓冲
audioBufferList = AudioBufferList(
mNumberBuffers: 1,
mBuffers: AudioBuffer(
mNumberChannels: 1,
mDataByteSize: 4096,
mData: malloc(4096)
)
)
super.init()
}
deinit {
stopRecord()
free(audioBufferList.mBuffers.mData)
}
/// 启动音频处理
/// - Returns: 是否成功启动
func startRecord() -> Bool {
print("[CustomAudioProcessor] 配置音频单元")
// 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除
var ioUnitDescription = AudioComponentDescription(
componentType: kAudioUnitType_Output,
componentSubType: kAudioUnitSubType_VoiceProcessingIO,
componentManufacturer: kAudioUnitManufacturer_Apple,
componentFlags: 0,
componentFlagsMask: 0
)
// 查找音频组件
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
print("[CustomAudioProcessor] 错误: 未找到音频组件")
return false
}
// 创建音频单元实例
if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") {
ioUnit = nil
return false
}
// 启用输入端口
var enableInput: UInt32 = 1
let kInputBus: AudioUnitElement = 1
let kOutputBus: AudioUnitElement = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Input, kInputBus, &enableInput,
UInt32(MemoryLayout<UInt32>.size)), "启用输入端口") {
return false
}
// 禁用输出端口 (我们只需要输入)
var enableOutput: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Output, kOutputBus,
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "禁用输出端口") {
return false
}
// 设置缓冲区分配标志
var flag: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置缓冲区分配标志") {
return false
}
// 设置音频格式
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size)
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") {
return false
}
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") {
return false
}
// 启用回音消除
if isEchoCancellationEnabled {
var echoCancellation: UInt32 = 1
AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing,
kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout<UInt32>.size))
}
// 设置输入回调 - 当有新音频数据时调用
var inputCallback = AURenderCallbackStruct(
inputProc: CustomAudioProcessor.onAudioDataAvailable,
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
)
if checkError(AudioUnitSetProperty(ioUnit!,
kAudioOutputUnitProperty_SetInputCallback,
kAudioUnitScope_Global, kInputBus,
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") {
return false
}
// 初始化音频单元
var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
while hasError {
Thread.sleep(forTimeInterval: 0.1)
hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
}
// 启动音频单元
hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元")
print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")")
return !hasError
}
/// 停止音频处理
func stopRecord() {
print("[CustomAudioProcessor] 停止音频处理器")
if let ioUnit = ioUnit {
// 停止音频单元
_ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元")
// 关闭音频单元
_ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元")
_ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元")
self.ioUnit = nil
}
// 清空音频数据缓冲
audioListQueue.sync {
audioList.removeAll()
}
}
/// 音频数据回调 - 当有新的音频数据可用时调用
private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
// 获取实例
let processor = Unmanaged<CustomAudioProcessor>.fromOpaque(inRefCon).takeUnretainedValue()
// 计算预期数据大小
let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame
// 确保缓冲区足够大
if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
}
// 渲染音频数据
let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp,
inBusNumber, inNumberFrames, &processor.audioBufferList),
"渲染音频数据")
// 将Int16数据转换为浮点数据进行处理
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
let buffer = processor.audioBufferList.mBuffers
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) {
// 归一化到[-1.0, 1.0]范围
audioDataFloat[j] = Float(bufferData[j]) / 32768.0
}
// 应用附加处理 (如有需要)
// processor.applyAdditionalProcessing(&audioDataFloat)
// 保存处理后的数据
if status == noErr {
processor.audioListQueue.async {
processor.audioList.append(contentsOf: audioDataFloat)
}
}
return status
}
/// 读取处理后的音频数据
/// - Parameter bytes: 输出字节数组
/// - Returns: 读取的字节数
func read(bytes: inout [UInt8]) -> Int {
return audioListQueue.sync {
// 如果没有数据,返回0
if audioList.isEmpty {
return 0
}
// 确保有足够的数据 (至少1280个样本)
if audioList.count < 1280 {
return 0
}
// 读取一帧数据 (1280个样本)
let frameLength = 1280
let buffer = Array(audioList.prefix(frameLength))
audioList.removeFirst(frameLength)
// 将浮点数据转回Int16格式
var int16Data = buffer.map { Int16($0 * 32767) }
// 转换为字节数组
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
bytes = [UInt8](data)
// 每个样本2字节 (16位PCM)
return frameLength * 2
}
}
/// 检查错误并打印日志
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 是否发生错误
private func checkError(_ status: OSStatus, _ operation: String) -> Bool {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
return true
}
return false
}
/// 检查OSStatus并返回状态
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 原始状态
private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
}
return status
}
}

152
ios/Runner/AudioSessionManager.swift

@ -190,9 +190,9 @@ import UIKit
// NSLog("[AudioSessionManager] 配置音频会话: 类别=\(category), 模式=\(mode)")
// 暂时停用当前会话(如果正在活动)
// if isActive {
// try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
// }
if isActive {
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
}
// 设置新的分类、模式和选项
try audioSession.setCategory(category, mode: mode, options: options)
@ -214,9 +214,9 @@ import UIKit
isConfigured = true
// 检查并设置首选的输入设备(如果需要)
// if category == .playAndRecord || category == .record {
// try configurePreferredInput()
// }
if category == .playAndRecord || category == .record {
try configurePreferredInput()
}
return
} catch {
NSLog("[AudioSessionManager] 配置音频会话失败: \(error.localizedDescription)")
@ -420,35 +420,141 @@ import UIKit
/// 配置用于语音交互(同时支持ASR和TTS)的音频会话
/// 此函数综合优化语音识别和语音合成,适用于需要双向交互的场景
/// 为Azure语音识别专门配置的音频会话
/// 此函数优化Azure语音识别的音频处理配置,提供更强的回音消除能力
/// - Parameters:
/// - force: 是否强制重新配置
/// - configureAdditionalSettings: 配置完成后进行额外的音频设置
/// - Returns: 配置是否成功
@objc func configureForVoiceInteraction(force: Bool = false) -> Bool {
@objc func configureForAzureSpeechRecognition(force: Bool = false, configureAdditionalSettings: Bool = true) -> Bool {
NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话")
do {
// 配置音频会话以支持语音交互
// 使用playAndRecord类别允许同时录音和播放
// 使用spokenAudio模式优化语音交互
// 配置音频会话以优化语音识别
try configureSession(
category: .playAndRecord,
mode: .spokenAudio,
category: .playAndRecord, // 允许同时录音和播放
mode: .voiceChat, // 使用voiceChat模式获得最佳回音消除效果
options: [
.allowBluetooth, // 允许蓝牙设备
.defaultToSpeaker, // 默认使用扬声器
.mixWithOthers // 允许与其他应用混音
],
force: force // 是否强制重新配置
)
// 额外的优化配置
if configureAdditionalSettings {
// 获取当前是否使用耳机
let currentRoute = audioSession.currentRoute
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
}
// 设置采样率为16kHz(Azure Speech API推荐)
try audioSession.setPreferredSampleRate(16000.0)
// 设置较小的缓冲区大小以减少延迟
try audioSession.setPreferredIOBufferDuration(0.01)
// 根据是否有耳机连接调整输入增益
if !hasHeadphones {
// 无耳机时降低输入增益以减少回音
try audioSession.setInputGain(0.8)
NSLog("[AudioSessionManager] 启用扬声器回音消除优化")
} else {
// 使用耳机时可以使用较高增益
try audioSession.setInputGain(1.0)
NSLog("[AudioSessionManager] 检测到耳机连接,应用耳机模式")
}
}
NSLog("[AudioSessionManager] 已成功配置Azure语音识别的音频会话")
return true
} catch {
NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话失败: \(error.localizedDescription)")
return false
}
}
@objc func configureForVoiceInteraction(force: Bool = false, configureAdditionalSettings: Bool = true) -> Bool {
NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话")
do {
// 配置音频会话以优化语音识别
try configureSession(
category: .playAndRecord, // 允许同时录音和播放
mode: .voiceChat, // 使用voiceChat模式获得最佳回音消除效果
options: [
.allowBluetoothA2DP, // 允许通过蓝牙A2DP连接进行高质量音频输出
// .allowBluetooth, // 允许通过蓝牙SCO连接进行输入和输出
// .defaultToSpeaker, // 默认使用扬声器输出
// .mixWithOthers
.allowBluetooth, // 允许蓝牙设备
.defaultToSpeaker, // 默认使用扬声器
.mixWithOthers // 允许与其他应用混音
],
force: force // 强制重新配置,确保设置生效
force: force // 是否强制重新配置
)
// 设置首选输入设备
// try configurePreferredInput()
// 额外的优化配置
if configureAdditionalSettings {
// 获取当前是否使用耳机
let currentRoute = audioSession.currentRoute
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
}
// 设置采样率为16kHz(Azure Speech API推荐)
try audioSession.setPreferredSampleRate(16000.0)
// 设置较小的缓冲区大小以减少延迟
try audioSession.setPreferredIOBufferDuration(0.01)
// 根据是否有耳机连接调整输入增益
if !hasHeadphones {
// 无耳机时降低输入增益以减少回音
try audioSession.setInputGain(0.8)
NSLog("[AudioSessionManager] 启用扬声器回音消除优化")
} else {
// 使用耳机时可以使用较高增益
try audioSession.setInputGain(1.0)
NSLog("[AudioSessionManager] 检测到耳机连接,应用耳机模式")
}
}
NSLog("[AudioSessionManager] 已配置语音交互(ASR+TTS)的音频会话")
NSLog("[AudioSessionManager] 已成功配置Azure语音识别的音频会话")
return true
} catch {
NSLog("[AudioSessionManager] 配置语音交互(ASR+TTS)的音频会话失败: \(error.localizedDescription)")
NSLog("[AudioSessionManager] 配置Azure语音识别的音频会话失败: \(error.localizedDescription)")
return false
}
}
/// 配置用于语音交互(同时支持ASR和TTS)的音频会话
/// 此函数综合优化语音识别和语音合成,适用于需要双向交互的场景
/// - Returns: 配置是否成功
// @objc func configureForVoiceInteraction(force: Bool = false) -> Bool {
// do {
// // 配置音频会话以支持语音交互
// // 使用playAndRecord类别允许同时录音和播放
// // 使用spokenAudio模式优化语音交互
// try configureSession(
// category: .playAndRecord,
// mode: .voiceChat,
// options: [
// // .allowBluetoothA2DP, // 允许通过蓝牙A2DP连接进行高质量音频输出
// .allowBluetooth, // 允许通过蓝牙SCO连接进行输入和输出
// .defaultToSpeaker, // 默认使用扬声器输出
// .mixWithOthers
// ],
// force: force // 强制重新配置,确保设置生效
// )
// // 设置首选输入设备
// try configurePreferredInput()
// NSLog("[AudioSessionManager] 已配置语音交互(ASR+TTS)的音频会话")
// return true
// } catch {
// NSLog("[AudioSessionManager] 配置语音交互(ASR+TTS)的音频会话失败: \(error.localizedDescription)")
// return false
// }
// }
}

373
ios/Runner/AzureAsrHelper.swift

@ -1,6 +1,7 @@
import Foundation
import MicrosoftCognitiveServicesSpeech
import AVFoundation
import AudioToolbox
/// Azure ASR工具类,负责实现语音识别服务接口
@available(iOS 13.0, *)
@ -19,6 +20,12 @@ class AzureAsrHelper: NSObject {
private var recognizer: SPXSpeechRecognizer?
private var audioConfig: SPXAudioConfiguration?
/// 添加自定义音频处理相关
private var pushStream: SPXPushAudioInputStream?
private var audioProcessor: CustomAudioProcessor?
private var isProcessingAudio = false
private var audioProcessingTimer: Timer?
/// 状态标志
private var isInitialized = false
private var _isContinuousRecognitionActive = false
@ -108,8 +115,12 @@ class AzureAsrHelper: NSObject {
// 设置音频输入参数
try setupAudioSession()
// 直接使用麦克风音频配置
audioConfig = try SPXAudioConfiguration()
// 创建自定义音频流和处理器,替代默认的麦克风输入
pushStream = try SPXPushAudioInputStream()
audioConfig = try SPXAudioConfiguration(streamInput: pushStream!)
// 初始化自定义音频处理器
audioProcessor = CustomAudioProcessor()
// 设置语言配置
if isAutoDetectLanguage {
@ -147,11 +158,15 @@ class AzureAsrHelper: NSObject {
/// 设置音频会话
private func setupAudioSession() throws {
print("[AzureAsrHelper] 开始配置音频会话...")
audioSessionManager.configureForVoiceInteraction()
// 使用AudioSessionManager配置音频会话,使用专门为Azure ASR优化的配置
let success = audioSessionManager.configureForAzureSpeechRecognition(force: true)
if !success {
print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败")
}
}
/// 设置所有回调
private func setupAllCallbacks() {
guard let recognizer = recognizer else { return }
@ -244,6 +259,9 @@ class AzureAsrHelper: NSObject {
}
do {
// 启动音频处理
startAudioProcessing()
// 通知会话开始
eventHandler?(["type": "sessionStarted"])
@ -251,6 +269,9 @@ class AzureAsrHelper: NSObject {
try recognizer?.recognizeOnceAsync { [weak self] result in
guard let self = self else { return }
// 停止音频处理
self.stopAudioProcessing()
if result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: result)
self.eventHandler?(["type": "result",
@ -276,6 +297,7 @@ class AzureAsrHelper: NSObject {
} catch {
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
eventHandler?(["type": "error", "message": "识别异常: \(error.localizedDescription)"])
stopAudioProcessing()
return false
}
}
@ -311,7 +333,9 @@ class AzureAsrHelper: NSObject {
return false
}
}
// 启动音频处理
startAudioProcessing()
// 尝试启动连续识别
do {
@ -319,13 +343,15 @@ class AzureAsrHelper: NSObject {
try recognizer?.startContinuousRecognition()
_isContinuousRecognitionActive = true
print("[AzureAsrHelper] 连续识别已启动")
return true
} catch {
_isContinuousRecognitionActive = false
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
// 停止音频处理
stopAudioProcessing()
// 发送错误通知
eventHandler?(["type": "error", "message": "开始连续识别失败: \(error.localizedDescription)"])
@ -337,6 +363,9 @@ class AzureAsrHelper: NSObject {
/// 停止连续语音识别
/// - Returns: 是否成功停止识别
func stopContinuousRecognition() -> Bool {
// 停止音频处理
stopAudioProcessing()
// 检查是否初始化
if !isInitialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
@ -362,14 +391,7 @@ class AzureAsrHelper: NSObject {
// 先标记为非活跃状态,防止重复调用
_isContinuousRecognitionActive = false
// do {
// try recognizer.stopContinuousRecognition()
// } catch {
// print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)")
// }
// // 异步执行停止操作,避免阻塞主线程
// 异步执行停止操作,避免阻塞主线程
DispatchQueue.global(qos: .userInitiated).async { [weak self] in
guard let self = self else { return }
@ -405,6 +427,9 @@ class AzureAsrHelper: NSObject {
/// 释放资源
func dispose() {
// 停止音频处理
stopAudioProcessing()
// 尝试停止所有识别操作
if _isContinuousRecognitionActive {
do {
@ -422,6 +447,8 @@ class AzureAsrHelper: NSObject {
recognizer = nil
audioConfig = nil
speechConfig = nil
pushStream = nil
audioProcessor = nil
// 重置状态
isInitialized = false
@ -448,4 +475,320 @@ class AzureAsrHelper: NSObject {
@objc func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) {
self.eventHandler = handler
}
// MARK: - 音频处理
/// 开始音频处理
private func startAudioProcessing() {
guard !isProcessingAudio, let audioProcessor = audioProcessor else { return }
isProcessingAudio = true
// 启动音频处理器
if !audioProcessor.startRecord() {
print("[AzureAsrHelper] 错误: 启动音频处理器失败")
eventHandler?(["type": "error", "message": "启动音频处理器失败"])
return
}
// 启动音频处理定时器
audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in
guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else {
return
}
// 读取处理后的音频数据
var bytes = [UInt8](repeating: 0, count: 2560)
let bytesRead = processor.read(bytes: &bytes)
if bytesRead > 0 {
// 推送数据到Azure语音服务
let data = Data(bytes: bytes, count: bytesRead)
stream.write(data)
// 通知音频数据可用(可选)
// self.eventHandler?(["type": "audioData", "data": bytes])
}
}
print("[AzureAsrHelper] 音频处理已启动")
}
/// 停止音频处理
private func stopAudioProcessing() {
// 停止定时器
audioProcessingTimer?.invalidate()
audioProcessingTimer = nil
// 停止音频处理器
audioProcessor?.stopRecord()
isProcessingAudio = false
print("[AzureAsrHelper] 音频处理已停止")
}
}
// MARK: - 自定义音频处理器
@available(iOS 13.0, *)
class CustomAudioProcessor: NSObject {
// 音频单元
private var ioUnit: AudioUnit?
// 音频格式
private var audioFormat: AudioStreamBasicDescription
// 音频缓冲
private var audioBufferList: AudioBufferList
private var audioList: [Float] = []
private let audioListQueue = DispatchQueue(label: "audioListQueue")
// 回音消除状态
private var isEchoCancellationEnabled = true
override init() {
// 设置音频格式 - 16kHz, 16位, 单声道
audioFormat = AudioStreamBasicDescription(
mSampleRate: 16000.0,
mFormatID: kAudioFormatLinearPCM,
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
mBytesPerPacket: 2,
mFramesPerPacket: 1,
mBytesPerFrame: 2,
mChannelsPerFrame: 1,
mBitsPerChannel: 16,
mReserved: 0
)
// 初始化音频缓冲
audioBufferList = AudioBufferList(
mNumberBuffers: 1,
mBuffers: AudioBuffer(
mNumberChannels: 1,
mDataByteSize: 4096,
mData: malloc(4096)
)
)
super.init()
}
deinit {
stopRecord()
free(audioBufferList.mBuffers.mData)
}
/// 启动音频处理
/// - Returns: 是否成功启动
func startRecord() -> Bool {
print("[CustomAudioProcessor] 配置音频单元")
// 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除
var ioUnitDescription = AudioComponentDescription(
componentType: kAudioUnitType_Output,
componentSubType: kAudioUnitSubType_VoiceProcessingIO,
componentManufacturer: kAudioUnitManufacturer_Apple,
componentFlags: 0,
componentFlagsMask: 0
)
// 查找音频组件
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
print("[CustomAudioProcessor] 错误: 未找到音频组件")
return false
}
// 创建音频单元实例
if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") {
ioUnit = nil
return false
}
// 启用输入端口
var enableInput: UInt32 = 1
let kInputBus: AudioUnitElement = 1
let kOutputBus: AudioUnitElement = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Input, kInputBus, &enableInput,
UInt32(MemoryLayout<UInt32>.size)), "启用输入端口") {
return false
}
// 禁用输出端口 (我们只需要输入)
var enableOutput: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Output, kOutputBus,
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "禁用输出端口") {
return false
}
// 设置缓冲区分配标志
var flag: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置缓冲区分配标志") {
return false
}
// 设置音频格式
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size)
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") {
return false
}
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") {
return false
}
// 启用回音消除 - 注意: kAUVoiceIOProperty_BypassVoiceProcessing值为1时表示绕过处理,值为0表示启用处理
if isEchoCancellationEnabled {
var echoCancellation: UInt32 = 0 // 0表示不绕过,即启用回音消除
AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing,
kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout<UInt32>.size))
}
// 设置输入回调 - 当有新音频数据时调用
var inputCallback = AURenderCallbackStruct(
inputProc: CustomAudioProcessor.onAudioDataAvailable,
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
)
if checkError(AudioUnitSetProperty(ioUnit!,
kAudioOutputUnitProperty_SetInputCallback,
kAudioUnitScope_Global, kInputBus,
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") {
return false
}
// 初始化音频单元
var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
while hasError {
Thread.sleep(forTimeInterval: 0.1)
hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
}
// 启动音频单元
hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元")
print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")")
return !hasError
}
/// 停止音频处理
func stopRecord() {
print("[CustomAudioProcessor] 停止音频处理器")
if let ioUnit = ioUnit {
// 停止音频单元
_ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元")
// 关闭音频单元
_ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元")
_ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元")
self.ioUnit = nil
}
// 清空音频数据缓冲
audioListQueue.sync {
audioList.removeAll()
}
}
/// 音频数据回调 - 当有新的音频数据可用时调用
private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
// 获取实例
let processor = Unmanaged<CustomAudioProcessor>.fromOpaque(inRefCon).takeUnretainedValue()
// 计算预期数据大小
let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame
// 确保缓冲区足够大
if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
}
// 渲染音频数据
let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp,
inBusNumber, inNumberFrames, &processor.audioBufferList),
"渲染音频数据")
// 将Int16数据转换为浮点数据进行处理
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
let buffer = processor.audioBufferList.mBuffers
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) {
// 归一化到[-1.0, 1.0]范围
audioDataFloat[j] = Float(bufferData[j]) / 32768.0
}
// 保存处理后的数据
if status == noErr {
processor.audioListQueue.async {
processor.audioList.append(contentsOf: audioDataFloat)
}
}
return status
}
/// 读取处理后的音频数据
/// - Parameter bytes: 输出字节数组
/// - Returns: 读取的字节数
func read(bytes: inout [UInt8]) -> Int {
return audioListQueue.sync {
// 如果没有数据,返回0
if audioList.isEmpty {
return 0
}
// 确保有足够的数据 (至少1280个样本)
if audioList.count < 1280 {
return 0
}
// 读取一帧数据 (1280个样本)
let frameLength = 1280
let buffer = Array(audioList.prefix(frameLength))
audioList.removeFirst(frameLength)
// 将浮点数据转回Int16格式
var int16Data = buffer.map { Int16($0 * 32767) }
// 转换为字节数组
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
bytes = [UInt8](data)
// 每个样本2字节 (16位PCM)
return frameLength * 2
}
}
/// 检查错误并打印日志
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 是否发生错误
private func checkError(_ status: OSStatus, _ operation: String) -> Bool {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
return true
}
return false
}
/// 检查OSStatus并返回状态
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 原始状态
private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
}
return status
}
}

2
ios/Runner/ClassicBluetoothHelper.swift

@ -80,7 +80,7 @@ import UIKit
NotificationCenter.default.removeObserver(self)
// 添加新的监听器
audioSessionManager.addRouteChangeListener(self, selector: #selector(handleRouteChange(_:)))
BluetoothMediaButtonHelper.shared.startButtonListening()
// BluetoothMediaButtonHelper.shared.startButtonListening()
}
// 处理音频路由变化

Loading…
Cancel
Save