|
|
@ -1,254 +1,6 @@ |
|
|
import Foundation |
|
|
import Foundation |
|
|
import MicrosoftCognitiveServicesSpeech |
|
|
import MicrosoftCognitiveServicesSpeech |
|
|
import AVFoundation |
|
|
import AVFoundation |
|
|
import AudioToolbox |
|
|
|
|
|
|
|
|
|
|
|
/// 自定义麦克风流实现,优化ASR语音输入捕获 |
|
|
|
|
|
class AzureMicrophoneStream: NSObject { |
|
|
|
|
|
var ioUnit: AudioUnit? |
|
|
|
|
|
var audioFormat: AudioStreamBasicDescription |
|
|
|
|
|
var audioBufferList: AudioBufferList |
|
|
|
|
|
var audioList: [Float] = [] |
|
|
|
|
|
let audioListQueue = DispatchQueue(label: "azureAudioListQueue") |
|
|
|
|
|
private var isActive = false |
|
|
|
|
|
|
|
|
|
|
|
override init() { |
|
|
|
|
|
// 音频会话配置 |
|
|
|
|
|
let audioSession = AVAudioSession.sharedInstance() |
|
|
|
|
|
do { |
|
|
|
|
|
try audioSession.setCategory(.playAndRecord, |
|
|
|
|
|
mode: .voiceChat, |
|
|
|
|
|
options: [.allowBluetooth, .defaultToSpeaker, .mixWithOthers]) |
|
|
|
|
|
try audioSession.setActive(true) |
|
|
|
|
|
print("[AzureMicrophoneStream] 音频会话配置成功") |
|
|
|
|
|
} catch { |
|
|
|
|
|
print("[AzureMicrophoneStream] 配置音频会话失败: \(error.localizedDescription)") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 设置音频格式为16KHz 16位单声道PCM |
|
|
|
|
|
audioFormat = AudioStreamBasicDescription( |
|
|
|
|
|
mSampleRate: 16000.0, |
|
|
|
|
|
mFormatID: kAudioFormatLinearPCM, |
|
|
|
|
|
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, |
|
|
|
|
|
mBytesPerPacket: 2, |
|
|
|
|
|
mFramesPerPacket: 1, |
|
|
|
|
|
mBytesPerFrame: 2, |
|
|
|
|
|
mChannelsPerFrame: 1, |
|
|
|
|
|
mBitsPerChannel: 16, |
|
|
|
|
|
mReserved: 0 |
|
|
|
|
|
) |
|
|
|
|
|
|
|
|
|
|
|
audioBufferList = AudioBufferList( |
|
|
|
|
|
mNumberBuffers: 1, |
|
|
|
|
|
mBuffers: AudioBuffer( |
|
|
|
|
|
mNumberChannels: audioFormat.mChannelsPerFrame, |
|
|
|
|
|
mDataByteSize: 4096, |
|
|
|
|
|
mData: malloc(4096) |
|
|
|
|
|
) |
|
|
|
|
|
) |
|
|
|
|
|
|
|
|
|
|
|
super.init() |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
func start() -> Bool { |
|
|
|
|
|
if isActive { |
|
|
|
|
|
return true // 已经在运行 |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if setupAudioUnit() { |
|
|
|
|
|
isActive = startAudioUnit() |
|
|
|
|
|
return isActive |
|
|
|
|
|
} |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
private func setupAudioUnit() -> Bool { |
|
|
|
|
|
print("[AzureMicrophoneStream] 设置音频单元") |
|
|
|
|
|
|
|
|
|
|
|
var ioUnitDescription = AudioComponentDescription( |
|
|
|
|
|
componentType: kAudioUnitType_Output, |
|
|
|
|
|
componentSubType: kAudioUnitSubType_VoiceProcessingIO, |
|
|
|
|
|
componentManufacturer: kAudioUnitManufacturer_Apple, |
|
|
|
|
|
componentFlags: 0, |
|
|
|
|
|
componentFlagsMask: 0 |
|
|
|
|
|
) |
|
|
|
|
|
|
|
|
|
|
|
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { |
|
|
|
|
|
print("[AzureMicrophoneStream] 无法找到音频组件") |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if CheckHasError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建IO单元") { |
|
|
|
|
|
ioUnit = nil |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
var enableInput: UInt32 = 1 |
|
|
|
|
|
let kInputBus: AudioUnitElement = 1 |
|
|
|
|
|
let kOutputBus: AudioUnitElement = 0 |
|
|
|
|
|
|
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|
|
|
|
|
kAudioUnitScope_Input, kInputBus, &enableInput, |
|
|
|
|
|
UInt32(MemoryLayout<UInt32>.size)), "设置输入总线的EnableIO属性") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
var enableOutput: UInt32 = 0 |
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|
|
|
|
|
kAudioUnitScope_Output, kOutputBus, |
|
|
|
|
|
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "设置输出总线的EnableIO属性") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
var flag: UInt32 = 0 |
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, |
|
|
|
|
|
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置ShouldAllocateBuffer属性") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size) |
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|
|
|
|
|
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线的StreamFormat属性") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|
|
|
|
|
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线的StreamFormat属性") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
var inputCallback = AURenderCallbackStruct( |
|
|
|
|
|
inputProc: AzureMicrophoneStream.OnRecordedDataIsAvailable, |
|
|
|
|
|
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) |
|
|
|
|
|
) |
|
|
|
|
|
|
|
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, |
|
|
|
|
|
kAudioOutputUnitProperty_SetInputCallback, |
|
|
|
|
|
kAudioUnitScope_Global, kInputBus, |
|
|
|
|
|
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
var hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "初始化IO单元") |
|
|
|
|
|
if hasError { |
|
|
|
|
|
// 如果初始化失败,重试一次 |
|
|
|
|
|
Thread.sleep(forTimeInterval: 0.5) |
|
|
|
|
|
hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "重试初始化IO单元") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureMicrophoneStream] 音频单元设置\(hasError ? "失败" : "成功")") |
|
|
|
|
|
return !hasError |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
private func startAudioUnit() -> Bool { |
|
|
|
|
|
print("[AzureMicrophoneStream] 启动音频单元") |
|
|
|
|
|
guard let ioUnit = ioUnit else { |
|
|
|
|
|
print("[AzureMicrophoneStream] IO单元未初始化") |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
return !CheckHasError(AudioOutputUnitStart(ioUnit), "启动IO单元") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
func stop() { |
|
|
|
|
|
print("[AzureMicrophoneStream] 停止音频单元") |
|
|
|
|
|
guard isActive, let ioUnit = ioUnit else { return } |
|
|
|
|
|
|
|
|
|
|
|
_ = CheckHasError(AudioOutputUnitStop(ioUnit), "停止IO单元") |
|
|
|
|
|
isActive = false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
func dispose() { |
|
|
|
|
|
stop() |
|
|
|
|
|
|
|
|
|
|
|
if let ioUnit = ioUnit { |
|
|
|
|
|
_ = CheckHasError(AudioUnitUninitialize(ioUnit), "反初始化IO单元") |
|
|
|
|
|
_ = CheckHasError(AudioComponentInstanceDispose(ioUnit), "释放IO单元") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 释放缓冲区 |
|
|
|
|
|
if let buffer = audioBufferList.mBuffers.mData { |
|
|
|
|
|
free(buffer) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
self.ioUnit = nil |
|
|
|
|
|
|
|
|
|
|
|
// 清空音频数据 |
|
|
|
|
|
audioListQueue.sync { |
|
|
|
|
|
audioList.removeAll() |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
static let OnRecordedDataIsAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in |
|
|
|
|
|
let wrapper = Unmanaged<AzureMicrophoneStream>.fromOpaque(inRefCon).takeUnretainedValue() |
|
|
|
|
|
let expectedDataByteSize = inNumberFrames * wrapper.audioFormat.mBytesPerFrame |
|
|
|
|
|
|
|
|
|
|
|
if wrapper.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { |
|
|
|
|
|
wrapper.audioBufferList.mBuffers.mData = realloc(wrapper.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) |
|
|
|
|
|
wrapper.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
let status = wrapper.CheckErrorStatus(AudioUnitRender(wrapper.ioUnit!, ioActionFlags, inTimeStamp, |
|
|
|
|
|
inBusNumber, inNumberFrames, &wrapper.audioBufferList), |
|
|
|
|
|
"AudioUnitRender调用") |
|
|
|
|
|
|
|
|
|
|
|
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) |
|
|
|
|
|
let buffer = wrapper.audioBufferList.mBuffers |
|
|
|
|
|
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) |
|
|
|
|
|
|
|
|
|
|
|
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) { |
|
|
|
|
|
audioDataFloat[j] = Float(bufferData[j]) / 32768.0 // 归一化到[-1.0, 1.0]范围 |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if status == noErr { |
|
|
|
|
|
wrapper.audioListQueue.async { |
|
|
|
|
|
wrapper.audioList.append(contentsOf: audioDataFloat) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
return status |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
private func CheckHasError(_ status: OSStatus, _ operation: String) -> Bool { |
|
|
|
|
|
if status != noErr { |
|
|
|
|
|
print("[AzureMicrophoneStream] \(operation)失败: \(status)") |
|
|
|
|
|
return true |
|
|
|
|
|
} |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
private func CheckErrorStatus(_ status: OSStatus, _ operation: String) -> OSStatus { |
|
|
|
|
|
if status != noErr { |
|
|
|
|
|
print("[AzureMicrophoneStream] \(operation)失败: \(status)") |
|
|
|
|
|
} |
|
|
|
|
|
return status |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 读取音频数据,适配Azure SDK |
|
|
|
|
|
func read(bytes: inout [UInt8]) -> Int { |
|
|
|
|
|
return audioListQueue.sync { |
|
|
|
|
|
if audioList.isEmpty { |
|
|
|
|
|
return 0 |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 确保有足够的数据 |
|
|
|
|
|
let minFrames = 1280 |
|
|
|
|
|
if audioList.count < minFrames { |
|
|
|
|
|
return 0 |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
let frameLength = minFrames |
|
|
|
|
|
|
|
|
|
|
|
let buffer = Array(audioList.prefix(frameLength)) |
|
|
|
|
|
audioList.removeFirst(frameLength) |
|
|
|
|
|
|
|
|
|
|
|
// 转换为Int16数据 |
|
|
|
|
|
var int16Data = buffer.map { Int16($0 * 32767) } |
|
|
|
|
|
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) |
|
|
|
|
|
bytes = [UInt8](data) |
|
|
|
|
|
|
|
|
|
|
|
return frameLength * 2 // 每个样本2字节(16位PCM) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/// Azure ASR工具类,负责实现语音识别服务接口 |
|
|
/// Azure ASR工具类,负责实现语音识别服务接口 |
|
|
@available(iOS 13.0, *) |
|
|
@available(iOS 13.0, *) |
|
|
@ -267,10 +19,6 @@ class AzureAsrHelper: NSObject { |
|
|
private var recognizer: SPXSpeechRecognizer? |
|
|
private var recognizer: SPXSpeechRecognizer? |
|
|
private var audioConfig: SPXAudioConfiguration? |
|
|
private var audioConfig: SPXAudioConfiguration? |
|
|
|
|
|
|
|
|
/// 麦克风流 |
|
|
|
|
|
private var microphoneStream: AzureMicrophoneStream? |
|
|
|
|
|
private var pushStreamConfig: SPXPushAudioInputStream? |
|
|
|
|
|
|
|
|
|
|
|
/// 状态标志 |
|
|
/// 状态标志 |
|
|
private var isInitialized = false |
|
|
private var isInitialized = false |
|
|
private var _isContinuousRecognitionActive = false |
|
|
private var _isContinuousRecognitionActive = false |
|
|
@ -329,16 +77,18 @@ class AzureAsrHelper: NSObject { |
|
|
currentLanguage = self.supportedLanguages[0] |
|
|
currentLanguage = self.supportedLanguages[0] |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 创建麦克风流 |
|
|
// 创建识别器和设置回调 |
|
|
microphoneStream = AzureMicrophoneStream() |
|
|
if !createRecognizerAndSetupCallbacks() { |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
|
|
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
|
|
isInitialized = true |
|
|
isInitialized = true |
|
|
return true |
|
|
return true |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
/// 重置 recognizer |
|
|
/// 创建识别器并设置回调 |
|
|
private func resetRecognizer() -> Bool { |
|
|
private func createRecognizerAndSetupCallbacks() -> Bool { |
|
|
// 释放之前的 recognizer |
|
|
// 释放之前的 recognizer |
|
|
recognizer = nil |
|
|
recognizer = nil |
|
|
audioConfig = nil |
|
|
audioConfig = nil |
|
|
@ -347,11 +97,11 @@ class AzureAsrHelper: NSObject { |
|
|
// 创建语音配置 |
|
|
// 创建语音配置 |
|
|
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|
|
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|
|
|
|
|
|
|
|
// 创建推送流 |
|
|
// 设置音频输入参数 |
|
|
pushStreamConfig = try SPXPushAudioInputStream() |
|
|
try setupAudioSession() |
|
|
|
|
|
|
|
|
// 创建音频配置,使用推送流 |
|
|
// 直接使用麦克风音频配置 |
|
|
audioConfig = try SPXAudioConfiguration(streamInput: pushStreamConfig!) |
|
|
audioConfig = try SPXAudioConfiguration() |
|
|
|
|
|
|
|
|
// 设置语言配置 |
|
|
// 设置语言配置 |
|
|
if isAutoDetectLanguage { |
|
|
if isAutoDetectLanguage { |
|
|
@ -375,65 +125,121 @@ class AzureAsrHelper: NSObject { |
|
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 设置所有回调 |
|
|
|
|
|
setupAllCallbacks() |
|
|
|
|
|
|
|
|
return true |
|
|
return true |
|
|
} catch { |
|
|
} catch { |
|
|
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") |
|
|
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") |
|
|
eventHandler("error", ["message": "重置识别器失败: \(error.localizedDescription)"]) |
|
|
eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) |
|
|
return false |
|
|
return false |
|
|
} |
|
|
} |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
/// 启动音频捕获和数据推送 |
|
|
/// 设置音频会话 |
|
|
private func startAudioStream() -> Bool { |
|
|
private func setupAudioSession() throws { |
|
|
guard let micStream = microphoneStream else { |
|
|
let audioSession = AVAudioSession.sharedInstance() |
|
|
print("[AzureAsrHelper] 错误: 麦克风流未初始化") |
|
|
|
|
|
return false |
|
|
// 使用playAndRecord类别允许同时录音和播放 |
|
|
|
|
|
try audioSession.setCategory(.playAndRecord, |
|
|
|
|
|
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 |
|
|
|
|
|
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay]) |
|
|
|
|
|
|
|
|
|
|
|
// 设置首选的输入和输出 |
|
|
|
|
|
let currentRoute = audioSession.currentRoute |
|
|
|
|
|
|
|
|
|
|
|
// 获取当前是否连接了耳机或外部麦克风 |
|
|
|
|
|
let hasHeadphones = currentRoute.outputs.contains { |
|
|
|
|
|
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 启动麦克风 |
|
|
// 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 |
|
|
if !micStream.start() { |
|
|
if !hasHeadphones { |
|
|
print("[AzureAsrHelper] 错误: 启动麦克风流失败") |
|
|
try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 |
|
|
return false |
|
|
|
|
|
|
|
|
// 启用回音消除和噪声抑制 |
|
|
|
|
|
try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 |
|
|
|
|
|
} else { |
|
|
|
|
|
// 耳机模式,可以使用不同的设置 |
|
|
|
|
|
try audioSession.setMode(.voiceChat) |
|
|
|
|
|
try audioSession.setInputGain(1.0) |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 创建并启动音频推送线程 |
|
|
// 设置合适的采样率 |
|
|
DispatchQueue.global(qos: .userInitiated).async { [weak self] in |
|
|
try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 |
|
|
guard let self = self, let pushStream = self.pushStreamConfig else { return } |
|
|
try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 |
|
|
|
|
|
|
|
|
|
|
|
// 激活音频会话 |
|
|
|
|
|
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/// 设置所有回调 |
|
|
|
|
|
private func setupAllCallbacks() { |
|
|
|
|
|
guard let recognizer = recognizer else { return } |
|
|
|
|
|
|
|
|
|
|
|
// 最终识别结果 |
|
|
|
|
|
recognizer.addRecognizedEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
var isRunning = true |
|
|
if event.result.reason == SPXResultReason.recognizedSpeech { |
|
|
var audioBuffer = [UInt8](repeating: 0, count: 16000) |
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
|
|
|
|
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
|
|
|
|
self.eventHandler("result", [ |
|
|
|
|
|
"text": event.result.text ?? "", |
|
|
|
|
|
"detectedLanguage": detectedLanguage |
|
|
|
|
|
]) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 识别中事件 |
|
|
|
|
|
recognizer.addRecognizingEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
while isRunning { |
|
|
if event.result.reason == SPXResultReason.recognizingSpeech { |
|
|
autoreleasepool { |
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
|
// 读取麦克风数据 |
|
|
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
|
let bytesRead = micStream.read(bytes: &audioBuffer) |
|
|
self.eventHandler("recognizing", [ |
|
|
|
|
|
"text": event.result.text ?? "", |
|
|
if bytesRead > 0 { |
|
|
"detectedLanguage": detectedLanguage |
|
|
do { |
|
|
]) |
|
|
// 推送音频数据到Azure识别流 |
|
|
|
|
|
let data = Data(bytes: audioBuffer, count: bytesRead) |
|
|
|
|
|
try pushStream.write(data) |
|
|
|
|
|
} catch { |
|
|
|
|
|
print("[AzureAsrHelper] 推送音频数据失败: \(error.localizedDescription)") |
|
|
|
|
|
isRunning = false |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 检查是否应该继续捕获 |
|
|
|
|
|
if !self._isContinuousRecognitionActive { |
|
|
|
|
|
isRunning = false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 添加适当的休眠以避免过度消耗CPU |
|
|
|
|
|
if bytesRead == 0 { |
|
|
|
|
|
Thread.sleep(forTimeInterval: 0.01) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
} |
|
|
print("[AzureAsrHelper] 音频推送线程已停止") |
|
|
|
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
return true |
|
|
// 会话事件 |
|
|
|
|
|
recognizer.addSessionStartedEventHandler { [weak self] _, _ in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] 识别会话已开始") |
|
|
|
|
|
self._isContinuousRecognitionActive = true |
|
|
|
|
|
self.eventHandler("sessionStarted", [:]) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
recognizer.addSessionStoppedEventHandler { [weak self] _, _ in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] 识别会话已结束") |
|
|
|
|
|
self._isContinuousRecognitionActive = false |
|
|
|
|
|
self.eventHandler("sessionStopped", [:]) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 取消事件 |
|
|
|
|
|
recognizer.addCanceledEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
let reason = event.reason.rawValue |
|
|
|
|
|
let errorDetails = event.errorDetails ?? "未知错误" |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] 识别取消: \(errorDetails)") |
|
|
|
|
|
|
|
|
|
|
|
self.eventHandler("canceled", [ |
|
|
|
|
|
"reason": reason, |
|
|
|
|
|
"errorDetails": errorDetails |
|
|
|
|
|
]) |
|
|
|
|
|
|
|
|
|
|
|
self._isContinuousRecognitionActive = false |
|
|
|
|
|
} |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
/// 执行一次性语音识别 |
|
|
/// 执行一次性语音识别 |
|
|
@ -450,37 +256,12 @@ class AzureAsrHelper: NSObject { |
|
|
stopContinuousRecognition() |
|
|
stopContinuousRecognition() |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 重置 recognizer |
|
|
// 确保识别器已创建 |
|
|
if !resetRecognizer() { |
|
|
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 启动音频流 |
|
|
|
|
|
if !startAudioStream() { |
|
|
|
|
|
return false |
|
|
return false |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
do { |
|
|
do { |
|
|
// 设置回调 |
|
|
|
|
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
if event.result.reason == SPXResultReason.recognizedSpeech { |
|
|
|
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
|
|
|
|
self.eventHandler("result", [ |
|
|
|
|
|
"text": event.result.text ?? "", |
|
|
|
|
|
"detectedLanguage": detectedLanguage |
|
|
|
|
|
]) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
recognizer?.addCanceledEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
let errorDetails = event.errorDetails ?? "未知错误" |
|
|
|
|
|
self.eventHandler("error", ["message": "识别异常: \(errorDetails)"]) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 通知会话开始 |
|
|
// 通知会话开始 |
|
|
eventHandler("sessionStarted", [:]) |
|
|
eventHandler("sessionStarted", [:]) |
|
|
|
|
|
|
|
|
@ -488,15 +269,15 @@ class AzureAsrHelper: NSObject { |
|
|
try recognizer?.recognizeOnceAsync { [weak self] result in |
|
|
try recognizer?.recognizeOnceAsync { [weak self] result in |
|
|
guard let self = self else { return } |
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
// 停止麦克风流 |
|
|
|
|
|
self.microphoneStream?.stop() |
|
|
|
|
|
|
|
|
|
|
|
if result.reason == SPXResultReason.recognizedSpeech { |
|
|
if result.reason == SPXResultReason.recognizedSpeech { |
|
|
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
|
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
|
self.eventHandler("result", [ |
|
|
self.eventHandler("result", [ |
|
|
"text": result.text ?? "", |
|
|
"text": result.text ?? "", |
|
|
"detectedLanguage": detectedLanguage |
|
|
"detectedLanguage": detectedLanguage |
|
|
]) |
|
|
]) |
|
|
|
|
|
} else if result.reason == SPXResultReason.noMatch { |
|
|
|
|
|
print("[AzureAsrHelper] 无匹配结果") |
|
|
|
|
|
self.eventHandler("noMatch", [:]) |
|
|
} else if result.reason == SPXResultReason.canceled { |
|
|
} else if result.reason == SPXResultReason.canceled { |
|
|
do { |
|
|
do { |
|
|
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) |
|
|
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) |
|
|
@ -513,7 +294,6 @@ class AzureAsrHelper: NSObject { |
|
|
} catch { |
|
|
} catch { |
|
|
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") |
|
|
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") |
|
|
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) |
|
|
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) |
|
|
microphoneStream?.stop() |
|
|
|
|
|
return false |
|
|
return false |
|
|
} |
|
|
} |
|
|
} |
|
|
} |
|
|
@ -532,36 +312,29 @@ class AzureAsrHelper: NSObject { |
|
|
stopContinuousRecognition() |
|
|
stopContinuousRecognition() |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 重置 recognizer |
|
|
// 确保识别器已创建 |
|
|
if !resetRecognizer() { |
|
|
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|
|
return false |
|
|
return false |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 重新确保音频设置正确 |
|
|
|
|
|
do { |
|
|
|
|
|
try setupAudioSession() |
|
|
|
|
|
} catch { |
|
|
|
|
|
print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
do { |
|
|
do { |
|
|
// 设置识别事件处理 |
|
|
|
|
|
setupContinuousRecognitionCallbacks() |
|
|
|
|
|
|
|
|
|
|
|
// 启动连续识别 |
|
|
// 启动连续识别 |
|
|
try recognizer?.startContinuousRecognition() |
|
|
try recognizer?.startContinuousRecognition() |
|
|
_isContinuousRecognitionActive = true |
|
|
_isContinuousRecognitionActive = true |
|
|
|
|
|
|
|
|
// 启动音频流 |
|
|
|
|
|
if !startAudioStream() { |
|
|
|
|
|
try recognizer?.stopContinuousRecognition() |
|
|
|
|
|
_isContinuousRecognitionActive = false |
|
|
|
|
|
return false |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 通知会话开始 |
|
|
|
|
|
eventHandler("sessionStarted", [:]) |
|
|
|
|
|
|
|
|
|
|
|
print("[AzureAsrHelper] 连续识别开始") |
|
|
print("[AzureAsrHelper] 连续识别开始") |
|
|
return true |
|
|
return true |
|
|
} catch { |
|
|
} catch { |
|
|
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
|
|
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
|
|
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) |
|
|
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) |
|
|
_isContinuousRecognitionActive = false |
|
|
_isContinuousRecognitionActive = false |
|
|
microphoneStream?.stop() |
|
|
|
|
|
return false |
|
|
return false |
|
|
} |
|
|
} |
|
|
} |
|
|
} |
|
|
@ -573,13 +346,9 @@ class AzureAsrHelper: NSObject { |
|
|
return true |
|
|
return true |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 停止麦克风流 |
|
|
|
|
|
microphoneStream?.stop() |
|
|
|
|
|
|
|
|
|
|
|
do { |
|
|
do { |
|
|
try recognizer?.stopContinuousRecognition() |
|
|
try recognizer?.stopContinuousRecognition() |
|
|
_isContinuousRecognitionActive = false |
|
|
_isContinuousRecognitionActive = false |
|
|
eventHandler("sessionStopped", [:]) |
|
|
|
|
|
print("[AzureAsrHelper] 连续识别已停止") |
|
|
print("[AzureAsrHelper] 连续识别已停止") |
|
|
return true |
|
|
return true |
|
|
} catch { |
|
|
} catch { |
|
|
@ -605,94 +374,23 @@ class AzureAsrHelper: NSObject { |
|
|
stopContinuousRecognition() |
|
|
stopContinuousRecognition() |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 关闭麦克风流 |
|
|
// 释放音频会话 |
|
|
microphoneStream?.dispose() |
|
|
do { |
|
|
microphoneStream = nil |
|
|
try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) |
|
|
|
|
|
} catch { |
|
|
// 关闭推送流 |
|
|
print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") |
|
|
if let pushStream = pushStreamConfig { |
|
|
|
|
|
do { |
|
|
|
|
|
try pushStream.close() |
|
|
|
|
|
} catch { |
|
|
|
|
|
print("[AzureAsrHelper] 关闭推送流失败: \(error.localizedDescription)") |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// 释放资源 |
|
|
// 释放资源 |
|
|
recognizer = nil |
|
|
recognizer = nil |
|
|
speechConfig = nil |
|
|
speechConfig = nil |
|
|
audioConfig = nil |
|
|
audioConfig = nil |
|
|
pushStreamConfig = nil |
|
|
|
|
|
|
|
|
|
|
|
// 重置状态 |
|
|
// 重置状态 |
|
|
_isContinuousRecognitionActive = false |
|
|
_isContinuousRecognitionActive = false |
|
|
isInitialized = false |
|
|
isInitialized = false |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
// MARK: - 私有辅助方法 |
|
|
|
|
|
|
|
|
|
|
|
/// 设置连续识别回调 |
|
|
|
|
|
private func setupContinuousRecognitionCallbacks() { |
|
|
|
|
|
// 最终识别结果 |
|
|
|
|
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
if event.result.reason == SPXResultReason.recognizedSpeech { |
|
|
|
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
|
|
|
|
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
|
|
|
|
self.eventHandler("result", [ |
|
|
|
|
|
"text": event.result.text ?? "", |
|
|
|
|
|
"detectedLanguage": detectedLanguage |
|
|
|
|
|
]) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 识别中事件 |
|
|
|
|
|
recognizer?.addRecognizingEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
if event.result.reason == SPXResultReason.recognizingSpeech { |
|
|
|
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
|
|
|
|
print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
|
|
|
|
self.eventHandler("recognizing", [ |
|
|
|
|
|
"text": event.result.text ?? "", |
|
|
|
|
|
"detectedLanguage": detectedLanguage |
|
|
|
|
|
]) |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 会话事件 |
|
|
|
|
|
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
self._isContinuousRecognitionActive = true |
|
|
|
|
|
self.eventHandler("sessionStarted", [:]) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
self._isContinuousRecognitionActive = false |
|
|
|
|
|
self.eventHandler("sessionStopped", [:]) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 取消事件 |
|
|
|
|
|
recognizer?.addCanceledEventHandler { [weak self] _, event in |
|
|
|
|
|
guard let self = self else { return } |
|
|
|
|
|
|
|
|
|
|
|
let reason = event.reason.rawValue |
|
|
|
|
|
let errorDetails = event.errorDetails ?? "" |
|
|
|
|
|
|
|
|
|
|
|
self.eventHandler("canceled", [ |
|
|
|
|
|
"reason": reason, |
|
|
|
|
|
"errorDetails": errorDetails |
|
|
|
|
|
]) |
|
|
|
|
|
|
|
|
|
|
|
self._isContinuousRecognitionActive = false |
|
|
|
|
|
self.microphoneStream?.stop() |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/// 从结果中获取检测到的语言 |
|
|
/// 从结果中获取检测到的语言 |
|
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
|
if isAutoDetectLanguage { |
|
|
if isAutoDetectLanguage { |
|
|
|