You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
710 lines
26 KiB
710 lines
26 KiB
import Foundation
|
|
import MicrosoftCognitiveServicesSpeech
|
|
import AVFoundation
|
|
import AudioToolbox
|
|
|
|
/// 自定义麦克风流实现,优化ASR语音输入捕获
|
|
class AzureMicrophoneStream: NSObject {
|
|
var ioUnit: AudioUnit?
|
|
var audioFormat: AudioStreamBasicDescription
|
|
var audioBufferList: AudioBufferList
|
|
var audioList: [Float] = []
|
|
let audioListQueue = DispatchQueue(label: "azureAudioListQueue")
|
|
private var isActive = false
|
|
|
|
override init() {
|
|
// 音频会话配置
|
|
let audioSession = AVAudioSession.sharedInstance()
|
|
do {
|
|
try audioSession.setCategory(.playAndRecord,
|
|
mode: .voiceChat,
|
|
options: [.allowBluetooth, .defaultToSpeaker, .mixWithOthers])
|
|
try audioSession.setActive(true)
|
|
print("[AzureMicrophoneStream] 音频会话配置成功")
|
|
} catch {
|
|
print("[AzureMicrophoneStream] 配置音频会话失败: \(error.localizedDescription)")
|
|
}
|
|
|
|
// 设置音频格式为16KHz 16位单声道PCM
|
|
audioFormat = AudioStreamBasicDescription(
|
|
mSampleRate: 16000.0,
|
|
mFormatID: kAudioFormatLinearPCM,
|
|
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
|
|
mBytesPerPacket: 2,
|
|
mFramesPerPacket: 1,
|
|
mBytesPerFrame: 2,
|
|
mChannelsPerFrame: 1,
|
|
mBitsPerChannel: 16,
|
|
mReserved: 0
|
|
)
|
|
|
|
audioBufferList = AudioBufferList(
|
|
mNumberBuffers: 1,
|
|
mBuffers: AudioBuffer(
|
|
mNumberChannels: audioFormat.mChannelsPerFrame,
|
|
mDataByteSize: 4096,
|
|
mData: malloc(4096)
|
|
)
|
|
)
|
|
|
|
super.init()
|
|
}
|
|
|
|
func start() -> Bool {
|
|
if isActive {
|
|
return true // 已经在运行
|
|
}
|
|
|
|
if setupAudioUnit() {
|
|
isActive = startAudioUnit()
|
|
return isActive
|
|
}
|
|
return false
|
|
}
|
|
|
|
private func setupAudioUnit() -> Bool {
|
|
print("[AzureMicrophoneStream] 设置音频单元")
|
|
|
|
var ioUnitDescription = AudioComponentDescription(
|
|
componentType: kAudioUnitType_Output,
|
|
componentSubType: kAudioUnitSubType_VoiceProcessingIO,
|
|
componentManufacturer: kAudioUnitManufacturer_Apple,
|
|
componentFlags: 0,
|
|
componentFlagsMask: 0
|
|
)
|
|
|
|
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
|
|
print("[AzureMicrophoneStream] 无法找到音频组件")
|
|
return false
|
|
}
|
|
|
|
if CheckHasError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建IO单元") {
|
|
ioUnit = nil
|
|
return false
|
|
}
|
|
|
|
var enableInput: UInt32 = 1
|
|
let kInputBus: AudioUnitElement = 1
|
|
let kOutputBus: AudioUnitElement = 0
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
|
|
kAudioUnitScope_Input, kInputBus, &enableInput,
|
|
UInt32(MemoryLayout<UInt32>.size)), "设置输入总线的EnableIO属性") {
|
|
return false
|
|
}
|
|
|
|
var enableOutput: UInt32 = 0
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
|
|
kAudioUnitScope_Output, kOutputBus,
|
|
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "设置输出总线的EnableIO属性") {
|
|
return false
|
|
}
|
|
|
|
var flag: UInt32 = 0
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
|
|
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置ShouldAllocateBuffer属性") {
|
|
return false
|
|
}
|
|
|
|
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size)
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
|
|
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线的StreamFormat属性") {
|
|
return false
|
|
}
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
|
|
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线的StreamFormat属性") {
|
|
return false
|
|
}
|
|
|
|
var inputCallback = AURenderCallbackStruct(
|
|
inputProc: AzureMicrophoneStream.OnRecordedDataIsAvailable,
|
|
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
|
|
)
|
|
|
|
if CheckHasError(AudioUnitSetProperty(ioUnit!,
|
|
kAudioOutputUnitProperty_SetInputCallback,
|
|
kAudioUnitScope_Global, kInputBus,
|
|
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") {
|
|
return false
|
|
}
|
|
|
|
var hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "初始化IO单元")
|
|
if hasError {
|
|
// 如果初始化失败,重试一次
|
|
Thread.sleep(forTimeInterval: 0.5)
|
|
hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "重试初始化IO单元")
|
|
}
|
|
|
|
print("[AzureMicrophoneStream] 音频单元设置\(hasError ? "失败" : "成功")")
|
|
return !hasError
|
|
}
|
|
|
|
private func startAudioUnit() -> Bool {
|
|
print("[AzureMicrophoneStream] 启动音频单元")
|
|
guard let ioUnit = ioUnit else {
|
|
print("[AzureMicrophoneStream] IO单元未初始化")
|
|
return false
|
|
}
|
|
return !CheckHasError(AudioOutputUnitStart(ioUnit), "启动IO单元")
|
|
}
|
|
|
|
func stop() {
|
|
print("[AzureMicrophoneStream] 停止音频单元")
|
|
guard isActive, let ioUnit = ioUnit else { return }
|
|
|
|
_ = CheckHasError(AudioOutputUnitStop(ioUnit), "停止IO单元")
|
|
isActive = false
|
|
}
|
|
|
|
func dispose() {
|
|
stop()
|
|
|
|
if let ioUnit = ioUnit {
|
|
_ = CheckHasError(AudioUnitUninitialize(ioUnit), "反初始化IO单元")
|
|
_ = CheckHasError(AudioComponentInstanceDispose(ioUnit), "释放IO单元")
|
|
}
|
|
|
|
// 释放缓冲区
|
|
if let buffer = audioBufferList.mBuffers.mData {
|
|
free(buffer)
|
|
}
|
|
|
|
self.ioUnit = nil
|
|
|
|
// 清空音频数据
|
|
audioListQueue.sync {
|
|
audioList.removeAll()
|
|
}
|
|
}
|
|
|
|
static let OnRecordedDataIsAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
|
|
let wrapper = Unmanaged<AzureMicrophoneStream>.fromOpaque(inRefCon).takeUnretainedValue()
|
|
let expectedDataByteSize = inNumberFrames * wrapper.audioFormat.mBytesPerFrame
|
|
|
|
if wrapper.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
|
|
wrapper.audioBufferList.mBuffers.mData = realloc(wrapper.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
|
|
wrapper.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
|
|
}
|
|
|
|
let status = wrapper.CheckErrorStatus(AudioUnitRender(wrapper.ioUnit!, ioActionFlags, inTimeStamp,
|
|
inBusNumber, inNumberFrames, &wrapper.audioBufferList),
|
|
"AudioUnitRender调用")
|
|
|
|
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
|
|
let buffer = wrapper.audioBufferList.mBuffers
|
|
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
|
|
|
|
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) {
|
|
audioDataFloat[j] = Float(bufferData[j]) / 32768.0 // 归一化到[-1.0, 1.0]范围
|
|
}
|
|
|
|
if status == noErr {
|
|
wrapper.audioListQueue.async {
|
|
wrapper.audioList.append(contentsOf: audioDataFloat)
|
|
}
|
|
}
|
|
return status
|
|
}
|
|
|
|
private func CheckHasError(_ status: OSStatus, _ operation: String) -> Bool {
|
|
if status != noErr {
|
|
print("[AzureMicrophoneStream] \(operation)失败: \(status)")
|
|
return true
|
|
}
|
|
return false
|
|
}
|
|
|
|
private func CheckErrorStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
|
|
if status != noErr {
|
|
print("[AzureMicrophoneStream] \(operation)失败: \(status)")
|
|
}
|
|
return status
|
|
}
|
|
|
|
// 读取音频数据,适配Azure SDK
|
|
func read(bytes: inout [UInt8]) -> Int {
|
|
return audioListQueue.sync {
|
|
if audioList.isEmpty {
|
|
return 0
|
|
}
|
|
|
|
// 确保有足够的数据
|
|
let minFrames = 1280
|
|
if audioList.count < minFrames {
|
|
return 0
|
|
}
|
|
|
|
let frameLength = minFrames
|
|
|
|
let buffer = Array(audioList.prefix(frameLength))
|
|
audioList.removeFirst(frameLength)
|
|
|
|
// 转换为Int16数据
|
|
var int16Data = buffer.map { Int16($0 * 32767) }
|
|
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
|
|
bytes = [UInt8](data)
|
|
|
|
return frameLength * 2 // 每个样本2字节(16位PCM)
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Azure ASR工具类,负责实现语音识别服务接口
|
|
@available(iOS 13.0, *)
|
|
class AzureAsrHelper: NSObject {
|
|
// MARK: - 属性
|
|
|
|
/// 事件处理回调
|
|
private var eventHandler: (String, [String: Any]) -> Void
|
|
|
|
/// 语音配置信息
|
|
private var speechSubscriptionKey: String = ""
|
|
private var serviceRegion: String = ""
|
|
|
|
/// 语音识别相关
|
|
private var speechConfig: SPXSpeechConfiguration?
|
|
private var recognizer: SPXSpeechRecognizer?
|
|
private var audioConfig: SPXAudioConfiguration?
|
|
|
|
/// 麦克风流
|
|
private var microphoneStream: AzureMicrophoneStream?
|
|
private var pushStreamConfig: SPXPushAudioInputStream?
|
|
|
|
/// 状态标志
|
|
private var isInitialized = false
|
|
private var _isContinuousRecognitionActive = false
|
|
|
|
/// 当前语言和支持的语言
|
|
private var currentLanguage = "zh-CN"
|
|
private var supportedLanguages: [String] = ["zh-CN", "en-US"]
|
|
private var isAutoDetectLanguage = false
|
|
|
|
// MARK: - 初始化
|
|
|
|
init(eventHandler: @escaping (String, [String: Any]) -> Void) {
|
|
self.eventHandler = eventHandler
|
|
super.init()
|
|
}
|
|
|
|
deinit {
|
|
dispose()
|
|
}
|
|
|
|
// MARK: - ASR Service 接口实现
|
|
|
|
/// 初始化语音识别服务
|
|
/// - Parameters:
|
|
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
|
|
/// - serviceRegion: Azure 服务区域 (如 eastasia)
|
|
/// - supportedLanguages: 支持的语言代码数组 (可选)
|
|
/// - Returns: 初始化是否成功
|
|
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool {
|
|
print("[AzureAsrHelper] 初始化 Azure 语音服务")
|
|
|
|
// 检查配置是否为空
|
|
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
|
|
print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
|
|
eventHandler("error", ["message": "Azure 配置信息不完整"])
|
|
return false
|
|
}
|
|
|
|
// 释放之前的资源
|
|
dispose()
|
|
|
|
// 记录配置信息
|
|
self.speechSubscriptionKey = speechSubscriptionKey
|
|
self.serviceRegion = serviceRegion
|
|
|
|
// 设置语言
|
|
if let languages = supportedLanguages, !languages.isEmpty {
|
|
self.supportedLanguages = languages
|
|
}
|
|
|
|
// 根据支持的语言数量决定是否启用自动语言检测
|
|
isAutoDetectLanguage = self.supportedLanguages.count >= 2
|
|
|
|
// 如果只有一种语言,设置为当前语言
|
|
if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty {
|
|
currentLanguage = self.supportedLanguages[0]
|
|
}
|
|
|
|
// 创建麦克风流
|
|
microphoneStream = AzureMicrophoneStream()
|
|
|
|
print("[AzureAsrHelper] Azure 语音服务初始化成功")
|
|
isInitialized = true
|
|
return true
|
|
}
|
|
|
|
/// 重置 recognizer
|
|
private func resetRecognizer() -> Bool {
|
|
// 释放之前的 recognizer
|
|
recognizer = nil
|
|
audioConfig = nil
|
|
|
|
do {
|
|
// 创建语音配置
|
|
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
|
|
|
|
// 创建推送流
|
|
pushStreamConfig = try SPXPushAudioInputStream()
|
|
|
|
// 创建音频配置,使用推送流
|
|
audioConfig = try SPXAudioConfiguration(streamInput: pushStreamConfig!)
|
|
|
|
// 设置语言配置
|
|
if isAutoDetectLanguage {
|
|
// 设置自动语言检测
|
|
speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode)
|
|
|
|
// 创建自动语言检测配置
|
|
let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages)
|
|
|
|
// 创建识别器
|
|
recognizer = try SPXSpeechRecognizer(
|
|
speechConfiguration: speechConfig!,
|
|
autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig,
|
|
audioConfiguration: audioConfig!
|
|
)
|
|
} else {
|
|
// 设置指定的识别语言
|
|
speechConfig?.speechRecognitionLanguage = currentLanguage
|
|
|
|
// 创建识别器
|
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
|
|
}
|
|
|
|
return true
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)")
|
|
eventHandler("error", ["message": "重置识别器失败: \(error.localizedDescription)"])
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 启动音频捕获和数据推送
|
|
private func startAudioStream() -> Bool {
|
|
guard let micStream = microphoneStream else {
|
|
print("[AzureAsrHelper] 错误: 麦克风流未初始化")
|
|
return false
|
|
}
|
|
|
|
// 启动麦克风
|
|
if !micStream.start() {
|
|
print("[AzureAsrHelper] 错误: 启动麦克风流失败")
|
|
return false
|
|
}
|
|
|
|
// 创建并启动音频推送线程
|
|
DispatchQueue.global(qos: .userInitiated).async { [weak self] in
|
|
guard let self = self, let pushStream = self.pushStreamConfig else { return }
|
|
|
|
var isRunning = true
|
|
var audioBuffer = [UInt8](repeating: 0, count: 16000)
|
|
|
|
while isRunning {
|
|
autoreleasepool {
|
|
// 读取麦克风数据
|
|
let bytesRead = micStream.read(bytes: &audioBuffer)
|
|
|
|
if bytesRead > 0 {
|
|
do {
|
|
// 推送音频数据到Azure识别流
|
|
let data = Data(bytes: audioBuffer, count: bytesRead)
|
|
try pushStream.write(data)
|
|
} catch {
|
|
print("[AzureAsrHelper] 推送音频数据失败: \(error.localizedDescription)")
|
|
isRunning = false
|
|
}
|
|
}
|
|
|
|
// 检查是否应该继续捕获
|
|
if !self._isContinuousRecognitionActive {
|
|
isRunning = false
|
|
}
|
|
|
|
// 添加适当的休眠以避免过度消耗CPU
|
|
if bytesRead == 0 {
|
|
Thread.sleep(forTimeInterval: 0.01)
|
|
}
|
|
}
|
|
}
|
|
print("[AzureAsrHelper] 音频推送线程已停止")
|
|
}
|
|
|
|
return true
|
|
}
|
|
|
|
/// 执行一次性语音识别
|
|
/// - Returns: 是否成功启动识别
|
|
func recognizeOnce() -> Bool {
|
|
if !isInitialized {
|
|
print("[AzureAsrHelper] 错误: 语音服务未初始化")
|
|
eventHandler("error", ["message": "语音服务未初始化"])
|
|
return false
|
|
}
|
|
|
|
// 如果正在连续识别,先停止
|
|
if _isContinuousRecognitionActive {
|
|
stopContinuousRecognition()
|
|
}
|
|
|
|
// 重置 recognizer
|
|
if !resetRecognizer() {
|
|
return false
|
|
}
|
|
|
|
// 启动音频流
|
|
if !startAudioStream() {
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 设置回调
|
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in
|
|
guard let self = self else { return }
|
|
|
|
if event.result.reason == SPXResultReason.recognizedSpeech {
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result)
|
|
self.eventHandler("result", [
|
|
"text": event.result.text ?? "",
|
|
"detectedLanguage": detectedLanguage
|
|
])
|
|
}
|
|
}
|
|
|
|
recognizer?.addCanceledEventHandler { [weak self] _, event in
|
|
guard let self = self else { return }
|
|
|
|
let errorDetails = event.errorDetails ?? "未知错误"
|
|
self.eventHandler("error", ["message": "识别异常: \(errorDetails)"])
|
|
}
|
|
|
|
// 通知会话开始
|
|
eventHandler("sessionStarted", [:])
|
|
|
|
// 执行识别
|
|
try recognizer?.recognizeOnceAsync { [weak self] result in
|
|
guard let self = self else { return }
|
|
|
|
// 停止麦克风流
|
|
self.microphoneStream?.stop()
|
|
|
|
if result.reason == SPXResultReason.recognizedSpeech {
|
|
let detectedLanguage = self.getDetectedLanguage(from: result)
|
|
self.eventHandler("result", [
|
|
"text": result.text ?? "",
|
|
"detectedLanguage": detectedLanguage
|
|
])
|
|
} else if result.reason == SPXResultReason.canceled {
|
|
do {
|
|
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result)
|
|
let errorDetails = details.errorDetails ?? "未知错误"
|
|
self.eventHandler("error", ["message": "识别取消: \(errorDetails)"])
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)")
|
|
self.eventHandler("error", ["message": "识别取消,无法获取详细原因"])
|
|
}
|
|
}
|
|
}
|
|
|
|
return true
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
|
|
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"])
|
|
microphoneStream?.stop()
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 开始连续语音识别
|
|
/// - Returns: 是否成功启动识别
|
|
func startContinuousRecognition() -> Bool {
|
|
if !isInitialized {
|
|
print("[AzureAsrHelper] 错误: 语音服务未初始化")
|
|
eventHandler("error", ["message": "语音服务未初始化"])
|
|
return false
|
|
}
|
|
|
|
// 如果已经在进行连续识别,先停止
|
|
if _isContinuousRecognitionActive {
|
|
stopContinuousRecognition()
|
|
}
|
|
|
|
// 重置 recognizer
|
|
if !resetRecognizer() {
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 设置识别事件处理
|
|
setupContinuousRecognitionCallbacks()
|
|
|
|
// 启动连续识别
|
|
try recognizer?.startContinuousRecognition()
|
|
_isContinuousRecognitionActive = true
|
|
|
|
// 启动音频流
|
|
if !startAudioStream() {
|
|
try recognizer?.stopContinuousRecognition()
|
|
_isContinuousRecognitionActive = false
|
|
return false
|
|
}
|
|
|
|
// 通知会话开始
|
|
eventHandler("sessionStarted", [:])
|
|
|
|
print("[AzureAsrHelper] 连续识别开始")
|
|
return true
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
|
|
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"])
|
|
_isContinuousRecognitionActive = false
|
|
microphoneStream?.stop()
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 停止连续语音识别
|
|
/// - Returns: 是否成功停止识别
|
|
func stopContinuousRecognition() -> Bool {
|
|
if !_isContinuousRecognitionActive || recognizer == nil {
|
|
return true
|
|
}
|
|
|
|
// 停止麦克风流
|
|
microphoneStream?.stop()
|
|
|
|
do {
|
|
try recognizer?.stopContinuousRecognition()
|
|
_isContinuousRecognitionActive = false
|
|
eventHandler("sessionStopped", [:])
|
|
print("[AzureAsrHelper] 连续识别已停止")
|
|
return true
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)")
|
|
eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"])
|
|
_isContinuousRecognitionActive = false
|
|
return false
|
|
}
|
|
}
|
|
|
|
/// 检查连续识别是否活跃
|
|
/// - Returns: 连续识别是否处于活跃状态
|
|
func isContinuousRecognitionActive() -> Bool {
|
|
return _isContinuousRecognitionActive
|
|
}
|
|
|
|
/// 释放资源
|
|
func dispose() {
|
|
print("[AzureAsrHelper] 释放资源")
|
|
|
|
// 停止连续识别
|
|
if _isContinuousRecognitionActive {
|
|
stopContinuousRecognition()
|
|
}
|
|
|
|
// 关闭麦克风流
|
|
microphoneStream?.dispose()
|
|
microphoneStream = nil
|
|
|
|
// 关闭推送流
|
|
if let pushStream = pushStreamConfig {
|
|
do {
|
|
try pushStream.close()
|
|
} catch {
|
|
print("[AzureAsrHelper] 关闭推送流失败: \(error.localizedDescription)")
|
|
}
|
|
}
|
|
|
|
// 释放资源
|
|
recognizer = nil
|
|
speechConfig = nil
|
|
audioConfig = nil
|
|
pushStreamConfig = nil
|
|
|
|
// 重置状态
|
|
_isContinuousRecognitionActive = false
|
|
isInitialized = false
|
|
}
|
|
|
|
// MARK: - 私有辅助方法
|
|
|
|
/// 设置连续识别回调
|
|
private func setupContinuousRecognitionCallbacks() {
|
|
// 最终识别结果
|
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in
|
|
guard let self = self else { return }
|
|
|
|
if event.result.reason == SPXResultReason.recognizedSpeech {
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result)
|
|
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
|
|
self.eventHandler("result", [
|
|
"text": event.result.text ?? "",
|
|
"detectedLanguage": detectedLanguage
|
|
])
|
|
}
|
|
}
|
|
|
|
// 识别中事件
|
|
recognizer?.addRecognizingEventHandler { [weak self] _, event in
|
|
guard let self = self else { return }
|
|
|
|
if event.result.reason == SPXResultReason.recognizingSpeech {
|
|
let detectedLanguage = self.getDetectedLanguage(from: event.result)
|
|
print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
|
|
self.eventHandler("recognizing", [
|
|
"text": event.result.text ?? "",
|
|
"detectedLanguage": detectedLanguage
|
|
])
|
|
}
|
|
}
|
|
|
|
// 会话事件
|
|
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
|
|
self._isContinuousRecognitionActive = true
|
|
self.eventHandler("sessionStarted", [:])
|
|
}
|
|
|
|
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in
|
|
guard let self = self else { return }
|
|
|
|
self._isContinuousRecognitionActive = false
|
|
self.eventHandler("sessionStopped", [:])
|
|
}
|
|
|
|
// 取消事件
|
|
recognizer?.addCanceledEventHandler { [weak self] _, event in
|
|
guard let self = self else { return }
|
|
|
|
let reason = event.reason.rawValue
|
|
let errorDetails = event.errorDetails ?? ""
|
|
|
|
self.eventHandler("canceled", [
|
|
"reason": reason,
|
|
"errorDetails": errorDetails
|
|
])
|
|
|
|
self._isContinuousRecognitionActive = false
|
|
self.microphoneStream?.stop()
|
|
}
|
|
}
|
|
|
|
/// 从结果中获取检测到的语言
|
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
|
|
if isAutoDetectLanguage {
|
|
do {
|
|
let langResult = try SPXAutoDetectSourceLanguageResult(result)
|
|
return langResult.language ?? currentLanguage
|
|
} catch {
|
|
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)")
|
|
return currentLanguage
|
|
}
|
|
} else {
|
|
return currentLanguage
|
|
}
|
|
}
|
|
}
|