From e4d17b94ddb99930f00b3c70eec0d1db2d401cd1 Mon Sep 17 00:00:00 2001 From: wolfplus Date: Wed, 19 Mar 2025 19:22:42 +0000 Subject: [PATCH] tts asr --- azure/ios/Classes/AzureAsrHelper.swift | 560 ++++-------------- azure/ios/Classes/AzureTtsHelper.swift | 63 +- .../SwiftAzureSpeechRecognitionPlugin.swift | 4 - lib/modules/home/views/home_view.dart | 6 +- 4 files changed, 186 insertions(+), 447 deletions(-) diff --git a/azure/ios/Classes/AzureAsrHelper.swift b/azure/ios/Classes/AzureAsrHelper.swift index 12bb57caf..4bd02368f 100644 --- a/azure/ios/Classes/AzureAsrHelper.swift +++ b/azure/ios/Classes/AzureAsrHelper.swift @@ -1,254 +1,6 @@ import Foundation import MicrosoftCognitiveServicesSpeech import AVFoundation -import AudioToolbox - -/// 自定义麦克风流实现,优化ASR语音输入捕获 -class AzureMicrophoneStream: NSObject { - var ioUnit: AudioUnit? - var audioFormat: AudioStreamBasicDescription - var audioBufferList: AudioBufferList - var audioList: [Float] = [] - let audioListQueue = DispatchQueue(label: "azureAudioListQueue") - private var isActive = false - - override init() { - // 音频会话配置 - let audioSession = AVAudioSession.sharedInstance() - do { - try audioSession.setCategory(.playAndRecord, - mode: .voiceChat, - options: [.allowBluetooth, .defaultToSpeaker, .mixWithOthers]) - try audioSession.setActive(true) - print("[AzureMicrophoneStream] 音频会话配置成功") - } catch { - print("[AzureMicrophoneStream] 配置音频会话失败: \(error.localizedDescription)") - } - - // 设置音频格式为16KHz 16位单声道PCM - audioFormat = AudioStreamBasicDescription( - mSampleRate: 16000.0, - mFormatID: kAudioFormatLinearPCM, - mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, - mBytesPerPacket: 2, - mFramesPerPacket: 1, - mBytesPerFrame: 2, - mChannelsPerFrame: 1, - mBitsPerChannel: 16, - mReserved: 0 - ) - - audioBufferList = AudioBufferList( - mNumberBuffers: 1, - mBuffers: AudioBuffer( - mNumberChannels: audioFormat.mChannelsPerFrame, - mDataByteSize: 4096, - mData: malloc(4096) - ) - ) - - super.init() - } - - func start() -> Bool { - if isActive { - return true // 已经在运行 - } - - if setupAudioUnit() { - isActive = startAudioUnit() - return isActive - } - return false - } - - private func setupAudioUnit() -> Bool { - print("[AzureMicrophoneStream] 设置音频单元") - - var ioUnitDescription = AudioComponentDescription( - componentType: kAudioUnitType_Output, - componentSubType: kAudioUnitSubType_VoiceProcessingIO, - componentManufacturer: kAudioUnitManufacturer_Apple, - componentFlags: 0, - componentFlagsMask: 0 - ) - - guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { - print("[AzureMicrophoneStream] 无法找到音频组件") - return false - } - - if CheckHasError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建IO单元") { - ioUnit = nil - return false - } - - var enableInput: UInt32 = 1 - let kInputBus: AudioUnitElement = 1 - let kOutputBus: AudioUnitElement = 0 - - if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Input, kInputBus, &enableInput, - UInt32(MemoryLayout.size)), "设置输入总线的EnableIO属性") { - return false - } - - var enableOutput: UInt32 = 0 - if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Output, kOutputBus, - &enableOutput, UInt32(MemoryLayout.size)), "设置输出总线的EnableIO属性") { - return false - } - - var flag: UInt32 = 0 - if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, - kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置ShouldAllocateBuffer属性") { - return false - } - - let size = UInt32(MemoryLayout.size) - if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线的StreamFormat属性") { - return false - } - - if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线的StreamFormat属性") { - return false - } - - var inputCallback = AURenderCallbackStruct( - inputProc: AzureMicrophoneStream.OnRecordedDataIsAvailable, - inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) - ) - - if CheckHasError(AudioUnitSetProperty(ioUnit!, - kAudioOutputUnitProperty_SetInputCallback, - kAudioUnitScope_Global, kInputBus, - &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { - return false - } - - var hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "初始化IO单元") - if hasError { - // 如果初始化失败,重试一次 - Thread.sleep(forTimeInterval: 0.5) - hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "重试初始化IO单元") - } - - print("[AzureMicrophoneStream] 音频单元设置\(hasError ? "失败" : "成功")") - return !hasError - } - - private func startAudioUnit() -> Bool { - print("[AzureMicrophoneStream] 启动音频单元") - guard let ioUnit = ioUnit else { - print("[AzureMicrophoneStream] IO单元未初始化") - return false - } - return !CheckHasError(AudioOutputUnitStart(ioUnit), "启动IO单元") - } - - func stop() { - print("[AzureMicrophoneStream] 停止音频单元") - guard isActive, let ioUnit = ioUnit else { return } - - _ = CheckHasError(AudioOutputUnitStop(ioUnit), "停止IO单元") - isActive = false - } - - func dispose() { - stop() - - if let ioUnit = ioUnit { - _ = CheckHasError(AudioUnitUninitialize(ioUnit), "反初始化IO单元") - _ = CheckHasError(AudioComponentInstanceDispose(ioUnit), "释放IO单元") - } - - // 释放缓冲区 - if let buffer = audioBufferList.mBuffers.mData { - free(buffer) - } - - self.ioUnit = nil - - // 清空音频数据 - audioListQueue.sync { - audioList.removeAll() - } - } - - static let OnRecordedDataIsAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in - let wrapper = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() - let expectedDataByteSize = inNumberFrames * wrapper.audioFormat.mBytesPerFrame - - if wrapper.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { - wrapper.audioBufferList.mBuffers.mData = realloc(wrapper.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) - wrapper.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize - } - - let status = wrapper.CheckErrorStatus(AudioUnitRender(wrapper.ioUnit!, ioActionFlags, inTimeStamp, - inBusNumber, inNumberFrames, &wrapper.audioBufferList), - "AudioUnitRender调用") - - var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) - let buffer = wrapper.audioBufferList.mBuffers - let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) - - for j in 0...size)) { - audioDataFloat[j] = Float(bufferData[j]) / 32768.0 // 归一化到[-1.0, 1.0]范围 - } - - if status == noErr { - wrapper.audioListQueue.async { - wrapper.audioList.append(contentsOf: audioDataFloat) - } - } - return status - } - - private func CheckHasError(_ status: OSStatus, _ operation: String) -> Bool { - if status != noErr { - print("[AzureMicrophoneStream] \(operation)失败: \(status)") - return true - } - return false - } - - private func CheckErrorStatus(_ status: OSStatus, _ operation: String) -> OSStatus { - if status != noErr { - print("[AzureMicrophoneStream] \(operation)失败: \(status)") - } - return status - } - - // 读取音频数据,适配Azure SDK - func read(bytes: inout [UInt8]) -> Int { - return audioListQueue.sync { - if audioList.isEmpty { - return 0 - } - - // 确保有足够的数据 - let minFrames = 1280 - if audioList.count < minFrames { - return 0 - } - - let frameLength = minFrames - - let buffer = Array(audioList.prefix(frameLength)) - audioList.removeFirst(frameLength) - - // 转换为Int16数据 - var int16Data = buffer.map { Int16($0 * 32767) } - let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) - bytes = [UInt8](data) - - return frameLength * 2 // 每个样本2字节(16位PCM) - } - } -} /// Azure ASR工具类,负责实现语音识别服务接口 @available(iOS 13.0, *) @@ -267,10 +19,6 @@ class AzureAsrHelper: NSObject { private var recognizer: SPXSpeechRecognizer? private var audioConfig: SPXAudioConfiguration? - /// 麦克风流 - private var microphoneStream: AzureMicrophoneStream? - private var pushStreamConfig: SPXPushAudioInputStream? - /// 状态标志 private var isInitialized = false private var _isContinuousRecognitionActive = false @@ -329,16 +77,18 @@ class AzureAsrHelper: NSObject { currentLanguage = self.supportedLanguages[0] } - // 创建麦克风流 - microphoneStream = AzureMicrophoneStream() + // 创建识别器和设置回调 + if !createRecognizerAndSetupCallbacks() { + return false + } print("[AzureAsrHelper] Azure 语音服务初始化成功") isInitialized = true return true } - /// 重置 recognizer - private func resetRecognizer() -> Bool { + /// 创建识别器并设置回调 + private func createRecognizerAndSetupCallbacks() -> Bool { // 释放之前的 recognizer recognizer = nil audioConfig = nil @@ -347,11 +97,11 @@ class AzureAsrHelper: NSObject { // 创建语音配置 speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - // 创建推送流 - pushStreamConfig = try SPXPushAudioInputStream() + // 设置音频输入参数 + try setupAudioSession() - // 创建音频配置,使用推送流 - audioConfig = try SPXAudioConfiguration(streamInput: pushStreamConfig!) + // 直接使用麦克风音频配置 + audioConfig = try SPXAudioConfiguration() // 设置语言配置 if isAutoDetectLanguage { @@ -375,65 +125,121 @@ class AzureAsrHelper: NSObject { recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) } + // 设置所有回调 + setupAllCallbacks() + return true } catch { - print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") - eventHandler("error", ["message": "重置识别器失败: \(error.localizedDescription)"]) + print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") + eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) return false } } - /// 启动音频捕获和数据推送 - private func startAudioStream() -> Bool { - guard let micStream = microphoneStream else { - print("[AzureAsrHelper] 错误: 麦克风流未初始化") - return false + /// 设置音频会话 + private func setupAudioSession() throws { + let audioSession = AVAudioSession.sharedInstance() + + // 使用playAndRecord类别允许同时录音和播放 + try audioSession.setCategory(.playAndRecord, + mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 + options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay]) + + // 设置首选的输入和输出 + let currentRoute = audioSession.currentRoute + + // 获取当前是否连接了耳机或外部麦克风 + let hasHeadphones = currentRoute.outputs.contains { + $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP } - // 启动麦克风 - if !micStream.start() { - print("[AzureAsrHelper] 错误: 启动麦克风流失败") - return false + // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 + if !hasHeadphones { + try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 + + // 启用回音消除和噪声抑制 + try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 + } else { + // 耳机模式,可以使用不同的设置 + try audioSession.setMode(.voiceChat) + try audioSession.setInputGain(1.0) } - // 创建并启动音频推送线程 - DispatchQueue.global(qos: .userInitiated).async { [weak self] in - guard let self = self, let pushStream = self.pushStreamConfig else { return } + // 设置合适的采样率 + try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 + try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 + + // 激活音频会话 + try audioSession.setActive(true, options: .notifyOthersOnDeactivation) + + print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") + } + + /// 设置所有回调 + private func setupAllCallbacks() { + guard let recognizer = recognizer else { return } + + // 最终识别结果 + recognizer.addRecognizedEventHandler { [weak self] _, event in + guard let self = self else { return } - var isRunning = true - var audioBuffer = [UInt8](repeating: 0, count: 16000) + if event.result.reason == SPXResultReason.recognizedSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + self.eventHandler("result", [ + "text": event.result.text ?? "", + "detectedLanguage": detectedLanguage + ]) + } + } + + // 识别中事件 + recognizer.addRecognizingEventHandler { [weak self] _, event in + guard let self = self else { return } - while isRunning { - autoreleasepool { - // 读取麦克风数据 - let bytesRead = micStream.read(bytes: &audioBuffer) - - if bytesRead > 0 { - do { - // 推送音频数据到Azure识别流 - let data = Data(bytes: audioBuffer, count: bytesRead) - try pushStream.write(data) - } catch { - print("[AzureAsrHelper] 推送音频数据失败: \(error.localizedDescription)") - isRunning = false - } - } - - // 检查是否应该继续捕获 - if !self._isContinuousRecognitionActive { - isRunning = false - } - - // 添加适当的休眠以避免过度消耗CPU - if bytesRead == 0 { - Thread.sleep(forTimeInterval: 0.01) - } - } + if event.result.reason == SPXResultReason.recognizingSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + self.eventHandler("recognizing", [ + "text": event.result.text ?? "", + "detectedLanguage": detectedLanguage + ]) } - print("[AzureAsrHelper] 音频推送线程已停止") } - return true + // 会话事件 + recognizer.addSessionStartedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + print("[AzureAsrHelper] 识别会话已开始") + self._isContinuousRecognitionActive = true + self.eventHandler("sessionStarted", [:]) + } + + recognizer.addSessionStoppedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + print("[AzureAsrHelper] 识别会话已结束") + self._isContinuousRecognitionActive = false + self.eventHandler("sessionStopped", [:]) + } + + // 取消事件 + recognizer.addCanceledEventHandler { [weak self] _, event in + guard let self = self else { return } + + let reason = event.reason.rawValue + let errorDetails = event.errorDetails ?? "未知错误" + + print("[AzureAsrHelper] 识别取消: \(errorDetails)") + + self.eventHandler("canceled", [ + "reason": reason, + "errorDetails": errorDetails + ]) + + self._isContinuousRecognitionActive = false + } } /// 执行一次性语音识别 @@ -450,37 +256,12 @@ class AzureAsrHelper: NSObject { stopContinuousRecognition() } - // 重置 recognizer - if !resetRecognizer() { - return false - } - - // 启动音频流 - if !startAudioStream() { + // 确保识别器已创建 + if recognizer == nil && !createRecognizerAndSetupCallbacks() { return false } do { - // 设置回调 - recognizer?.addRecognizedEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - self.eventHandler("result", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - recognizer?.addCanceledEventHandler { [weak self] _, event in - guard let self = self else { return } - - let errorDetails = event.errorDetails ?? "未知错误" - self.eventHandler("error", ["message": "识别异常: \(errorDetails)"]) - } - // 通知会话开始 eventHandler("sessionStarted", [:]) @@ -488,15 +269,15 @@ class AzureAsrHelper: NSObject { try recognizer?.recognizeOnceAsync { [weak self] result in guard let self = self else { return } - // 停止麦克风流 - self.microphoneStream?.stop() - if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) self.eventHandler("result", [ "text": result.text ?? "", "detectedLanguage": detectedLanguage ]) + } else if result.reason == SPXResultReason.noMatch { + print("[AzureAsrHelper] 无匹配结果") + self.eventHandler("noMatch", [:]) } else if result.reason == SPXResultReason.canceled { do { let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) @@ -513,7 +294,6 @@ class AzureAsrHelper: NSObject { } catch { print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) - microphoneStream?.stop() return false } } @@ -532,36 +312,29 @@ class AzureAsrHelper: NSObject { stopContinuousRecognition() } - // 重置 recognizer - if !resetRecognizer() { + // 确保识别器已创建 + if recognizer == nil && !createRecognizerAndSetupCallbacks() { return false } + // 重新确保音频设置正确 + do { + try setupAudioSession() + } catch { + print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") + } + do { - // 设置识别事件处理 - setupContinuousRecognitionCallbacks() - // 启动连续识别 try recognizer?.startContinuousRecognition() _isContinuousRecognitionActive = true - // 启动音频流 - if !startAudioStream() { - try recognizer?.stopContinuousRecognition() - _isContinuousRecognitionActive = false - return false - } - - // 通知会话开始 - eventHandler("sessionStarted", [:]) - print("[AzureAsrHelper] 连续识别开始") return true } catch { print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) _isContinuousRecognitionActive = false - microphoneStream?.stop() return false } } @@ -573,13 +346,9 @@ class AzureAsrHelper: NSObject { return true } - // 停止麦克风流 - microphoneStream?.stop() - do { try recognizer?.stopContinuousRecognition() _isContinuousRecognitionActive = false - eventHandler("sessionStopped", [:]) print("[AzureAsrHelper] 连续识别已停止") return true } catch { @@ -605,94 +374,23 @@ class AzureAsrHelper: NSObject { stopContinuousRecognition() } - // 关闭麦克风流 - microphoneStream?.dispose() - microphoneStream = nil - - // 关闭推送流 - if let pushStream = pushStreamConfig { - do { - try pushStream.close() - } catch { - print("[AzureAsrHelper] 关闭推送流失败: \(error.localizedDescription)") - } + // 释放音频会话 + do { + try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) + } catch { + print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") } // 释放资源 recognizer = nil speechConfig = nil audioConfig = nil - pushStreamConfig = nil // 重置状态 _isContinuousRecognitionActive = false isInitialized = false } - // MARK: - 私有辅助方法 - - /// 设置连续识别回调 - private func setupContinuousRecognitionCallbacks() { - // 最终识别结果 - recognizer?.addRecognizedEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("result", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 识别中事件 - recognizer?.addRecognizingEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizingSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("recognizing", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 会话事件 - recognizer?.addSessionStartedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - self._isContinuousRecognitionActive = true - self.eventHandler("sessionStarted", [:]) - } - - recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - self._isContinuousRecognitionActive = false - self.eventHandler("sessionStopped", [:]) - } - - // 取消事件 - recognizer?.addCanceledEventHandler { [weak self] _, event in - guard let self = self else { return } - - let reason = event.reason.rawValue - let errorDetails = event.errorDetails ?? "" - - self.eventHandler("canceled", [ - "reason": reason, - "errorDetails": errorDetails - ]) - - self._isContinuousRecognitionActive = false - self.microphoneStream?.stop() - } - } - /// 从结果中获取检测到的语言 private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { if isAutoDetectLanguage { diff --git a/azure/ios/Classes/AzureTtsHelper.swift b/azure/ios/Classes/AzureTtsHelper.swift index 650a4eeb8..589df617c 100644 --- a/azure/ios/Classes/AzureTtsHelper.swift +++ b/azure/ios/Classes/AzureTtsHelper.swift @@ -26,6 +26,9 @@ class AzureTtsHelper: NSObject { /// 当前是否正在播放 private var _isSpeaking = false + /// 音频会话配置 + private var isAudioSessionConfigured = false + // MARK: - 语音设置 /// 当前语音 @@ -82,13 +85,15 @@ class AzureTtsHelper: NSObject { self.speechSubscriptionKey = speechSubscriptionKey self.serviceRegion = serviceRegion + // 配置音频会话 + if !configureAudioSession() { + print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化") + } + do { // 创建语音配置 speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - // 设置语音合成输出格式 - // speechConfig?.setSpeechSynthesisOutputFormat(.audio24Khz48KBitRateMonoMp3) - // 设置默认语音 let defaultVoice = getDefaultVoiceForLanguage(language) currentVoice = defaultVoice @@ -111,6 +116,46 @@ class AzureTtsHelper: NSObject { } } + /// 配置音频会话 + private func configureAudioSession() -> Bool { + let audioSession = AVAudioSession.sharedInstance() + do { + // 使用playback类别,但支持混合和空中播放 + try audioSession.setCategory(.playback, + mode: .spokenAudio, + options: [.mixWithOthers, .allowAirPlay, .duckOthers]) + + // 根据设备类型选择最佳配置 + let currentRoute = audioSession.currentRoute + let hasHeadphones = currentRoute.outputs.contains { + $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP + } + + // 优化音频路由 + if hasHeadphones { + // 耳机模式,使用默认设置 + try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟 + } else { + // 扬声器模式 + try audioSession.setPreferredIOBufferDuration(0.005) + } + + // 避免完全激活音频会话,因为ASR可能已经激活 + // 这里使用setActive(false)是为了不与ASR冲突 + if !audioSession.isOtherAudioPlaying { + try audioSession.setActive(true, options: .notifyOthersOnDeactivation) + } + + isAudioSessionConfigured = true + print("[AzureTtsHelper] 音频会话配置成功") + return true + } catch { + print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)") + isAudioSessionConfigured = false + return false + } + } + /// 设置语音 /// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") /// - Returns: 设置是否成功 @@ -181,6 +226,11 @@ class AzureTtsHelper: NSObject { return true } + // 确保音频会话已配置 + if !isAudioSessionConfigured { + _ = configureAudioSession() + } + print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") // 生成SSML @@ -203,8 +253,6 @@ class AzureTtsHelper: NSObject { Task { do { - print("[AzureTtsHelper] 开始语音合成") - // 使用异步方法进行合成并直接播放 _ = try await synthesizer.startSpeakingSsml(text) @@ -257,6 +305,7 @@ class AzureTtsHelper: NSObject { isInitialized = false _isSpeaking = false + isAudioSessionConfigured = false print("[AzureTtsHelper] TTS 引擎已释放") } @@ -311,12 +360,12 @@ class AzureTtsHelper: NSObject { // 合成开始事件 synthesizer.addSynthesisStartedEventHandler { _, _ in - print("[AzureTtsHelper] 语音合成开始") + // print("[AzureTtsHelper] 语音合成开始") } // 合成中事件 synthesizer.addSynthesizingEventHandler { _, _ in - print("[AzureTtsHelper] 语音合成中") + // print("[AzureTtsHelper] 语音合成中") } } diff --git a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift index de40821fe..ea393e215 100644 --- a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift +++ b/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift @@ -42,12 +42,10 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { // 新增直接处理方法调用的函数 private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - print("[AzurePlugin] 处理TTS方法调用: \(call.method)") handleTtsMethod(call, result) } private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - print("[AzurePlugin] 处理ASR方法调用: \(call.method)") handleAsrMethod(call, result) } @@ -74,7 +72,6 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { } private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - print("[AzurePlugin] 处理ASR方法: \(call.method)") let args = call.arguments as? Dictionary @@ -136,7 +133,6 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { } private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - print("[AzurePlugin] 处理TTS方法: \(call.method)") let args = call.arguments as? Dictionary diff --git a/lib/modules/home/views/home_view.dart b/lib/modules/home/views/home_view.dart index b5567e65e..4cc01cd56 100644 --- a/lib/modules/home/views/home_view.dart +++ b/lib/modules/home/views/home_view.dart @@ -101,11 +101,7 @@ class HomeView extends GetView { SizedBox(height: 16.h), _buildPremiumAILayout(), - // 在合适的位置添加测试按钮 - ElevatedButton( - onPressed: () => Get.toNamed(Routes.ASR_TEST), - child: const Text('Azure语音识别测试'), - ), + ], ), ),