Browse Source

tts asr

newdev_shunjiawei
wolfplus 2 years ago
parent
commit
e4d17b94dd
  1. 560
      azure/ios/Classes/AzureAsrHelper.swift
  2. 63
      azure/ios/Classes/AzureTtsHelper.swift
  3. 4
      azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift
  4. 6
      lib/modules/home/views/home_view.dart

560
azure/ios/Classes/AzureAsrHelper.swift

@ -1,254 +1,6 @@
import Foundation import Foundation
import MicrosoftCognitiveServicesSpeech import MicrosoftCognitiveServicesSpeech
import AVFoundation import AVFoundation
import AudioToolbox
/// 自定义麦克风流实现,优化ASR语音输入捕获
class AzureMicrophoneStream: NSObject {
var ioUnit: AudioUnit?
var audioFormat: AudioStreamBasicDescription
var audioBufferList: AudioBufferList
var audioList: [Float] = []
let audioListQueue = DispatchQueue(label: "azureAudioListQueue")
private var isActive = false
override init() {
// 音频会话配置
let audioSession = AVAudioSession.sharedInstance()
do {
try audioSession.setCategory(.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth, .defaultToSpeaker, .mixWithOthers])
try audioSession.setActive(true)
print("[AzureMicrophoneStream] 音频会话配置成功")
} catch {
print("[AzureMicrophoneStream] 配置音频会话失败: \(error.localizedDescription)")
}
// 设置音频格式为16KHz 16位单声道PCM
audioFormat = AudioStreamBasicDescription(
mSampleRate: 16000.0,
mFormatID: kAudioFormatLinearPCM,
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
mBytesPerPacket: 2,
mFramesPerPacket: 1,
mBytesPerFrame: 2,
mChannelsPerFrame: 1,
mBitsPerChannel: 16,
mReserved: 0
)
audioBufferList = AudioBufferList(
mNumberBuffers: 1,
mBuffers: AudioBuffer(
mNumberChannels: audioFormat.mChannelsPerFrame,
mDataByteSize: 4096,
mData: malloc(4096)
)
)
super.init()
}
func start() -> Bool {
if isActive {
return true // 已经在运行
}
if setupAudioUnit() {
isActive = startAudioUnit()
return isActive
}
return false
}
private func setupAudioUnit() -> Bool {
print("[AzureMicrophoneStream] 设置音频单元")
var ioUnitDescription = AudioComponentDescription(
componentType: kAudioUnitType_Output,
componentSubType: kAudioUnitSubType_VoiceProcessingIO,
componentManufacturer: kAudioUnitManufacturer_Apple,
componentFlags: 0,
componentFlagsMask: 0
)
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
print("[AzureMicrophoneStream] 无法找到音频组件")
return false
}
if CheckHasError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建IO单元") {
ioUnit = nil
return false
}
var enableInput: UInt32 = 1
let kInputBus: AudioUnitElement = 1
let kOutputBus: AudioUnitElement = 0
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Input, kInputBus, &enableInput,
UInt32(MemoryLayout<UInt32>.size)), "设置输入总线的EnableIO属性") {
return false
}
var enableOutput: UInt32 = 0
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Output, kOutputBus,
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "设置输出总线的EnableIO属性") {
return false
}
var flag: UInt32 = 0
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置ShouldAllocateBuffer属性") {
return false
}
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size)
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线的StreamFormat属性") {
return false
}
if CheckHasError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线的StreamFormat属性") {
return false
}
var inputCallback = AURenderCallbackStruct(
inputProc: AzureMicrophoneStream.OnRecordedDataIsAvailable,
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
)
if CheckHasError(AudioUnitSetProperty(ioUnit!,
kAudioOutputUnitProperty_SetInputCallback,
kAudioUnitScope_Global, kInputBus,
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") {
return false
}
var hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "初始化IO单元")
if hasError {
// 如果初始化失败,重试一次
Thread.sleep(forTimeInterval: 0.5)
hasError = CheckHasError(AudioUnitInitialize(ioUnit!), "重试初始化IO单元")
}
print("[AzureMicrophoneStream] 音频单元设置\(hasError ? "失败" : "成功")")
return !hasError
}
private func startAudioUnit() -> Bool {
print("[AzureMicrophoneStream] 启动音频单元")
guard let ioUnit = ioUnit else {
print("[AzureMicrophoneStream] IO单元未初始化")
return false
}
return !CheckHasError(AudioOutputUnitStart(ioUnit), "启动IO单元")
}
func stop() {
print("[AzureMicrophoneStream] 停止音频单元")
guard isActive, let ioUnit = ioUnit else { return }
_ = CheckHasError(AudioOutputUnitStop(ioUnit), "停止IO单元")
isActive = false
}
func dispose() {
stop()
if let ioUnit = ioUnit {
_ = CheckHasError(AudioUnitUninitialize(ioUnit), "反初始化IO单元")
_ = CheckHasError(AudioComponentInstanceDispose(ioUnit), "释放IO单元")
}
// 释放缓冲区
if let buffer = audioBufferList.mBuffers.mData {
free(buffer)
}
self.ioUnit = nil
// 清空音频数据
audioListQueue.sync {
audioList.removeAll()
}
}
static let OnRecordedDataIsAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
let wrapper = Unmanaged<AzureMicrophoneStream>.fromOpaque(inRefCon).takeUnretainedValue()
let expectedDataByteSize = inNumberFrames * wrapper.audioFormat.mBytesPerFrame
if wrapper.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
wrapper.audioBufferList.mBuffers.mData = realloc(wrapper.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
wrapper.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
}
let status = wrapper.CheckErrorStatus(AudioUnitRender(wrapper.ioUnit!, ioActionFlags, inTimeStamp,
inBusNumber, inNumberFrames, &wrapper.audioBufferList),
"AudioUnitRender调用")
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
let buffer = wrapper.audioBufferList.mBuffers
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) {
audioDataFloat[j] = Float(bufferData[j]) / 32768.0 // 归一化到[-1.0, 1.0]范围
}
if status == noErr {
wrapper.audioListQueue.async {
wrapper.audioList.append(contentsOf: audioDataFloat)
}
}
return status
}
private func CheckHasError(_ status: OSStatus, _ operation: String) -> Bool {
if status != noErr {
print("[AzureMicrophoneStream] \(operation)失败: \(status)")
return true
}
return false
}
private func CheckErrorStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
if status != noErr {
print("[AzureMicrophoneStream] \(operation)失败: \(status)")
}
return status
}
// 读取音频数据,适配Azure SDK
func read(bytes: inout [UInt8]) -> Int {
return audioListQueue.sync {
if audioList.isEmpty {
return 0
}
// 确保有足够的数据
let minFrames = 1280
if audioList.count < minFrames {
return 0
}
let frameLength = minFrames
let buffer = Array(audioList.prefix(frameLength))
audioList.removeFirst(frameLength)
// 转换为Int16数据
var int16Data = buffer.map { Int16($0 * 32767) }
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
bytes = [UInt8](data)
return frameLength * 2 // 每个样本2字节(16位PCM)
}
}
}
/// Azure ASR工具类,负责实现语音识别服务接口 /// Azure ASR工具类,负责实现语音识别服务接口
@available(iOS 13.0, *) @available(iOS 13.0, *)
@ -267,10 +19,6 @@ class AzureAsrHelper: NSObject {
private var recognizer: SPXSpeechRecognizer? private var recognizer: SPXSpeechRecognizer?
private var audioConfig: SPXAudioConfiguration? private var audioConfig: SPXAudioConfiguration?
/// 麦克风流
private var microphoneStream: AzureMicrophoneStream?
private var pushStreamConfig: SPXPushAudioInputStream?
/// 状态标志 /// 状态标志
private var isInitialized = false private var isInitialized = false
private var _isContinuousRecognitionActive = false private var _isContinuousRecognitionActive = false
@ -329,16 +77,18 @@ class AzureAsrHelper: NSObject {
currentLanguage = self.supportedLanguages[0] currentLanguage = self.supportedLanguages[0]
} }
// 创建麦克风流 // 创建识别器和设置回调
microphoneStream = AzureMicrophoneStream() if !createRecognizerAndSetupCallbacks() {
return false
}
print("[AzureAsrHelper] Azure 语音服务初始化成功") print("[AzureAsrHelper] Azure 语音服务初始化成功")
isInitialized = true isInitialized = true
return true return true
} }
/// 重置 recognizer /// 创建识别器并设置回调
private func resetRecognizer() -> Bool { private func createRecognizerAndSetupCallbacks() -> Bool {
// 释放之前的 recognizer // 释放之前的 recognizer
recognizer = nil recognizer = nil
audioConfig = nil audioConfig = nil
@ -347,11 +97,11 @@ class AzureAsrHelper: NSObject {
// 创建语音配置 // 创建语音配置
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
// 创建推送流 // 设置音频输入参数
pushStreamConfig = try SPXPushAudioInputStream() try setupAudioSession()
// 创建音频配置,使用推送流 // 直接使用麦克风音频配置
audioConfig = try SPXAudioConfiguration(streamInput: pushStreamConfig!) audioConfig = try SPXAudioConfiguration()
// 设置语言配置 // 设置语言配置
if isAutoDetectLanguage { if isAutoDetectLanguage {
@ -375,65 +125,121 @@ class AzureAsrHelper: NSObject {
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
} }
// 设置所有回调
setupAllCallbacks()
return true return true
} catch { } catch {
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)")
eventHandler("error", ["message": "重置识别器失败: \(error.localizedDescription)"]) eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"])
return false return false
} }
} }
/// 启动音频捕获和数据推送 /// 设置音频会话
private func startAudioStream() -> Bool { private func setupAudioSession() throws {
guard let micStream = microphoneStream else { let audioSession = AVAudioSession.sharedInstance()
print("[AzureAsrHelper] 错误: 麦克风流未初始化")
return false // 使用playAndRecord类别允许同时录音和播放
try audioSession.setCategory(.playAndRecord,
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay])
// 设置首选的输入和输出
let currentRoute = audioSession.currentRoute
// 获取当前是否连接了耳机或外部麦克风
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
} }
// 启动麦克风 // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除
if !micStream.start() { if !hasHeadphones {
print("[AzureAsrHelper] 错误: 启动麦克风流失败") try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除
return false
// 启用回音消除和噪声抑制
try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性
} else {
// 耳机模式,可以使用不同的设置
try audioSession.setMode(.voiceChat)
try audioSession.setInputGain(1.0)
} }
// 创建并启动音频推送线程 // 设置合适的采样率
DispatchQueue.global(qos: .userInitiated).async { [weak self] in try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率
guard let self = self, let pushStream = self.pushStreamConfig else { return } try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟
// 激活音频会话
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除")
}
/// 设置所有回调
private func setupAllCallbacks() {
guard let recognizer = recognizer else { return }
// 最终识别结果
recognizer.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
var isRunning = true if event.result.reason == SPXResultReason.recognizedSpeech {
var audioBuffer = [UInt8](repeating: 0, count: 16000) let detectedLanguage = self.getDetectedLanguage(from: event.result)
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
self.eventHandler("result", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
// 识别中事件
recognizer.addRecognizingEventHandler { [weak self] _, event in
guard let self = self else { return }
while isRunning { if event.result.reason == SPXResultReason.recognizingSpeech {
autoreleasepool { let detectedLanguage = self.getDetectedLanguage(from: event.result)
// 读取麦克风数据 // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
let bytesRead = micStream.read(bytes: &audioBuffer) self.eventHandler("recognizing", [
"text": event.result.text ?? "",
if bytesRead > 0 { "detectedLanguage": detectedLanguage
do { ])
// 推送音频数据到Azure识别流
let data = Data(bytes: audioBuffer, count: bytesRead)
try pushStream.write(data)
} catch {
print("[AzureAsrHelper] 推送音频数据失败: \(error.localizedDescription)")
isRunning = false
}
}
// 检查是否应该继续捕获
if !self._isContinuousRecognitionActive {
isRunning = false
}
// 添加适当的休眠以避免过度消耗CPU
if bytesRead == 0 {
Thread.sleep(forTimeInterval: 0.01)
}
}
} }
print("[AzureAsrHelper] 音频推送线程已停止")
} }
return true // 会话事件
recognizer.addSessionStartedEventHandler { [weak self] _, _ in
guard let self = self else { return }
print("[AzureAsrHelper] 识别会话已开始")
self._isContinuousRecognitionActive = true
self.eventHandler("sessionStarted", [:])
}
recognizer.addSessionStoppedEventHandler { [weak self] _, _ in
guard let self = self else { return }
print("[AzureAsrHelper] 识别会话已结束")
self._isContinuousRecognitionActive = false
self.eventHandler("sessionStopped", [:])
}
// 取消事件
recognizer.addCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
let reason = event.reason.rawValue
let errorDetails = event.errorDetails ?? "未知错误"
print("[AzureAsrHelper] 识别取消: \(errorDetails)")
self.eventHandler("canceled", [
"reason": reason,
"errorDetails": errorDetails
])
self._isContinuousRecognitionActive = false
}
} }
/// 执行一次性语音识别 /// 执行一次性语音识别
@ -450,37 +256,12 @@ class AzureAsrHelper: NSObject {
stopContinuousRecognition() stopContinuousRecognition()
} }
// 重置 recognizer // 确保识别器已创建
if !resetRecognizer() { if recognizer == nil && !createRecognizerAndSetupCallbacks() {
return false
}
// 启动音频流
if !startAudioStream() {
return false return false
} }
do { do {
// 设置回调
recognizer?.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
self.eventHandler("result", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
recognizer?.addCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
let errorDetails = event.errorDetails ?? "未知错误"
self.eventHandler("error", ["message": "识别异常: \(errorDetails)"])
}
// 通知会话开始 // 通知会话开始
eventHandler("sessionStarted", [:]) eventHandler("sessionStarted", [:])
@ -488,15 +269,15 @@ class AzureAsrHelper: NSObject {
try recognizer?.recognizeOnceAsync { [weak self] result in try recognizer?.recognizeOnceAsync { [weak self] result in
guard let self = self else { return } guard let self = self else { return }
// 停止麦克风流
self.microphoneStream?.stop()
if result.reason == SPXResultReason.recognizedSpeech { if result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: result) let detectedLanguage = self.getDetectedLanguage(from: result)
self.eventHandler("result", [ self.eventHandler("result", [
"text": result.text ?? "", "text": result.text ?? "",
"detectedLanguage": detectedLanguage "detectedLanguage": detectedLanguage
]) ])
} else if result.reason == SPXResultReason.noMatch {
print("[AzureAsrHelper] 无匹配结果")
self.eventHandler("noMatch", [:])
} else if result.reason == SPXResultReason.canceled { } else if result.reason == SPXResultReason.canceled {
do { do {
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result)
@ -513,7 +294,6 @@ class AzureAsrHelper: NSObject {
} catch { } catch {
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"])
microphoneStream?.stop()
return false return false
} }
} }
@ -532,36 +312,29 @@ class AzureAsrHelper: NSObject {
stopContinuousRecognition() stopContinuousRecognition()
} }
// 重置 recognizer // 确保识别器已创建
if !resetRecognizer() { if recognizer == nil && !createRecognizerAndSetupCallbacks() {
return false return false
} }
// 重新确保音频设置正确
do {
try setupAudioSession()
} catch {
print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)")
}
do { do {
// 设置识别事件处理
setupContinuousRecognitionCallbacks()
// 启动连续识别 // 启动连续识别
try recognizer?.startContinuousRecognition() try recognizer?.startContinuousRecognition()
_isContinuousRecognitionActive = true _isContinuousRecognitionActive = true
// 启动音频流
if !startAudioStream() {
try recognizer?.stopContinuousRecognition()
_isContinuousRecognitionActive = false
return false
}
// 通知会话开始
eventHandler("sessionStarted", [:])
print("[AzureAsrHelper] 连续识别开始") print("[AzureAsrHelper] 连续识别开始")
return true return true
} catch { } catch {
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"])
_isContinuousRecognitionActive = false _isContinuousRecognitionActive = false
microphoneStream?.stop()
return false return false
} }
} }
@ -573,13 +346,9 @@ class AzureAsrHelper: NSObject {
return true return true
} }
// 停止麦克风流
microphoneStream?.stop()
do { do {
try recognizer?.stopContinuousRecognition() try recognizer?.stopContinuousRecognition()
_isContinuousRecognitionActive = false _isContinuousRecognitionActive = false
eventHandler("sessionStopped", [:])
print("[AzureAsrHelper] 连续识别已停止") print("[AzureAsrHelper] 连续识别已停止")
return true return true
} catch { } catch {
@ -605,94 +374,23 @@ class AzureAsrHelper: NSObject {
stopContinuousRecognition() stopContinuousRecognition()
} }
// 关闭麦克风流 // 释放音频会话
microphoneStream?.dispose() do {
microphoneStream = nil try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation)
} catch {
// 关闭推送流 print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)")
if let pushStream = pushStreamConfig {
do {
try pushStream.close()
} catch {
print("[AzureAsrHelper] 关闭推送流失败: \(error.localizedDescription)")
}
} }
// 释放资源 // 释放资源
recognizer = nil recognizer = nil
speechConfig = nil speechConfig = nil
audioConfig = nil audioConfig = nil
pushStreamConfig = nil
// 重置状态 // 重置状态
_isContinuousRecognitionActive = false _isContinuousRecognitionActive = false
isInitialized = false isInitialized = false
} }
// MARK: - 私有辅助方法
/// 设置连续识别回调
private func setupContinuousRecognitionCallbacks() {
// 最终识别结果
recognizer?.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
self.eventHandler("result", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
// 识别中事件
recognizer?.addRecognizingEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == SPXResultReason.recognizingSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
self.eventHandler("recognizing", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
// 会话事件
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in
guard let self = self else { return }
self._isContinuousRecognitionActive = true
self.eventHandler("sessionStarted", [:])
}
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in
guard let self = self else { return }
self._isContinuousRecognitionActive = false
self.eventHandler("sessionStopped", [:])
}
// 取消事件
recognizer?.addCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
let reason = event.reason.rawValue
let errorDetails = event.errorDetails ?? ""
self.eventHandler("canceled", [
"reason": reason,
"errorDetails": errorDetails
])
self._isContinuousRecognitionActive = false
self.microphoneStream?.stop()
}
}
/// 从结果中获取检测到的语言 /// 从结果中获取检测到的语言
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
if isAutoDetectLanguage { if isAutoDetectLanguage {

63
azure/ios/Classes/AzureTtsHelper.swift

@ -26,6 +26,9 @@ class AzureTtsHelper: NSObject {
/// 当前是否正在播放 /// 当前是否正在播放
private var _isSpeaking = false private var _isSpeaking = false
/// 音频会话配置
private var isAudioSessionConfigured = false
// MARK: - 语音设置 // MARK: - 语音设置
/// 当前语音 /// 当前语音
@ -82,13 +85,15 @@ class AzureTtsHelper: NSObject {
self.speechSubscriptionKey = speechSubscriptionKey self.speechSubscriptionKey = speechSubscriptionKey
self.serviceRegion = serviceRegion self.serviceRegion = serviceRegion
// 配置音频会话
if !configureAudioSession() {
print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化")
}
do { do {
// 创建语音配置 // 创建语音配置
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置语音合成输出格式
// speechConfig?.setSpeechSynthesisOutputFormat(.audio24Khz48KBitRateMonoMp3)
// 设置默认语音 // 设置默认语音
let defaultVoice = getDefaultVoiceForLanguage(language) let defaultVoice = getDefaultVoiceForLanguage(language)
currentVoice = defaultVoice currentVoice = defaultVoice
@ -111,6 +116,46 @@ class AzureTtsHelper: NSObject {
} }
} }
/// 配置音频会话
private func configureAudioSession() -> Bool {
let audioSession = AVAudioSession.sharedInstance()
do {
// 使用playback类别,但支持混合和空中播放
try audioSession.setCategory(.playback,
mode: .spokenAudio,
options: [.mixWithOthers, .allowAirPlay, .duckOthers])
// 根据设备类型选择最佳配置
let currentRoute = audioSession.currentRoute
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
}
// 优化音频路由
if hasHeadphones {
// 耳机模式,使用默认设置
try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟
} else {
// 扬声器模式
try audioSession.setPreferredIOBufferDuration(0.005)
}
// 避免完全激活音频会话,因为ASR可能已经激活
// 这里使用setActive(false)是为了不与ASR冲突
if !audioSession.isOtherAudioPlaying {
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
}
isAudioSessionConfigured = true
print("[AzureTtsHelper] 音频会话配置成功")
return true
} catch {
print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)")
isAudioSessionConfigured = false
return false
}
}
/// 设置语音 /// 设置语音
/// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") /// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural")
/// - Returns: 设置是否成功 /// - Returns: 设置是否成功
@ -181,6 +226,11 @@ class AzureTtsHelper: NSObject {
return true return true
} }
// 确保音频会话已配置
if !isAudioSessionConfigured {
_ = configureAudioSession()
}
print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...")
// 生成SSML // 生成SSML
@ -203,8 +253,6 @@ class AzureTtsHelper: NSObject {
Task { Task {
do { do {
print("[AzureTtsHelper] 开始语音合成")
// 使用异步方法进行合成并直接播放 // 使用异步方法进行合成并直接播放
_ = try await synthesizer.startSpeakingSsml(text) _ = try await synthesizer.startSpeakingSsml(text)
@ -257,6 +305,7 @@ class AzureTtsHelper: NSObject {
isInitialized = false isInitialized = false
_isSpeaking = false _isSpeaking = false
isAudioSessionConfigured = false
print("[AzureTtsHelper] TTS 引擎已释放") print("[AzureTtsHelper] TTS 引擎已释放")
} }
@ -311,12 +360,12 @@ class AzureTtsHelper: NSObject {
// 合成开始事件 // 合成开始事件
synthesizer.addSynthesisStartedEventHandler { _, _ in synthesizer.addSynthesisStartedEventHandler { _, _ in
print("[AzureTtsHelper] 语音合成开始") // print("[AzureTtsHelper] 语音合成开始")
} }
// 合成中事件 // 合成中事件
synthesizer.addSynthesizingEventHandler { _, _ in synthesizer.addSynthesizingEventHandler { _, _ in
print("[AzureTtsHelper] 语音合成中") // print("[AzureTtsHelper] 语音合成中")
} }
} }

4
azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift

@ -42,12 +42,10 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
// 新增直接处理方法调用的函数 // 新增直接处理方法调用的函数
private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
print("[AzurePlugin] 处理TTS方法调用: \(call.method)")
handleTtsMethod(call, result) handleTtsMethod(call, result)
} }
private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
print("[AzurePlugin] 处理ASR方法调用: \(call.method)")
handleAsrMethod(call, result) handleAsrMethod(call, result)
} }
@ -74,7 +72,6 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
} }
private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
print("[AzurePlugin] 处理ASR方法: \(call.method)")
let args = call.arguments as? Dictionary<String, Any> let args = call.arguments as? Dictionary<String, Any>
@ -136,7 +133,6 @@ public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
} }
private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
print("[AzurePlugin] 处理TTS方法: \(call.method)")
let args = call.arguments as? Dictionary<String, Any> let args = call.arguments as? Dictionary<String, Any>

6
lib/modules/home/views/home_view.dart

@ -101,11 +101,7 @@ class HomeView extends GetView<HomeController> {
SizedBox(height: 16.h), SizedBox(height: 16.h),
_buildPremiumAILayout(), _buildPremiumAILayout(),
// 在合适的位置添加测试按钮
ElevatedButton(
onPressed: () => Get.toNamed(Routes.ASR_TEST),
child: const Text('Azure语音识别测试'),
),
], ],
), ),
), ),

Loading…
Cancel
Save