Browse Source

fix(tts): 修复蓝牙音频路由和播放状态同步问题

优化音频会话配置逻辑,避免在通话中错误禁用蓝牙音频
重构TTS播放状态管理,确保播放事件与状态同步
修复流式播放完成事件触发时机,避免过早通知
weicu
liwei1dao 9 months ago
parent
commit
3f5173f28f
  1. 19
      local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift
  2. 738
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift
  3. 64
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift

19
local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift

@ -407,19 +407,27 @@ class AgentServiceImpl: NSObject {
os_log("通话模式:禁用空闲检测", log: logger, type: .info)
return
case "ble_wakeup":
azureAsrHelper?.disableBluetoothAudio()
if !isSpeaking {
azureAsrHelper?.disableBluetoothAudio()
}
// BLE唤醒模式:使用默认的空闲检测
startIdleCheck(idleSeconds: maxIdleSeconds)
case "push_to_talk":
azureAsrHelper?.disableBluetoothAudio()
if !isSpeaking {
azureAsrHelper?.disableBluetoothAudio()
}
// 按住说话模式:使用默认的空闲检测
startIdleCheck(idleSeconds: maxIdleSeconds)
case "normal":
azureAsrHelper?.disableBluetoothAudio()
if !isSpeaking {
azureAsrHelper?.disableBluetoothAudio()
}
// 普通模式:使用默认的空闲检测
startIdleCheck(idleSeconds: maxIdleSeconds)
default:
azureAsrHelper?.disableBluetoothAudio()
if !isSpeaking {
azureAsrHelper?.disableBluetoothAudio()
}
// 未知模式:使用默认的空闲检测
os_log("未知识别模式: %{public}@,使用默认空闲检测", log: logger, type: .error, mode)
startIdleCheck(idleSeconds: maxIdleSeconds)
@ -1970,11 +1978,11 @@ extension AgentServiceImpl: TtsEventListener {
sendEvent(name: "tts_started", data: ["status": "started"])
case .synthesisCompleted:
isSpeaking = false
restartIdleCheck()
sendEvent(name: "tts_completed", data: ["status": "completed"])
case .playbackStarted:
isSpeaking = true
restartIdleCheck()
sendEvent(name: "playback_started", data: ["status": "playback_started"])
audioPlayer?.stopAwaitSound()
@ -1983,6 +1991,7 @@ extension AgentServiceImpl: TtsEventListener {
BleService.shared.closeCodec()
}
case .playbackCompleted:
isSpeaking = false
restartIdleCheck()
if(isInterrupt){//播报结束。打开解码器
BleService.shared.openB1Encoder()

738
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift

@ -5,7 +5,7 @@ import os.log
// 自定义语音处理组件,提供音频流处理等功能
import speech
public class AzureTtsHelper: NSObject, ITtsService {
public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate {
private let tag = "AzureTtsHelper"
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper")
@ -48,10 +48,30 @@ public class AzureTtsHelper: NSObject, ITtsService {
// 自定义音频输出流
private var customAudioOutputStream: SPXPushAudioOutputStream?
private var useInternalPlayer = true
private var useInternalPlayer = false
private var micCapture: MicrophoneCapture!
// 记录最后播放的文本
private var lastSpokenText: String?
// 播放相关:队列与播放器
private var playbackQueue: [Data] = []
private var audioPlayer: AVAudioPlayer?
private var isPlaying = false
private var currentSynthesisBuffer = Data()
private var currentTempFileURL: URL?
private var isUsingPushStreamCapture = false
// 流式播放相关:边合成边播放(16kHz/16bit/mono PCM)
private let audioPlaybackQueue = DispatchQueue(label: "com.azure.tts.playback", qos: .userInitiated)
private var audioEngine: AVAudioEngine?
private var playerNode: AVAudioPlayerNode?
private var pcmFormat: AVAudioFormat?
private var pcmPendingData = Data()
private var hasStrippedWavHeader = false
private var hasNotifiedPlaybackStartedForStream = false
private var activeStreamSynthesisCount = 0
private var scheduledBufferCount = 0
private let streamChunkBytes = 1600
private var suppressStreamPlayback = false
/**
* 初始化语音合成服务
@ -65,7 +85,7 @@ public class AzureTtsHelper: NSObject, ITtsService {
currentLanguage = language
// 设置音频输出格式 - 与Android一致
speechConfig?.setPropertyTo("riff-16khz-16bit-mono-pcm",
speechConfig?.setPropertyTo("raw-16khz-16bit-mono-pcm",
byName: "SpeechServiceConnection_SynthOutputFormat")
// 设置低延迟属性
@ -164,6 +184,7 @@ public class AzureTtsHelper: NSObject, ITtsService {
}
// 异步处理
os_log("调用链: speakOnce 入队合成 session=%{public}@ 文本长度=%{public}d", log: log, type: .info, sessionid, cleanedText.count)
enqueueSynthesisTask {
self.performSynthesis(sessionid: sessionid, text: cleanedText)
}
@ -258,15 +279,34 @@ public class AzureTtsHelper: NSObject, ITtsService {
/**
* 停止语音合成
* - Parameters: 无
* - Returns: 是否成功发起停止流程
* - Throws: 无(内部捕获 SDK/音频会话异常并转为 error 事件)
*/
public func stop() -> Bool {
speaking = false
sessionid = ""
streamBuffer = ""
lastSpokenText = ""
// 重置计数器
pendingTextCount = 0
lastSpokenText = nil
resetStreamingPlaybackState()
if let player = audioPlayer {
player.stop()
}
audioPlayer = nil
isPlaying = false
playbackQueue.removeAll()
currentSynthesisBuffer = Data()
if let url = currentTempFileURL {
try? FileManager.default.removeItem(at: url)
currentTempFileURL = nil
}
pendingTextCount = 0
currentSessionTextCount = 0
// 清空所有待处理任务并重置处理状态
taskLock.lock()
pendingTasks.removeAll()
@ -274,16 +314,8 @@ public class AzureTtsHelper: NSObject, ITtsService {
taskLock.unlock()
do {
print("liwei-------------停止语音合成播放")
try self.synthesizer?.stopSpeaking()
try synthesizer?.stopSpeaking()
self.notifyEvent(eventType: .synthesisCanceled)
if !micCapture.isCapturing{
// 停止时设置音频会话为非活跃状态
// try AVAudioSession.sharedInstance().setActive(false)
}
print("音频会话已设置为非活跃状态")
} catch {
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription)
self.notifyEvent(eventType: .error, params: [
@ -293,13 +325,10 @@ public class AzureTtsHelper: NSObject, ITtsService {
}
synthesisQueue.async {
do {
// self.synthesizer = nil
try self.recreateSynthesizer()
}catch {
os_log("重置语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription)
}
self.recreateSynthesizer()
}
deactivateAudioSessionAfterTTSIfIdle()
return true
}
@ -446,6 +475,14 @@ public class AzureTtsHelper: NSObject, ITtsService {
}
speaking = true
// lastSpokenText = text
if !applyAudioSessionForTTS() {
speaking = false
notifyEvent(eventType: .error, params: [
"errorCode": "AUDIO_SESSION_FAILED",
"errorMessage": "TTS音频会话激活失败"
])
return
}
// 生成SSML
let ssml = generateOptimizedSsml(text)
@ -479,17 +516,45 @@ public class AzureTtsHelper: NSObject, ITtsService {
do {
// 强制释放旧的合成器
synthesizer = nil
isUsingPushStreamCapture = false
// 创建音频配置
let audioConfig: SPXAudioConfiguration?
if !useInternalPlayer || customAudioOutputStream != nil {
audioConfig = try SPXAudioConfiguration(streamOutput: customAudioOutputStream ?? SPXPushAudioOutputStream())
let stream: SPXPushAudioOutputStream
if let provided = customAudioOutputStream {
stream = provided
} else {
isUsingPushStreamCapture = true
// 使用写入/关闭回调创建推送输出流,避免无效参数错误
var totalBytes: Int = 0
let created = SPXPushAudioOutputStream(writeHandler: { [weak self] data -> UInt in
guard let self = self else { return 0 }
totalBytes += data.count
self.handleSynthesizedAudioChunk(data)
return UInt(data.count)
}, closeHandler: { [weak self] in
guard let self = self else { return }
os_log("调用链: 输出流关闭 累计字节=%{public}d", log: self.log, type: .info, totalBytes)
totalBytes = 0
})
// 如果创建失败,抛出以进入 catch
guard let nonNilStream = created else {
throw NSError(domain: "AzureTtsHelper", code: -1, userInfo: [NSLocalizedDescriptionKey: "创建推送输出流失败"])
}
stream = nonNilStream
}
audioConfig = try SPXAudioConfiguration(streamOutput: stream)
} else {
audioConfig = nil // 使用默认扬声器
audioConfig = nil
}
// 创建合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig!)
if let audioConfig = audioConfig {
synthesizer = try SPXSpeechSynthesizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig)
} else {
synthesizer = try SPXSpeechSynthesizer(speechConfig!)
}
setupEventListeners()
} catch {
os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription)
@ -505,35 +570,53 @@ public class AzureTtsHelper: NSObject, ITtsService {
*/
private func setupEventListeners() {
synthesizer?.addSynthesisStartedEventHandler { [weak self] _, _ in
self?.notifyEvent(eventType: .synthesisStarted)
self?.notifyEvent(eventType: .playbackStarted)
guard let self = self else { return }
self.currentSynthesisBuffer = Data()
if self.suppressStreamPlayback { return }
if self.isUsingPushStreamCapture {
self.audioPlaybackQueue.async {
self.activeStreamSynthesisCount += 1
self.prepareStreamingQueueForNewSynthesisIfNeeded()
}
}
os_log("调用链: 合成开始 session=%{public}@", log: self.log, type: .info, self.sessionid)
self.notifyEvent(eventType: .synthesisStarted)
}
synthesizer?.addSynthesizingEventHandler { [weak self] _, event in
if let audioData = event.result.audioData {
self?.notifyAudioData(audioData)
guard let self = self else { return }
os_log("调用链: 合成中 接收音频片段=%{public}dB session=%{public}@", log: self.log, type: .debug, audioData.count, self.sessionid)
if !self.isUsingPushStreamCapture {
self.currentSynthesisBuffer.append(audioData)
self.notifyAudioData(audioData)
}
}
}
synthesizer?.addSynthesisCompletedEventHandler { [weak self] _, _ in
guard let self = self else { return }
self.speaking = false
if self.suppressStreamPlayback { return }
// 将当前合成的音频加入播放队列
if !self.currentSynthesisBuffer.isEmpty {
os_log("调用链: 合成完成 入队播放 数据长度=%{public}dB session=%{public}@", log: self.log, type: .info, self.currentSynthesisBuffer.count, self.sessionid)
self.enqueuePlaybackItem(self.currentSynthesisBuffer)
self.currentSynthesisBuffer = Data()
}
if self.isUsingPushStreamCapture {
self.audioPlaybackQueue.async {
self.activeStreamSynthesisCount = max(0, self.activeStreamSynthesisCount - 1)
self.scheduleAvailablePcmBuffers()
self.checkStreamPlaybackCompletedIfNeeded()
}
}
// 减少待处理文本计数
self.pendingTextCount = max(0, self.pendingTextCount - 1)
// 通知合成完成
self.notifyEvent(eventType: .synthesisCompleted)
self.notifyEvent(eventType: .playbackCompleted)
// 检查是否为最后一个文本播放完毕
if self.pendingTextCount == 0 && self.pendingTasks.isEmpty && !micCapture.isCapturing{
do {
// try AVAudioSession.sharedInstance().setActive(false)
print("最后一个文本播放完毕,音频会话已设置为非活跃状态")
} catch {
print("设置音频会话为非活跃状态失败: \(error.localizedDescription)")
}
}
// 尝试开始播放
self.startPlaybackIfNeeded()
}
synthesizer?.addSynthesisCanceledEventHandler { [weak self] _, event in
@ -542,6 +625,14 @@ public class AzureTtsHelper: NSObject, ITtsService {
// 减少待处理文本计数
self.pendingTextCount = max(0, self.pendingTextCount - 1)
if self.suppressStreamPlayback { return }
if self.isUsingPushStreamCapture {
self.audioPlaybackQueue.async {
self.activeStreamSynthesisCount = max(0, self.activeStreamSynthesisCount - 1)
self.checkStreamPlaybackCompletedIfNeeded()
}
}
var params: [String: Any] = [:]
if let details = try? SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: event.result) {
@ -552,6 +643,390 @@ public class AzureTtsHelper: NSObject, ITtsService {
self.notifyEvent(eventType: .synthesisCanceled, params: params)
}
}
/**
* 入队播放数据
* - Parameters:
* - data: 合成的音频数据(RIFF PCM)
* - Returns: 无
* - Throws: 无
*/
private func enqueuePlaybackItem(_ data: Data) {
playbackQueue.append(data)
}
/**
* 处理合成输出的音频片段(用于流式播放)
* - Parameters:
* - data: SDK 推送输出流回调提供的音频片段数据
* - Returns: 无
* - Throws: 无
*/
private func handleSynthesizedAudioChunk(_ data: Data) {
if suppressStreamPlayback { return }
if !speaking { return }
notifyAudioData(data)
guard isUsingPushStreamCapture else { return }
audioPlaybackQueue.async {
if !self.speaking { return }
let pcm = self.stripWavHeaderIfNeeded(data)
if !pcm.isEmpty {
self.pcmPendingData.append(pcm)
self.scheduleAvailablePcmBuffers()
}
}
}
/**
* 新一轮合成开始时准备流式播放队列(不打断正在播放的内容)
* - Returns: 无
* - Throws: 无
*/
private func prepareStreamingQueueForNewSynthesisIfNeeded() {
if pcmPendingData.isEmpty, scheduledBufferCount == 0, playerNode?.isPlaying != true {
hasStrippedWavHeader = false
hasNotifiedPlaybackStartedForStream = false
}
}
/**
* 停止并清理流式播放资源(用于 stop/dispose)
* - Returns: 无
* - Throws: 无
*/
private func resetStreamingPlaybackState() {
audioPlaybackQueue.sync {
self.pcmPendingData = Data()
self.hasStrippedWavHeader = false
self.hasNotifiedPlaybackStartedForStream = false
self.activeStreamSynthesisCount = 0
self.scheduledBufferCount = 0
self.playerNode?.stop()
self.audioEngine?.stop()
self.audioEngine = nil
self.playerNode = nil
self.pcmFormat = nil
}
}
/**
* 将待播 PCM 数据按固定块切分并调度到 AVAudioPlayerNode
* - Returns: 无
* - Throws: 无
*/
private func scheduleAvailablePcmBuffers() {
guard ensureStreamingEngineIfNeeded() else { return }
while pcmPendingData.count >= streamChunkBytes {
let chunk = pcmPendingData.prefix(streamChunkBytes)
pcmPendingData.removeFirst(streamChunkBytes)
guard let buffer = makePcmBuffer(from: Data(chunk)) else { continue }
scheduledBufferCount += 1
if !hasNotifiedPlaybackStartedForStream {
hasNotifiedPlaybackStartedForStream = true
os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid)
notifyEvent(eventType: .playbackStarted)
}
playerNode?.scheduleBuffer(buffer, completionHandler: { [weak self] in
guard let self = self else { return }
self.audioPlaybackQueue.async {
self.scheduledBufferCount = max(0, self.scheduledBufferCount - 1)
self.checkStreamPlaybackCompletedIfNeeded()
}
})
}
if activeStreamSynthesisCount == 0, !pcmPendingData.isEmpty {
if pcmPendingData.count % 2 != 0 {
pcmPendingData.removeLast()
}
if !pcmPendingData.isEmpty, let buffer = makePcmBuffer(from: pcmPendingData) {
pcmPendingData.removeAll(keepingCapacity: true)
scheduledBufferCount += 1
if !hasNotifiedPlaybackStartedForStream {
hasNotifiedPlaybackStartedForStream = true
os_log("调用链: 流式播放开始 session=%{public}@", log: log, type: .info, sessionid)
notifyEvent(eventType: .playbackStarted)
}
playerNode?.scheduleBuffer(buffer, completionHandler: { [weak self] in
guard let self = self else { return }
self.audioPlaybackQueue.async {
self.scheduledBufferCount = max(0, self.scheduledBufferCount - 1)
self.checkStreamPlaybackCompletedIfNeeded()
}
})
}
}
if playerNode?.isPlaying == false {
playerNode?.play()
}
}
/**
* 当合成结束且播放缓冲耗尽时触发播放完成事件
* - Returns: 无
* - Throws: 无
*/
private func checkStreamPlaybackCompletedIfNeeded() {
if activeStreamSynthesisCount == 0, pcmPendingData.isEmpty, scheduledBufferCount == 0 {
hasNotifiedPlaybackStartedForStream = false
os_log("调用链: 流式播放完成 session=%{public}@", log: log, type: .info, sessionid)
notifyEvent(eventType: .playbackCompleted)
deactivateAudioSessionAfterTTSIfIdle()
}
}
/**
* 确保 AVAudioEngine/AVAudioPlayerNode 已准备好用于 16kHz/16bit/mono PCM 播放
* - Returns: 初始化是否成功
* - Throws: 无
*/
private func ensureStreamingEngineIfNeeded() -> Bool {
if !applyAudioSessionForTTS() {
os_log("流式播放初始化: 激活音频会话失败", log: log, type: .error)
return false
}
if let engine = audioEngine, let node = playerNode, let format = pcmFormat {
if !engine.isRunning {
do {
try engine.start()
} catch {
audioEngine = nil
playerNode = nil
pcmFormat = nil
}
}
if audioEngine != nil, playerNode != nil, pcmFormat != nil {
if node.isPlaying == false, engine.isRunning {
node.play()
}
_ = format
return true
}
}
let engine = AVAudioEngine()
let node = AVAudioPlayerNode()
guard let format = AVAudioFormat(standardFormatWithSampleRate: 16000, channels: 1) else {
os_log("流式播放初始化失败: 创建音频格式失败", log: log, type: .error)
return false
}
engine.attach(node)
engine.connect(node, to: engine.mainMixerNode, format: format)
do {
try engine.start()
} catch {
os_log("流式播放初始化失败: %{public}@", log: log, type: .error, error.localizedDescription)
return false
}
audioEngine = engine
playerNode = node
pcmFormat = format
return true
}
/**
* 将 PCM 数据(16kHz/16bit/mono)转换为可调度的 AVAudioPCMBuffer
* - Parameters:
* - data: PCM 裸数据(小端 Int16)
* - Returns: 可用于 scheduleBuffer 的 AVAudioPCMBuffer,失败返回 nil
* - Throws: 无
*/
private func makePcmBuffer(from data: Data) -> AVAudioPCMBuffer? {
guard let format = pcmFormat else { return nil }
let frameCount = data.count / 2
guard frameCount > 0 else { return nil }
guard let buffer = AVAudioPCMBuffer(pcmFormat: format, frameCapacity: AVAudioFrameCount(frameCount)) else { return nil }
buffer.frameLength = AVAudioFrameCount(frameCount)
guard let dst = buffer.floatChannelData else { return nil }
data.withUnsafeBytes { raw in
guard let src = raw.bindMemory(to: Int16.self).baseAddress else { return }
for i in 0..<frameCount {
dst[0][i] = Float(src[i]) / 32768.0
}
}
return buffer
}
/**
* 如数据包含 WAV 头则剥离,返回 PCM 部分(用于推送流的首包兼容)
* - Parameters:
* - data: 推送流返回的原始数据
* - Returns: PCM 裸数据(可能为空)
* - Throws: 无
*/
private func stripWavHeaderIfNeeded(_ data: Data) -> Data {
if hasStrippedWavHeader { return data }
if data.count >= 12,
String(data: data.subdata(in: 0..<4), encoding: .ascii) == "RIFF",
String(data: data.subdata(in: 8..<12), encoding: .ascii) == "WAVE" {
let marker = Data("data".utf8)
if let range = data.range(of: marker, options: [], in: 0..<min(data.count, 512)) {
let start = range.lowerBound + 8
if start <= data.count {
hasStrippedWavHeader = true
return data.subdata(in: start..<data.count)
}
}
if data.count > 44 {
hasStrippedWavHeader = true
return data.subdata(in: 44..<data.count)
}
return Data()
}
hasStrippedWavHeader = true
return data
}
/**
* 如有需要则开始播放队列
* - Returns: 无
* - Throws: 无
*/
private func startPlaybackIfNeeded() {
guard !isPlaying, !playbackQueue.isEmpty else { return }
let next = playbackQueue.first!
let playableData = ensureWavDataForPlayback(next)
do {
// 先尝试直接用内存数据播放,失败再降级为临时文件播放(兼容性更好)
do {
audioPlayer = try AVAudioPlayer(data: playableData)
} catch {
os_log("内存播放初始化失败,降级为临时文件: %{public}@", log: log, type: .error, error.localizedDescription)
if let url = createTempWavFile(playableData) {
currentTempFileURL = url
audioPlayer = try AVAudioPlayer(contentsOf: url)
} else {
throw error
}
}
audioPlayer?.delegate = self
audioPlayer?.prepareToPlay()
isPlaying = true
os_log("调用链: 开始播放 队首长度=%{public}dB session=%{public}@", log: log, type: .info, next.count, sessionid)
notifyEvent(eventType: .playbackStarted)
audioPlayer?.play()
} catch {
os_log("播放初始化失败: %{public}@", log: log, type: .error, error.localizedDescription)
// 异常情况下移除该条并尝试播放下一条
playbackQueue.removeFirst()
isPlaying = false
startPlaybackIfNeeded()
}
}
/**
* 确保可被 AVAudioPlayer 识别的 WAV 数据
* - Parameters:
* - data: 可能为 WAV 或 PCM 的音频数据
* - Returns: WAV 数据(必要时为 PCM 补 WAV 头)
* - Throws: 无
*/
private func ensureWavDataForPlayback(_ data: Data) -> Data {
if isWavData(data) {
return data
}
os_log("调用链: 音频无WAV头,补WAV头后播放 bytes=%{public}d", log: log, type: .info, data.count)
let header = makeWavHeader(pcmDataSize: UInt32(data.count), sampleRate: 16000, channels: 1, bitsPerSample: 16)
var out = Data()
out.append(header)
out.append(data)
return out
}
/**
* 判断数据是否为 WAV(RIFF/WAVE)
* - Parameters:
* - data: 音频数据
* - Returns: 是否为 WAV
* - Throws: 无
*/
private func isWavData(_ data: Data) -> Bool {
guard data.count >= 12 else { return false }
let riff = data.subdata(in: 0..<4)
let wave = data.subdata(in: 8..<12)
return String(data: riff, encoding: .ascii) == "RIFF" && String(data: wave, encoding: .ascii) == "WAVE"
}
/**
* 生成 WAV 头(PCM)
* - Parameters:
* - pcmDataSize: PCM 数据长度(字节)
* - sampleRate: 采样率
* - channels: 声道数
* - bitsPerSample: 位深
* - Returns: WAV 文件头数据
* - Throws: 无
*/
private func makeWavHeader(pcmDataSize: UInt32, sampleRate: UInt32, channels: UInt16, bitsPerSample: UInt16) -> Data {
let byteRate = sampleRate * UInt32(channels) * UInt32(bitsPerSample) / 8
let blockAlign = channels * bitsPerSample / 8
let chunkSize = 36 + pcmDataSize
var data = Data()
data.append(contentsOf: Array("RIFF".utf8))
data.append(contentsOf: withUnsafeBytes(of: chunkSize.littleEndian, Array.init))
data.append(contentsOf: Array("WAVE".utf8))
data.append(contentsOf: Array("fmt ".utf8))
var subchunk1Size: UInt32 = 16
data.append(contentsOf: withUnsafeBytes(of: subchunk1Size.littleEndian, Array.init))
var audioFormat: UInt16 = 1
data.append(contentsOf: withUnsafeBytes(of: audioFormat.littleEndian, Array.init))
data.append(contentsOf: withUnsafeBytes(of: channels.littleEndian, Array.init))
data.append(contentsOf: withUnsafeBytes(of: sampleRate.littleEndian, Array.init))
data.append(contentsOf: withUnsafeBytes(of: byteRate.littleEndian, Array.init))
data.append(contentsOf: withUnsafeBytes(of: blockAlign.littleEndian, Array.init))
data.append(contentsOf: withUnsafeBytes(of: bitsPerSample.littleEndian, Array.init))
data.append(contentsOf: Array("data".utf8))
data.append(contentsOf: withUnsafeBytes(of: pcmDataSize.littleEndian, Array.init))
return data
}
/**
* 创建临时WAV文件并返回URL
* - Parameters:
* - data: WAV数据(RIFF 16kHz/16bit/mono)
* - Returns: 成功返回文件URL,失败返回nil
* - Throws: 无
*/
private func createTempWavFile(_ data: Data) -> URL? {
let tmpDir = URL(fileURLWithPath: NSTemporaryDirectory(), isDirectory: true)
let fileURL = tmpDir.appendingPathComponent("azure_tts_\(UUID().uuidString).wav")
do {
try data.write(to: fileURL, options: .atomic)
os_log("调用链: 写入临时文件 path=%{public}@", log: log, type: .debug, fileURL.path)
return fileURL
} catch {
os_log("写入临时文件失败: %{public}@", log: log, type: .error, error.localizedDescription)
return nil
}
}
/**
* 播放完成回调
* - Parameters:
* - player: AVAudioPlayer 实例
* - successfully: 是否成功播放完成
* - Returns: 无
* - Throws: 无
*/
public func audioPlayerDidFinishPlaying(_ player: AVAudioPlayer, successfully flag: Bool) {
// 移除队首
if !playbackQueue.isEmpty {
playbackQueue.removeFirst()
}
// 清理临时文件
if let url = currentTempFileURL {
try? FileManager.default.removeItem(at: url)
currentTempFileURL = nil
}
isPlaying = false
os_log("调用链: 播放完成 成功=%{public}@ session=%{public}@", log: log, type: .info, flag.description, sessionid)
notifyEvent(eventType: .playbackCompleted)
// 如果还有剩余,继续播放
startPlaybackIfNeeded()
deactivateAudioSessionAfterTTSIfIdle()
}
/**
* 通知事件
@ -587,6 +1062,8 @@ public class AzureTtsHelper: NSObject, ITtsService {
"""
synthesisQueue.async {
self.suppressStreamPlayback = true
defer { self.suppressStreamPlayback = false }
_ = try? self.synthesizer?.startSpeakingSsml(warmupSsml)
}
}
@ -688,12 +1165,97 @@ public func resetTextCount() {
* 音频会话并发防护和状态
*/
private let audioSessionQueue = DispatchQueue(label: "com.azure.tts.audioSession")
private let audioSessionQueueKey = DispatchSpecificKey<UInt8>()
private var audioSessionQueueKeySet = false
private var audioInterrupted = false
private var audioSessionObserversAdded = false
// 新增:路由变化的去抖与配置重入保护标记
private var routeChangeDebounceWorkItem: DispatchWorkItem?
private var isApplyingAudioSessionConfig = false
/**
* 确保音频会话队列已设置 specific key(用于避免同队列 sync 死锁)
* - Parameters: 无
* - Returns: 无
* - Throws: 无
*/
private func ensureAudioSessionQueueSpecificKeySet() {
if audioSessionQueueKeySet { return }
audioSessionQueueKeySet = true
audioSessionQueue.setSpecific(key: audioSessionQueueKey, value: 1)
}
/**
* 同步应用 TTS 音频会话配置(用于需要立即生效的播放/引擎初始化路径)
* - Parameters: 无
* - Returns: 是否成功激活音频会话
* - Throws: 无(内部捕获 AVAudioSession 相关异常并记录日志)
*/
private func applyAudioSessionForTTS() -> Bool {
var success = true
ensureAudioSessionQueueSpecificKeySet()
let applyBody = {
if self.isApplyingAudioSessionConfig { return }
self.isApplyingAudioSessionConfig = true
defer { self.isApplyingAudioSessionConfig = false }
let audioSession = AVAudioSession.sharedInstance()
let outputs: [AVAudioSessionPortDescription] = audioSession.currentRoute.outputs
let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP })
let otherAudioPlaying = audioSession.isOtherAudioPlaying
let desiredCategory: AVAudioSession.Category
let desiredMode: AVAudioSession.Mode
var desiredOptions: AVAudioSession.CategoryOptions = []
if self.isTelephonyActive() || hasBluetoothHFP {
desiredCategory = .playback
desiredMode = .default
desiredOptions.insert(.duckOthers)
} else {
desiredCategory = .playback
desiredMode = .spokenAudio
if otherAudioPlaying {
desiredOptions.insert(.duckOthers)
desiredOptions.insert(.interruptSpokenAudioAndMixWithOthers)
}
}
do {
for attempt in 0..<3 {
do {
if audioSession.category != desiredCategory || audioSession.mode != desiredMode || audioSession.categoryOptions != desiredOptions {
try audioSession.setCategory(desiredCategory, mode: desiredMode, options: desiredOptions)
}
try audioSession.setActive(true)
break
} catch {
if attempt == 2 {
throw error
}
Thread.sleep(forTimeInterval: 0.05)
}
}
} catch {
do {
try audioSession.setCategory(.playback, mode: .default, options: [])
try audioSession.setActive(true)
} catch {
success = false
os_log("TTS 音频会话配置失败: %{public}@", log: self.log, type: .error, error.localizedDescription)
}
}
}
if DispatchQueue.getSpecific(key: audioSessionQueueKey) != nil {
applyBody()
} else {
audioSessionQueue.sync(execute: applyBody)
}
return success
}
/**
* 监听并处理音频中断与路由变化(通话/Siri/蓝牙切换等)
* - 目的:在通话期间或路由变化时安全地停止/重配,避免 AVAudioSession 断言导致崩溃
@ -729,25 +1291,26 @@ private func isTelephonyActive() -> Bool {
}
/**
* 安全配置音频会话用于 TTS,兼容通话场景
* - 通话中:使用 .playback + .duckOthers,避免 .voiceChat/.allowBluetoothA2DP 导致断言
* - 非通话:使用 .playAndRecord + .voiceChat + .allowBluetooth(不使用 A2DP)
* 安全配置音频会话用于 TTS
* - 播放优先走媒体声道(A2DP),避免切换到通话声道(HFP)
* - 外部音乐播放时:不启用 mix,默认会打断外部音频;同时可 duck 确保可听见
* - 录音进行中:保持 playAndRecord + A2DP 输出 + 内置麦输入,避免切到 HFP
* - 通话中:保持保守策略,避免触发系统断言或破坏通话链路
* - Parameters: 无
* - Returns: 无
* - Throws: 无(内部捕获 AVAudioSession 相关异常并记录日志)
*/
/// 为 TTS 安全配置音频会话:
/// - 使用异步队列,避免在同一串行队列上 sync 导致死锁
/// - 根据当前音频路由(蓝牙 A2DP / 蓝牙 HFP / 有线耳机)选择更合适的类别与模式
/// - 优先使用 .playback + .spokenAudio + .allowBluetoothA2DP(媒体声道),避免进入 HFP
/// - 外部音频播放时启用 spoken audio 策略,确保能在 A2DP 上播放
/// - 仅当类别或模式发生变化时才 setCategory,降低 AVAudioSessionRouteChangeReason.categoryChange 的触发概率
private func safeConfigureAudioSessionForTTS() {
private func safeConfigureAudioSessionForTTS() {
// 使用异步,避免串行队列重入自锁
audioSessionQueue.async { [weak self] in
guard let self = self else { return }
// 防重入:配置中不重复进入
if self.isApplyingAudioSessionConfig { return }
self.isApplyingAudioSessionConfig = true
defer { self.isApplyingAudioSessionConfig = false }
let audioSession = AVAudioSession.sharedInstance()
// 显式指定 outputs 的类型,确保闭包参数类型可推断
@ -757,44 +1320,49 @@ private func isTelephonyActive() -> Bool {
// 保持使用 audioSessionQueue.async,防止在串行队列上自锁。
// 仅在类别或模式发生变化时调用 setCategory,减少 categoryChange 的回调风暴。
let hasBluetoothHFP = outputs.contains(where: { $0.portType == .bluetoothHFP })
let hasBluetoothA2DP = outputs.contains(where: { $0.portType == .bluetoothA2DP })
let hasWiredHeadphones = outputs.contains(where: { $0.portType == .headphones || $0.portType == .headsetMic })
// 计算目标配置
let desiredCategory: AVAudioSession.Category
let desiredMode: AVAudioSession.Mode
var desiredOptions: AVAudioSession.CategoryOptions = [.mixWithOthers]
if self.isTelephonyActive() || hasBluetoothHFP {
// 通话活跃或存在 HFP(语音)链路:使用语音聊天模式
desiredCategory = .playAndRecord
desiredMode = .voiceChat
desiredOptions.insert(.allowBluetooth) // HFP 需 allowBluetooth
} else if hasBluetoothA2DP || hasWiredHeadphones {
// 纯播放链路耳机(A2DP/有线耳机):避免强制切语音链路
desiredCategory = .playback
desiredMode = .default
desiredOptions.insert(.allowBluetoothA2DP)
} else {
// 默认:本机扬声器的语音链路
desiredCategory = .playAndRecord
desiredMode = .voiceChat
desiredOptions.insert(.defaultToSpeaker)
desiredOptions.insert(.allowBluetooth) // 允许外接麦/耳机(非 HFP 也不冲突)
}
_ = hasBluetoothHFP
_ = audioSession
do {
// 仅当类别或模式变化时设置,减少无谓的 categoryChange
if audioSession.category != desiredCategory || audioSession.mode != desiredMode {
try audioSession.setCategory(desiredCategory, mode: desiredMode, options: desiredOptions)
}
// 激活会话(通常不会引发死锁),失败时仅记录错误
try audioSession.setActive(true)
} catch {
os_log("TTS 音频会话配置失败: %{public}@", log: self.log, type: .error, error.localizedDescription)
}
_ = self.applyAudioSessionForTTS()
}
}
/**
* 在 TTS 完全空闲后释放音频会话(用于恢复外部音乐)
* - Parameters: 无
* - Returns: 无
* - Throws: 无(内部捕获 AVAudioSession 异常并忽略)
*/
private func deactivateAudioSessionAfterTTSIfIdle() {
ensureAudioSessionQueueSpecificKeySet()
let work = { [weak self] in
guard let self = self else { return }
if self.speaking { return }
if self.isPlaying { return }
if !self.playbackQueue.isEmpty { return }
if self.pendingTextCount > 0 { return }
if !self.pendingTasks.isEmpty { return }
if self.activeStreamSynthesisCount > 0 { return }
if !self.pcmPendingData.isEmpty { return }
if self.scheduledBufferCount > 0 { return }
self.audioPlaybackQueue.async { [weak self] in
guard let self = self else { return }
if self.playerNode?.isPlaying == true { return }
self.playerNode?.stop()
self.audioEngine?.stop()
self.audioEngine = nil
self.playerNode = nil
self.pcmFormat = nil
}
let session = AVAudioSession.sharedInstance()
_ = try? session.setActive(false, options: .notifyOthersOnDeactivation)
}
if DispatchQueue.getSpecific(key: audioSessionQueueKey) != nil {
work()
} else {
audioSessionQueue.async(execute: work)
}
}
/**

64
local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift

@ -94,11 +94,6 @@ public class MicrophoneCapture: NSObject {
do {
try audioSession.setPreferredSampleRate(sampleRate)
try audioSession.setPreferredIOBufferDuration(0.005) // 5ms缓冲
// 优先设置蓝牙音频路由
configureAudioRoute()
try audioSession.setActive(true)
print("音频会话配置成功")
} catch {
print("音频会话配置失败: \(error.localizedDescription)")
@ -153,11 +148,10 @@ public class MicrophoneCapture: NSObject {
// 取消扬声器强制输出,让音频通过蓝牙耳机输出
do {
// 设置音频会话参数
try audioSession.setCategory(.playAndRecord,
try audioSession.setCategory(.playback,
mode: .videoChat,
options: [.allowBluetoothA2DP, // 允许蓝牙耳机,不占用hfp链路
.mixWithOthers,
.allowBluetooth]) // 添加音频优先级控制
options: [.allowBluetoothA2DP, // 仅允许A2DP,不启用HFP
.mixWithOthers])
try audioSession.overrideOutputAudioPort(.none)
print("蓝牙模式:音频输出设置为蓝牙耳机")
@ -214,6 +208,18 @@ public class MicrophoneCapture: NSObject {
// 检查麦克风权限
switch audioSession.recordPermission {
case .granted:
do {
try audioSession.setCategory(.record,
mode: .default,
options: [])
if let inputs = audioSession.availableInputs,
let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try? audioSession.setPreferredInput(builtInMic)
}
try audioSession.setActive(true)
} catch {
throw error
}
try setupAudioEngine()
case .denied:
throw NSError(domain: "麦克风权限被拒绝", code: 0)
@ -221,6 +227,19 @@ public class MicrophoneCapture: NSObject {
audioSession.requestRecordPermission { granted in
if granted {
do {
do {
try audioSession.setCategory(.record,
mode: .default,
options: [])
if let inputs = audioSession.availableInputs,
let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try? audioSession.setPreferredInput(builtInMic)
}
try audioSession.setActive(true)
} catch {
print("音频会话配置失败: \(error.localizedDescription)")
return
}
try self.setupAudioEngine()
} catch {
print("音频引擎设置失败: \(error.localizedDescription)")
@ -444,29 +463,8 @@ public class MicrophoneCapture: NSObject {
// 安全地处理音频会话
guard let audioSession = audioSession else { return }
do {
try audioSession.setActive(false)
if self.hasBluetoothDevices {
print("耳机")
// 设置音频会话参数
try audioSession.setCategory(.playback,
mode: .videoChat,
options: [
.allowBluetoothA2DP,
.allowBluetooth
]) // 添加音频优先级控制
} else {
print("手机")
try audioSession.setCategory(.playback,
mode: .videoChat,
options: [
.mixWithOthers,
.defaultToSpeaker])
}
try audioSession.overrideOutputAudioPort(.none)
try audioSession.setActive(true)
do {
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
} catch {
print("音频会话配置失败: \(error.localizedDescription)")
}
@ -487,4 +485,4 @@ public class MicrophoneCapture: NSObject {
}
}
}

Loading…
Cancel
Save