Browse Source

feat(语音处理): 实现电话模式音频路由优化

添加电话模式支持,优化音频路由和麦克风选择逻辑
- 在AgentServiceImpl中新增电话模式设置
- AzureAsrHelper和AzureTtsHelper添加电话模式状态管理
- SimpleAudioReceiver支持优先使用内置麦克风
- 改进音频会话配置和错误处理
- 增强麦克风权限请求和格式转换稳定性
weicu
liwei1dao 9 months ago
parent
commit
337a6f659b
  1. 11
      local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift
  2. 21
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift
  3. 30
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift
  4. 185
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift
  5. 46
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

11
local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift

@ -402,7 +402,8 @@ class AgentServiceImpl: NSObject {
// 根据模式决定空闲检测策略
switch mode {
case "phone_call":
azureAsrHelper?.restoreOriginalAudioState()
_ = azureAsrHelper?.setPhoneCallMode(enabled: true)
_ = azureTtsHelper?.setPhoneCallMode(enabled: true)
// 通话模式:禁用空闲检测,保持持续激活
os_log("通话模式:禁用空闲检测", log: logger, type: .info)
return
@ -523,6 +524,9 @@ class AgentServiceImpl: NSObject {
// 设置当前识别模式
currentRecognitionMode = mode
os_log("设置语音识别模式: %{public}@", log: logger, type: .info, mode)
_ = azureAsrHelper?.setPhoneCallMode(enabled: mode == "phone_call")
_ = azureTtsHelper?.setPhoneCallMode(enabled: mode == "phone_call")
_ = azureAsrHelper?.setPreferBuiltInMic(enabled: mode == "phone_call" || mode == "push_to_talk")
let audioSourceType: AzureAsrHelper.AudioSourceType = useBle ? .external : .microphone
@ -579,6 +583,11 @@ class AgentServiceImpl: NSObject {
os_log("停止语音识别,当前模式: %{public}@", log: logger, type: .info, currentRecognitionMode)
let success = azureAsrHelper?.stopContinuousRecognition() ?? false
if currentRecognitionMode == "phone_call" {
_ = azureAsrHelper?.setPhoneCallMode(enabled: false)
_ = azureTtsHelper?.setPhoneCallMode(enabled: false)
}
_ = azureAsrHelper?.setPreferBuiltInMic(enabled: false)
if !success && isStartingRecognition {
// 启动尚未完成,先记录一次待停止请求,onSessionStarted 到来后立即 stop

21
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

@ -73,6 +73,7 @@ public class AzureAsrHelper: NSObject {
}
public var audioSourceType = AudioSourceType.microphone
private var isPhoneCallMode = false
// 音频处理
// private var externalAudioStream: ExternalAudioPullStream?
@ -80,6 +81,23 @@ public class AzureAsrHelper: NSObject {
// 音频处理
public var audioStream: SimpleAudioReceiver?
/**
* 设置 AI 电话/按住说话场景下是否优先使用手机内置麦克风
* @param enabled 是否启用
* @return 是否设置成功
* @throws 无
*/
public func setPreferBuiltInMic(enabled: Bool) -> Bool {
audioStream?.setPreferBuiltInMic(enabled)
return true
}
public func setPhoneCallMode(enabled: Bool) -> Bool {
isPhoneCallMode = enabled
audioStream?.setPhoneCallMode(enabled)
return true
}
public func asrProvider() -> String {
return useXunfei ? "xunfei" : "azure"
}
@ -237,6 +255,7 @@ public class AzureAsrHelper: NSObject {
if audioStream == nil {
audioStream = SimpleAudioReceiver()
audioStream?.initAudioRecord()
audioStream?.setPhoneCallMode(isPhoneCallMode)
}
// 预创建音频配置
if let pushStream = audioStream?.pushAudioStream {
@ -768,6 +787,7 @@ public class AzureAsrHelper: NSObject {
if audioStream == nil {
audioStream = SimpleAudioReceiver()
audioStream?.initAudioRecord() // 确保调用初始化
audioStream?.setPhoneCallMode(isPhoneCallMode)
}
// 检查音频配置是否已存在
@ -1208,4 +1228,3 @@ private class AudioDataCallbackProxy: SimpleAudioReceiver.AudioDataCallback {

30
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift

@ -78,6 +78,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate {
private var pushStreamCallbackCount = 0
private var synthEventAudioCount = 0
private var isExtaudioSource = false
private var isPhoneCallMode = false
@ -440,6 +441,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate {
return true
}
public func setPhoneCallMode(enabled: Bool) -> Bool {
isPhoneCallMode = enabled
if isInitialized {
safeConfigureAudioSessionForTTS()
}
return true
}
/**
* 设置音频输出设备
*/
@ -1361,15 +1370,20 @@ private func applyAudioSessionForTTS() -> Bool {
let desiredMode: AVAudioSession.Mode
var desiredOptions: AVAudioSession.CategoryOptions = []
if self.isTelephonyActive() || hasBluetoothHFP {
if self.isTelephonyActive() {
desiredCategory = .playback
desiredMode = .default
desiredOptions.insert(.duckOthers)
} else if isRecording {
desiredCategory = .playAndRecord
desiredMode = self.isExtaudioSource ? .default : .videoChat
desiredOptions.insert(.allowBluetooth)
desiredOptions.insert(.allowBluetoothA2DP)
if self.isPhoneCallMode {
desiredMode = .voiceChat
desiredOptions.insert(.allowBluetooth)
} else {
desiredMode = self.isExtaudioSource ? .default : .videoChat
desiredOptions.insert(.allowBluetooth)
desiredOptions.insert(.allowBluetoothA2DP)
}
if self.isExtaudioSource {
if otherAudioPlaying {
desiredOptions.insert(.duckOthers)
@ -1380,6 +1394,10 @@ private func applyAudioSessionForTTS() -> Bool {
desiredOptions.insert(.defaultToSpeaker)
}
}
} else if hasBluetoothHFP {
desiredCategory = .playback
desiredMode = .default
desiredOptions.insert(.duckOthers)
} else {
desiredCategory = .playback
desiredMode = .spokenAudio
@ -1399,6 +1417,9 @@ private func applyAudioSessionForTTS() -> Bool {
_ = try? audioSession.setPreferredIOBufferDuration(0.02)
}
try audioSession.setActive(true)
if self.isPhoneCallMode, desiredCategory == .playAndRecord, !hasHeadphones {
_ = try? audioSession.overrideOutputAudioPort(.speaker)
}
break
} catch {
if attempt == 2 {
@ -1511,6 +1532,7 @@ private func deactivateAudioSessionAfterTTSIfIdle() {
ensureAudioSessionQueueSpecificKeySet()
let work = { [weak self] in
guard let self = self else { return }
if self.micCapture?.isCapturing == true { return }
if self.speaking { return }
if self.isPlaying { return }
if !self.playbackQueue.isEmpty { return }

185
local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift

@ -206,49 +206,55 @@ public class MicrophoneCapture: NSObject {
}
// 检查麦克风权限
let permission: AVAudioSession.RecordPermission
switch audioSession.recordPermission {
case .granted:
do {
try audioSession.setCategory(.record,
mode: .default,
options: [])
if let inputs = audioSession.availableInputs,
let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try? audioSession.setPreferredInput(builtInMic)
}
try audioSession.setActive(true)
} catch {
throw error
}
try setupAudioEngine()
permission = .granted
case .denied:
throw NSError(domain: "麦克风权限被拒绝", code: 0)
permission = .denied
case .undetermined:
let semaphore = DispatchSemaphore(value: 0)
var grantedResult = false
audioSession.requestRecordPermission { granted in
if granted {
do {
do {
try audioSession.setCategory(.record,
mode: .default,
options: [])
if let inputs = audioSession.availableInputs,
let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try? audioSession.setPreferredInput(builtInMic)
}
try audioSession.setActive(true)
} catch {
print("音频会话配置失败: \(error.localizedDescription)")
return
}
try self.setupAudioEngine()
} catch {
print("音频引擎设置失败: \(error.localizedDescription)")
}
}
grantedResult = granted
semaphore.signal()
}
let waitResult = semaphore.wait(timeout: .now() + 60)
if waitResult == .timedOut {
throw NSError(domain: "麦克风权限请求超时", code: 2)
}
permission = grantedResult ? .granted : .denied
@unknown default:
throw NSError(domain: "未知权限状态", code: 1)
}
guard permission == .granted else {
throw NSError(domain: "麦克风权限被拒绝", code: 0)
}
do {
print("startCapture: 激活前 category=\(audioSession.category.rawValue) mode=\(audioSession.mode.rawValue) options=\(audioSession.categoryOptions)")
print("startCapture: 激活前 sampleRate=\(audioSession.sampleRate) ioBuffer=\(audioSession.ioBufferDuration)")
print("startCapture: 激活前 route inputs=\(audioSession.currentRoute.inputs.map { $0.portType.rawValue }) outputs=\(audioSession.currentRoute.outputs.map { $0.portType.rawValue })")
if let inputs = audioSession.availableInputs,
let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try? audioSession.setPreferredInput(builtInMic)
}
try audioSession.setActive(true)
print("startCapture: 激活后 category=\(audioSession.category.rawValue) mode=\(audioSession.mode.rawValue) options=\(audioSession.categoryOptions)")
print("startCapture: 激活后 sampleRate=\(audioSession.sampleRate) ioBuffer=\(audioSession.ioBufferDuration)")
print("startCapture: 激活后 route inputs=\(audioSession.currentRoute.inputs.map { $0.portType.rawValue }) outputs=\(audioSession.currentRoute.outputs.map { $0.portType.rawValue })")
} catch {
throw error
}
isCapturing = true
do {
try setupAudioEngine()
} catch {
isCapturing = false
cleanupAudioEngine()
throw error
}
}
/**
@ -299,36 +305,51 @@ public class MicrophoneCapture: NSObject {
if let audioSession = audioSession {
sampleRate = audioSession.sampleRate
}
// 设置音频格式
let inputFormat = audioInputNode.inputFormat(forBus: 0)
// iOS 13+ 启用语音处理
// iOS 13+ 启用语音处理(可能在部分路由/模式下失败,失败不应阻断录音启动)
if #available(iOS 13.0, *) {
try audioInputNode.setVoiceProcessingEnabled(true)
do {
try audioInputNode.setVoiceProcessingEnabled(true)
} catch {
if let nsError = error as NSError? {
print("启用语音处理失败: domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)")
} else {
print("启用语音处理失败: \(error.localizedDescription)")
}
}
}
// 检查输入格式是否符合Microsoft要求
let needsConversion = !isMicrosoftCompatibleFormat(inputFormat)
// 需要格式转换
guard let targetFormat = audioFormat,
let converter = AVAudioConverter(from: inputFormat, to: targetFormat) else {
// 等待并获取有效的输入格式:在路由/类别切换瞬间可能出现 0Hz,直接 installTap 会触发系统断言崩溃
guard let inputFormat = waitForValidInputFormat(inputNode: audioInputNode, maxAttempts: 10, retryIntervalSeconds: 0.03) else {
let currentSession = AVAudioSession.sharedInstance()
let inputs = currentSession.currentRoute.inputs.map { $0.portType.rawValue }.joined(separator: ",")
let outputs = currentSession.currentRoute.outputs.map { $0.portType.rawValue }.joined(separator: ",")
throw NSError(domain: "AudioSetup", code: 3, userInfo: [
NSLocalizedDescriptionKey: "输入音频格式无效(sampleRate/channelCount),可能处于路由或类别切换中",
"sessionCategory": currentSession.category.rawValue,
"sessionMode": currentSession.mode.rawValue,
"sessionSampleRate": currentSession.sampleRate,
"routeInputs": inputs,
"routeOutputs": outputs
])
}
// 需要格式转换(输出固定 16k/16bit/mono PCM,匹配 PushStream 默认/常用配置)
guard let targetFormat = audioFormat else {
throw NSError(domain: "AudioSetup", code: 2, userInfo: [NSLocalizedDescriptionKey: "目标音频格式为空"])
}
guard let converter = AVAudioConverter(from: inputFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2, userInfo: [NSLocalizedDescriptionKey: "音频格式转换器创建失败"])
}
print("输入格式不符合Microsoft要求,进行格式转换")
print("输入格式: \(inputFormat.sampleRate)Hz, \(inputFormat.commonFormat.rawValue)")
print("目标格式: \(targetFormat.sampleRate)Hz, \(targetFormat.commonFormat.rawValue)")
// 添加tap进行格式转换
print("输入格式: \(inputFormat.sampleRate)Hz, \(inputFormat.commonFormat.rawValue), ch=\(inputFormat.channelCount), interleaved=\(inputFormat.isInterleaved)")
print("目标格式: \(targetFormat.sampleRate)Hz, \(targetFormat.commonFormat.rawValue), ch=\(targetFormat.channelCount), interleaved=\(targetFormat.isInterleaved)")
audioInputNode.installTap(onBus: 0,
bufferSize: 1024,
format: inputFormat) { [weak self] buffer, when in
bufferSize: 1024,
format: inputFormat) { [weak self] buffer, _ in
guard let strongSelf = self else { return }
// 修复:添加安全检查,避免强制解包崩溃
guard let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
frameCapacity: AVAudioFrameCount(
@ -338,36 +359,59 @@ public class MicrophoneCapture: NSObject {
print("创建转换缓冲区失败")
return
}
var error: NSError?
// 执行音频格式转换
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
withInputFrom: { _, outStatus in
outStatus.pointee = .haveData
return buffer
}
)
// 转换成功且无错误
if status == .haveData, error == nil {
let data = strongSelf.audioBufferToData(convertedBuffer)
strongSelf.audioDataHandler?(data)
} else if let error = error {
print("音频格式转换失败: \(error.localizedDescription)")
print("音频格式转换失败: domain=\(error.domain) code=\(error.code) desc=\(error.localizedDescription)")
}
}
audioEngine.prepare()
// 启动引擎
do {
try audioEngine.start()
isCapturing = true
} catch {
print("音频引擎启动失败: \(error.localizedDescription)")
if let nsError = error as NSError? {
print("音频引擎启动失败: domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)")
} else {
print("音频引擎启动失败: \(error.localizedDescription)")
}
throw error
}
}
/**
* 等待并获取有效的输入音频格式
* - 参数:
* - inputNode: 输入节点(通常为 AVAudioEngine.inputNode)
* - maxAttempts: 最大重试次数
* - retryIntervalSeconds: 每次重试间隔(秒)
* - 返回值: 若在重试窗口内拿到有效格式则返回 AVAudioFormat,否则返回 nil
* - 异常: 无(内部不抛出异常)
*/
private func waitForValidInputFormat(inputNode: AVAudioInputNode, maxAttempts: Int, retryIntervalSeconds: TimeInterval) -> AVAudioFormat? {
for attempt in 1...maxAttempts {
let format = inputNode.inputFormat(forBus: 0)
if format.sampleRate > 0, format.channelCount > 0 {
return format
}
print("输入格式无效,等待重试 attempt=\(attempt) sampleRate=\(format.sampleRate) ch=\(format.channelCount)")
Thread.sleep(forTimeInterval: retryIntervalSeconds)
}
return nil
}
/**
* 安全清理音频引擎资源
@ -401,6 +445,15 @@ public class MicrophoneCapture: NSObject {
count: dataLength
)
}
// Handle 32-bit integer format
else if let int32Data = buffer.int32ChannelData {
let int32Buffer = int32Data.pointee
var int16Array = [Int16](repeating: 0, count: frameLength)
for i in 0..<frameLength {
int16Array[i] = Int16(max(Int32(Int16.min), min(Int32(Int16.max), int32Buffer[i] >> 16)))
}
return Data(bytes: int16Array, count: dataLength)
}
// Handle float format
else if let floatData = buffer.floatChannelData {
var int16Array = [Int16](repeating: 0, count: frameLength)

46
local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

@ -56,6 +56,8 @@ public class SimpleAudioReceiver: NSObject {
public var recordfile: RecordFile?
public var isHeadphones = true
public var isReceiverHFP = false // 是否是接收HFP音频
public var isPhoneCallMode = false // AI电话模式:录音+播放强制走电话链路(voiceChat/HFP)
public var preferBuiltInMic = false // 耳机连接时优先使用手机内置麦克风录音
/**
* 初始化
@ -134,8 +136,7 @@ public class SimpleAudioReceiver: NSObject {
print("音源设制---------- startAudioRecord=audioSourceType:\(audioSourceType)")
switch audioSourceType {
case .microphone:
// 唤醒后禁用通话音道(HFP),改为手机麦克风 + A2DP 输出
isReceiverHFP = false
isReceiverHFP = isPhoneCallMode
// 动态检测是否存在耳机/蓝牙输出,用于后续路由选项计算
isHeadphones = detectHeadphonesOutput()
runMicrophoneCapture()
@ -144,6 +145,20 @@ public class SimpleAudioReceiver: NSObject {
runExternalCapture()
}
}
public func setPhoneCallMode(_ enabled: Bool) {
isPhoneCallMode = enabled
}
/**
* 设置是否优先使用手机内置麦克风录音
* @param enabled 是否启用
* @return 无
* @throws 无
*/
public func setPreferBuiltInMic(_ enabled: Bool) {
preferBuiltInMic = enabled
}
/**
* 检测是否存在蓝牙HFP输入设备
@ -327,7 +342,15 @@ public class SimpleAudioReceiver: NSObject {
let desiredCategory: AVAudioSession.Category
let desiredMode: AVAudioSession.Mode
let desiredOptions: AVAudioSession.CategoryOptions
if isReceiverHFP {
if isPhoneCallMode {
desiredCategory = .playAndRecord
desiredMode = .voiceChat
if isHeadphones {
desiredOptions = [.allowBluetooth]
} else {
desiredOptions = [.allowBluetooth, .defaultToSpeaker]
}
} else if isReceiverHFP {
if isTelephonyActive() {
// 通话中:录音优先,避免 A2DP(仅播放),改用 HFP(.allowBluetooth)
desiredCategory = .record
@ -357,7 +380,7 @@ public class SimpleAudioReceiver: NSObject {
// 非通话:保持原有逻辑,但使用 .allowBluetooth 以支持录音(而不是 A2DP)
desiredCategory = .playAndRecord
desiredMode = .videoChat
desiredOptions = [.allowBluetoothA2DP, .mixWithOthers]
desiredOptions = [.allowBluetooth, .allowBluetoothA2DP, .mixWithOthers]
//desiredOptions = [.mixWithOthers, .defaultToSpeaker]
}else{
desiredCategory = .playAndRecord
@ -390,9 +413,13 @@ public class SimpleAudioReceiver: NSObject {
}
if hasBtOutput {
try audioSession.overrideOutputAudioPort(.none)
} else if isPhoneCallMode, !isHeadphones {
try audioSession.overrideOutputAudioPort(.speaker)
}
if let inputs = audioSession.availableInputs {
if isReceiverHFP, let btInput = inputs.first(where: { $0.portType == .bluetoothHFP }) {
if preferBuiltInMic, let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try audioSession.setPreferredInput(builtInMic)
} else if isReceiverHFP, let btInput = inputs.first(where: { $0.portType == .bluetoothHFP }) {
try audioSession.setPreferredInput(btInput)
} else if let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) {
try audioSession.setPreferredInput(builtInMic)
@ -429,9 +456,14 @@ public class SimpleAudioReceiver: NSObject {
// 启动录音引擎
do {
print("resumeRecord")
try micCapture.startCapture()
try safeConfigureAudioSessionForMic()
try micCapture.startCapture()
} catch {
os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription)
if let nsError = error as NSError? {
os_log("Failed to start audio engine: domain=%{public}@ code=%{public}d desc=%{public}@", log: log, type: .error, nsError.domain, nsError.code, nsError.localizedDescription)
} else {
os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription)
}
}
}

Loading…
Cancel
Save