From 337a6f659bbab8d5a93f7d2c5fc9211c86d483d1 Mon Sep 17 00:00:00 2001 From: liwei1dao Date: Mon, 19 Jan 2026 21:25:13 +0800 Subject: [PATCH] =?UTF-8?q?feat(=E8=AF=AD=E9=9F=B3=E5=A4=84=E7=90=86):=20?= =?UTF-8?q?=E5=AE=9E=E7=8E=B0=E7=94=B5=E8=AF=9D=E6=A8=A1=E5=BC=8F=E9=9F=B3?= =?UTF-8?q?=E9=A2=91=E8=B7=AF=E7=94=B1=E4=BC=98=E5=8C=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 添加电话模式支持,优化音频路由和麦克风选择逻辑 - 在AgentServiceImpl中新增电话模式设置 - AzureAsrHelper和AzureTtsHelper添加电话模式状态管理 - SimpleAudioReceiver支持优先使用内置麦克风 - 改进音频会话配置和错误处理 - 增强麦克风权限请求和格式转换稳定性 --- .../agent_service/AgentServiceImpl.swift | 11 +- .../Sources/azure_speech/AzureAsrHelper.swift | 21 +- .../Sources/azure_speech/AzureTtsHelper.swift | 30 ++- .../Sources/tools/MicrophoneCapture.swift | 185 +++++++++++------- .../Sources/tools/SimpleAudioReceiver.swift | 46 ++++- 5 files changed, 214 insertions(+), 79 deletions(-) diff --git a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift index 08f7f4b7f..8914a89bb 100644 --- a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift +++ b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift @@ -402,7 +402,8 @@ class AgentServiceImpl: NSObject { // 根据模式决定空闲检测策略 switch mode { case "phone_call": - azureAsrHelper?.restoreOriginalAudioState() + _ = azureAsrHelper?.setPhoneCallMode(enabled: true) + _ = azureTtsHelper?.setPhoneCallMode(enabled: true) // 通话模式:禁用空闲检测,保持持续激活 os_log("通话模式:禁用空闲检测", log: logger, type: .info) return @@ -523,6 +524,9 @@ class AgentServiceImpl: NSObject { // 设置当前识别模式 currentRecognitionMode = mode os_log("设置语音识别模式: %{public}@", log: logger, type: .info, mode) + _ = azureAsrHelper?.setPhoneCallMode(enabled: mode == "phone_call") + _ = azureTtsHelper?.setPhoneCallMode(enabled: mode == "phone_call") + _ = azureAsrHelper?.setPreferBuiltInMic(enabled: mode == "phone_call" || mode == "push_to_talk") let audioSourceType: AzureAsrHelper.AudioSourceType = useBle ? .external : .microphone @@ -579,6 +583,11 @@ class AgentServiceImpl: NSObject { os_log("停止语音识别,当前模式: %{public}@", log: logger, type: .info, currentRecognitionMode) let success = azureAsrHelper?.stopContinuousRecognition() ?? false + if currentRecognitionMode == "phone_call" { + _ = azureAsrHelper?.setPhoneCallMode(enabled: false) + _ = azureTtsHelper?.setPhoneCallMode(enabled: false) + } + _ = azureAsrHelper?.setPreferBuiltInMic(enabled: false) if !success && isStartingRecognition { // 启动尚未完成,先记录一次待停止请求,onSessionStarted 到来后立即 stop diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index 3ea1fb6a0..a69b0c385 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -73,6 +73,7 @@ public class AzureAsrHelper: NSObject { } public var audioSourceType = AudioSourceType.microphone + private var isPhoneCallMode = false // 音频处理 // private var externalAudioStream: ExternalAudioPullStream? @@ -80,6 +81,23 @@ public class AzureAsrHelper: NSObject { // 音频处理 public var audioStream: SimpleAudioReceiver? + /** + * 设置 AI 电话/按住说话场景下是否优先使用手机内置麦克风 + * @param enabled 是否启用 + * @return 是否设置成功 + * @throws 无 + */ + public func setPreferBuiltInMic(enabled: Bool) -> Bool { + audioStream?.setPreferBuiltInMic(enabled) + return true + } + + public func setPhoneCallMode(enabled: Bool) -> Bool { + isPhoneCallMode = enabled + audioStream?.setPhoneCallMode(enabled) + return true + } + public func asrProvider() -> String { return useXunfei ? "xunfei" : "azure" } @@ -237,6 +255,7 @@ public class AzureAsrHelper: NSObject { if audioStream == nil { audioStream = SimpleAudioReceiver() audioStream?.initAudioRecord() + audioStream?.setPhoneCallMode(isPhoneCallMode) } // 预创建音频配置 if let pushStream = audioStream?.pushAudioStream { @@ -768,6 +787,7 @@ public class AzureAsrHelper: NSObject { if audioStream == nil { audioStream = SimpleAudioReceiver() audioStream?.initAudioRecord() // 确保调用初始化 + audioStream?.setPhoneCallMode(isPhoneCallMode) } // 检查音频配置是否已存在 @@ -1208,4 +1228,3 @@ private class AudioDataCallbackProxy: SimpleAudioReceiver.AudioDataCallback { - diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift index b59031000..aee7d4930 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift @@ -78,6 +78,7 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { private var pushStreamCallbackCount = 0 private var synthEventAudioCount = 0 private var isExtaudioSource = false + private var isPhoneCallMode = false @@ -440,6 +441,14 @@ public class AzureTtsHelper: NSObject, ITtsService, AVAudioPlayerDelegate { return true } + public func setPhoneCallMode(enabled: Bool) -> Bool { + isPhoneCallMode = enabled + if isInitialized { + safeConfigureAudioSessionForTTS() + } + return true + } + /** * 设置音频输出设备 */ @@ -1361,15 +1370,20 @@ private func applyAudioSessionForTTS() -> Bool { let desiredMode: AVAudioSession.Mode var desiredOptions: AVAudioSession.CategoryOptions = [] - if self.isTelephonyActive() || hasBluetoothHFP { + if self.isTelephonyActive() { desiredCategory = .playback desiredMode = .default desiredOptions.insert(.duckOthers) } else if isRecording { desiredCategory = .playAndRecord - desiredMode = self.isExtaudioSource ? .default : .videoChat - desiredOptions.insert(.allowBluetooth) - desiredOptions.insert(.allowBluetoothA2DP) + if self.isPhoneCallMode { + desiredMode = .voiceChat + desiredOptions.insert(.allowBluetooth) + } else { + desiredMode = self.isExtaudioSource ? .default : .videoChat + desiredOptions.insert(.allowBluetooth) + desiredOptions.insert(.allowBluetoothA2DP) + } if self.isExtaudioSource { if otherAudioPlaying { desiredOptions.insert(.duckOthers) @@ -1380,6 +1394,10 @@ private func applyAudioSessionForTTS() -> Bool { desiredOptions.insert(.defaultToSpeaker) } } + } else if hasBluetoothHFP { + desiredCategory = .playback + desiredMode = .default + desiredOptions.insert(.duckOthers) } else { desiredCategory = .playback desiredMode = .spokenAudio @@ -1399,6 +1417,9 @@ private func applyAudioSessionForTTS() -> Bool { _ = try? audioSession.setPreferredIOBufferDuration(0.02) } try audioSession.setActive(true) + if self.isPhoneCallMode, desiredCategory == .playAndRecord, !hasHeadphones { + _ = try? audioSession.overrideOutputAudioPort(.speaker) + } break } catch { if attempt == 2 { @@ -1511,6 +1532,7 @@ private func deactivateAudioSessionAfterTTSIfIdle() { ensureAudioSessionQueueSpecificKeySet() let work = { [weak self] in guard let self = self else { return } + if self.micCapture?.isCapturing == true { return } if self.speaking { return } if self.isPlaying { return } if !self.playbackQueue.isEmpty { return } diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift index df55d1727..b379819de 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/MicrophoneCapture.swift @@ -206,49 +206,55 @@ public class MicrophoneCapture: NSObject { } // 检查麦克风权限 + let permission: AVAudioSession.RecordPermission switch audioSession.recordPermission { case .granted: - do { - try audioSession.setCategory(.record, - mode: .default, - options: []) - if let inputs = audioSession.availableInputs, - let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { - try? audioSession.setPreferredInput(builtInMic) - } - try audioSession.setActive(true) - } catch { - throw error - } - try setupAudioEngine() + permission = .granted case .denied: - throw NSError(domain: "麦克风权限被拒绝", code: 0) + permission = .denied case .undetermined: + let semaphore = DispatchSemaphore(value: 0) + var grantedResult = false audioSession.requestRecordPermission { granted in - if granted { - do { - do { - try audioSession.setCategory(.record, - mode: .default, - options: []) - if let inputs = audioSession.availableInputs, - let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { - try? audioSession.setPreferredInput(builtInMic) - } - try audioSession.setActive(true) - } catch { - print("音频会话配置失败: \(error.localizedDescription)") - return - } - try self.setupAudioEngine() - } catch { - print("音频引擎设置失败: \(error.localizedDescription)") - } - } + grantedResult = granted + semaphore.signal() + } + let waitResult = semaphore.wait(timeout: .now() + 60) + if waitResult == .timedOut { + throw NSError(domain: "麦克风权限请求超时", code: 2) } + permission = grantedResult ? .granted : .denied @unknown default: throw NSError(domain: "未知权限状态", code: 1) } + + guard permission == .granted else { + throw NSError(domain: "麦克风权限被拒绝", code: 0) + } + + do { + print("startCapture: 激活前 category=\(audioSession.category.rawValue) mode=\(audioSession.mode.rawValue) options=\(audioSession.categoryOptions)") + print("startCapture: 激活前 sampleRate=\(audioSession.sampleRate) ioBuffer=\(audioSession.ioBufferDuration)") + print("startCapture: 激活前 route inputs=\(audioSession.currentRoute.inputs.map { $0.portType.rawValue }) outputs=\(audioSession.currentRoute.outputs.map { $0.portType.rawValue })") + if let inputs = audioSession.availableInputs, + let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { + try? audioSession.setPreferredInput(builtInMic) + } + try audioSession.setActive(true) + print("startCapture: 激活后 category=\(audioSession.category.rawValue) mode=\(audioSession.mode.rawValue) options=\(audioSession.categoryOptions)") + print("startCapture: 激活后 sampleRate=\(audioSession.sampleRate) ioBuffer=\(audioSession.ioBufferDuration)") + print("startCapture: 激活后 route inputs=\(audioSession.currentRoute.inputs.map { $0.portType.rawValue }) outputs=\(audioSession.currentRoute.outputs.map { $0.portType.rawValue })") + } catch { + throw error + } + isCapturing = true + do { + try setupAudioEngine() + } catch { + isCapturing = false + cleanupAudioEngine() + throw error + } } /** @@ -299,36 +305,51 @@ public class MicrophoneCapture: NSObject { if let audioSession = audioSession { sampleRate = audioSession.sampleRate } - - // 设置音频格式 - let inputFormat = audioInputNode.inputFormat(forBus: 0) - - // iOS 13+ 启用语音处理 + + // iOS 13+ 启用语音处理(可能在部分路由/模式下失败,失败不应阻断录音启动) if #available(iOS 13.0, *) { - try audioInputNode.setVoiceProcessingEnabled(true) + do { + try audioInputNode.setVoiceProcessingEnabled(true) + } catch { + if let nsError = error as NSError? { + print("启用语音处理失败: domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)") + } else { + print("启用语音处理失败: \(error.localizedDescription)") + } + } } - - // 检查输入格式是否符合Microsoft要求 - let needsConversion = !isMicrosoftCompatibleFormat(inputFormat) - - // 需要格式转换 - guard let targetFormat = audioFormat, - let converter = AVAudioConverter(from: inputFormat, to: targetFormat) else { + + // 等待并获取有效的输入格式:在路由/类别切换瞬间可能出现 0Hz,直接 installTap 会触发系统断言崩溃 + guard let inputFormat = waitForValidInputFormat(inputNode: audioInputNode, maxAttempts: 10, retryIntervalSeconds: 0.03) else { + let currentSession = AVAudioSession.sharedInstance() + let inputs = currentSession.currentRoute.inputs.map { $0.portType.rawValue }.joined(separator: ",") + let outputs = currentSession.currentRoute.outputs.map { $0.portType.rawValue }.joined(separator: ",") + throw NSError(domain: "AudioSetup", code: 3, userInfo: [ + NSLocalizedDescriptionKey: "输入音频格式无效(sampleRate/channelCount),可能处于路由或类别切换中", + "sessionCategory": currentSession.category.rawValue, + "sessionMode": currentSession.mode.rawValue, + "sessionSampleRate": currentSession.sampleRate, + "routeInputs": inputs, + "routeOutputs": outputs + ]) + } + + // 需要格式转换(输出固定 16k/16bit/mono PCM,匹配 PushStream 默认/常用配置) + guard let targetFormat = audioFormat else { + throw NSError(domain: "AudioSetup", code: 2, userInfo: [NSLocalizedDescriptionKey: "目标音频格式为空"]) + } + guard let converter = AVAudioConverter(from: inputFormat, to: targetFormat) else { throw NSError(domain: "AudioSetup", code: 2, userInfo: [NSLocalizedDescriptionKey: "音频格式转换器创建失败"]) } - - print("输入格式不符合Microsoft要求,进行格式转换") - print("输入格式: \(inputFormat.sampleRate)Hz, \(inputFormat.commonFormat.rawValue)") - print("目标格式: \(targetFormat.sampleRate)Hz, \(targetFormat.commonFormat.rawValue)") - - // 添加tap进行格式转换 + + print("输入格式: \(inputFormat.sampleRate)Hz, \(inputFormat.commonFormat.rawValue), ch=\(inputFormat.channelCount), interleaved=\(inputFormat.isInterleaved)") + print("目标格式: \(targetFormat.sampleRate)Hz, \(targetFormat.commonFormat.rawValue), ch=\(targetFormat.channelCount), interleaved=\(targetFormat.isInterleaved)") + audioInputNode.installTap(onBus: 0, - bufferSize: 1024, - format: inputFormat) { [weak self] buffer, when in - + bufferSize: 1024, + format: inputFormat) { [weak self] buffer, _ in guard let strongSelf = self else { return } - - // 修复:添加安全检查,避免强制解包崩溃 + guard let convertedBuffer = AVAudioPCMBuffer( pcmFormat: targetFormat, frameCapacity: AVAudioFrameCount( @@ -338,36 +359,59 @@ public class MicrophoneCapture: NSObject { print("创建转换缓冲区失败") return } - + var error: NSError? - // 执行音频格式转换 let status = converter.convert( to: convertedBuffer, error: &error, - withInputFrom: { inNumPackets, outStatus in + withInputFrom: { _, outStatus in outStatus.pointee = .haveData return buffer } ) - - // 转换成功且无错误 + if status == .haveData, error == nil { let data = strongSelf.audioBufferToData(convertedBuffer) strongSelf.audioDataHandler?(data) } else if let error = error { - print("音频格式转换失败: \(error.localizedDescription)") + print("音频格式转换失败: domain=\(error.domain) code=\(error.code) desc=\(error.localizedDescription)") } } + audioEngine.prepare() // 启动引擎 do { try audioEngine.start() - isCapturing = true } catch { - print("音频引擎启动失败: \(error.localizedDescription)") + if let nsError = error as NSError? { + print("音频引擎启动失败: domain=\(nsError.domain) code=\(nsError.code) desc=\(nsError.localizedDescription)") + } else { + print("音频引擎启动失败: \(error.localizedDescription)") + } throw error } } + + /** + * 等待并获取有效的输入音频格式 + * - 参数: + * - inputNode: 输入节点(通常为 AVAudioEngine.inputNode) + * - maxAttempts: 最大重试次数 + * - retryIntervalSeconds: 每次重试间隔(秒) + * - 返回值: 若在重试窗口内拿到有效格式则返回 AVAudioFormat,否则返回 nil + * - 异常: 无(内部不抛出异常) + */ + private func waitForValidInputFormat(inputNode: AVAudioInputNode, maxAttempts: Int, retryIntervalSeconds: TimeInterval) -> AVAudioFormat? { + for attempt in 1...maxAttempts { + let format = inputNode.inputFormat(forBus: 0) + if format.sampleRate > 0, format.channelCount > 0 { + return format + } + print("输入格式无效,等待重试 attempt=\(attempt) sampleRate=\(format.sampleRate) ch=\(format.channelCount)") + Thread.sleep(forTimeInterval: retryIntervalSeconds) + } + return nil + } /** * 安全清理音频引擎资源 @@ -401,6 +445,15 @@ public class MicrophoneCapture: NSObject { count: dataLength ) } + // Handle 32-bit integer format + else if let int32Data = buffer.int32ChannelData { + let int32Buffer = int32Data.pointee + var int16Array = [Int16](repeating: 0, count: frameLength) + for i in 0..> 16))) + } + return Data(bytes: int16Array, count: dataLength) + } // Handle float format else if let floatData = buffer.floatChannelData { var int16Array = [Int16](repeating: 0, count: frameLength) diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift index 0ae23d2e4..b5a0a8a5f 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift @@ -56,6 +56,8 @@ public class SimpleAudioReceiver: NSObject { public var recordfile: RecordFile? public var isHeadphones = true public var isReceiverHFP = false // 是否是接收HFP音频 + public var isPhoneCallMode = false // AI电话模式:录音+播放强制走电话链路(voiceChat/HFP) + public var preferBuiltInMic = false // 耳机连接时优先使用手机内置麦克风录音 /** * 初始化 @@ -134,8 +136,7 @@ public class SimpleAudioReceiver: NSObject { print("音源设制---------- startAudioRecord=audioSourceType:\(audioSourceType)") switch audioSourceType { case .microphone: - // 唤醒后禁用通话音道(HFP),改为手机麦克风 + A2DP 输出 - isReceiverHFP = false + isReceiverHFP = isPhoneCallMode // 动态检测是否存在耳机/蓝牙输出,用于后续路由选项计算 isHeadphones = detectHeadphonesOutput() runMicrophoneCapture() @@ -144,6 +145,20 @@ public class SimpleAudioReceiver: NSObject { runExternalCapture() } } + + public func setPhoneCallMode(_ enabled: Bool) { + isPhoneCallMode = enabled + } + + /** + * 设置是否优先使用手机内置麦克风录音 + * @param enabled 是否启用 + * @return 无 + * @throws 无 + */ + public func setPreferBuiltInMic(_ enabled: Bool) { + preferBuiltInMic = enabled + } /** * 检测是否存在蓝牙HFP输入设备 @@ -327,7 +342,15 @@ public class SimpleAudioReceiver: NSObject { let desiredCategory: AVAudioSession.Category let desiredMode: AVAudioSession.Mode let desiredOptions: AVAudioSession.CategoryOptions - if isReceiverHFP { + if isPhoneCallMode { + desiredCategory = .playAndRecord + desiredMode = .voiceChat + if isHeadphones { + desiredOptions = [.allowBluetooth] + } else { + desiredOptions = [.allowBluetooth, .defaultToSpeaker] + } + } else if isReceiverHFP { if isTelephonyActive() { // 通话中:录音优先,避免 A2DP(仅播放),改用 HFP(.allowBluetooth) desiredCategory = .record @@ -357,7 +380,7 @@ public class SimpleAudioReceiver: NSObject { // 非通话:保持原有逻辑,但使用 .allowBluetooth 以支持录音(而不是 A2DP) desiredCategory = .playAndRecord desiredMode = .videoChat - desiredOptions = [.allowBluetoothA2DP, .mixWithOthers] + desiredOptions = [.allowBluetooth, .allowBluetoothA2DP, .mixWithOthers] //desiredOptions = [.mixWithOthers, .defaultToSpeaker] }else{ desiredCategory = .playAndRecord @@ -390,9 +413,13 @@ public class SimpleAudioReceiver: NSObject { } if hasBtOutput { try audioSession.overrideOutputAudioPort(.none) + } else if isPhoneCallMode, !isHeadphones { + try audioSession.overrideOutputAudioPort(.speaker) } if let inputs = audioSession.availableInputs { - if isReceiverHFP, let btInput = inputs.first(where: { $0.portType == .bluetoothHFP }) { + if preferBuiltInMic, let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { + try audioSession.setPreferredInput(builtInMic) + } else if isReceiverHFP, let btInput = inputs.first(where: { $0.portType == .bluetoothHFP }) { try audioSession.setPreferredInput(btInput) } else if let builtInMic = inputs.first(where: { $0.portType == .builtInMic }) { try audioSession.setPreferredInput(builtInMic) @@ -429,9 +456,14 @@ public class SimpleAudioReceiver: NSObject { // 启动录音引擎 do { print("resumeRecord") - try micCapture.startCapture() + try safeConfigureAudioSessionForMic() + try micCapture.startCapture() } catch { - os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription) + if let nsError = error as NSError? { + os_log("Failed to start audio engine: domain=%{public}@ code=%{public}d desc=%{public}@", log: log, type: .error, nsError.domain, nsError.code, nsError.localizedDescription) + } else { + os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription) + } } }