From e79f6ccf3b650e50cfaaefed8220a8bdc8c249c5 Mon Sep 17 00:00:00 2001 From: wolfplus2048 <30993207+wolfplus2048@users.noreply.github.com> Date: Sun, 22 Jun 2025 13:16:24 +0100 Subject: [PATCH 1/3] add --- .../azure_speech/AzureTtsHelper.kt | 5 - .../bytedance_speech/BytedanceAudioPlayer.kt | 315 ++++++++++++------ .../chat_api/ChatApiService.kt | 3 +- .../kotlin/com/deep_voice/speech/TtsEvents.kt | 1 - 4 files changed, 214 insertions(+), 110 deletions(-) diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 0465e5e4c..ea9d42588 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -232,11 +232,6 @@ class AzureTtsHelper(private val context: Context) : ITtsService { manager.isSpeakerphoneOn = false // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 } - AudioOutputDevice.EARPIECE -> { - // 强制使用听筒 - manager.mode = AudioManager.MODE_IN_COMMUNICATION - manager.isSpeakerphoneOn = false - } } } } diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt index a3f2c972b..ac7cf0afd 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt @@ -1,12 +1,16 @@ package com.deep_voice.bytedance_speech +import android.content.BroadcastReceiver import android.content.Context +import android.content.Intent +import android.content.IntentFilter import android.media.AudioAttributes +import android.media.AudioDeviceCallback +import android.media.AudioDeviceInfo import android.media.AudioFormat import android.media.AudioManager import android.media.AudioTrack import android.os.Build -import android.util.Log import com.deep_voice.speech.tts.AudioDataListener import com.deep_voice.speech.tts.AudioOutputDevice @@ -29,6 +33,11 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { private var audioOutputDevice = AudioOutputDevice.DEFAULT // 音频输出设备 private var audioManager: AudioManager? = null + // 设备监听相关 + private var deviceCallback: AudioDeviceCallback? = null + private var noisyReceiver: BroadcastReceiver? = null + private var isMonitoringDevices = false + // 回调 private var onPlayStarted: (() -> Unit)? = null private var onPlayCompleted: (() -> Unit)? = null @@ -36,7 +45,6 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { // 播放位置监听器 private val playbackListener = object : AudioTrack.OnPlaybackPositionUpdateListener { override fun onMarkerReached(track: AudioTrack) { - Log.d(TAG, "播放到达标记位置: ${track.playbackHeadPosition}") onPlayCompleted?.invoke() } @@ -46,7 +54,6 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { } init { - Log.d(TAG, "BytedanceAudioPlayer 初始化") audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager initAudioTrack() } @@ -56,24 +63,19 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { * 开始新的播放会话 */ fun startSession() { - Log.d(TAG, "开始新会话") - - // 重置状态 isFirstData = true totalBytesWritten = 0 sessionActive = true - // 重置 AudioTrack audioTrack?.let { track -> if (track.state == AudioTrack.STATE_INITIALIZED) { - // 停止并清空缓冲区 track.pause() track.flush() - // 重新开始播放 track.play() - Log.d(TAG, "AudioTrack 已重置") } - } ?: initAudioTrack() // 如果没有初始化,则初始化 + } ?: initAudioTrack() + + startDeviceMonitoring() } /** @@ -82,28 +84,20 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { fun endSession() { if (!sessionActive) return - Log.d(TAG, "数据流结束,总共写入字节数: $totalBytesWritten") sessionActive = false audioTrack?.let { track -> if (totalBytesWritten > 0) { - // 计算总帧数(16-bit 单声道,每帧2字节) val totalFrames = totalBytesWritten / 2 - - // 设置标记位置 try { track.setNotificationMarkerPosition(totalFrames) - Log.d(TAG, "设置播放完成标记位置: $totalFrames") } catch (e: Exception) { - Log.e(TAG, "设置标记失败: ${e.message}") - // 设置失败时,使用延迟触发作为后备 val durationMs = (totalFrames * 1000L) / SAMPLE_RATE android.os.Handler(android.os.Looper.getMainLooper()).postDelayed({ onPlayCompleted?.invoke() }, durationMs + 500) } } else { - // 没有数据,直接触发完成 onPlayCompleted?.invoke() } } @@ -113,37 +107,26 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { * 接收音频数据 */ override fun onAudioData(data: ByteArray) { - if (data.isEmpty()) return - if (!sessionActive) { - Log.w(TAG, "收到音频数据但会话未激活,忽略数据") - return - } - if (audioTrack == null || audioTrack?.state != AudioTrack.STATE_INITIALIZED) { - Log.w(TAG, "收到音频数据但 AudioTrack 未初始化或状态不正确,忽略数据") - return - } + if (data.isEmpty() || !sessionActive) return + if (audioTrack?.state != AudioTrack.STATE_INITIALIZED) return + try { audioTrack?.let { track -> - if (track.state == AudioTrack.STATE_INITIALIZED) { - val bytesWritten = track.write(data, 0, data.size) + val bytesWritten = track.write(data, 0, data.size) + + if (bytesWritten > 0) { + if (isFirstData) { + isFirstData = false + onPlayStarted?.invoke() + } - if (bytesWritten > 0) { - // 第一次写入数据时自动触发开始回调 - if (isFirstData) { - isFirstData = false - onPlayStarted?.invoke() - Log.d(TAG, "播放开始") - } - - // 累计写入字节数 - if (sessionActive) { - totalBytesWritten += bytesWritten - } + if (sessionActive) { + totalBytesWritten += bytesWritten } } } } catch (e: Exception) { - Log.e(TAG, "写入音频数据失败: ${e.message}") + // 忽略写入失败 } } @@ -151,8 +134,6 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { * 停止播放 */ fun stop() { - Log.d(TAG, "停止播放") - sessionActive = false audioTrack?.let { track -> @@ -162,7 +143,8 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { } } - // 如果已经开始播放,触发完成回调 + stopDeviceMonitoring() + if (!isFirstData) { onPlayCompleted?.invoke() } @@ -172,12 +154,10 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { * 释放资源 */ fun release() { - Log.d(TAG, "释放资源") - stop() - audioTrack?.release() audioTrack = null + stopDeviceMonitoring() } /** @@ -214,13 +194,14 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { fun setAudioOutputDevice(device: AudioOutputDevice) { if (audioOutputDevice != device) { audioOutputDevice = device - // 如果AudioTrack已初始化,需要重新创建以应用新的输出设备设置 - if (audioTrack != null) { - val wasPlaying = isPlaying() - initAudioTrack() - if (wasPlaying) { - audioTrack?.play() - } + + // 应用新的音频路由设置 + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { + // API 31+ 可以动态更改设备 + setPreferredDeviceForTrack() + } else { + // API 30 及以下需要重新配置 + configureAudioRouting() } } } @@ -229,43 +210,19 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { * 初始化 AudioTrack */ private fun initAudioTrack() { - // 释放旧实例 audioTrack?.release() - // 计算缓冲区大小 val minBufferSize = AudioTrack.getMinBufferSize(SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT) val bufferSize = minBufferSize * 2 - // 根据输出设备配置AudioAttributes - val audioAttributes = when (audioOutputDevice) { - AudioOutputDevice.EARPIECE -> { - // 听筒模式 - if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { - AudioAttributes.Builder() - .setUsage(AudioAttributes.USAGE_VOICE_COMMUNICATION) - .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH) - .build() - } else { - null - } - } - else -> { - // 默认、耳机、扬声器模式 - if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { + audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { + AudioTrack.Builder() + .setAudioAttributes( AudioAttributes.Builder() .setUsage(AudioAttributes.USAGE_MEDIA) .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH) .build() - } else { - null - } - } - } - - // 创建 AudioTrack - audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M && audioAttributes != null) { - AudioTrack.Builder() - .setAudioAttributes(audioAttributes) + ) .setAudioFormat( AudioFormat.Builder() .setEncoding(AUDIO_FORMAT) @@ -278,12 +235,8 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { .build() } else { @Suppress("DEPRECATION") - val streamType = when (audioOutputDevice) { - AudioOutputDevice.EARPIECE -> AudioManager.STREAM_VOICE_CALL - else -> AudioManager.STREAM_MUSIC - } AudioTrack( - streamType, + AudioManager.STREAM_MUSIC, SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT, @@ -292,46 +245,202 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { ) } - // 设置播放位置监听器 audioTrack?.setPlaybackPositionUpdateListener(playbackListener) - // 配置音频路由 - configureAudioRouting() + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { + setPreferredDeviceForTrack() + } else { + configureAudioRouting() + } - // 开始播放 audioTrack?.play() + } + + /** + * API 31+ 使用 setPreferredDevice 设置音频输出设备 + */ + private fun setPreferredDeviceForTrack() { + if (Build.VERSION.SDK_INT < Build.VERSION_CODES.S) return - Log.d(TAG, "AudioTrack 初始化成功") + audioManager?.let { manager -> + audioTrack?.let { track -> + when (audioOutputDevice) { + AudioOutputDevice.DEFAULT -> { + track.setPreferredDevice(null) + manager.isSpeakerphoneOn = false + manager.mode = AudioManager.MODE_NORMAL + } + + AudioOutputDevice.SPEAKER -> { + val speaker = manager.getDevices(AudioManager.GET_DEVICES_OUTPUTS) + .firstOrNull { it.type == AudioDeviceInfo.TYPE_BUILTIN_SPEAKER } + speaker?.let { track.setPreferredDevice(it) } + } + + AudioOutputDevice.HEADPHONES -> { + val headphones = manager.getDevices(AudioManager.GET_DEVICES_OUTPUTS) + .firstOrNull { device -> + device.type == AudioDeviceInfo.TYPE_WIRED_HEADSET || + device.type == AudioDeviceInfo.TYPE_WIRED_HEADPHONES || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_A2DP || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_SCO + } + headphones?.let { track.setPreferredDevice(it) } + ?: track.setPreferredDevice(null) + } + } + } + } } /** - * 配置音频路由 + * 配置音频路由(API 30 及以下) */ private fun configureAudioRouting() { audioManager?.let { manager -> when (audioOutputDevice) { AudioOutputDevice.DEFAULT -> { - // 默认模式:系统自动选择 manager.mode = AudioManager.MODE_NORMAL manager.isSpeakerphoneOn = false } + AudioOutputDevice.SPEAKER -> { - // 强制使用扬声器 manager.mode = AudioManager.MODE_NORMAL manager.isSpeakerphoneOn = true } + AudioOutputDevice.HEADPHONES -> { - // 强制使用耳机(如果已连接) manager.mode = AudioManager.MODE_NORMAL manager.isSpeakerphoneOn = false - // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 } - AudioOutputDevice.EARPIECE -> { - // 强制使用听筒 - manager.mode = AudioManager.MODE_IN_COMMUNICATION - manager.isSpeakerphoneOn = false + } + } + } + + /** + * 开始监听设备变化 + */ + private fun startDeviceMonitoring() { + if (isMonitoringDevices) return + + // API 23+ 使用 AudioDeviceCallback + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { + deviceCallback = object : AudioDeviceCallback() { + override fun onAudioDevicesAdded(addedDevices: Array) { + super.onAudioDevicesAdded(addedDevices) + handleDeviceAdded(addedDevices) + } + + override fun onAudioDevicesRemoved(removedDevices: Array) { + super.onAudioDevicesRemoved(removedDevices) + handleDeviceRemoved(removedDevices) } } + audioManager?.registerAudioDeviceCallback(deviceCallback, null) + } + + // 所有版本都监听 NOISY 广播(耳机拔出) + noisyReceiver = object : BroadcastReceiver() { + override fun onReceive(context: Context, intent: Intent) { + if (intent.action == AudioManager.ACTION_AUDIO_BECOMING_NOISY) { + handleNoisyAudioEvent() + } + } + } + context.registerReceiver(noisyReceiver, IntentFilter(AudioManager.ACTION_AUDIO_BECOMING_NOISY)) + + isMonitoringDevices = true + } + + /** + * 停止监听设备变化 + */ + private fun stopDeviceMonitoring() { + if (!isMonitoringDevices) return + + // 注销 AudioDeviceCallback + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { + deviceCallback?.let { + audioManager?.unregisterAudioDeviceCallback(it) + } + deviceCallback = null + } + + // 注销广播接收器 + noisyReceiver?.let { + try { + context.unregisterReceiver(it) + } catch (e: Exception) { + // 忽略已注销的异常 + } + } + noisyReceiver = null + + isMonitoringDevices = false + } + + /** + * 处理设备添加 + */ + private fun handleDeviceAdded(devices: Array) { + val hasHeadphones = devices.any { device -> + device.type == AudioDeviceInfo.TYPE_WIRED_HEADSET || + device.type == AudioDeviceInfo.TYPE_WIRED_HEADPHONES || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_A2DP || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_SCO + } + + if (hasHeadphones) { + switchToDefaultMode() + } + } + + /** + * 处理设备移除 + */ + private fun handleDeviceRemoved(devices: Array) { + val hasHeadphones = devices.any { device -> + device.type == AudioDeviceInfo.TYPE_WIRED_HEADSET || + device.type == AudioDeviceInfo.TYPE_WIRED_HEADPHONES || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_A2DP || + device.type == AudioDeviceInfo.TYPE_BLUETOOTH_SCO + } + + if (hasHeadphones) { + switchToSpeakerMode() + } + } + + /** + * 处理音频变得嘈杂事件(通常是耳机拔出) + */ + private fun handleNoisyAudioEvent() { + switchToSpeakerMode() + } + + /** + * 切换到 DEFAULT 模式 + */ + private fun switchToDefaultMode() { + audioOutputDevice = AudioOutputDevice.DEFAULT + + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { + setPreferredDeviceForTrack() + } else { + configureAudioRouting() + } + } + + /** + * 切换到扬声器模式 + */ + private fun switchToSpeakerMode() { + audioOutputDevice = AudioOutputDevice.SPEAKER + + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { + setPreferredDeviceForTrack() + } else { + configureAudioRouting() } } } \ No newline at end of file diff --git a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt index 13f74a898..55485623b 100644 --- a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt +++ b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt @@ -218,7 +218,8 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor mcpConfigJson = mcpServer // 异步初始化MCP客户端 launch { - initializeMcpClient(mcpServer) + // initializeMcpClient(mcpServer) + initializeMcpClient("{}") } isInitialized = apiKey.isNotEmpty() diff --git a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt index 6758c6f05..47bea05bf 100644 --- a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt +++ b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt @@ -67,5 +67,4 @@ enum class AudioOutputDevice { DEFAULT, // 默认(如果有耳机选耳机,否则使用系统扬声器) HEADPHONES, // 强制使用耳机 SPEAKER, // 强制使用扬声器 - EARPIECE // 强制使用听筒 } \ No newline at end of file From 2f4744357fa40edc7af741e03450397c956f1063 Mon Sep 17 00:00:00 2001 From: wolfplus2048 <30993207+wolfplus2048@users.noreply.github.com> Date: Sun, 22 Jun 2025 14:03:52 +0100 Subject: [PATCH 2/3] before job --- .../agent/controllers/agent_controller.dart | 22 ++++++-- .../agent_service/AgentService.kt | 53 +++++++++++-------- .../agent_service/AgentServiceImpl.swift | 8 +++ .../agent_service/lib/agent_service.dart | 10 ++++ .../chat_api/SystemFunctionHandler.kt | 22 ++++---- 5 files changed, 76 insertions(+), 39 deletions(-) diff --git a/lib/modules/agent/controllers/agent_controller.dart b/lib/modules/agent/controllers/agent_controller.dart index 2bbaa3d78..cc98107c4 100644 --- a/lib/modules/agent/controllers/agent_controller.dart +++ b/lib/modules/agent/controllers/agent_controller.dart @@ -253,6 +253,7 @@ class AgentController extends GetxController { case AgentServiceEventType.recognizing: final text = event.data['text'] ?? ''; currentText.value = text; // 保留当前文本,以便其他地方使用 + Logger.i(TAG, '识别中间结果: $text'); if (text.isNotEmpty) { // 查找是否有正在识别中的消息 final index = messages @@ -280,7 +281,7 @@ class AgentController extends GetxController { case AgentServiceEventType.recognitionResult: final text = event.data['text'] ?? ''; - Logger.i(TAG, '识别结果: $text'); + Logger.i(TAG, '识别最终结果: $text'); // 查找是否有正在识别中的消息 final index = messages.lastIndexWhere((msg) => msg.isRecognizing && msg.isUser); @@ -313,16 +314,27 @@ class AgentController extends GetxController { break; case AgentServiceEventType.ttsStarted: - isSpeaking.value = true; - Logger.i(TAG, 'TTS开始播放'); + // isSpeaking.value = true; + // Logger.i(TAG, 'TTS开始播放'); break; case AgentServiceEventType.ttsCompleted: case AgentServiceEventType.ttsStopped: case AgentServiceEventType.ttsCanceled: + // isSpeaking.value = false; + + // Logger.i(TAG, 'TTS停止播放, $event.type'); + break; + + case AgentServiceEventType.playbackStarted: + isSpeaking.value = true; + Logger.i(TAG, 'TTS播放开始'); + break; + + case AgentServiceEventType.playbackCompleted: isSpeaking.value = false; - Logger.i(TAG, 'TTS停止播放, $event.type'); + Logger.i(TAG, 'TTS播放完成'); break; case AgentServiceEventType.imageProcessing: @@ -338,7 +350,7 @@ class AgentController extends GetxController { final token = event.data['token'] ?? ''; final responseId = event.data['responseId'] ?? ''; - + Logger.i(TAG, 'AI回复Token: $token, responseId: $responseId'); if (token.isNotEmpty) { // 如果是新的回复或者响应ID改变,创建新消息 if (_isNewAssistantResponse || diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt index d9d2c9700..ea0a1b19a 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt @@ -266,6 +266,14 @@ object AgentService : CoroutineScope { restartIdleCheck() sendEvent("tts_canceled", mapOf("status" to "canceled")) } + TtsEventType.PLAYBACK_STARTED -> { + restartIdleCheck() + sendEvent("playback_started", mapOf("status" to "playback_started")) + } + TtsEventType.PLAYBACK_COMPLETED -> { + restartIdleCheck() + sendEvent("playback_completed", mapOf("status" to "playback_completed")) + } TtsEventType.ERROR -> { isTtsSpeaking = false restartIdleCheck() @@ -400,21 +408,21 @@ object AgentService : CoroutineScope { if (isTtsSpeaking || isAiStreaming) { val currentTime = System.currentTimeMillis() - // 防抖处理:避免过于频繁的打断 - if (currentTime - lastInterruptTime < INTERRUPT_DEBOUNCE_MS) { - return - } + // // 防抖处理:避免过于频繁的打断 + // if (currentTime - lastInterruptTime < INTERRUPT_DEBOUNCE_MS) { + // return + // } - // 简单过滤:太短的内容可能是噪音 - if (recognizing.trim().length < 2) { - return - } + // // 简单过滤:太短的内容可能是噪音 + // if (recognizing.trim().length < 2) { + // return + // } - // 过滤纯语气词 - val trimmedText = recognizing.trim().lowercase() - if (FILLER_WORDS.contains(trimmedText)) { - return - } + // // 过滤纯语气词 + // val trimmedText = recognizing.trim().lowercase() + // if (FILLER_WORDS.contains(trimmedText)) { + // return + // } // 执行打断 lastInterruptTime = currentTime @@ -520,18 +528,17 @@ object AgentService : CoroutineScope { val startTime = System.currentTimeMillis() // 并行执行停止操作,加快响应速度 - launch { stopTts() } - launch { stopAiStream() } + stopTts() + stopAiStream() // 记录打断耗时 - launch { - delay(100) // 短暂延迟后计算耗时 - val duration = System.currentTimeMillis() - startTime - sendEvent("response_interrupted", mapOf( - "status" to "interrupted", - "duration_ms" to duration - )) - } + + val duration = System.currentTimeMillis() - startTime + sendEvent("response_interrupted", mapOf( + "status" to "interrupted", + "duration_ms" to duration + )) + } } diff --git a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift index e5035bf17..0dc9119f6 100644 --- a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift +++ b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift @@ -909,6 +909,14 @@ extension AgentServiceImpl: TtsEventListener { restartIdleCheck() sendEvent(name: "tts_completed", data: ["status": "completed"]) + case .playbackStarted: + restartIdleCheck() + sendEvent(name: "playback_started", data: ["status": "playback_started"]) + + case .playbackCompleted: + restartIdleCheck() + sendEvent(name: "playback_completed", data: ["status": "playback_completed"]) + case .synthesisCanceled: isSpeaking = false restartIdleCheck() diff --git a/local_plugins/agent_service/lib/agent_service.dart b/local_plugins/agent_service/lib/agent_service.dart index 4ee6907cb..fc1579a05 100644 --- a/local_plugins/agent_service/lib/agent_service.dart +++ b/local_plugins/agent_service/lib/agent_service.dart @@ -38,6 +38,12 @@ enum AgentServiceEventType { /// TTS停止 ttsStopped, + /// 播放开始 + playbackStarted, + + /// 播放完成 + playbackCompleted, + /// 响应中断 responseInterrupted, @@ -147,6 +153,10 @@ class AgentService { return AgentServiceEventType.ttsCanceled; case 'tts_stopped': return AgentServiceEventType.ttsStopped; + case 'playback_started': + return AgentServiceEventType.playbackStarted; + case 'playback_completed': + return AgentServiceEventType.playbackCompleted; case 'response_interrupted': return AgentServiceEventType.responseInterrupted; case 'assistant_token': diff --git a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/SystemFunctionHandler.kt b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/SystemFunctionHandler.kt index 1c5b7de75..24f56af0d 100644 --- a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/SystemFunctionHandler.kt +++ b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/SystemFunctionHandler.kt @@ -47,17 +47,17 @@ class SystemFunctionHandler(private val context: Context? = null) { handler = ExitInteractionHandler(context) ) - // 注册翻译模式函数 - client.registerLocalFunction( - name = "enter_translation_mode", - description = "用户请求进入实时翻译模式时,启动实时翻译功能", - parameters = mapOf( - "type" to "object", - "properties" to emptyMap(), - "required" to emptyList() - ), - handler = TranslationModeHandler(context) - ) + // // 注册翻译模式函数 + // client.registerLocalFunction( + // name = "enter_translation_mode", + // description = "用户请求进入实时翻译模式时,启动实时翻译功能", + // parameters = mapOf( + // "type" to "object", + // "properties" to emptyMap(), + // "required" to emptyList() + // ), + // handler = TranslationModeHandler(context) + // ) // 注册发送短信函数 client.registerLocalFunction( From bcda9422e8d6ee5a49caec1b5ff031b369611bb0 Mon Sep 17 00:00:00 2001 From: wolfplus2048 <30993207+wolfplus2048@users.noreply.github.com> Date: Sun, 22 Jun 2025 15:47:04 +0100 Subject: [PATCH 3/3] add --- .../agent/controllers/agent_controller.dart | 2 +- .../agent_service/AgentService.kt | 113 ++++++++++-------- .../bytedance_speech/BytedanceAudioPlayer.kt | 40 +++++-- .../chat_api/ChatApiService.kt | 47 +++++++- .../yunqiinnovation/chat_api/MCPSubClient.kt | 4 +- 5 files changed, 140 insertions(+), 66 deletions(-) diff --git a/lib/modules/agent/controllers/agent_controller.dart b/lib/modules/agent/controllers/agent_controller.dart index cc98107c4..7cdebb046 100644 --- a/lib/modules/agent/controllers/agent_controller.dart +++ b/lib/modules/agent/controllers/agent_controller.dart @@ -350,7 +350,7 @@ class AgentController extends GetxController { final token = event.data['token'] ?? ''; final responseId = event.data['responseId'] ?? ''; - Logger.i(TAG, 'AI回复Token: $token, responseId: $responseId'); + // Logger.i(TAG, 'AI回复Token: $token, responseId: $responseId'); if (token.isNotEmpty) { // 如果是新的回复或者响应ID改变,创建新消息 if (_isNewAssistantResponse || diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt index ea0a1b19a..e18de276f 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt @@ -16,6 +16,9 @@ import java.util.Collections import kotlin.coroutines.CoroutineContext import com.yunqiinnovation.ble_service.BleService import android.media.MediaPlayer +import java.util.concurrent.atomic.AtomicBoolean +import kotlinx.coroutines.sync.Mutex +import kotlinx.coroutines.sync.withLock import com.deep_voice.speech.tts.TtsEvent import com.deep_voice.speech.tts.TtsEventListener import com.deep_voice.speech.tts.TtsEventType @@ -43,7 +46,7 @@ object AgentService : CoroutineScope { // 协程相关 private val job = SupervisorJob() override val coroutineContext: CoroutineContext - get() = Dispatchers.Main + job + get() = Dispatchers.IO + job // 上下文和监听器 private lateinit var context: Context @@ -73,16 +76,24 @@ object AgentService : CoroutineScope { // 音频播放器 private var audioPlayer: AudioPlayer? = null // 初始化音频播放器 - // 状态 - private var isInitialized = false - var isRecognitionActive = false - private set - var isTtsSpeaking = false - private set - var hasSpeechDetected = false - private set - var isAiStreaming = false - private set + // 状态 - 使用原子类型确保线程安全 + private val _isInitialized = AtomicBoolean(false) + val isInitialized: Boolean get() = _isInitialized.get() + + private val _isRecognitionActive = AtomicBoolean(false) + val isRecognitionActive: Boolean get() = _isRecognitionActive.get() + + private val _isTtsSpeaking = AtomicBoolean(false) + val isTtsSpeaking: Boolean get() = _isTtsSpeaking.get() + + private val _hasSpeechDetected = AtomicBoolean(false) + val hasSpeechDetected: Boolean get() = _hasSpeechDetected.get() + + private val _isAiStreaming = AtomicBoolean(false) + val isAiStreaming: Boolean get() = _isAiStreaming.get() + + // 用于保护复杂状态操作的互斥锁 + private val stateMutex = Mutex() // AI流生成相关 private var currentAiJob: Job? = null @@ -175,7 +186,7 @@ object AgentService : CoroutineScope { // 加载最近的聊天记录 loadChatHistory() - isInitialized = true + _isInitialized.set(true) return true } catch (e: Exception) { Log.e(TAG, "初始化失败: ${e.message}") @@ -201,6 +212,7 @@ object AgentService : CoroutineScope { // 释放ChatAPI服务 if (::chatApiService.isInitialized) { chatApiService.cancelCurrentStream() + chatApiService.dispose() } // 释放TTS服务 @@ -209,7 +221,7 @@ object AgentService : CoroutineScope { job.cancel() clearListeners() - isInitialized = false + _isInitialized.set(false) } catch (e: Exception) { Log.e(TAG, "释放资源异常: ${e.message}") } @@ -252,17 +264,17 @@ object AgentService : CoroutineScope { override fun onEvent(event: TtsEvent) { when (event.type) { TtsEventType.SYNTHESIS_STARTED -> { - isTtsSpeaking = true + _isTtsSpeaking.set(true) restartIdleCheck() sendEvent("tts_started", mapOf("status" to "started")) } TtsEventType.SYNTHESIS_COMPLETED -> { - isTtsSpeaking = false + _isTtsSpeaking.set(false) restartIdleCheck() sendEvent("tts_completed", mapOf("status" to "completed")) } TtsEventType.SYNTHESIS_CANCELED -> { - isTtsSpeaking = false + _isTtsSpeaking.set(false) restartIdleCheck() sendEvent("tts_canceled", mapOf("status" to "canceled")) } @@ -275,7 +287,7 @@ object AgentService : CoroutineScope { sendEvent("playback_completed", mapOf("status" to "playback_completed")) } TtsEventType.ERROR -> { - isTtsSpeaking = false + _isTtsSpeaking.set(false) restartIdleCheck() val params = event.params val code = params["errorCode"] as? String ?: "UNKNOWN_ERROR" @@ -379,8 +391,8 @@ object AgentService : CoroutineScope { return false } - isRecognitionActive = true - hasSpeechDetected = false + _isRecognitionActive.set(true) + _hasSpeechDetected.set(false) try { val audioSourceType = if (isExternalActive) { @@ -393,7 +405,7 @@ object AgentService : CoroutineScope { if (recognizing.isNotEmpty()) { // 检测到语音,更新状态 val previousHasSpeech = hasSpeechDetected - hasSpeechDetected = true + _hasSpeechDetected.set(true) if (!previousHasSpeech) { restartIdleCheck() @@ -408,21 +420,21 @@ object AgentService : CoroutineScope { if (isTtsSpeaking || isAiStreaming) { val currentTime = System.currentTimeMillis() - // // 防抖处理:避免过于频繁的打断 - // if (currentTime - lastInterruptTime < INTERRUPT_DEBOUNCE_MS) { - // return - // } + // 防抖处理:避免过于频繁的打断 + if (currentTime - lastInterruptTime < INTERRUPT_DEBOUNCE_MS) { + return + } - // // 简单过滤:太短的内容可能是噪音 - // if (recognizing.trim().length < 2) { - // return - // } + // 简单过滤:太短的内容可能是噪音 + if (recognizing.trim().length < 2) { + return + } - // // 过滤纯语气词 - // val trimmedText = recognizing.trim().lowercase() - // if (FILLER_WORDS.contains(trimmedText)) { - // return - // } + // 过滤纯语气词 + val trimmedText = recognizing.trim().lowercase() + if (FILLER_WORDS.contains(trimmedText)) { + return + } // 执行打断 lastInterruptTime = currentTime @@ -443,7 +455,7 @@ object AgentService : CoroutineScope { // 重置状态,继续识别 val previousHasSpeech = hasSpeechDetected - hasSpeechDetected = false + _hasSpeechDetected.set(false) if (previousHasSpeech) { restartIdleCheck() @@ -460,13 +472,13 @@ object AgentService : CoroutineScope { override fun onSessionStopped() { sendEvent("recognition_stopped", mapOf("status" to "stopped")) - isRecognitionActive = false + _isRecognitionActive.set(false) stopIdleCheck() audioPlayer?.playAudio(R.raw.stop) } override fun onCanceled(reason: String, errorDetails: String) { - isRecognitionActive = false + _isRecognitionActive.set(false) stopIdleCheck() BleService.closeCodec() sendEvent("recognition_canceled", mapOf( @@ -476,7 +488,7 @@ object AgentService : CoroutineScope { } override fun onError(error: String) { - isRecognitionActive = false + _isRecognitionActive.set(false) stopIdleCheck() BleService.closeCodec() Log.e(TAG, "语音识别错误: $error") @@ -488,7 +500,7 @@ object AgentService : CoroutineScope { }, audioSourceType) return true } catch (e: Exception) { - isRecognitionActive = false + _isRecognitionActive.set(false) Log.e(TAG, "启动语音识别失败: ${e.message}") sendEvent("error", mapOf( "code" to "RECOGNITION_START_ERROR", @@ -510,11 +522,11 @@ object AgentService : CoroutineScope { try { azureAsrHelper?.stopContinuousRecognition() BleService.closeCodec() - isRecognitionActive = false + _isRecognitionActive.set(false) stopIdleCheck() } catch (e: Exception) { Log.e(TAG, "停止语音识别异常: ${e.message}") - isRecognitionActive = false + _isRecognitionActive.set(false) stopIdleCheck() } } @@ -549,7 +561,7 @@ object AgentService : CoroutineScope { if (isAiStreaming) { try { // 先更新状态,避免回调时的状态不一致 - isAiStreaming = false + _isAiStreaming.set(false) // 取消当前AI生成任务 currentAiJob?.cancel() @@ -561,7 +573,7 @@ object AgentService : CoroutineScope { } catch (e: Exception) { Log.e(TAG, "停止AI流输出异常", e) // 确保状态被重置,即使发生异常 - isAiStreaming = false + _isAiStreaming.set(false) currentAiJob = null } } @@ -655,7 +667,7 @@ object AgentService : CoroutineScope { currentAiJob = launch { try { // 设置状态为正在流式输出 - isAiStreaming = true + _isAiStreaming.set(true) // 使用历史记录作为上下文发送到OpenAI val responseBuilder = StringBuilder() @@ -753,7 +765,7 @@ object AgentService : CoroutineScope { // 保存聊天记录 saveChatMessage(displayText, response,aiMetadata,userMetadata.toString()) // 标记AI流式输出已完成 - isAiStreaming = false + _isAiStreaming.set(false) currentAiJob = null } @@ -765,7 +777,7 @@ object AgentService : CoroutineScope { )) // 标记AI流式输出已完成 - isAiStreaming = false + _isAiStreaming.set(false) currentAiJob = null } @@ -809,7 +821,7 @@ object AgentService : CoroutineScope { )) // 确保状态被重置 - isAiStreaming = false + _isAiStreaming.set(false) currentAiJob = null } } @@ -873,7 +885,7 @@ object AgentService : CoroutineScope { if (text.isEmpty()) return // 更新状态 - isTtsSpeaking = true + _isTtsSpeaking.set(true) restartIdleCheck() // 状态变化,重启检测 // 直接调用TTS,无需协程包装 @@ -887,7 +899,7 @@ object AgentService : CoroutineScope { if (isTtsSpeaking) { ttsService?.stop() - isTtsSpeaking = false + _isTtsSpeaking.set(false) restartIdleCheck() // 状态变化,重启检测 sendEvent("tts_stopped", mapOf("status" to "stopped")) } @@ -964,9 +976,8 @@ object AgentService : CoroutineScope { * 发送事件 */ private fun sendEvent(eventName: String, data: Map) { - // 使用协程确保在主线程上执行 - launch { - // 我们已在主线程上下文中启动协程,无需再切换线程 + // 切换到主线程执行监听器回调,避免 UI 更新问题 + launch(Dispatchers.Main) { listeners.forEach { listener -> try { listener.onEvent(eventName, data) diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt index ac7cf0afd..b3cffcbd1 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt @@ -24,6 +24,7 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { private const val SAMPLE_RATE = 24000 private const val CHANNEL_CONFIG = AudioFormat.CHANNEL_OUT_MONO private const val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT + private const val VOLUME_REDUCTION = 0.7f // 降低音量以减少回音 } private var audioTrack: AudioTrack? = null @@ -206,6 +207,7 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { } } + /** * 初始化 AudioTrack */ @@ -213,13 +215,18 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { audioTrack?.release() val minBufferSize = AudioTrack.getMinBufferSize(SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT) - val bufferSize = minBufferSize * 2 + val bufferSize = minBufferSize * 2 // 使用较小的缓冲区以降低延迟 audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { - AudioTrack.Builder() + val builder = AudioTrack.Builder() .setAudioAttributes( AudioAttributes.Builder() - .setUsage(AudioAttributes.USAGE_MEDIA) + .setUsage( + // 扬声器模式下默认使用语音通信模式以启用回声消除 + if (audioOutputDevice == AudioOutputDevice.SPEAKER) + AudioAttributes.USAGE_VOICE_COMMUNICATION + else AudioAttributes.USAGE_MEDIA + ) .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH) .build() ) @@ -232,7 +239,13 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { ) .setBufferSizeInBytes(bufferSize) .setTransferMode(AudioTrack.MODE_STREAM) - .build() + + // API 26+ 设置低延迟模式 + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.O) { + builder.setPerformanceMode(AudioTrack.PERFORMANCE_MODE_LOW_LATENCY) + } + + builder.build() } else { @Suppress("DEPRECATION") AudioTrack( @@ -247,6 +260,9 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { audioTrack?.setPlaybackPositionUpdateListener(playbackListener) + // 设置音量以减少回音 + audioTrack?.setVolume(VOLUME_REDUCTION) + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { setPreferredDeviceForTrack() } else { @@ -268,13 +284,17 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { AudioOutputDevice.DEFAULT -> { track.setPreferredDevice(null) manager.isSpeakerphoneOn = false - manager.mode = AudioManager.MODE_NORMAL + // 默认模式下使用通信模式以获得更好的音频处理 + manager.mode = AudioManager.MODE_IN_COMMUNICATION } AudioOutputDevice.SPEAKER -> { val speaker = manager.getDevices(AudioManager.GET_DEVICES_OUTPUTS) .firstOrNull { it.type == AudioDeviceInfo.TYPE_BUILTIN_SPEAKER } speaker?.let { track.setPreferredDevice(it) } + // 扬声器模式下始终启用通信模式以获得回声消除 + manager.mode = AudioManager.MODE_IN_COMMUNICATION + manager.isSpeakerphoneOn = true } AudioOutputDevice.HEADPHONES -> { @@ -287,6 +307,9 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { } headphones?.let { track.setPreferredDevice(it) } ?: track.setPreferredDevice(null) + // 耳机模式下可以使用普通模式 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false } } } @@ -300,16 +323,19 @@ class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { audioManager?.let { manager -> when (audioOutputDevice) { AudioOutputDevice.DEFAULT -> { - manager.mode = AudioManager.MODE_NORMAL + // 默认模式下使用通信模式以获得更好的音频处理 + manager.mode = AudioManager.MODE_IN_COMMUNICATION manager.isSpeakerphoneOn = false } AudioOutputDevice.SPEAKER -> { - manager.mode = AudioManager.MODE_NORMAL + // 扬声器模式下始终启用通信模式以获得回声消除 + manager.mode = AudioManager.MODE_IN_COMMUNICATION manager.isSpeakerphoneOn = true } AudioOutputDevice.HEADPHONES -> { + // 耳机模式下可以使用普通模式 manager.mode = AudioManager.MODE_NORMAL manager.isSpeakerphoneOn = false } diff --git a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt index 55485623b..ee7fae90a 100644 --- a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt +++ b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt @@ -111,6 +111,10 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor override val coroutineContext: CoroutineContext = Dispatchers.IO + SupervisorJob() + companion object { + private const val TAG = "ChatApiService" + } + // MARK: - 属性 private var baseUrl = "https://api.openai.com/v1/" private var apiKey = "" @@ -218,8 +222,8 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor mcpConfigJson = mcpServer // 异步初始化MCP客户端 launch { - // initializeMcpClient(mcpServer) - initializeMcpClient("{}") + initializeMcpClient(mcpServer) + // initializeMcpClient("{}") } isInitialized = apiKey.isNotEmpty() @@ -514,8 +518,8 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor ) // 通知上层工具调用事件 currSessionCallback?.onFunctionCall(convertMapToJsonObject(functionCall)) - // 在后台队列处理工具调用 - launch { + // 在当前协程作用域内处理工具调用,使用async确保生命周期管理 + val toolCallDeferred = async { try { if (sessionid == currSessionId) { // 通过MCP客户端处理工具调用 @@ -587,6 +591,15 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor } } + // 等待工具调用完成,确保生命周期管理 + try { + toolCallDeferred.await() + } catch (e: CancellationException) { + // 协程被取消,确保子任务也被取消 + toolCallDeferred.cancel() + throw e + } + return true } @@ -780,7 +793,7 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor description = description, parameters = parameters ) - Log.d("ChatApiService", "AI携带工具: $name, 参数定义: $parameters") + // Log.d("ChatApiService", "AI携带工具: $name, 参数定义: $parameters") tools.add(tool) } catch (e: Exception) { @@ -813,6 +826,30 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor } } + /** + * 释放所有资源 + * 统一的资源管理方法,确保所有协程被正确取消 + */ + fun dispose() { + runBlocking { + // 取消当前流式任务 + currentStreamJob?.cancelAndJoin() + currentStreamJob = null + + // 关闭MCP客户端 + closeMcpClient() + + // 取消所有子协程 + coroutineContext[Job]?.cancelChildren() + + // 清理其他资源 + currSessionCallback = null + currSessionId = "" + currentMessages = emptyList() + toolCalls.clear() + } + } + /** * 处理MCP工具调用 * diff --git a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/MCPSubClient.kt b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/MCPSubClient.kt index 3fe6b926a..77393a44a 100644 --- a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/MCPSubClient.kt +++ b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/MCPSubClient.kt @@ -129,7 +129,7 @@ class MCPSubClient( "properties" to emptyMap(), "required" to emptyList() ) - Log.w(TAG, "解析工具数据: ${tool.name} ${parametersMap}") + // Log.w(TAG, "解析工具数据: ${tool.name} ${parametersMap}") val toolMap = mapOf( "type" to "function", "function" to mapOf( @@ -175,7 +175,7 @@ class MCPSubClient( name = name, arguments = argumentsJson ) - + Log.d(TAG, "[$serverId] 调用工具 '$name',参数: ${argumentsJson.toString()}") // 调用工具 val result = mcpClient?.callTool(request)