diff --git a/lib/modules/agent/controllers/agent_controller.dart b/lib/modules/agent/controllers/agent_controller.dart index e0379d657..b55455446 100644 --- a/lib/modules/agent/controllers/agent_controller.dart +++ b/lib/modules/agent/controllers/agent_controller.dart @@ -1,22 +1,28 @@ import 'dart:async'; import 'dart:convert'; +import 'dart:io'; import 'package:flutter/material.dart'; import 'package:get/get.dart'; import 'package:agent_service/agent_service.dart'; import 'package:logger/logger.dart'; import 'package:chat_storage/chat_storage.dart'; +import 'package:image_picker/image_picker.dart'; class Message { final bool isUser; final String text; final DateTime timestamp; final bool isRecognizing; // 是否为语音识别中的临时消息 + final bool hasImage; // 是否包含图片 + final String? imagePath; // 图片路径 Message({ required this.isUser, required this.text, DateTime? timestamp, this.isRecognizing = false, + this.hasImage = false, + this.imagePath, }) : this.timestamp = timestamp ?? DateTime.now(); } @@ -38,12 +44,19 @@ class AgentController extends GetxController { final isSpeaking = false.obs; final isProcessing = false.obs; + // 图片处理状态 + final isImageProcessing = false.obs; + // 输入模式控制 final isTextInputMode = true.obs; // 当前输入的文本内容 final currentText = ''.obs; + // 图片输入相关 + final isImageInputActive = false.obs; // 是否处于图片输入模式 + final selectedImagePath = Rx(null); // 当前选择的图片路径 + // 流事件订阅 StreamSubscription? _eventSubscription; @@ -217,6 +230,14 @@ class AgentController extends GetxController { case AgentServiceEventType.ttsCanceled: isSpeaking.value = false; break; + + case AgentServiceEventType.imageProcessing: + isImageProcessing.value = true; + break; + + case AgentServiceEventType.imageReady: + // 图片准备就绪,但AI还没有开始处理,保持processing状态 + break; case AgentServiceEventType.assistantToken: if (!isProcessing.value) isProcessing.value = true; @@ -249,6 +270,7 @@ class AgentController extends GetxController { case AgentServiceEventType.assistantResponse: isProcessing.value = false; + isImageProcessing.value = false; // 标记当前回复完成 _isNewAssistantResponse = true; break; @@ -257,6 +279,7 @@ class AgentController extends GetxController { isListening.value = false; isSpeaking.value = false; isProcessing.value = false; + isImageProcessing.value = false; // 移除临时的识别消息 final index = messages.lastIndexWhere((msg) => msg.isRecognizing && msg.isUser); @@ -331,6 +354,17 @@ class AgentController extends GetxController { // 发送文本消息 Future sendTextMessage() async { + // 如果图片输入是活跃的,使用选中的图片和文本框的内容 + if (isImageInputActive.value && selectedImagePath.value != null) { + final text = textController.text.trim(); + final imagePath = selectedImagePath.value!; // 使用!强制断言非空 + await sendImageMessage(imagePath, text: text); + + // 清除图片输入状态 + clearImageInput(); + return; + } + final text = textController.text.trim(); if (text.isEmpty) return; @@ -352,6 +386,50 @@ class AgentController extends GetxController { } } + // 发送图片消息 + Future sendImageMessage(String imagePath, {String? text}) async { + if (imagePath.isEmpty || !File(imagePath).existsSync()) { + logger.e('图片不存在: $imagePath'); + return; + } + + // 显示的文本,如果没有提供则使用默认值 + final displayText = text?.isNotEmpty == true ? text! : '[图片]'; + + // 添加用户消息 + final message = Message( + isUser: true, + text: displayText, + hasImage: true, + imagePath: imagePath, + ); + messages.add(message); + + // 清空文本输入 + textController.clear(); + + // 滚动到底部 + _scrollToBottom(); + + try { + isProcessing.value = true; + isImageProcessing.value = true; + // 标记为新的AI回复 + _isNewAssistantResponse = true; + + // 调用Agent Service处理图片 + await AgentService.processImageInput( + imagePath, + text: text ?? '', + speakResponse: true, + ); + } catch (e) { + logger.e('处理图片失败: $e'); + isProcessing.value = false; + isImageProcessing.value = false; + } + } + // 开始语音输入 Future startVoiceInput() async { if (isListening.value) return; @@ -395,6 +473,12 @@ class AgentController extends GetxController { // 切换输入模式 void toggleInputMode() { + // 如果当前在图片输入模式,先清除 + if (isImageInputActive.value) { + clearImageInput(); + return; + } + isTextInputMode.toggle(); // 切换到语音模式时,直接开始语音输入 @@ -406,4 +490,83 @@ class AgentController extends GetxController { stopVoiceInput(); } } + + // 从相册选择图片 + Future pickImage() async { + try { + final ImagePicker picker = ImagePicker(); + final XFile? image = await picker.pickImage( + source: ImageSource.gallery, + imageQuality: 80, // 设置图片质量 + maxWidth: 1024, // 限制最大宽度 + maxHeight: 1024, // 限制最大高度 + ); + + if (image != null) { + // 设置已选择的图片 + selectedImagePath.value = image.path; + // 激活图片输入模式 + isImageInputActive.value = true; + } + } catch (e) { + logger.e('选择图片失败: $e'); + } + } + + // 拍照获取图片 + Future takePhoto() async { + try { + final ImagePicker picker = ImagePicker(); + final XFile? photo = await picker.pickImage( + source: ImageSource.camera, + imageQuality: 80, // 设置图片质量 + maxWidth: 1024, // 限制最大宽度 + maxHeight: 1024, // 限制最大高度 + ); + + if (photo != null) { + // 设置已选择的图片 + selectedImagePath.value = photo.path; + // 激活图片输入模式 + isImageInputActive.value = true; + } + } catch (e) { + logger.e('拍照失败: $e'); + } + } + + // 清除图片输入状态 + void clearImageInput() { + selectedImagePath.value = null; + isImageInputActive.value = false; + } + + // 使用预设提示词处理图片 + void useImagePrompt(String prompt) { + // 确保selectedImagePath.value不为null + if (selectedImagePath.value == null) return; + + // 填充文本框同时直接发送消息 + textController.text = prompt; + // 立即发送图片和提示词 + final imagePath = selectedImagePath.value!; + sendImageMessage(imagePath, text: prompt); + // 发送后清除图片输入状态 + clearImageInput(); + } + + // 预设提示词:总结图片内容 + void summarizeImage() { + useImagePrompt('请总结这张图片的内容'); + } + + // 预设提示词:抽取图像文字 + void extractTextFromImage() { + useImagePrompt('请提取这张图片中的所有文字'); + } + + // 预设提示词:翻译图像文字 + void translateImageText() { + useImagePrompt('请翻译这张图片中的文字'); + } } \ No newline at end of file diff --git a/lib/modules/agent/views/agent_view.dart b/lib/modules/agent/views/agent_view.dart index c74a2f97f..8189ec0b7 100644 --- a/lib/modules/agent/views/agent_view.dart +++ b/lib/modules/agent/views/agent_view.dart @@ -1,3 +1,4 @@ +import 'dart:io'; import 'package:flutter/material.dart'; import 'package:get/get.dart'; import '../controllers/agent_controller.dart'; @@ -90,6 +91,8 @@ class AgentView extends StatelessWidget { message: message.text, timestamp: message.timestamp, isRecognizing: message.isRecognizing, + hasImage: message.hasImage, + imagePath: message.imagePath, ); }, ), @@ -109,145 +112,265 @@ class AgentView extends StatelessWidget { ), ], ), - child: Row( + child: Column( + mainAxisSize: MainAxisSize.min, children: [ - // 键盘/语音切换按钮 + // 图片预览区域 - 仅在图片输入模式下显示 Obx(() { - final isTextMode = controller.isTextInputMode.value; - return GestureDetector( - onTap: () { - controller.toggleInputMode(); - }, - child: Container( - width: 48, - height: 48, - decoration: BoxDecoration( - color: Colors.grey[200], - borderRadius: BorderRadius.circular(24), - ), - child: Icon( - isTextMode ? Icons.mic_none : Icons.keyboard, - color: Colors.grey[600], - ), - ), - ); - }), - - const SizedBox(width: 8), - - // 文本输入框或语音波形图 - Expanded( - child: Obx(() { - final isTextMode = controller.isTextInputMode.value; - - if (isTextMode) { - // 文本输入模式 - return Container( - padding: const EdgeInsets.symmetric(horizontal: 16), - decoration: BoxDecoration( - color: Colors.grey[200], - borderRadius: BorderRadius.circular(24), - ), - child: Row( + if (controller.isImageInputActive.value && controller.selectedImagePath.value != null) { + return Column( + mainAxisSize: MainAxisSize.min, + children: [ + // 图片预览和关闭按钮 + Stack( children: [ - Expanded( - child: TextField( - controller: controller.textController, - decoration: const InputDecoration( - hintText: '输入消息...', - border: InputBorder.none, - ), - onSubmitted: (_) => controller.sendTextMessage(), + ClipRRect( + borderRadius: BorderRadius.circular(12), + child: Image.file( + File(controller.selectedImagePath.value!), + height: 150, + width: double.infinity, + fit: BoxFit.cover, ), ), - GestureDetector( - onTap: () => controller.sendTextMessage(), - child: Icon( - Icons.send, - color: Theme.of(context).primaryColor, + Positioned( + top: 8, + right: 8, + child: GestureDetector( + onTap: () => controller.clearImageInput(), + child: Container( + padding: EdgeInsets.all(4), + decoration: BoxDecoration( + color: Colors.black.withOpacity(0.6), + shape: BoxShape.circle, + ), + child: Icon( + Icons.close, + color: Colors.white, + size: 16, + ), + ), ), ), ], ), - ); - } else { - // 语音输入模式 - 始终显示波形图 + + // 预设提示选项 + SizedBox(height: 8), + SingleChildScrollView( + scrollDirection: Axis.horizontal, + child: Row( + children: [ + _buildPromptChip( + context: context, + label: '总结图片内容', + onTap: () => controller.summarizeImage(), + ), + SizedBox(width: 8), + _buildPromptChip( + context: context, + label: '抽取图像文字', + onTap: () => controller.extractTextFromImage(), + ), + SizedBox(width: 8), + _buildPromptChip( + context: context, + label: '翻译图像文字', + onTap: () => controller.translateImageText(), + ), + ], + ), + ), + SizedBox(height: 12), + ], + ); + } else { + return SizedBox.shrink(); + } + }), + + // 输入控件行 + Row( + children: [ + // 键盘/语音切换按钮 + Obx(() { + final isTextMode = controller.isTextInputMode.value; return GestureDetector( onTap: () { - if (controller.isListening.value) { - controller.stopVoiceInput(); - } else { - controller.startVoiceInput(); - } + controller.toggleInputMode(); }, child: Container( + width: 48, height: 48, - padding: const EdgeInsets.symmetric(horizontal: 16), decoration: BoxDecoration( color: Colors.grey[200], borderRadius: BorderRadius.circular(24), ), - child: Row( - children: [ - // 语音波形图 - 始终显示 - Expanded( - child: Row( - mainAxisAlignment: MainAxisAlignment.start, - children: [ - ...List.generate(10, (index) { - return _buildSoundBar(index); - }), - const Spacer(), - // "正在聆听"文字 - Obx(() { - final isListening = controller.isListening.value; - return Row( - children: [ - Icon( - Icons.mic, - size: 14, - color: isListening - ? Theme.of(context).primaryColor - : Colors.grey, - ), - const SizedBox(width: 4), - Text( - '正在聆听', - style: TextStyle( - fontSize: 12, - color: isListening - ? Theme.of(context).primaryColor - : Colors.grey, - ), - ), - ], - ); - }), - ], - ), - ), - ], + child: Icon( + isTextMode ? Icons.mic_none : Icons.keyboard, + color: Colors.grey[600], ), ), ); - } - }), - ), - - const SizedBox(width: 8), - - // 只显示添加按钮,不再显示停止按钮 - Container( - width: 48, - height: 48, - decoration: BoxDecoration( - color: Colors.grey[200], - borderRadius: BorderRadius.circular(24), - ), - child: Icon( - Icons.add, - color: Colors.grey[600], - ), + }), + + const SizedBox(width: 8), + + // 文本输入框或语音波形图 + Expanded( + child: Obx(() { + final isTextMode = controller.isTextInputMode.value; + + if (isTextMode) { + // 文本输入模式 - 注意观察图片模式时的高度变化 + return Container( + padding: const EdgeInsets.symmetric(horizontal: 16), + decoration: BoxDecoration( + color: Colors.grey[200], + borderRadius: BorderRadius.circular(24), + ), + child: Row( + children: [ + Expanded( + child: Obx(() { + // 图片模式下使用更高的输入框 + final isImageActive = controller.isImageInputActive.value; + + return TextField( + controller: controller.textController, + maxLines: isImageActive ? 2 : 1, + minLines: isImageActive ? 2 : 1, + decoration: InputDecoration( + hintText: isImageActive ? '添加对图片的问题或描述...' : '输入消息...', + border: InputBorder.none, + contentPadding: EdgeInsets.symmetric(vertical: isImageActive ? 12 : 0), + ), + onSubmitted: (_) => controller.sendTextMessage(), + ); + }), + ), + GestureDetector( + onTap: () => controller.sendTextMessage(), + child: Icon( + Icons.send, + color: Theme.of(context).primaryColor, + ), + ), + ], + ), + ); + } else { + // 语音输入模式 - 始终显示波形图 + return GestureDetector( + onTap: () { + if (controller.isListening.value) { + controller.stopVoiceInput(); + } else { + controller.startVoiceInput(); + } + }, + child: Container( + height: 48, + padding: const EdgeInsets.symmetric(horizontal: 16), + decoration: BoxDecoration( + color: Colors.grey[200], + borderRadius: BorderRadius.circular(24), + ), + child: Row( + children: [ + // 语音波形图 - 始终显示 + Expanded( + child: Row( + mainAxisAlignment: MainAxisAlignment.start, + children: [ + ...List.generate(10, (index) { + return _buildSoundBar(index); + }), + const Spacer(), + // "正在聆听"文字 + Obx(() { + final isListening = controller.isListening.value; + return Row( + children: [ + Icon( + Icons.mic, + size: 14, + color: isListening + ? Theme.of(context).primaryColor + : Colors.grey, + ), + const SizedBox(width: 4), + Text( + '正在聆听', + style: TextStyle( + fontSize: 12, + color: isListening + ? Theme.of(context).primaryColor + : Colors.grey, + ), + ), + ], + ); + }), + ], + ), + ), + ], + ), + ), + ); + } + }), + ), + + const SizedBox(width: 8), + + // 图片选择按钮 + Container( + width: 48, + height: 48, + decoration: BoxDecoration( + color: Colors.grey[200], + borderRadius: BorderRadius.circular(24), + ), + child: PopupMenuButton( + icon: Icon( + Icons.add_photo_alternate_outlined, + color: Colors.grey[600], + ), + padding: EdgeInsets.zero, + onSelected: (value) async { + if (value == 'camera') { + await controller.takePhoto(); + } else if (value == 'gallery') { + await controller.pickImage(); + } + }, + itemBuilder: (context) => [ + PopupMenuItem( + value: 'camera', + child: Row( + children: [ + Icon(Icons.camera_alt, color: Theme.of(context).primaryColor), + SizedBox(width: 8), + Text('拍照'), + ], + ), + ), + PopupMenuItem( + value: 'gallery', + child: Row( + children: [ + Icon(Icons.photo_library, color: Theme.of(context).primaryColor), + SizedBox(width: 8), + Text('从相册选择'), + ], + ), + ), + ], + ), + ), + ], ), ], ), @@ -257,6 +380,35 @@ class AgentView extends StatelessWidget { ); } + // 构建提示选项芯片 + Widget _buildPromptChip({ + required BuildContext context, + required String label, + required VoidCallback onTap, + }) { + return GestureDetector( + onTap: onTap, + child: Container( + padding: EdgeInsets.symmetric(horizontal: 12, vertical: 6), + decoration: BoxDecoration( + color: Theme.of(context).primaryColor.withOpacity(0.1), + borderRadius: BorderRadius.circular(16), + border: Border.all( + color: Theme.of(context).primaryColor.withOpacity(0.3), + ), + ), + child: Text( + label, + style: TextStyle( + fontSize: 13, + color: Theme.of(context).primaryColor, + fontWeight: FontWeight.w500, + ), + ), + ), + ); + } + // 构建声音柱 Widget _buildSoundBar(int index) { return Padding( diff --git a/lib/modules/agent/views/message_bubble.dart b/lib/modules/agent/views/message_bubble.dart index b7cd530f7..c7c4b581c 100644 --- a/lib/modules/agent/views/message_bubble.dart +++ b/lib/modules/agent/views/message_bubble.dart @@ -1,5 +1,6 @@ import 'package:flutter/material.dart'; import 'package:intl/intl.dart'; +import 'dart:io'; /// 消息气泡组件 class MessageBubble extends StatelessWidget { @@ -7,6 +8,8 @@ class MessageBubble extends StatelessWidget { final String message; final DateTime timestamp; final bool isRecognizing; + final bool hasImage; + final String? imagePath; const MessageBubble({ Key? key, @@ -14,6 +17,8 @@ class MessageBubble extends StatelessWidget { required this.message, required this.timestamp, this.isRecognizing = false, + this.hasImage = false, + this.imagePath, }) : super(key: key); @override @@ -33,7 +38,7 @@ class MessageBubble extends StatelessWidget { crossAxisAlignment: isUser ? CrossAxisAlignment.end : CrossAxisAlignment.start, children: [ Container( - padding: const EdgeInsets.all(12), + padding: hasImage ? (message == '[图片]' ? EdgeInsets.zero : const EdgeInsets.all(12)) : const EdgeInsets.all(12), decoration: BoxDecoration( color: _getBubbleColor(context), borderRadius: BorderRadius.circular(18).copyWith( @@ -131,6 +136,48 @@ class MessageBubble extends StatelessWidget { ); } + // 如果是图片消息,显示图片 + if (hasImage && imagePath != null) { + // 如果消息是默认的[图片]标记,只显示图片 + if (message == '[图片]') { + return ClipRRect( + borderRadius: BorderRadius.circular(16), + child: Image.file( + File(imagePath!), + width: 200, + height: 200, + fit: BoxFit.cover, + ), + ); + } + + // 如果消息是自定义文本,显示文本+图片 + return Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + Text( + message, + style: TextStyle( + color: isUser ? Colors.white : Colors.black87, + fontSize: 14, + height: 1.4, + fontWeight: FontWeight.w400, + ), + ), + const SizedBox(height: 8), + ClipRRect( + borderRadius: BorderRadius.circular(12), + child: Image.file( + File(imagePath!), + width: 200, + height: 200, + fit: BoxFit.cover, + ), + ), + ], + ); + } + // 判断消息内容,显示特殊卡片 if (!isUser) { if (message.contains('播放列表') || message.contains('音乐')) { diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt index 156e5d2f6..3b486cdcf 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt @@ -99,7 +99,7 @@ object AgentService : CoroutineScope { 4. 避免过长的列表,尽量将信息分成小段 5. 不要使用需要视觉展示的元素(如表格、图表或代码块) 6. 记住用户之前的对话内容,保持对话连贯 - + 7. 如果用户发送了图片,请根据图片内容和文字要求回答问题 你不仅可以回答知识性问题,还可以帮助用户设置提醒、提供建议,或进行轻松愉快的对话。 无论遇到什么问题,都要尽力以温暖、贴心的语气提供最佳帮助。 """.trimIndent() @@ -467,12 +467,6 @@ object AgentService : CoroutineScope { return false } - // 发送与语音识别结果相同格式的事件 - sendEvent("recognition_result", mapOf( - "text" to text, - "language" to "zh-CN", // 默认语言,未来可以从配置或检测中获取 - "inputType" to "text" - )) // 使用OpenAI处理文本 processWithOpenAI(text, speakResponse) @@ -499,6 +493,46 @@ object AgentService : CoroutineScope { ) { FileLogger.d(TAG, "用户问题: $text") + // 创建用户文本消息并处理 + val userMessage = openAIService.createUserMessage(text) + processWithOpenAIInternal(userMessage, text, speakResponse) + } + + /** + * 使用OpenAI处理图片 + * + * @param imageBase64 Base64编码的图片数据 + * @param text 可选的文本描述或问题 + * @param speakResponse 是否朗读回复 + */ + private fun processImageWithOpenAI( + imageBase64: String, + text: String = "", + speakResponse: Boolean = false + ) { + FileLogger.d(TAG, "处理图片输入: ${if (text.isEmpty()) "无附加文本" else "附带文本: $text"}") + + // 创建带图片的用户消息并处理 + val userMessage = openAIService.createUserMessageWithImage(text, imageBase64) + // 图片描述用于存储 + val displayText = text.ifEmpty { "[图片]" } + processWithOpenAIInternal(userMessage, displayText, speakResponse, true) + } + + /** + * 内部方法:通用的OpenAI处理逻辑 + * + * @param userMessage 用户消息(可以是文本或图片格式) + * @param displayText 用于显示和存储的文本 + * @param speakResponse 是否朗读回复 + * @param hasImage 是否包含图片 + */ + private fun processWithOpenAIInternal( + userMessage: JSONObject, + displayText: String, + speakResponse: Boolean = true, + hasImage: Boolean = false + ) { // 如果有正在进行的AI流式输出,先停止它 stopAiStream() @@ -508,12 +542,7 @@ object AgentService : CoroutineScope { // 设置状态为正在流式输出 isAiStreaming = true - // 创建用户消息 - val userMessage = openAIService.createUserMessage(text) - - // 将用户消息添加到历史记录 - addToHistoryMessages(userMessage) - + // 使用历史记录作为上下文发送到OpenAI val responseBuilder = StringBuilder() @@ -529,6 +558,36 @@ object AgentService : CoroutineScope { for (i in 0 until historyMessages.length()) { messagesWithSystemPrompt.put(historyMessages.getJSONObject(i)) } + messagesWithSystemPrompt.put(userMessage) + + // 将用户消息添加到历史记录(注意:要去掉图片数据再保存) + if (userMessage.has("content")) { + val content = userMessage.get("content") + // 检查是否是带图片的消息(即content是JSONArray而不是String) + if (content is JSONArray) { + // 提取文本内容 + var textContent = "" + for (i in 0 until content.length()) { + val item = content.getJSONObject(i) + if (item.getString("type") == "text") { + textContent = item.getString("text") + break + } + } + // 创建新的只含文本的消息 + val textOnlyMessage = JSONObject(userMessage.toString()) + textOnlyMessage.put("content", textContent.ifEmpty { "[图片]" }) + // 添加到历史记录 + addToHistoryMessages(textOnlyMessage) + } else { + // 普通文本消息,直接添加 + addToHistoryMessages(userMessage) + } + } else { + // 兜底处理,直接添加原消息 + addToHistoryMessages(userMessage) + } + openAIService.sendMessageStream( messages = messagesWithSystemPrompt, @@ -548,20 +607,23 @@ object AgentService : CoroutineScope { azureTtsHelper?.flushStream() } val response = responseBuilder.toString() - // FileLogger.d(TAG, "AI完整回复: $response, $speakResponse") - + if (response.isNotEmpty()) { - // 发送完整回复 - sendEvent("assistant_response", mapOf( + // 发送完整回复,包含是否有图片的标记 + val responseData = mutableMapOf( "text" to response, - "userInput" to text - )) + "userInput" to displayText + ) + if (hasImage) { + responseData["hasImage"] = true + } + sendEvent("assistant_response", responseData) // 添加AI回复到历史记录 addToHistoryMessages(openAIService.createAssistantMessage(response)) // 保存聊天记录 - saveChatMessage(text, response) + saveChatMessage(displayText, response) } // 标记AI流式输出已完成 @@ -602,9 +664,10 @@ object AgentService : CoroutineScope { ) } catch (e: Exception) { + val errorType = if (hasImage) "AI_IMAGE_PROCESS_ERROR" else "AI_PROCESS_ERROR" FileLogger.e(TAG, "AI处理出错", e) sendEvent("error", mapOf( - "code" to "AI_PROCESS_ERROR", + "code" to errorType, "message" to e.message.toString() )) @@ -712,11 +775,11 @@ object AgentService : CoroutineScope { sender = "assistant" ) - if (assistantMessageId != -1L) { - FileLogger.d(TAG, "聊天记录已保存:用户消息ID=$userMessageId, 助手消息ID=$assistantMessageId") - } else { - FileLogger.e(TAG, "保存助手消息失败") - } + // if (assistantMessageId != -1L) { + // FileLogger.d(TAG, "聊天记录已保存:用户消息ID=$userMessageId, 助手消息ID=$assistantMessageId") + // } else { + // FileLogger.e(TAG, "保存助手消息失败") + // } } else { FileLogger.e(TAG, "保存用户消息失败") } @@ -803,4 +866,73 @@ object AgentService : CoroutineScope { val result = context.checkCallingOrSelfPermission(permission) return result == android.content.pm.PackageManager.PERMISSION_GRANTED } + + /** + * 处理图片输入 + * 将图片与文本一起发送给AI进行处理 + * + * @param imagePath 图片文件路径 + * @param text 可选的文本描述或问题,默认为空 + * @param speakResponse 是否朗读回复,默认为false + * @return 是否成功开始处理 + */ + fun processImageInput(imagePath: String, text: String = "", speakResponse: Boolean = false): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "服务未初始化") + sendEvent("error", mapOf("code" to "NOT_INITIALIZED", "message" to "服务未初始化")) + return false + } + + if (imagePath.isEmpty()) { + FileLogger.e(TAG, "图片路径不能为空") + sendEvent("error", mapOf("code" to "EMPTY_IMAGE_PATH", "message" to "图片路径不能为空")) + return false + } + + // 通知开始处理图片 + sendEvent("image_processing", mapOf( + "status" to "processing", + "imagePath" to imagePath + )) + + // 使用协程处理耗时的图片转换操作 + launch(Dispatchers.IO) { + try { + // 将图片转换为Base64格式 + val imageBase64 = openAIService.fileToBase64(imagePath) + + if (imageBase64 == null) { + withContext(Dispatchers.Main) { + FileLogger.e(TAG, "图片转换失败: $imagePath") + sendEvent("error", mapOf( + "code" to "IMAGE_CONVERSION_FAILED", + "message" to "图片转换失败" + )) + } + return@launch + } + + // 通知图片准备完成 + withContext(Dispatchers.Main) { + sendEvent("image_ready", mapOf( + "status" to "ready", + "imagePath" to imagePath + )) + + // 使用OpenAI处理包含图片的消息 + processImageWithOpenAI(imageBase64, text, speakResponse) + } + } catch (e: Exception) { + withContext(Dispatchers.Main) { + FileLogger.e(TAG, "处理图片输入失败", e) + sendEvent("error", mapOf( + "code" to "IMAGE_PROCESSING_ERROR", + "message" to e.message.toString() + )) + } + } + } + + return true + } } \ No newline at end of file diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentServicePlugin.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentServicePlugin.kt index 9e8f9eb46..50d189240 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentServicePlugin.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentServicePlugin.kt @@ -76,6 +76,29 @@ class AgentServicePlugin : FlutterPlugin, MethodCallHandler, EventChannel.Stream result.error("TEXT_PROCESS_ERROR", "处理文本输入失败: ${e.message}", null) } } + "processImageInput" -> { + val imagePath = call.argument("imagePath") + if (imagePath.isNullOrEmpty()) { + result.error("INVALID_ARGUMENTS", "图片路径不能为空", null) + return + } + + val text = call.argument("text") ?: "" + val speakResponse = call.argument("speakResponse") ?: false + + try { + val isProcessing = AgentService.processImageInput( + imagePath = imagePath, + text = text, + speakResponse = speakResponse + ) + + result.success(isProcessing) + } catch (e: Exception) { + FileLogger.e(TAG, "处理图片输入失败", e) + result.error("IMAGE_PROCESS_ERROR", "处理图片输入失败: ${e.message}", null) + } + } "speakText" -> { val text = call.argument("text") if (text.isNullOrEmpty()) { @@ -110,7 +133,24 @@ class AgentServicePlugin : FlutterPlugin, MethodCallHandler, EventChannel.Stream result.error("CLEAR_HISTORY_ERROR", "清除聊天历史失败: ${e.message}", null) } } - + "interruptCurrentResponse" -> { + try { + AgentService.interruptCurrentResponse() + result.success(true) + } catch (e: Exception) { + FileLogger.e(TAG, "中断响应失败", e) + result.error("INTERRUPT_ERROR", "中断响应失败: ${e.message}", null) + } + } + "dispose" -> { + try { + AgentService.dispose() + result.success(true) + } catch (e: Exception) { + FileLogger.e(TAG, "释放资源失败", e) + result.error("DISPOSE_ERROR", "释放资源失败: ${e.message}", null) + } + } else -> { result.notImplemented() } diff --git a/local_plugins/agent_service/lib/agent_service.dart b/local_plugins/agent_service/lib/agent_service.dart index daa09ddb4..f4818bdbe 100644 --- a/local_plugins/agent_service/lib/agent_service.dart +++ b/local_plugins/agent_service/lib/agent_service.dart @@ -53,6 +53,12 @@ enum AgentServiceEventType { /// 函数调用结果 functionCallResult, + /// 图片处理中 + imageProcessing, + + /// 图片准备就绪 + imageReady, + /// 错误 error, @@ -151,6 +157,10 @@ class AgentService { return AgentServiceEventType.functionCall; case 'function_call_result': return AgentServiceEventType.functionCallResult; + case 'image_processing': + return AgentServiceEventType.imageProcessing; + case 'image_ready': + return AgentServiceEventType.imageReady; default: return AgentServiceEventType.unknown; } @@ -273,6 +283,25 @@ class AgentService { } } + /// 处理图片输入 + /// + /// [imagePath] 图片文件路径 + /// [text] 可选的文本描述或问题 + /// [speakResponse] 是否朗读响应 + /// 返回是否成功处理图片 + static Future processImageInput(String imagePath, {String text = "", bool speakResponse = false}) async { + try { + final bool result = await _channel.invokeMethod('processImageInput', { + 'imagePath': imagePath, + 'text': text, + 'speakResponse': speakResponse, + }); + return result; + } on PlatformException catch (e) { + throw AgentServiceException(e.code, e.message ?? '处理图片输入失败', e.details); + } + } + /// 释放资源 static Future dispose() async { try { diff --git a/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt index 98356f674..304771e21 100644 --- a/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt +++ b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt @@ -1,6 +1,9 @@ package com.yunqiinnovation.open_ai_service import android.content.Context +import android.graphics.Bitmap +import android.graphics.BitmapFactory +import android.util.Base64 import android.util.Log import com.fasterxml.jackson.core.type.TypeReference import com.fasterxml.jackson.databind.ObjectMapper @@ -14,6 +17,8 @@ import okhttp3.RequestBody.Companion.toRequestBody import okio.Buffer import org.json.JSONArray import org.json.JSONObject +import java.io.ByteArrayOutputStream +import java.io.File import java.io.IOException import java.util.concurrent.TimeUnit @@ -34,6 +39,7 @@ class OpenAIService(private val context: Context? = null) { private var apiKey: String = "" private var isInitialized = false private var model: String = "" // 默认模型 + private var visionModel: String = "doubao-1-5-vision-pro-32k-250115" // 默认模型 // MCP客户端 private var mcpClient: MCPClient? = null @@ -75,6 +81,101 @@ class OpenAIService(private val context: Context? = null) { return command.toString() } + /** + * 将文件转换为Base64字符串 + */ + fun fileToBase64(filePath: String, maxSizeKB: Int = 20480): String? { + try { + val file = File(filePath) + if (!file.exists() || !file.isFile) { + Log.e(TAG, "文件不存在: $filePath") + return null + } + + // 读取文件并压缩(如果需要) + val originalBitmap = BitmapFactory.decodeFile(filePath) + if (originalBitmap == null) { + Log.e(TAG, "无法解码图片: $filePath") + return null + } + + val outputStream = ByteArrayOutputStream() + var quality = 100 + var compressedBitmap = originalBitmap + + // 检查图片尺寸,限制最大为1024*1024 + val maxDimension = 1024 + if (originalBitmap.width > maxDimension || originalBitmap.height > maxDimension) { + Log.d(TAG, "图片尺寸超过限制,进行缩放: ${originalBitmap.width}x${originalBitmap.height} -> ${maxDimension}x${maxDimension}") + + // 计算缩放比例,保持纵横比 + val widthRatio = maxDimension.toFloat() / originalBitmap.width + val heightRatio = maxDimension.toFloat() / originalBitmap.height + val ratio = Math.min(widthRatio, heightRatio) + + val newWidth = (originalBitmap.width * ratio).toInt() + val newHeight = (originalBitmap.height * ratio).toInt() + + compressedBitmap = Bitmap.createScaledBitmap(originalBitmap, newWidth, newHeight, true) + Log.d(TAG, "缩放后图片尺寸: ${newWidth}x${newHeight}") + } + + // 如果原始图片太大,继续优化文件大小 + var fileSize = file.length() / 1024 // 转为KB + if (fileSize > maxSizeKB) { + val scale = Math.sqrt(maxSizeKB.toDouble() / fileSize) + val newWidth = (compressedBitmap.width * scale).toInt() + val newHeight = (compressedBitmap.height * scale).toInt() + compressedBitmap = Bitmap.createScaledBitmap(compressedBitmap, newWidth, newHeight, true) + quality = 85 + } + + // 压缩图片 + compressedBitmap.compress(Bitmap.CompressFormat.JPEG, quality, outputStream) + val imageBytes = outputStream.toByteArray() + + // 检查压缩后大小 + if (imageBytes.size / 1024 > maxSizeKB) { + Log.w(TAG, "压缩后图片仍然超出大小限制: ${imageBytes.size / 1024}KB > ${maxSizeKB}KB") + } + + // 转为Base64 + return Base64.encodeToString(imageBytes, Base64.NO_WRAP) + } catch (e: Exception) { + Log.e(TAG, "转换文件到Base64失败: ${e.message}", e) + return null + } + } + + /** + * 创建带图片的用户消息 + */ + fun createUserMessageWithImage(text: String, imageBase64: String): JSONObject { + // 创建包含文本和图片的内容数组 + val contentArray = JSONArray().apply { + // 添加文本部分 + if (text.isNotEmpty()) { + put(JSONObject().apply { + put("type", "text") + put("text", text) + }) + } + + // 添加图片部分 + put(JSONObject().apply { + put("type", "image_url") + put("image_url", JSONObject().apply { + put("url", "data:image/jpeg;base64,$imageBase64") + }) + }) + } + + return JSONObject().apply { + put("role", "user") + put("content", contentArray) + } + } + /** * 创建用户消息 */ @@ -371,8 +472,17 @@ class OpenAIService(private val context: Context? = null) { // 重置取消状态 isCanceled = false + // 检查最后一条消息内容中是否包含 "image_url",决定使用哪个模型 + var currentModel = model + if (messages.length() > 0) { + val lastMessage = messages.getJSONObject(messages.length() - 1) + if (lastMessage.has("content")) { + val content = lastMessage.getString("content") + currentModel = visionModel + } + } val requestBody = JSONObject().apply { - put("model", model) + put("model", currentModel) put("messages", messages) put("temperature", 0.7) put("max_tokens", 2000)