import 'dart:convert'; import 'dart:typed_data'; import 'package:get/get.dart'; import 'package:http/http.dart' as http; import '../../core/utils/logger.dart'; import '../../data/models/appconfig.dart'; import '../../modules/picture_translation/models/picture_translation_models.dart'; import 'language_manager.dart'; /// 阿里云百炼(通义千问 VL)图片翻译服务。 /// /// 用来替掉火山的 `TranslateImage`。两边能力并不完全等价,选它的理由与取舍: /// /// **火山 TranslateImage 会返回**:每个文本块的四点坐标、前景/背景平均色, /// 以及一张「把译文回贴到原图」的成品图。 /// **本实现返回**:识别原文、译文、检测语种;坐标/配色/回贴图**不提供**。 /// /// 这样取舍是因为 UI 实际只用到了文本、译文和检测语种 /// (`points` / `foreColor` / `backColor` 在整个 picture_translation 模块里 /// 从未被引用),而回贴图在视图层本来就是判空后条件渲染的, /// 给 null 会自动隐藏那一块,不会出错。 /// /// 模型走 DashScope 的 OpenAI 兼容接口。`qwen-vl-max-latest` 在当前 /// 账号下未开通(access_denied),实测可用的是 `qwen-vl-plus`, /// 因此默认用它,并允许通过配置覆盖。 class AlibabaImageTranslationService extends GetxService { static const String _tag = 'AlibabaImageTranslation'; static const String _defaultEndpoint = 'https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions'; static const String _defaultModel = 'qwen-vl-plus'; late String _apiKey; late String _endpoint; late String _model; bool _isInitialized = false; bool get isInitialized => _isInitialized; final LanguageManager _languageManager = Get.find(); @override Future onInit() async { await initialize(); super.onInit(); } Future initialize() async { if (_isInitialized) return true; try { _apiKey = AppConfig.cred('ast_alibaba', 'app_key', 'ALIBABA_OPENSPEECH_APP_KEY') ?? ''; // 模型/接口地址挂在 ast_alibaba 服务上(vl_model / vl_endpoint),和上面的 API Key 同一条; // cred() 对空串返回 null,所以留空自然落到默认值。 _endpoint = AppConfig.cred('ast_alibaba', 'vl_endpoint', 'ALIBABA_VL_ENDPOINT') ?? _defaultEndpoint; _model = AppConfig.cred('ast_alibaba', 'vl_model', 'ALIBABA_VL_MODEL') ?? _defaultModel; if (_apiKey.trim().isEmpty) { Logger.e(_tag, '未配置 ALIBABA_OPENSPEECH_APP_KEY,图片翻译不可用'); return false; } _isInitialized = true; Logger.i(_tag, '图片翻译服务初始化完成 model=$_model'); return true; } catch (e) { Logger.e(_tag, '初始化异常: $e'); return false; } } /// 识别图中文字并翻译成 [targetLanguageCode](ASR 代码,如 zh-CN / en-US)。 Future translateImage({ required Uint8List imageBytes, required String targetLanguageCode, }) async { if (!_isInitialized && !await initialize()) return null; try { final targetName = _languageManager.getChineseNameByAsrCode(targetLanguageCode) ?? targetLanguageCode; final b64 = base64.encode(imageBytes); // 明确要求纯 JSON。实测模型仍会套 ```json 围栏,所以下面解析时会剥掉。 final prompt = '识别图中所有文字,并翻译成$targetName。' '只返回 JSON,格式:' '{"detectedLanguage":"<源语言短代码,如 en/zh/ja>",' '"blocks":[{"text":"原文","translation":"译文"}]}。' '按阅读顺序输出,不要任何解释或代码块标记。'; final body = json.encode({ 'model': _model, 'messages': [ { 'role': 'user', 'content': [ { 'type': 'image_url', 'image_url': {'url': 'data:image/png;base64,$b64'} }, {'type': 'text', 'text': prompt}, ], } ], 'max_tokens': 2000, }); final response = await http .post(Uri.parse(_endpoint), headers: { 'Authorization': 'Bearer $_apiKey', 'Content-Type': 'application/json', }, body: body) .timeout(const Duration(seconds: 60), onTimeout: () => http.Response('{"error":"timeout"}', 408)); if (response.statusCode != 200) { Logger.e(_tag, '图片翻译 HTTP 错误: status=${response.statusCode}, body=${response.body}'); return null; } final decoded = json.decode(utf8.decode(response.bodyBytes)); final content = decoded?['choices']?[0]?['message']?['content'] as String?; if (content == null || content.trim().isEmpty) { Logger.e(_tag, '图片翻译返回空内容'); return null; } return _parseResult(content); } catch (e) { Logger.e(_tag, '图片翻译异常: $e'); return null; } } /// 解析模型输出。模型习惯把 JSON 包在 ```json 围栏里,先剥掉再解析; /// 万一还是解析不了,就退化成「整段当作一块识别文本」, /// 至少让用户看到内容而不是一个失败弹窗。 ImageTranslationResult? _parseResult(String content) { var text = content.trim(); if (text.startsWith('```')) { final firstBreak = text.indexOf('\n'); if (firstBreak != -1) text = text.substring(firstBreak + 1); final fenceEnd = text.lastIndexOf('```'); if (fenceEnd != -1) text = text.substring(0, fenceEnd); text = text.trim(); } try { final map = json.decode(text); if (map is Map) { final rawBlocks = map['blocks']; final detected = (map['detectedLanguage'] as String?) ?? ''; final blocks = []; if (rawBlocks is List) { for (final b in rawBlocks) { if (b is! Map) continue; final src = (b['text'] ?? '').toString(); final dst = (b['translation'] ?? '').toString(); if (src.isEmpty && dst.isEmpty) continue; blocks.add(ImageTextBlock( points: const [], // 千问 VL 不返回文本块坐标 detectedLanguage: detected, text: src, translation: dst, )); } } if (blocks.isNotEmpty) { Logger.i(_tag, '图片翻译成功:${blocks.length} 个文本块, 语种=$detected'); return ImageTranslationResult( translatedImageBytes: null, // 不提供回贴图,视图层判空后自动隐藏 blocks: blocks, detectedLanguage: detected, ); } } } catch (e) { Logger.w(_tag, 'JSON 解析失败,退化为纯文本展示: $e'); } if (text.isEmpty) return null; return ImageTranslationResult( translatedImageBytes: null, blocks: [ ImageTextBlock( points: const [], detectedLanguage: '', text: '', translation: text, ) ], detectedLanguage: '', ); } }