You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
 
 
 
 
 
 

198 lines
7.1 KiB

import 'dart:convert';
import 'dart:typed_data';
import 'package:get/get.dart';
import 'package:http/http.dart' as http;
import '../../core/utils/logger.dart';
import '../../data/models/appconfig.dart';
import '../../modules/picture_translation/models/picture_translation_models.dart';
import 'language_manager.dart';
/// 阿里云百炼(通义千问 VL)图片翻译服务。
///
/// 用来替掉火山的 `TranslateImage`。两边能力并不完全等价,选它的理由与取舍:
///
/// **火山 TranslateImage 会返回**:每个文本块的四点坐标、前景/背景平均色,
/// 以及一张「把译文回贴到原图」的成品图。
/// **本实现返回**:识别原文、译文、检测语种;坐标/配色/回贴图**不提供**。
///
/// 这样取舍是因为 UI 实际只用到了文本、译文和检测语种
/// (`points` / `foreColor` / `backColor` 在整个 picture_translation 模块里
/// 从未被引用),而回贴图在视图层本来就是判空后条件渲染的,
/// 给 null 会自动隐藏那一块,不会出错。
///
/// 模型走 DashScope 的 OpenAI 兼容接口。`qwen-vl-max-latest` 在当前
/// 账号下未开通(access_denied),实测可用的是 `qwen-vl-plus`,
/// 因此默认用它,并允许通过配置覆盖。
class AlibabaImageTranslationService extends GetxService {
static const String _tag = 'AlibabaImageTranslation';
static const String _defaultEndpoint =
'https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions';
static const String _defaultModel = 'qwen-vl-plus';
late String _apiKey;
late String _endpoint;
late String _model;
bool _isInitialized = false;
bool get isInitialized => _isInitialized;
final LanguageManager _languageManager = Get.find<LanguageManager>();
@override
Future<void> onInit() async {
await initialize();
super.onInit();
}
Future<bool> initialize() async {
if (_isInitialized) return true;
try {
_apiKey = AppConfig.cred('ast_alibaba', 'app_key', 'ALIBABA_OPENSPEECH_APP_KEY') ?? '';
// 模型/接口地址挂在 ast_alibaba 服务上(vl_model / vl_endpoint),和上面的 API Key 同一条;
// cred() 对空串返回 null,所以留空自然落到默认值。
_endpoint = AppConfig.cred('ast_alibaba', 'vl_endpoint', 'ALIBABA_VL_ENDPOINT') ?? _defaultEndpoint;
_model = AppConfig.cred('ast_alibaba', 'vl_model', 'ALIBABA_VL_MODEL') ?? _defaultModel;
if (_apiKey.trim().isEmpty) {
Logger.e(_tag, '未配置 ALIBABA_OPENSPEECH_APP_KEY,图片翻译不可用');
return false;
}
_isInitialized = true;
Logger.i(_tag, '图片翻译服务初始化完成 model=$_model');
return true;
} catch (e) {
Logger.e(_tag, '初始化异常: $e');
return false;
}
}
/// 识别图中文字并翻译成 [targetLanguageCode](ASR 代码,如 zh-CN / en-US)。
Future<ImageTranslationResult?> translateImage({
required Uint8List imageBytes,
required String targetLanguageCode,
}) async {
if (!_isInitialized && !await initialize()) return null;
try {
final targetName =
_languageManager.getChineseNameByAsrCode(targetLanguageCode) ??
targetLanguageCode;
final b64 = base64.encode(imageBytes);
// 明确要求纯 JSON。实测模型仍会套 ```json 围栏,所以下面解析时会剥掉。
final prompt = '识别图中所有文字,并翻译成$targetName。'
'只返回 JSON,格式:'
'{"detectedLanguage":"<源语言短代码,如 en/zh/ja>",'
'"blocks":[{"text":"原文","translation":"译文"}]}。'
'按阅读顺序输出,不要任何解释或代码块标记。';
final body = json.encode({
'model': _model,
'messages': [
{
'role': 'user',
'content': [
{
'type': 'image_url',
'image_url': {'url': 'data:image/png;base64,$b64'}
},
{'type': 'text', 'text': prompt},
],
}
],
'max_tokens': 2000,
});
final response = await http
.post(Uri.parse(_endpoint),
headers: {
'Authorization': 'Bearer $_apiKey',
'Content-Type': 'application/json',
},
body: body)
.timeout(const Duration(seconds: 60),
onTimeout: () => http.Response('{"error":"timeout"}', 408));
if (response.statusCode != 200) {
Logger.e(_tag,
'图片翻译 HTTP 错误: status=${response.statusCode}, body=${response.body}');
return null;
}
final decoded = json.decode(utf8.decode(response.bodyBytes));
final content =
decoded?['choices']?[0]?['message']?['content'] as String?;
if (content == null || content.trim().isEmpty) {
Logger.e(_tag, '图片翻译返回空内容');
return null;
}
return _parseResult(content);
} catch (e) {
Logger.e(_tag, '图片翻译异常: $e');
return null;
}
}
/// 解析模型输出。模型习惯把 JSON 包在 ```json 围栏里,先剥掉再解析;
/// 万一还是解析不了,就退化成「整段当作一块识别文本」,
/// 至少让用户看到内容而不是一个失败弹窗。
ImageTranslationResult? _parseResult(String content) {
var text = content.trim();
if (text.startsWith('```')) {
final firstBreak = text.indexOf('\n');
if (firstBreak != -1) text = text.substring(firstBreak + 1);
final fenceEnd = text.lastIndexOf('```');
if (fenceEnd != -1) text = text.substring(0, fenceEnd);
text = text.trim();
}
try {
final map = json.decode(text);
if (map is Map) {
final rawBlocks = map['blocks'];
final detected = (map['detectedLanguage'] as String?) ?? '';
final blocks = <ImageTextBlock>[];
if (rawBlocks is List) {
for (final b in rawBlocks) {
if (b is! Map) continue;
final src = (b['text'] ?? '').toString();
final dst = (b['translation'] ?? '').toString();
if (src.isEmpty && dst.isEmpty) continue;
blocks.add(ImageTextBlock(
points: const [], // 千问 VL 不返回文本块坐标
detectedLanguage: detected,
text: src,
translation: dst,
));
}
}
if (blocks.isNotEmpty) {
Logger.i(_tag, '图片翻译成功:${blocks.length} 个文本块, 语种=$detected');
return ImageTranslationResult(
translatedImageBytes: null, // 不提供回贴图,视图层判空后自动隐藏
blocks: blocks,
detectedLanguage: detected,
);
}
}
} catch (e) {
Logger.w(_tag, 'JSON 解析失败,退化为纯文本展示: $e');
}
if (text.isEmpty) return null;
return ImageTranslationResult(
translatedImageBytes: null,
blocks: [
ImageTextBlock(
points: const [],
detectedLanguage: '',
text: '',
translation: text,
)
],
detectedLanguage: '',
);
}
}