You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
201 lines
7.0 KiB
201 lines
7.0 KiB
import 'dart:convert';
|
|
import 'dart:typed_data';
|
|
|
|
import 'package:get/get.dart';
|
|
import 'package:http/http.dart' as http;
|
|
|
|
import '../../core/utils/logger.dart';
|
|
import '../../data/models/appconfig.dart';
|
|
import '../../modules/picture_translation/models/picture_translation_models.dart';
|
|
import 'language_manager.dart';
|
|
|
|
/// 阿里云百炼(通义千问 VL)图片翻译服务。
|
|
///
|
|
/// 用来替掉火山的 `TranslateImage`。两边能力并不完全等价,选它的理由与取舍:
|
|
///
|
|
/// **火山 TranslateImage 会返回**:每个文本块的四点坐标、前景/背景平均色,
|
|
/// 以及一张「把译文回贴到原图」的成品图。
|
|
/// **本实现返回**:识别原文、译文、检测语种;坐标/配色/回贴图**不提供**。
|
|
///
|
|
/// 这样取舍是因为 UI 实际只用到了文本、译文和检测语种
|
|
/// (`points` / `foreColor` / `backColor` 在整个 picture_translation 模块里
|
|
/// 从未被引用),而回贴图在视图层本来就是判空后条件渲染的,
|
|
/// 给 null 会自动隐藏那一块,不会出错。
|
|
///
|
|
/// 模型走 DashScope 的 OpenAI 兼容接口。`qwen-vl-max-latest` 在当前
|
|
/// 账号下未开通(access_denied),实测可用的是 `qwen-vl-plus`,
|
|
/// 因此默认用它,并允许通过配置覆盖。
|
|
class AlibabaImageTranslationService extends GetxService {
|
|
static const String _tag = 'AlibabaImageTranslation';
|
|
static const String _defaultEndpoint =
|
|
'https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions';
|
|
static const String _defaultModel = 'qwen-vl-plus';
|
|
|
|
late String _apiKey;
|
|
late String _endpoint;
|
|
late String _model;
|
|
|
|
bool _isInitialized = false;
|
|
bool get isInitialized => _isInitialized;
|
|
|
|
final LanguageManager _languageManager = Get.find<LanguageManager>();
|
|
|
|
@override
|
|
Future<void> onInit() async {
|
|
await initialize();
|
|
super.onInit();
|
|
}
|
|
|
|
Future<bool> initialize() async {
|
|
if (_isInitialized) return true;
|
|
try {
|
|
_apiKey = AppConfig.env('ALIBABA_OPENSPEECH_APP_KEY') ?? '';
|
|
_endpoint =
|
|
(AppConfig.env('ALIBABA_VL_ENDPOINT') ?? '').trim().isNotEmpty
|
|
? AppConfig.env('ALIBABA_VL_ENDPOINT')!
|
|
: _defaultEndpoint;
|
|
_model = (AppConfig.env('ALIBABA_VL_MODEL') ?? '').trim().isNotEmpty
|
|
? AppConfig.env('ALIBABA_VL_MODEL')!
|
|
: _defaultModel;
|
|
if (_apiKey.trim().isEmpty) {
|
|
Logger.e(_tag, '未配置 ALIBABA_OPENSPEECH_APP_KEY,图片翻译不可用');
|
|
return false;
|
|
}
|
|
_isInitialized = true;
|
|
Logger.i(_tag, '图片翻译服务初始化完成 model=$_model');
|
|
return true;
|
|
} catch (e) {
|
|
Logger.e(_tag, '初始化异常: $e');
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 识别图中文字并翻译成 [targetLanguageCode](ASR 代码,如 zh-CN / en-US)。
|
|
Future<ImageTranslationResult?> translateImage({
|
|
required Uint8List imageBytes,
|
|
required String targetLanguageCode,
|
|
}) async {
|
|
if (!_isInitialized && !await initialize()) return null;
|
|
|
|
try {
|
|
final targetName =
|
|
_languageManager.getChineseNameByAsrCode(targetLanguageCode) ??
|
|
targetLanguageCode;
|
|
final b64 = base64.encode(imageBytes);
|
|
|
|
// 明确要求纯 JSON。实测模型仍会套 ```json 围栏,所以下面解析时会剥掉。
|
|
final prompt = '识别图中所有文字,并翻译成$targetName。'
|
|
'只返回 JSON,格式:'
|
|
'{"detectedLanguage":"<源语言短代码,如 en/zh/ja>",'
|
|
'"blocks":[{"text":"原文","translation":"译文"}]}。'
|
|
'按阅读顺序输出,不要任何解释或代码块标记。';
|
|
|
|
final body = json.encode({
|
|
'model': _model,
|
|
'messages': [
|
|
{
|
|
'role': 'user',
|
|
'content': [
|
|
{
|
|
'type': 'image_url',
|
|
'image_url': {'url': 'data:image/png;base64,$b64'}
|
|
},
|
|
{'type': 'text', 'text': prompt},
|
|
],
|
|
}
|
|
],
|
|
'max_tokens': 2000,
|
|
});
|
|
|
|
final response = await http
|
|
.post(Uri.parse(_endpoint),
|
|
headers: {
|
|
'Authorization': 'Bearer $_apiKey',
|
|
'Content-Type': 'application/json',
|
|
},
|
|
body: body)
|
|
.timeout(const Duration(seconds: 60),
|
|
onTimeout: () => http.Response('{"error":"timeout"}', 408));
|
|
|
|
if (response.statusCode != 200) {
|
|
Logger.e(_tag,
|
|
'图片翻译 HTTP 错误: status=${response.statusCode}, body=${response.body}');
|
|
return null;
|
|
}
|
|
|
|
final decoded = json.decode(utf8.decode(response.bodyBytes));
|
|
final content =
|
|
decoded?['choices']?[0]?['message']?['content'] as String?;
|
|
if (content == null || content.trim().isEmpty) {
|
|
Logger.e(_tag, '图片翻译返回空内容');
|
|
return null;
|
|
}
|
|
|
|
return _parseResult(content);
|
|
} catch (e) {
|
|
Logger.e(_tag, '图片翻译异常: $e');
|
|
return null;
|
|
}
|
|
}
|
|
|
|
/// 解析模型输出。模型习惯把 JSON 包在 ```json 围栏里,先剥掉再解析;
|
|
/// 万一还是解析不了,就退化成「整段当作一块识别文本」,
|
|
/// 至少让用户看到内容而不是一个失败弹窗。
|
|
ImageTranslationResult? _parseResult(String content) {
|
|
var text = content.trim();
|
|
if (text.startsWith('```')) {
|
|
final firstBreak = text.indexOf('\n');
|
|
if (firstBreak != -1) text = text.substring(firstBreak + 1);
|
|
final fenceEnd = text.lastIndexOf('```');
|
|
if (fenceEnd != -1) text = text.substring(0, fenceEnd);
|
|
text = text.trim();
|
|
}
|
|
|
|
try {
|
|
final map = json.decode(text);
|
|
if (map is Map) {
|
|
final rawBlocks = map['blocks'];
|
|
final detected = (map['detectedLanguage'] as String?) ?? '';
|
|
final blocks = <ImageTextBlock>[];
|
|
if (rawBlocks is List) {
|
|
for (final b in rawBlocks) {
|
|
if (b is! Map) continue;
|
|
final src = (b['text'] ?? '').toString();
|
|
final dst = (b['translation'] ?? '').toString();
|
|
if (src.isEmpty && dst.isEmpty) continue;
|
|
blocks.add(ImageTextBlock(
|
|
points: const [], // 千问 VL 不返回文本块坐标
|
|
detectedLanguage: detected,
|
|
text: src,
|
|
translation: dst,
|
|
));
|
|
}
|
|
}
|
|
if (blocks.isNotEmpty) {
|
|
Logger.i(_tag, '图片翻译成功:${blocks.length} 个文本块, 语种=$detected');
|
|
return ImageTranslationResult(
|
|
translatedImageBytes: null, // 不提供回贴图,视图层判空后自动隐藏
|
|
blocks: blocks,
|
|
detectedLanguage: detected,
|
|
);
|
|
}
|
|
}
|
|
} catch (e) {
|
|
Logger.w(_tag, 'JSON 解析失败,退化为纯文本展示: $e');
|
|
}
|
|
|
|
if (text.isEmpty) return null;
|
|
return ImageTranslationResult(
|
|
translatedImageBytes: null,
|
|
blocks: [
|
|
ImageTextBlock(
|
|
points: const [],
|
|
detectedLanguage: '',
|
|
text: '',
|
|
translation: text,
|
|
)
|
|
],
|
|
detectedLanguage: '',
|
|
);
|
|
}
|
|
}
|
|
|