Browse Source

翻译延迟优化:接上阿里增量文本、面对面/同传去掉固定等待

通话翻译:阿里百炼端到端协议本来就是流式的(音频 response.audio.delta
一直在逐帧回灌),但增量文本订阅的事件名是错的——代码监听
response.audio_transcript.delta(字段 delta),而实时语音翻译
audio+text 模态实际发的是 response.audio_transcript.text(字段 text),
.delta 是 OpenAI Realtime / 通义 Omni 的写法。于是 onPartialText 一次
都不触发,译文只在 .done 时整句蹦出来,看着像"等说完才翻"。
现在两套名字都认、两个字段都取,并按首次出现的事件名锁定,防止服务端
兼容层双发导致文本累计两遍。Android / iOS 两端同改。

面对面 / 同声翻译(走 ASR → HTTP 机器翻译 → TTS 三段串行):
- 补上 Speech_SegmentationSilenceTimeoutMs=300,原来一行没设走 Azure
  默认 500ms,这是"说完到出译文"里唯一一段纯等待;
- 同传 / 音视频改单语 ASR,关掉连续语种识别(它要缓冲够音频、给候选
  语言逐个打分才敢出 final)。只有面对面需要判断"谁在说",仍保持双语;
- 中间结果翻译从"每 300ms 无条件发"改成 700ms + 至少多 4 个字 +
  CancelToken 撤旧请求 + 序号丢弃过期结果。中间译文只上屏不播报,
  原来一句 5 秒的话要打十几次真实服务商调用,还会跟 final 抢链路;
- TTS 提到 _updateExistingHistoryItem 最前,不再排在浮窗 / 写历史 /
  记统计那串 await 之后;
- startRecognition 先 await onInit 那次 changeTranslationMode,堵掉
  "进页面立刻点开始 → 落到默认 zh-CN/en-US"的竞态。

AuthInterceptor 的 DioExceptionType.cancel 分支不再弹红色"请求已取消"
——取消永远是客户端自己发起的,不改的话中间结果一接上就会一路狂弹。

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
main
Rodger-Wang 1 month ago
parent
commit
946204cc3c
  1. 7
      apps/client/lib/data/services/network/api.dart
  2. 19
      apps/client/lib/data/services/network/auth_interceptor.dart
  3. 2
      apps/client/lib/data/services/network/dio_manager.dart
  4. 9
      apps/client/lib/data/services/server_translation_service.dart
  5. 202
      apps/client/lib/modules/translation/controllers/translation_controller.dart
  6. 101
      apps/client/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AliyunBailianE2EHelper.kt
  7. 16
      apps/client/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt
  8. 96
      apps/client/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift
  9. 13
      apps/client/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

7
apps/client/lib/data/services/network/api.dart

@ -1,3 +1,4 @@
import 'package:dio/dio.dart';
import 'dio_manager.dart';
import 'nw_method.dart';
import 'package:get/get.dart';
@ -453,11 +454,15 @@ class Api {
// 实时机器翻译:服务端代调后台「第三方服务配置」里启用的 MT 服务。
// 云账号凭据因此不必下发到客户端。
static userTranslate([params]) {
//
// [cancelToken] 供同传/面对面的**中间结果**翻译使用:说话过程中会连着发好几次,
// 后一次发出时要把前一次撤掉,别让废弃的预览请求占着链路和服务商配额。
static userTranslate(params, {CancelToken? cancelToken}) {
return DioManager().request(
NWMethod.post,
'/api/home/user_translate',
params: params,
cancelToken: cancelToken,
);
}
}

19
apps/client/lib/data/services/network/auth_interceptor.dart

@ -5,6 +5,7 @@ import 'dart:convert';
import 'package:crypto/crypto.dart';
import 'package:get_storage/get_storage.dart';
import '../../../core/utils/logger.dart';
import 'entity/base_entity.dart';
class AuthInterceptor extends Interceptor {
@ -18,7 +19,12 @@ class AuthInterceptor extends Interceptor {
bool _shouldSuppressBusinessErrorSnackbar(RequestOptions options) {
final path = options.path;
return path == '/api/home/user_getinfo';
// user_binddevice 的业务错误由调用方自己提示:
// 恒玄链路(BesDeviceAuth)要弹的是「设备连接受限」,杰理的手动绑定按钮
// (DevicesController.bindAndConnect)弹的是 deviceUnauthorized。
// 不挡住的话用户会先看到一条后端的英文内部状态词(AuthorizeNoCanUse)。
return path == '/api/home/user_getinfo' ||
path == '/api/home/user_binddevice';
}
/// 后端在未登录时会给一些接口返回业务错误码,msg 字段就是英文内部状态词
@ -111,13 +117,10 @@ class AuthInterceptor extends Interceptor {
switch (err.type) {
case DioExceptionType.cancel:
{
Get.snackbar(
'error'.tr, // 错误
'requestCancellation'.tr,
snackPosition: SnackPosition.TOP,
backgroundColor: Colors.red.withOpacity(0.1),
colorText: Colors.red,
);
// 取消永远是客户端自己发起的(同传的中间结果预览翻译被后一发取代、
// OTA 下载被用户中止…),不是故障,用户不需要看到红色报错。
// 原来这里弹「请求已取消」,同传接上取消逻辑后会一路狂弹。
Logger.d('API', '请求已取消: ${err.requestOptions.path}');
}
case DioExceptionType.connectionTimeout:
{

2
apps/client/lib/data/services/network/dio_manager.dart

@ -78,11 +78,13 @@ class DioManager {
String path, {
Object? params,
Map<String, dynamic>? queryParameters,
CancelToken? cancelToken,
}) async {
Response response = await dio.request(
path,
data: params,
queryParameters: queryParameters,
cancelToken: cancelToken,
options: Options(method: nwMethodValues[method]),
);
return response.data;

9
apps/client/lib/data/services/server_translation_service.dart

@ -1,3 +1,4 @@
import 'package:dio/dio.dart';
import 'package:get/get.dart';
import '../../core/utils/logger.dart';
@ -36,10 +37,14 @@ class ServerTranslationService extends GetxService {
///
/// 语言码传 BCP-47(zh-CN / en-US)即可,服务端会归一成服务商要的短码。
/// [sourceLanguageCode] 传空串 = 自动检测,检测结果读 [lastDetectedFrom]。
/// [cancelToken] 给同传/面对面的**中间结果**预览翻译用:说话过程中会连发好几次,
/// 后一次发出时把前一次撤掉。被取消返回 null,且不打 warning——那是预期行为,
/// 不是失败。
Future<String?> translateText({
required String text,
required String sourceLanguageCode,
required String targetLanguageCode,
CancelToken? cancelToken,
}) async {
final src = text.trim();
if (src.isEmpty) return null;
@ -48,7 +53,7 @@ class ServerTranslationService extends GetxService {
'text': src,
'from': sourceLanguageCode,
'to': targetLanguageCode,
});
}, cancelToken: cancelToken);
if (data is! Map) return null;
final out = (data['text'] as String?)?.trim() ?? '';
lastProvider.value = (data['provider'] as String?) ?? '';
@ -60,6 +65,8 @@ class ServerTranslationService extends GetxService {
}
return out;
} catch (e) {
// 主动取消的中间结果请求:预期行为,静默返回。
if (e is DioException && e.type == DioExceptionType.cancel) return null;
// 服务端错误码见 errorcode.proto 5101-5103:
// 5101 未配置 MT 服务 / 5102 参数不合法 / 5103 服务商调用失败
Logger.w(_tag, '翻译失败($sourceLanguageCode→$targetLanguageCode): $e');

202
apps/client/lib/modules/translation/controllers/translation_controller.dart

@ -1,6 +1,7 @@
import 'dart:io';
import 'dart:async';
import 'package:dio/dio.dart' show CancelToken;
import 'package:flutter/material.dart';
import 'package:flutter/services.dart';
import 'package:get/get.dart';
@ -668,7 +669,12 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
_historyManager.initialize();
// 应用模式逻辑(含ASR/TTS配置等)
changeTranslationMode(initMode);
// ⚠️ 这是个 async 方法,里面才会按当前语言对去 initialize ASR。
// 不接住这个 Future 的话,用户进页面立刻点「开始」会抢在它前面:
// AzureAsrService.startContinuousRecognition 那条 `if (!_isInitialized)`
// 兜底分支**不带语言**,会落到默认的 ['zh-CN','en-US']——
// 选了别的语种就等于拿错的候选语言在做识别。startRecognition 会先 await 它。
_modeInitFuture = changeTranslationMode(initMode);
// 监听设备上报的通话翻译开关 (0x16 / 0x17)
_setupDeviceCallTranslationListener(initMode);
@ -1155,10 +1161,15 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
await _initServicesFuture;
_timerManager.startTimer();
final formattedTime = DateFormat('yyyyMMdd_HHmmss').format(DateTime.now());
// 通话翻译按 RecordingArchive 已经定好的 trans_call_ 前缀命名,
// 停止时才能归档进语音记录助手;其它模式暂不归档,命名保持原样。
final String filePath = currentMode.value == 'call'
? "${dir.path}/trans_call_$formattedTime.wav"
// 文件名即语音记录助手里的标题,前缀按 RecordingArchive 的约定走:
// trans_call_* 通话翻译
// trans_holder_* 同声翻译
// 其它模式(音视频/面对面)暂不归档,沿用本地化标题命名。
// ⚠️ 别把这里的前缀和下面 stopRecording 的归档条件改岔了——
// 命名对不上约定,归档进去的标题就成了「同声传译_2026...」这种,
// 列表里跟别的来源混在一起认不出。
final String filePath = _shouldArchiveRecording
? "${dir.path}/${_archivePrefix}_$formattedTime.wav"
: "${dir.path}/${currentModeTitle.value.tr}_$formattedTime.wav";
_currentRecordingPath = filePath;
@ -1197,15 +1208,31 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
isCreateRecord.value = false;
Logger.info('录音已停止,计时器已重置');
if (currentMode.value == 'call' && path != null && seconds > 0) {
await RecordingArchive.archive(
if (_shouldArchiveRecording && path != null && seconds > 0) {
// ⚠️ 不 await:归档里有一次 echomeet_addrecord 网络请求(Dio 30s
// connect + 30s receive),await 会把「点结束」一路堵到超时。
// 归档本身不依赖 controller(RecordingArchive 是静态的),
// 页面退了也能跑完。
unawaited(RecordingArchive.archive(
path: path,
seconds: seconds,
tag: 'TranslationController',
);
));
}
}
/// 这个模式的录音要不要进「语音记录助手」。
///
/// 通话翻译和同声翻译要;音视频翻译录的是系统声音、面对面翻译是逐句
/// 按轮次开停的(见 _startFaceToFaceRecognition),两者的归档语义还没定,
/// 先不动。
bool get _shouldArchiveRecording =>
currentMode.value == 'call' || currentMode.value == 'simultaneous';
/// 归档文件名前缀,取值见 RecordingArchive 的约定。
String get _archivePrefix =>
currentMode.value == 'call' ? 'trans_call' : 'trans_holder';
/// 暂停录音
Future<void> pauseRecording() async {
_timerManager.pauseTimer();
@ -1253,6 +1280,14 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
if (!await _checkPermissions()) return;
if (isRecognizing.value) return;
// 等 onInit 里那次按当前语言对做的 ASR 初始化落地,别让它被兜底的
// 默认 zh-CN/en-US 初始化抢先(见 onInit 中 _modeInitFuture 的说明)。
try {
await _modeInitFuture;
} catch (e) {
Logger.error('等待模式初始化失败(忽略): $e');
}
try {
await _initializeSession();
@ -1651,6 +1686,18 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
_startAsrActiveTracking();
isRecognizing.value = true;
// 同声翻译**默认开录音**:不录就没有文件,也就没有东西归档进「语音记录助手」,
// 而用户的预期是translate完能在助手里找到这段音频。
// 录音按钮仍然在,用户本轮可以随时关掉;下一轮开始时回到默认开。
//
// ⚠️ 只对同声翻译这么做,**通话翻译保持手动**:那录的是通话双方的声音,
// 默认开启属于「未经告知录音」,合规风险完全不是一个量级。
// 音视频翻译录的是系统播放声,面对面是逐句按轮次开停,两者归档语义都还没定,
// 一并保持原样。
if (currentMode.value == 'simultaneous' && !isRecording.value) {
isRecording.value = true;
}
if (isRecording.value && isRecognizing.value) {
startRecording();
}
@ -1970,6 +2017,14 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
Future<void> stopRecognition() async {
if (!isRecognizing.value) return;
// ⚠️ 先翻状态再清理,别放到最后。
// 底下这串(停录音、关 ASR 的 WebSocket、归档)随时可能要好几秒,而
// 「开始/结束」按钮是 Obx 盯着 isRecognizing 的——放在最末尾翻,用户点完
// 「结束」会盯着一个毫无反应的按钮,以为没点上又点一次。
// 提前翻不影响防重入:开头那行 `if (!isRecognizing.value) return` 正是靠
// 它挡住重复调用,提前翻只会更严。
isRecognizing.value = false;
try {
await stopRecording();
await _asrService.stopContinuousRecognition();
@ -1984,7 +2039,6 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
: sourceLanguageCode.value)) ??
'未知';
_recordAsrUsageStats(langName);
isRecognizing.value = false;
currentSessionId = null;
} catch (e) {
Logger.error('停止语音识别失败: ${e.toString()}');
@ -2034,16 +2088,29 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
_historyManager.scrollToBottom();
}
// 防抖处理翻译请求
// 中间结果的翻译只是屏幕上的预览:它不播 TTS,也不写历史、不记统计。
// 原来是「每 300ms 无条件发一次」,一句 5 秒的话就是十几次真实的服务商调用;
// 这些在途请求还会跟说完之后那一次 final 翻译抢同一条链路和同一份配额,
// 弱网时直接表现为「话说完了译文还要等很久」。
//
// 现在:间隔拉到 [_interimTranslateDebounce],且要求文本比上次真正发出去的
// 多出 [_interimMinGrowth] 个字才发(改写过的文本不受这条限制,照发)。
// 真正的取消发生在 [translateText] 里。
_translationDebounceTimer?.cancel();
_translationDebounceTimer = Timer(const Duration(milliseconds: 300), () {
if (translationHistory.isNotEmpty &&
translationHistory.last.isIntermediate &&
translationHistory.last.sourceText.isNotEmpty) {
translateText(translationHistory.last.sourceText,
translationHistory.last.timestamp,
isFinal: false);
}
_translationDebounceTimer = Timer(_interimTranslateDebounce, () {
if (translationHistory.isEmpty) return;
final last = translationHistory.last;
if (!last.isIntermediate || last.sourceText.isEmpty) return;
// 只是在上次发出去的文本后面又多了一两个字:不值得再翻一次。
// 文本被 ASR 改写过(不再以上次那段开头)则不受此限,照发。
final notGrownEnough = last.sourceText.startsWith(_lastInterimSentSource) &&
last.sourceText.length - _lastInterimSentSource.length <
_interimMinGrowth;
if (notGrownEnough) return;
_lastInterimSentSource = last.sourceText;
translateText(last.sourceText, last.timestamp, isFinal: false);
});
}
@ -2105,10 +2172,30 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
isTranslating.value = true;
String? translationResult;
// 中间结果(预览)可以被后来者取消;final 不行——它要写历史、要播报。
CancelToken? cancelToken;
final int seq;
if (isFinal) {
// 说完了,屏幕上的预览已经没有意义。把在途的中间请求全部撤掉,
// 免得它们晚一步回来,把 final 译文盖成半句。
_interimTranslateToken?.cancel('final result arrived');
_interimTranslateToken = null;
_lastInterimSentSource = '';
seq = ++_interimTranslateSeq;
} else {
_interimTranslateToken?.cancel('superseded by newer interim');
cancelToken = _interimTranslateToken = CancelToken();
seq = ++_interimTranslateSeq;
}
try {
final languageCodes = _determineLanguageCodes();
translationResult = await _performTranslation(
sourceText, languageCodes['source']!, languageCodes['target']!);
translationResult = await _performTranslation(sourceText,
languageCodes['source']!, languageCodes['target']!, cancelToken);
// 过期的中间结果直接丢:这期间已经有更新的中间请求、或者 final 发出去了,
// 再写回去只会让译文闪回上一句的半截。
if (!isFinal && seq != _interimTranslateSeq) return null;
if (translationResult != null) {
await _updateTranslationHistory(
@ -2117,7 +2204,9 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
} catch (e, st) {
Logger.e('Translation', '翻译失败: $e', e, st);
} finally {
isTranslating.value = false;
// 只有最新那一发才有资格熄掉 loading,否则被取消的旧请求会把
// 正在跑的 final 提前标记成"翻译完了"。
if (seq == _interimTranslateSeq) isTranslating.value = false;
}
return translationResult;
@ -2184,12 +2273,13 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
/// [sourceCode] 源语言代码
/// [targetCode] 目标语言代码
/// 返回翻译结果
Future<String?> _performTranslation(
String sourceText, String sourceCode, String targetCode) async {
Future<String?> _performTranslation(String sourceText, String sourceCode,
String targetCode, CancelToken? cancelToken) async {
return await _translationService.translateText(
text: sourceText,
sourceLanguageCode: sourceCode,
targetLanguageCode: targetCode,
cancelToken: cancelToken,
);
}
@ -2236,6 +2326,14 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
translationHistory[index].targetLanguageCode = languageCodes['target']!;
translationHistory.refresh();
// ⚠️ 播报必须排在最前面。
// 它是用户唯一「听得见」的一环,而下面的浮窗更新、写历史、记统计
// 全是 await 的平台通道 / 存储操作。原来 TTS 排在它们之后,
// 等于白白把播报推迟了这一整串的耗时。
if (isFinal) {
await _handleTtsPlayback(translationResult);
}
// 更新悬浮窗内容
if (isFloatingWindowEnabled.value) {
await _updateFloatingWindowWithTranslation(translationHistory[index]);
@ -2254,10 +2352,6 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
}
_scrollToBottom();
if (isFinal) {
await _handleTtsPlayback(translationResult);
}
}
/// 记录翻译统计
@ -2637,6 +2731,49 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
await _reinitializeAsrService(restartRecognition: wasRecognizing);
});
/// 当前模式该给 ASR 传哪几种语言。
///
/// 传 1 种 = 固定语种识别;传 2 种 = 原生打开 Azure 的 **连续语种识别**
/// (`LanguageIdMode=Continuous` + AutoDetectSourceLanguageConfig)。
/// 连续语种识别要先缓冲够音频、对候选语言逐个打分才敢出 final,
/// 这段等待直接加在「说完到出译文」上,是面对面/同传最大的一块固定延迟。
///
/// 所以只有真正需要判断「现在是谁在说」的模式才开:
/// - faceToFace:两个人轮流说,方向靠检测结果定,**必须**双语;
/// - simultaneous / audioVideo:单向,方向固定,开了纯亏延迟;
/// 固定语种时原生上报的 detectedLanguage 就是 supportedLanguages[0]
/// (见 AzureAsrHelper 的 onResult),[_determineLanguageCodes] 据此
/// 算出 shouldSwap=true,方向仍然正确。
///
/// ⚠️ 代价:同传下用户若改说目标语种,不会再被自动识别成反向。
/// 同传本来就是单向场景,需要反向请切面对面。
List<String> _asrLanguagesForCurrentMode() {
if (currentMode.value == 'faceToFace') {
return [sourceLanguageCode.value, targetLanguageCode.value];
}
return [sourceLanguageCode.value];
}
/// onInit 里那次 [changeTranslationMode] 的 Future,见 onInit 的说明。
Future<void>? _modeInitFuture;
// ---- 中间结果(说话过程中的预览译文)翻译的节流与取消,见 handleIntermediateResult ----
/// 中间结果翻译的防抖间隔。中间译文只上屏、不播报,值得为它省下请求。
static const Duration _interimTranslateDebounce = Duration(milliseconds: 700);
/// 相比上次真正发出去的文本,至少要多这么多个字才值得再翻一次。
static const int _interimMinGrowth = 4;
/// 在途的中间结果翻译请求,发新的之前撤掉它。
CancelToken? _interimTranslateToken;
/// 上一次真正发出去翻译的中间文本,用于增量门槛判断。
String _lastInterimSentSource = '';
/// 翻译请求序号:回来时不是最新的一发就丢弃,避免旧结果盖掉新译文。
int _interimTranslateSeq = 0;
/// 重新初始化ASR服务
/// [restartRecognition] 初始化完成后是否自动重新开始识别
Future<void> _reinitializeAsrService({bool restartRecognition = false}) async {
@ -2661,16 +2798,7 @@ class TranslationController extends GetxController with WidgetsBindingObserver {
}
isRecognizing.value = false;
late final List<String> asrSupportedLanguages;
if (currentMode.value == 'audioVideo') {
asrSupportedLanguages = [sourceLanguageCode.value];
} else {
asrSupportedLanguages = [
sourceLanguageCode.value,
targetLanguageCode.value
];
}
final List<String> asrSupportedLanguages = _asrLanguagesForCurrentMode();
if (currentMode.value == "call") {
await _initializeCallModeTranslationService();

101
apps/client/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AliyunBailianE2EHelper.kt

@ -68,7 +68,17 @@ class AliyunBailianE2EHelper(
val targetLanguage: String = "en",
// 音色 (e.g. "Cherry", "Kiki")
val voice: String = ""
val voice: String = "",
// 是否开启实时声音复刻(仅 qwen3.5-livetranslate-flash-realtime 支持)。
// 开启后译文听起来像**说话人本人**在说外语,男女音色天然对得上,
// 不需要再按性别挑音色。
val enableVoiceClone: Boolean = false,
// 复刻时机:never / once / always。
// ⚠️ 为 once 或 always 时,阿里要求 voice 必须是 "default",
// 见 sendSessionUpdate 里的强制覆盖。
val voiceCloneFrequency: String = "once"
)
private var conf: Config = Config()
@ -93,6 +103,11 @@ class AliyunBailianE2EHelper(
private val isStarted = AtomicBoolean(false)
private val scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
// 服务端实际使用的增量/终态文本事件名(首次出现即锁定,见 handleJsonMessage)。
// 两套命名(.text / .delta)只会用一套,锁定是为了防止服务端兼容层双发时文本翻倍。
private var partialTextEvent: String? = null
private var finalTextEvent: String? = null
// 诊断计数器
private var pushCount = 0L
private var pushDroppedCount = 0L
@ -116,6 +131,8 @@ class AliyunBailianE2EHelper(
pushDroppedCount = 0
recvMsgCount = 0
audioChunkBuffer = ByteArray(0)
partialTextEvent = null
finalTextEvent = null
recvTextBuffer.setLength(0)
fullTextBuffer.setLength(0)
fullAudioBuffer.reset()
@ -149,14 +166,20 @@ class AliyunBailianE2EHelper(
fullAudioBuffer.reset()
// 构造 URL,必须包含 model 参数
// wss://dashscope.aliyuncs.com/api-ws/v1/realtime?model=qwen3-livetranslate-flash-realtime
// wss://dashscope.aliyuncs.com/api-ws/v1/realtime?model=<模型名>
//
// ⚠️ 这里**必须用 conf.appId**。原来写死的是旧模型名
// qwen3-livetranslate-flash-realtime,于是 Dart 侧无论传什么模型都不生效——
// 表现是「3.5 和声音复刻都配好了却完全没效果」,而且不报任何错
// (连的还是旧模型,session.update 里的 enable_voice_clone 被旧模型忽略)。
val model = conf.appId.ifBlank { "qwen3-livetranslate-flash-realtime" }
val urlWithModel = if (conf.wsUrl.contains("?")) {
// 如果已经包含了参数,假设用户自己拼接了 model 或者其他参数
// 已经带参数:认为调用方自己拼好了 model
conf.wsUrl
} else {
// 强制拼接 model=qwen3-livetranslate-flash-realtime
"${conf.wsUrl}?model=qwen3-livetranslate-flash-realtime"
"${conf.wsUrl}?model=$model"
}
Log.d(TAG, "startContinuousConversation: url=$urlWithModel, voiceClone=${conf.enableVoiceClone}")
val request = Request.Builder()
.url(urlWithModel)
@ -259,7 +282,16 @@ class AliyunBailianE2EHelper(
translation.put("language", conf.targetLanguage) // 目标语言
session.put("translation", translation)
if (conf.voice.isNotEmpty()) {
if (conf.enableVoiceClone) {
// 实时声音复刻:译文用说话人本人的音色。
// ⚠️ frequency 为 once/always 时 voice 必须写 "default",
// 填具体音色名服务端会报错。所以这里**无条件覆盖** Dart 传来的音色。
session.put("voice", "default")
session.put("enable_voice_clone", true)
val cloneOptions = JSONObject()
cloneOptions.put("frequency", conf.voiceCloneFrequency.ifBlank { "once" })
session.put("voice_clone_options", cloneOptions)
} else if (conf.voice.isNotEmpty()) {
session.put("voice", conf.voice)
} else {
if(conf.targetLanguage != "yue"){
@ -378,25 +410,52 @@ class AliyunBailianE2EHelper(
}
// 翻译/生成结果 (Partial Text)
"response.audio_transcript.delta" -> {
val delta = json.optString("delta", "")
if (delta.isNotEmpty()) {
recvTextBuffer.append(delta)
Log.d(TAG, "onPartialText: $delta")
callback?.onPartialText(sessionId, delta)
//
// ⚠️ 增量文本的事件名有两套,之前只认错的那一套:
// 实时语音翻译(livetranslate):audio+text 模态是
// `response.audio_transcript.text`(字段 **text**),
// 纯文本模态是 `response.text.text`;
// OpenAI Realtime / 通义 Omni:`...delta`(字段 **delta**)。
// 阿里文档专门写了这两者不同。原实现只监听 `.delta`,于是
// onPartialText **一次都不会触发**,译文只在 `.done` 时整句蹦出来——
// 协议本身是流式的(音频 `response.audio.delta` 一直是逐帧回灌的),
// 只是文本没接上。这里两套都认、两个字段都取。
"response.audio_transcript.delta",
"response.audio_transcript.text",
"response.text.delta",
"response.text.text" -> {
// 万一服务端两套都发,只认第一次出现的那个名字,避免同一段文本累计两遍。
if (partialTextEvent == null) {
partialTextEvent = type
Log.i(TAG, "增量文本事件名锁定为: $type")
}
if (type == partialTextEvent) {
val delta = json.optString("delta", "")
.ifEmpty { json.optString("text", "") }
if (delta.isNotEmpty()) {
recvTextBuffer.append(delta)
Log.d(TAG, "onPartialText: $delta")
callback?.onPartialText(sessionId, delta)
}
}
}
// 翻译/生成结果 (Final Text)
"response.audio_transcript.done" -> {
val transcript = json.optString("transcript", "")
val finalText = if (transcript.isNotEmpty()) transcript else recvTextBuffer.toString()
Log.d(TAG, "sessionId: ${sessionId}, onFinalTranslatedText: $finalText")
callback?.onFinalTranslatedText(sessionId, finalText)
if (fullTextBuffer.isNotEmpty()) fullTextBuffer.append(" ")
fullTextBuffer.append(finalText)
recvTextBuffer.setLength(0)
// 同上,`.done` 也按模态分两个名字;字段名两种都取。
"response.audio_transcript.done",
"response.text.done" -> {
if (finalTextEvent == null) finalTextEvent = type
if (type == finalTextEvent) {
val transcript = json.optString("transcript", "")
.ifEmpty { json.optString("text", "") }
val finalText = if (transcript.isNotEmpty()) transcript else recvTextBuffer.toString()
Log.d(TAG, "sessionId: ${sessionId}, onFinalTranslatedText: $finalText")
callback?.onFinalTranslatedText(sessionId, finalText)
if (fullTextBuffer.isNotEmpty()) fullTextBuffer.append(" ")
fullTextBuffer.append(finalText)
recvTextBuffer.setLength(0)
}
}
// 翻译/生成音频 (Delta)

16
apps/client/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt

@ -24,6 +24,11 @@ import java.util.UUID
*/
class AzureAsrHelper(private val context: Context) {
companion object {
/** 断句静音等待(毫秒),见 initialize() 里的说明。 */
const val SEGMENTATION_SILENCE_TIMEOUT_MS = 300
}
private val tag = "AzureAsrHelper"
// 核心组件
@ -129,6 +134,17 @@ class AzureAsrHelper(private val context: Context) {
speechRecognitionLanguage = currentLanguage
}
// 断句静音等待:说完之后还要静默这么久,Azure 才吐 final。
// 原来一行都没设 → 走 Azure 默认 500ms,而这是「说完到出译文」里
// 唯一一段纯等待,直接决定面对面/同传的体感。收到 300ms,
// 与通话链路 AzureAsrToAsr.setRecognitionParameters 的默认值对齐。
//
// ⚠️ 别再往下调:值越小越容易把一句话切成好几段,
// 每段各送一次翻译,上下文丢了译文质量反而更差。合法区间 100~5000。
setProperty(
"Speech_SegmentationSilenceTimeoutMs",
SEGMENTATION_SILENCE_TIMEOUT_MS.toString()
)
}
// 初始化网络监听
networkMonitor.initialize(object : NetworkStateMonitor.NetworkStateListener {

96
apps/client/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift

@ -31,6 +31,15 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
var sourceLanguage: String = "zh"
var targetLanguage: String = "en"
var voice: String = ""
/// 是否开启实时声音复刻(仅 qwen3.5-livetranslate-flash-realtime 支持)。
/// 开启后译文听起来像**说话人本人**在说外语,男女音色天然对得上。
var enableVoiceClone: Bool = false
/// 复刻时机:never / once / always。
/// ⚠️ 为 once 或 always 时阿里要求 voice 必须是 "default",
/// 见 sendSessionUpdate 里的强制覆盖。
var voiceCloneFrequency: String = "once"
}
private let tag = "AliyunBailianE2EHelper"
@ -47,6 +56,10 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
private var fullAudioBuffer = Data()
private var recvTextBuffer = ""
// 服务端实际使用的增量/终态文本事件名(首次出现即锁定,见 handleJsonMessage)。
// 两套命名(.text / .delta)只会用一套,锁定是为了防止服务端兼容层双发时文本翻倍。
private var partialTextEvent: String? = nil
private var finalTextEvent: String? = nil
private var audioChunkBuffer = Data()
private var isStarted = false
@ -127,12 +140,20 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
return false
}
// ⚠️ 这里**必须用 conf.appId**。原来写死的是旧模型名
// qwen3-livetranslate-flash-realtime,于是上层无论传什么模型都不生效——
// 表现是「3.5 和声音复刻都配好了却完全没效果」,且不报任何错
// (连的还是旧模型,session.update 里的 enable_voice_clone 被它忽略)。
let model = conf.appId.isEmpty ? "qwen3-livetranslate-flash-realtime" : conf.appId
let urlWithModel: String
if conf.wsUrl.contains("?") {
// 已经带参数:认为调用方自己拼好了 model
urlWithModel = conf.wsUrl
} else {
urlWithModel = "\(conf.wsUrl)?model=qwen3-livetranslate-flash-realtime"
urlWithModel = "\(conf.wsUrl)?model=\(model)"
}
os_log("startSession: url=%{public}@ voiceClone=%d", log: log, type: .info,
urlWithModel, conf.enableVoiceClone ? 1 : 0)
guard let url = URL(string: urlWithModel) else {
os_log("Invalid wsUrl: %{public}@", log: log, type: .error, urlWithModel)
@ -144,6 +165,8 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
recvTextBuffer = ""
fullTextBuffer = ""
partialTextEvent = nil
finalTextEvent = nil
fullAudioBuffer.removeAll()
audioChunkBuffer.removeAll()
@ -174,7 +197,16 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
translation["language"] = conf.targetLanguage
session["translation"] = translation
if !conf.voice.isEmpty {
if conf.enableVoiceClone {
// 实时声音复刻:译文用说话人本人的音色。
// ⚠️ frequency 为 once/always 时 voice 必须写 "default",
// 填具体音色名服务端会报错。所以这里**无条件覆盖**上层传来的音色。
session["voice"] = "default"
session["enable_voice_clone"] = true
session["voice_clone_options"] = [
"frequency": conf.voiceCloneFrequency.isEmpty ? "once" : conf.voiceCloneFrequency
]
} else if !conf.voice.isEmpty {
session["voice"] = conf.voice
} else {
if conf.targetLanguage != "yue" {
@ -297,23 +329,53 @@ class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
os_log("onFinalSourceText: %{public}@", log: log, type: .info, finalTxt)
callback?.onFinalSourceText(sessionId: sessionId, finalText: finalTxt)
}
case "response.audio_transcript.delta":
if let delta = obj["delta"] as? String, !delta.isEmpty {
recvTextBuffer.append(delta)
// 高频事件,取消 info 日志
callback?.onPartialText(sessionId: sessionId, text: delta)
// ⚠️ 增量文本的事件名有两套,之前只认错的那一套:
// 实时语音翻译(livetranslate):audio+text 模态是
// `response.audio_transcript.text`(字段 **text**),
// 纯文本模态是 `response.text.text`;
// OpenAI Realtime / 通义 Omni:`...delta`(字段 **delta**)。
// 阿里文档专门写了这两者不同。原实现只监听 `.delta`,于是
// onPartialText **一次都不会触发**,译文只在 `.done` 时整句蹦出来——
// 协议本身是流式的(音频 `response.audio.delta` 一直是逐帧回灌的),
// 只是文本没接上。这里两套都认、两个字段都取。
// 与 Android 侧 AliyunBailianE2EHelper.kt 保持一致。
case "response.audio_transcript.delta",
"response.audio_transcript.text",
"response.text.delta",
"response.text.text":
// 万一服务端两套都发,只认第一次出现的那个名字,避免同一段文本累计两遍。
if partialTextEvent == nil {
partialTextEvent = type
os_log("增量文本事件名锁定为: %{public}@", log: log, type: .info, type)
}
case "response.audio_transcript.done":
let transcript = obj["transcript"] as? String ?? ""
let finalText = transcript.isEmpty ? recvTextBuffer : transcript
os_log("onFinalTranslatedText: %{public}@", log: log, type: .info, finalText)
callback?.onFinalTranslatedText(sessionId: sessionId, finalText: finalText)
if !fullTextBuffer.isEmpty {
fullTextBuffer.append(" ")
if type == partialTextEvent {
let deltaField = obj["delta"] as? String ?? ""
let textField = obj["text"] as? String ?? ""
let delta = deltaField.isEmpty ? textField : deltaField
if !delta.isEmpty {
recvTextBuffer.append(delta)
// 高频事件,取消 info 日志
callback?.onPartialText(sessionId: sessionId, text: delta)
}
}
// 同上,`.done` 也按模态分两个名字;字段名两种都取。
case "response.audio_transcript.done",
"response.text.done":
if finalTextEvent == nil { finalTextEvent = type }
if type == finalTextEvent {
let transcriptField = obj["transcript"] as? String ?? ""
let textField = obj["text"] as? String ?? ""
let transcript = transcriptField.isEmpty ? textField : transcriptField
let finalText = transcript.isEmpty ? recvTextBuffer : transcript
os_log("onFinalTranslatedText: %{public}@", log: log, type: .info, finalText)
callback?.onFinalTranslatedText(sessionId: sessionId, finalText: finalText)
if !fullTextBuffer.isEmpty {
fullTextBuffer.append(" ")
}
fullTextBuffer.append(finalText)
recvTextBuffer = ""
}
fullTextBuffer.append(finalText)
recvTextBuffer = ""
case "response.audio.delta":
if let base64Audio = obj["delta"] as? String, !base64Audio.isEmpty {
if let audioBytes = Data(base64Encoded: base64Audio) {

13
apps/client/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

@ -13,6 +13,9 @@ import speech
/// 基于微软Azure语音服务的ASR实现
/// 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-recognize-speech
public class AzureAsrHelper: NSObject {
/// 断句静音等待(毫秒),见 initialize() 里的说明。
static let segmentationSilenceTimeoutMs = 300
private let tag = "AzureAsrHelper"
// 日志对象
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureAsrHelper")
@ -194,6 +197,16 @@ public class AzureAsrHelper: NSObject {
os_log("固定语言模式,当前语言: %{public}@", log: log, type: .info, currentLanguage)
}
// 断句静音等待:说完之后还要静默这么久,Azure 才吐 final。
// 原来一行都没设 → 走 Azure 默认 500ms,而这是「说完到出译文」里
// 唯一一段纯等待,直接决定面对面/同传的体感。收到 300ms,
// 与 Android 侧 AzureAsrHelper.SEGMENTATION_SILENCE_TIMEOUT_MS 对齐。
//
// ⚠️ 别再往下调:值越小越容易把一句话切成好几段,
// 每段各送一次翻译,上下文丢了译文质量反而更差。合法区间 100~5000。
speechConfig?.setPropertyTo(
"\(Self.segmentationSilenceTimeoutMs)", byName: "Speech_SegmentationSilenceTimeoutMs")
// 预初始化音频组件
preInitializeAudioComponents()

Loading…
Cancel
Save