From 7193c758aa1b6937c5857dadc7636d031209298d Mon Sep 17 00:00:00 2001 From: fdp <1286779656@qq.com> Date: Tue, 29 Jul 2025 18:53:15 +0800 Subject: [PATCH] =?UTF-8?q?=E5=8A=A0=E5=85=A5=E9=80=9A=E8=AF=9D=E7=BB=93?= =?UTF-8?q?=E6=9E=9C?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- lib/data/services/asr_service.dart | 3 + lib/data/services/ast_service.dart | 185 +++++++++--------- .../speech_impl/azure_asr_service.dart | 17 +- .../speech_impl/azure_ast_service.dart | 164 +++++++++++++++- .../controllers/speech_test_controller.dart | 10 +- .../controllers/translation_controller.dart | 28 ++- .../translation/views/translation_view.dart | 171 ++++++++++------ .../azure_speech/AzureAsrHelper.kt | 2 +- .../azure_speech/AzureAsrToAsr.kt | 133 ++++++++++--- .../azure_speech/AzureSpeechPlugin.kt | 45 ++++- .../azure_speech/tools/RecordFile.kt | 15 +- .../yunqiinnovation/ble_service/BleService.kt | 4 +- 12 files changed, 581 insertions(+), 196 deletions(-) diff --git a/lib/data/services/asr_service.dart b/lib/data/services/asr_service.dart index 62928938d..54b507005 100644 --- a/lib/data/services/asr_service.dart +++ b/lib/data/services/asr_service.dart @@ -70,6 +70,9 @@ enum RecognitionEventType { /// 最终识别结果 finalResult, + /// 麦识别结果 + finalResult1, + /// 中间识别结果(实时反馈) intermediateResult, diff --git a/lib/data/services/ast_service.dart b/lib/data/services/ast_service.dart index 53671ced5..69cd23592 100644 --- a/lib/data/services/ast_service.dart +++ b/lib/data/services/ast_service.dart @@ -12,96 +12,105 @@ abstract class AstService { /// 开始录音 Future enableRecord(String filePath); - /// 停止录音 - Future stopContinuousTranslation(bool isSave); + /// + Future startContinuousTranslation(); + + /// + Future stopContinuousTranslation(); + + /// 返回一个包含识别事件的流 + Future> recognizeCallback(); /// 开始录音 Future path(String filePath); + + /// 释放资源 + Future dispose(); +} + +// 识别事件类型 +enum RecognitionEventType1 { + /// 最终识别结果 + finalResult, + + /// 中间识别结果(实时反馈) + intermediateResult, + + /// 音频 + onAudio, + + /// 会话开始 + sessionStarted, + + /// 会话结束 + sessionStopped, + + /// 识别取消 + canceled, + + /// 识别错误 + error, } -/// 识别事件类型 -// enum RecognitionEventType { -// /// 最终识别结果 -// finalResult, - -// /// 中间识别结果(实时反馈) -// intermediateResult, - -// /// 音频 -// onAudio, - -// /// 会话开始 -// sessionStarted, - -// /// 会话结束 -// sessionStopped, - -// /// 识别取消 -// canceled, - -// /// 识别错误 -// error, -// } - -/// 识别事件 -// class RecognitionEvent { -// /// 事件类型 -// final RecognitionEventType type; - -// /// 识别文本(仅在 finalResult 和 intermediateResult 类型中有效) -// final String text; - -// /// 检测到的语言 -// final String detectedLanguage; - -// /// 角色 -// final String role; - -// /// 原始音频 -// final Uint8List? audio; - -// /// 错误信息(仅在 error 和 canceled 类型中有效) -// final String error; - -// RecognitionEvent({ -// required this.type, -// this.text = '', -// this.detectedLanguage = '', -// this.role = '', -// this.audio, -// this.error = '', -// }); - -// /// 创建最终结果事件的快捷构造函数 -// factory RecognitionEvent.finalResult({ -// required String text, -// String detectedLanguage = '', -// }) { -// return RecognitionEvent( -// type: RecognitionEventType.finalResult, -// text: text, -// detectedLanguage: detectedLanguage, -// ); -// } - -// /// 创建错误事件的快捷构造函数 -// factory RecognitionEvent.error(String errorMessage) { -// return RecognitionEvent( -// type: RecognitionEventType.error, -// error: errorMessage, -// ); -// } - -// /// 检查是否为最终结果 -// bool get isFinalResult => type == RecognitionEventType.finalResult; - -// /// 检查是否为错误 -// bool get isError => -// type == RecognitionEventType.error || -// type == RecognitionEventType.canceled; - -// @override -// String toString() { -// return 'RecognitionEvent{type: $type, text: $text, detectedLanguage: $detectedLanguage, error: $error}'; -// } -// } +// 识别事件 +class RecognitionEvent1 { + /// 事件类型 + final RecognitionEventType1 type; + + /// 识别文本(仅在 finalResult 和 intermediateResult 类型中有效) + final String text; + + /// 检测到的语言 + final String detectedLanguage; + + /// 角色 + final String role; + + /// 原始音频 + final Uint8List? audio; + + /// 错误信息(仅在 error 和 canceled 类型中有效) + final String error; + + RecognitionEvent1({ + required this.type, + this.text = '', + this.detectedLanguage = '', + this.role = '', + this.audio, + this.error = '', + }); + + /// 创建最终结果事件的快捷构造函数 + factory RecognitionEvent1.finalResult({ + required String text, + String detectedLanguage = '', + }) { + return RecognitionEvent1( + type: RecognitionEventType1.finalResult, + text: text, + detectedLanguage: detectedLanguage, + ); + } + + /// 创建错误事件的快捷构造函数 + factory RecognitionEvent1.error(String errorMessage) { + return RecognitionEvent1( + type: RecognitionEventType1.error, + error: errorMessage, + ); + } + + /// 检查是否为最终结果 + bool get isFinalResult => type == RecognitionEventType1.finalResult; + + /// 检查是否为错误 + bool get isError => + type == RecognitionEventType1.error || + type == RecognitionEventType1.canceled; + + @override + String toString() { + return 'RecognitionEvent{type: $type, text: $text, detectedLanguage: $detectedLanguage, error: $error}'; + } +} diff --git a/lib/data/services/speech_impl/azure_asr_service.dart b/lib/data/services/speech_impl/azure_asr_service.dart index c02347552..832b83b5b 100644 --- a/lib/data/services/speech_impl/azure_asr_service.dart +++ b/lib/data/services/speech_impl/azure_asr_service.dart @@ -242,6 +242,18 @@ class AzureAsrService extends GetxService implements AsrService { detectedLanguage: detectedLanguage, )); break; + case 'result1': + final String text = eventMap['text'] as String? ?? ''; + final String detectedLanguage = + eventMap['detectedLanguage'] as String? ?? ''; + _latestRecognizedText = text; + _latestDetectedLanguage = detectedLanguage; + _eventStreamController?.add(RecognitionEvent( + type: RecognitionEventType.finalResult1, + text: text, + detectedLanguage: detectedLanguage, + )); + break; case 'recognizing': final String text = eventMap['text'] as String? ?? ''; @@ -404,7 +416,7 @@ class AzureAsrService extends GetxService implements AsrService { } /// 设置音频配置 - /// + /// /// [sampleRate] 采样率,默认16000 /// [channels] 声道数,默认1(单声道) Future setAudioConfig({ @@ -417,7 +429,8 @@ class AzureAsrService extends GetxService implements AsrService { 'channels': channels, }); - Logger.info('音频配置设置${result ? '成功' : '失败'}: 采样率=$sampleRate, 声道数=$channels'); + Logger.info( + '音频配置设置${result ? '成功' : '失败'}: 采样率=$sampleRate, 声道数=$channels'); return result; } catch (e) { Logger.error('设置音频配置失败: ${e.toString()}'); diff --git a/lib/data/services/speech_impl/azure_ast_service.dart b/lib/data/services/speech_impl/azure_ast_service.dart index 6e9a4af0a..f66945d9f 100644 --- a/lib/data/services/speech_impl/azure_ast_service.dart +++ b/lib/data/services/speech_impl/azure_ast_service.dart @@ -34,10 +34,10 @@ class AzureAstService extends GetxService implements AstService { @override List get supportedLanguages => _defaultSupportedLanguages; - // 连续识别相关 - // bool _isContinuousRecognitionActive = false; - // StreamController? _eventStreamController; - // StreamSubscription? _eventSubscription; + //连续识别相关 + bool _isContinuousRecognitionActive = false; + StreamController? _eventStreamController; + StreamSubscription? _eventSubscription; // 最新的识别结果 String _latestRecognizedText = ''; @@ -54,6 +54,109 @@ class AzureAstService extends GetxService implements AstService { _loadConfig(); } + /// 设置事件通道 + void _setupEventChannel() { + _eventSubscription?.cancel(); + _eventSubscription = _eventChannel.receiveBroadcastStream().listen((event) { + if (event is Map) { + _handleRecognitionEvent(event); + } + }, onError: _handleRecognitionError); + } + + /// 处理来自原生端的识别事件 + void _handleRecognitionEvent(dynamic event) { + if (event is! Map || _eventStreamController == null) return; + + final Map eventMap = event; + final String eventType = eventMap['type'] as String? ?? ''; + + // 添加日志帮助调试 + + switch (eventType) { + case 'result': + final String text = eventMap['text'] as String? ?? ''; + final String detectedLanguage = + eventMap['detectedLanguage'] as String? ?? ''; + _latestRecognizedText = text; + _latestDetectedLanguage = detectedLanguage; + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.finalResult, + text: text, + detectedLanguage: detectedLanguage, + )); + break; + + case 'recognizing': + final String text = eventMap['text'] as String? ?? ''; + final String detectedLanguage = + eventMap['detectedLanguage'] as String? ?? ''; + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.intermediateResult, + text: text, + detectedLanguage: detectedLanguage, + )); + break; + case 'onAudio': + final Uint8List data = eventMap['data'] as Uint8List? ?? Uint8List(0); + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.onAudio, + audio: data, + )); + break; + case 'sessionStarted': + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.sessionStarted, + )); + break; + + case 'sessionStopped': + _isContinuousRecognitionActive = false; + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.sessionStopped, + )); + break; + + case 'canceled': + _isContinuousRecognitionActive = false; + final String reason = eventMap['reason'] as String? ?? ''; + final String errorDetails = eventMap['errorDetails'] as String? ?? ''; + + if (reason.isNotEmpty || errorDetails.isNotEmpty) { + Logger.error('识别取消: $reason - ${errorDetails.toString()}'); + } + + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.canceled, + error: '$reason: $errorDetails', + )); + break; + + case 'error': + final String error = eventMap['message'] as String? ?? ''; + Logger.error('识别错误: ${error.toString()}'); + _eventStreamController?.add(RecognitionEvent1( + type: RecognitionEventType1.error, + error: error, + )); + break; + } + } + + /// 处理识别事件流错误 + void _handleRecognitionError(Object error) { + Logger.error('识别事件流错误: ${error.toString()}'); + _eventStreamController?.addError(error); + _cleanupEventStream(); + } + + /// 清理事件流资源 + void _cleanupEventStream() { + _eventStreamController?.close(); + _eventStreamController = null; + _isContinuousRecognitionActive = false; + } + /// 从环境变量加载配置 void _loadConfig() { // final _env = _storage.read("ENV") as Map; @@ -70,6 +173,27 @@ class AzureAstService extends GetxService implements AstService { } } + @override + Future> recognizeCallback() async { + if (!_isInitialized) { + await initialize(); + } + try { + _eventStreamController = StreamController.broadcast(); + + // 开始连续识别 + final bool result = await _channel.invokeMethod('recognizeCallback'); + if (!result) { + _cleanupEventStream(); + } + + return _eventStreamController!.stream; + } catch (e) { + Logger.error('开始连续语音识别失败: ${e.toString()}'); + rethrow; + } + } + @override Future enableRecord(String filePath) async { try { @@ -99,12 +223,23 @@ class AzureAstService extends GetxService implements AstService { } @override - Future stopContinuousTranslation(bool isSave) async { + Future startContinuousTranslation() async { try { final bool result = - await _channel.invokeMethod('stopContinuousTranslation', { - 'isSave': isSave, - }); + await _channel.invokeMethod('startContinuousTranslation'); + + return result; + } catch (e) { + Logger.error('停止录音: ${e.toString()}'); + rethrow; + } + } + + @override + Future stopContinuousTranslation() async { + try { + final bool result = + await _channel.invokeMethod('stopContinuousTranslation'); return result; } catch (e) { @@ -113,6 +248,18 @@ class AzureAstService extends GetxService implements AstService { } } + @override + Future dispose() async { + try { + final bool result = await _channel.invokeMethod('dispose'); + + return; + } catch (e) { + Logger.error('停止录音: ${e.toString()}'); + rethrow; + } + } + @override Future initialize({ List? supportedLanguages, @@ -153,6 +300,7 @@ class AzureAstService extends GetxService implements AstService { }); _isInitialized = result; + _setupEventChannel(); Logger.info('Azure 语音识别服务初始化${result ? '成功' : '失败'}'); return result; } catch (e) { diff --git a/lib/modules/speech_test/controllers/speech_test_controller.dart b/lib/modules/speech_test/controllers/speech_test_controller.dart index ec1833c90..5bcfc83e0 100644 --- a/lib/modules/speech_test/controllers/speech_test_controller.dart +++ b/lib/modules/speech_test/controllers/speech_test_controller.dart @@ -164,7 +164,15 @@ class SpeechTestController extends GetxController { } recognitionStatus.value = '识别完成'; break; - + case RecognitionEventType.finalResult1: + // 更新最终结果,显示为正常文本 + finalResult.value = event.text; + intermediateResult.value = ''; // 清空中间结果 + if (event.detectedLanguage.isNotEmpty) { + detectedLanguage.value = event.detectedLanguage; + } + recognitionStatus.value = '识别完成'; + break; case RecognitionEventType.error: isListening.value = false; recognitionStatus.value = '识别错误: ${event.error}'; diff --git a/lib/modules/translation/controllers/translation_controller.dart b/lib/modules/translation/controllers/translation_controller.dart index 1627b595a..c5894390a 100644 --- a/lib/modules/translation/controllers/translation_controller.dart +++ b/lib/modules/translation/controllers/translation_controller.dart @@ -107,6 +107,11 @@ class TranslationController extends GetxController { var appDir; var dir; +// 最新的语音识别文本 + final latestRecognitionText = ''.obs; + +// 检测到的语言代码 + final detectedLanguageCode1 = ''.obs; //int num = 0; Future changeTranslationMode(String mode) async { currentMode.value = mode; @@ -182,7 +187,6 @@ class TranslationController extends GetxController { StreamSubscription? _recognitionSubscription; StreamSubscription? _voiceInteractionSubscription; Timer? _translationDebounceTimer; - // 获取支持的语言列表 Map get supportedLanguages => _languageManager.getChineseNameToAsrCodeMap(); @@ -216,7 +220,8 @@ class TranslationController extends GetxController { // 初始化服务 _initServices(); - + // 初始化语音翻译服务 + _initializeCallModeTranslationService(); // 🔑 reverse模式下不需要手动滚动,ListView会自然显示最新内容 // 参考Agent模块的实现模式 } @@ -350,6 +355,7 @@ class TranslationController extends GetxController { _recognitionSubscription?.cancel(); _recognitionSubscription = recognitionStream.listen(_handleRecognitionEvent); + final bool storage = await _requestStoragePermission(); if (!storage) { Logger.d("Permission", "未授予存储权限,无法提供录音功能"); @@ -386,7 +392,7 @@ class TranslationController extends GetxController { // }); // 重新初始化ASR服务以支持通话音频源 await _astService.initialize(supportedLanguages: callModeLanguages); - await _astService.path("${dir.path}/8_mic.wav"); + // 配置实时翻译参数 await _configureCallModeTranslation(); @@ -617,10 +623,9 @@ class TranslationController extends GetxController { bleManager.openA2DPDecoder(); _audioSourceType = true; isTtsEnabled.value = false; + await _astService.startContinuousTranslation(); + // await _astService.path("${dir.path}/8_mic.wav"); Logger.info('发送ble系统mic和dac(音乐或者通话远端)声音'); - - // 初始化语音翻译服务 - await _initializeCallModeTranslationService(); } else { // 开始连续语音识别 _audioSourceType = false; @@ -650,13 +655,22 @@ class TranslationController extends GetxController { if (event.type == RecognitionEventType.finalResult && event.text.isNotEmpty) { detectedLanguageCode = event.detectedLanguage; + print("处理识别事件:${event.text}"); handleFinalResult(event.text); } else if (event.type == RecognitionEventType.intermediateResult && event.text.isNotEmpty) { if (event.detectedLanguage.isNotEmpty) { detectedLanguageCode = event.detectedLanguage; } + print("处理识别事件:${event.text}"); handleIntermediateResult(event.text); + } else if (event.type == RecognitionEventType.finalResult1 && + event.text.isNotEmpty) { + if (event.detectedLanguage.isNotEmpty) { + detectedLanguageCode1.value = event.detectedLanguage; + } + latestRecognitionText.value = event.text; + print("处理麦:${event.text}"); } else if (event.type == RecognitionEventType.error) { isRecognizing.value = false; Logger.error('语音识别错误: ${event.error}'); @@ -672,7 +686,7 @@ class TranslationController extends GetxController { try { await _asrService.stopContinuousRecognition(); - await _astService.stopContinuousTranslation(true); + await _astService.stopContinuousTranslation(); // 停止ASR活跃时长计时 _stopAsrActiveTracking(); diff --git a/lib/modules/translation/views/translation_view.dart b/lib/modules/translation/views/translation_view.dart index d69278376..47cf1d2e8 100644 --- a/lib/modules/translation/views/translation_view.dart +++ b/lib/modules/translation/views/translation_view.dart @@ -459,67 +459,128 @@ class TranslationView extends GetView { ], ), child: Obx(() { - // 同时模式显示双按钮 - if (controller.currentMode.value == 'faceToFace') { - return Row( - mainAxisAlignment: MainAxisAlignment.center, - children: [ - // 耳机图标按钮 (说话者1) - _buildHoldToTalkButton( - isDarkMode: isDarkMode, - speakerId: 1, - activeSpeaker: controller.activeSpeaker.value, - icon: Icons.headset, // 耳机图标 - label: "headsetUser".tr, // 耳机用户 + return Column( + mainAxisSize: MainAxisSize.min, + children: [ + // 原有的按钮区域 + if (controller.currentMode.value == 'faceToFace') + Row( + mainAxisAlignment: MainAxisAlignment.center, + children: [ + // 耳机图标按钮 (说话者1) + _buildHoldToTalkButton( + isDarkMode: isDarkMode, + speakerId: 1, + activeSpeaker: controller.activeSpeaker.value, + icon: Icons.headset, // 耳机图标 + label: "headsetUser".tr, // 耳机用户 + ), + SizedBox(width: 40.w), + // 手机图标按钮 (说话者2) + _buildHoldToTalkButton( + isDarkMode: isDarkMode, + speakerId: 2, + activeSpeaker: controller.activeSpeaker.value, + icon: Icons.phone_android, // 手机图标 + label: "phoneUser".tr, // 手机用户 + ), + ], + ) + else + Row( + mainAxisAlignment: MainAxisAlignment.center, + children: [ + // 主要控制按钮 - 开始/暂停语音识别 + Obx(() { + final isRecognizing = controller.isRecognizing.value; + return GestureDetector( + onTap: () { + if (isRecognizing) { + controller.stopAll(); + } else { + controller.startRecognition(); + } + }, + child: Container( + width: 60.w, + height: 60.w, + decoration: BoxDecoration( + color: isRecognizing + ? (isDarkMode ? Colors.red[700] : Colors.red) + : (isDarkMode ? AppColors.primary : Colors.blue), + shape: BoxShape.circle, + ), + child: Icon( + isRecognizing ? Icons.pause : Icons.play_arrow, + color: Colors.white, + size: 30.sp, + ), + ), + ); + }), + ], ), - SizedBox(width: 40.w), - // 手机图标按钮 (说话者2) - _buildHoldToTalkButton( - isDarkMode: isDarkMode, - speakerId: 2, - activeSpeaker: controller.activeSpeaker.value, - icon: Icons.phone_android, // 手机图标 - label: "phoneUser".tr, // 手机用户 + + // 添加间距 + SizedBox(height: 16.h), + + // 语音识别结果显示框 + Container( + width: double.infinity, + padding: EdgeInsets.all(12.w), + decoration: BoxDecoration( + color: isDarkMode ? Colors.grey[800] : Colors.grey[100], + borderRadius: BorderRadius.circular(8.r), + border: Border.all( + color: isDarkMode ? Colors.grey[600]! : Colors.grey[300]!, + width: 1, + ), ), - ], - ); - } - // 普通模式显示单按钮 - else { - return Row( - mainAxisAlignment: MainAxisAlignment.center, - children: [ - // 主要控制按钮 - 开始/暂停语音识别 - Obx(() { - final isRecognizing = controller.isRecognizing.value; - return GestureDetector( - onTap: () { - if (isRecognizing) { - controller.stopAll(); - } else { - controller.startRecognition(); - } - }, - child: Container( - width: 60.w, - height: 60.w, - decoration: BoxDecoration( - color: isRecognizing - ? (isDarkMode ? Colors.red[700] : Colors.red) - : (isDarkMode ? AppColors.primary : Colors.blue), - shape: BoxShape.circle, + child: Obx(() { + // 显示最新的语音识别结果 + final recognitionText = controller.latestRecognitionText.value; + final detectedLanguage = controller.detectedLanguageCode1.value; + + return Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + // 标题 + Text( + '麦头语音识别结果', + style: TextStyle( + fontSize: 12.sp, + color: isDarkMode ? Colors.grey[400] : Colors.grey[600], + fontWeight: FontWeight.w500, + ), ), - child: Icon( - isRecognizing ? Icons.pause : Icons.play_arrow, - color: Colors.white, - size: 30.sp, + SizedBox(height: 4.h), + // 识别文本 - 这里显示 finalResult1 的内容 + Text( + recognitionText.isEmpty ? '等待语音输入...' : recognitionText, + style: TextStyle( + fontSize: 14.sp, + color: isDarkMode ? Colors.white : Colors.black87, + height: 1.3, + ), ), - ), + // 检测到的语言(如果有) + if (detectedLanguage.isNotEmpty) ...[ + SizedBox(height: 4.h), + Text( + '检测语言: $detectedLanguage', + style: TextStyle( + fontSize: 10.sp, + color: + isDarkMode ? Colors.grey[500] : Colors.grey[500], + ), + ), + ], + ], ); }), - ], - ); - } + ), + ], + ); }), ); } diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt index 0d85eda9c..17efe005a 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -144,7 +144,7 @@ class AzureAsrHelper(private val context: Context) { setProperty("Speech_SegmentationStrategy", "Time") } // 录音文件类 - recordfile = RecordFile; + recordfile = RecordFile(); return true diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt index ad89b9409..6c8d80041 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt @@ -64,7 +64,7 @@ class IntegratedSpeechTranslationService( private var audioProcessor: AudioProcessor? = null private var audioConfig: AudioConfig? = null // 录音文件处理 - var recordfile: RecordFile? = null + var recordfile1: RecordFile? = null var filePath: String? = null // 配置管理 private var serviceConfig = ServiceConfiguration() @@ -91,7 +91,9 @@ class IntegratedSpeechTranslationService( var enableContinuousRecognition: Boolean = true, var enableAutoLanguageDetection: Boolean = false, var maxRetryAttempts: Int = 3, - var translationTimeout: Long = 10000L + var translationTimeout: Long = 10000L, + // 新增:控制是否播放合成的音频 + var enableAudioPlayback: Boolean = false ) /** @@ -122,6 +124,8 @@ class IntegratedSpeechTranslationService( fun onRecognitionStopped() fun onStateChanged(component: String, isActive: Boolean) fun onError(component: String, error: String) + // 新增:返回合成的音频数据 + fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) } /** @@ -190,8 +194,8 @@ class IntegratedSpeechTranslationService( withContext(Dispatchers.Main) { callback.onServiceInitialized() } -startContinuousTranslation() - recordfile = RecordFile; + //startContinuousTranslation() + recordfile1 = RecordFile(); return@withContext true } catch (e: Exception) { Log.e(TAG, "初始化失败", e) @@ -281,7 +285,7 @@ startContinuousTranslation() recognizing.addEventListener { _, event -> if (event.result.text.isNotEmpty()) { val confidence = extractConfidence(event.result) - Log.d(TAG, "识别中事件: ${event.result.text}") + Log.d(TAG, "识别中事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") eventCallback?.onRecognizing( event.result.text, serviceConfig.sourceLanguage, @@ -301,7 +305,7 @@ startContinuousTranslation() serviceConfig.sourceLanguage, confidence ) - Log.d(TAG, "识别完成事件: ${event.result.text}") + Log.d(TAG, "识别完成事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") // 触发翻译流程 launch { processTranslationAndSynthesis(event.result.text) @@ -353,8 +357,16 @@ startContinuousTranslation() */ private fun setupSpeechSynthesizer() { synthesizer?.close() - synthesizer = - SpeechSynthesizer(speechConfig, AudioConfig.fromDefaultSpeakerOutput()).apply { + + // 根据配置决定音频输出方式 + val audioConfig = if (serviceConfig.enableAudioPlayback) { + AudioConfig.fromDefaultSpeakerOutput() + } else { + // 不输出到扬声器,只生成音频数据 + null + } + + synthesizer = SpeechSynthesizer(speechConfig, audioConfig).apply { // 合成开始事件 SynthesisStarted.addEventListener { _, _ -> @@ -374,8 +386,7 @@ startContinuousTranslation() // 合成完成事件 SynthesisCompleted.addEventListener { _, event -> Log.d(TAG, "合成完成事件") - serviceState.isSynthesizing.set(false) - eventCallback?.onSynthesisCompleted("") + // eventCallback?.onSynthesisCompleted("") eventCallback?.onStateChanged("Synthesis", false) } @@ -389,13 +400,13 @@ startContinuousTranslation() } } - Log.d(TAG, "语音合成器设置完成") + Log.d(TAG, "语音合成器设置完成,播放模式: ${serviceConfig.enableAudioPlayback}") } /** * 开始连续语音翻译 */ - suspend fun startContinuousTranslation(): Boolean { + fun startContinuousTranslation(): Boolean { if (!serviceState.isInitialized.get()) { eventCallback?.onError("Service", "服务未初始化") return false @@ -426,10 +437,10 @@ startContinuousTranslation() fun enableRecord(filePath: String) { Log.i(TAG, "开启录音:") - recordfile!!.closeFile(true) + recordfile1!!.closeFile(true) - recordfile!!.creatingFiles(filePath) + recordfile1!!.creatingFiles(filePath) } @@ -440,7 +451,7 @@ startContinuousTranslation() try { recognizer?.stopContinuousRecognitionAsync() audioProcessor?.stopRecording() - recordfile!!.closeFile(true) + recordfile1!!.closeFile(true) // 等待当前处理完成 while (serviceState.isTranslating.get() || serviceState.isSynthesizing.get()) { Thread.sleep(100) @@ -510,11 +521,13 @@ startContinuousTranslation() /** * 语音合成 + * @param text 要合成的文本 + * @return 返回合成的音频数据,如果合成失败则返回null */ - private suspend fun synthesizeText(text: String) { + private suspend fun synthesizeText(text: String): ByteArray? { if (serviceState.isSynthesizing.get()) { Log.w(TAG, "语音合成正在进行中") - return + return null } try { @@ -524,20 +537,32 @@ startContinuousTranslation() val ssml = generateOptimizedSsml(text) - // 使用同步方法确保合成完成后再播放 - val result = synthesizer?.SpeakSsml(ssml) + // 始终使用异步方法获取音频数据 + val result = synthesizer?.SpeakSsmlAsync(ssml)?.get() if (result?.reason == ResultReason.SynthesizingAudioCompleted) { - Log.d(TAG, "语音合成成功,音频已播放") - eventCallback?.onSynthesisCompleted(text) + val playbackStatus = if (serviceConfig.enableAudioPlayback) "音频已播放" else "音频已合成但未播放" + Log.d(TAG, "语音合成成功,$playbackStatus") + + // 获取音频数据 + val audioData = result.audioData + if (audioData != null && audioData.isNotEmpty()) { + // 通过回调返回音频数据 + eventCallback?.onSynthesisAudioGenerated(text, audioData) + Log.d(TAG, "音频数据大小: ${audioData.size} 字节") + } + + return audioData } else { Log.e(TAG, "语音合成失败: ${result?.reason}") eventCallback?.onSynthesisFailed(text, "合成失败: ${result?.reason}") + return null } } catch (e: Exception) { Log.e(TAG, "语音合成失败", e) eventCallback?.onSynthesisFailed(text, "合成失败: ${e.message}") + return null } finally { serviceState.isSynthesizing.set(false) } @@ -693,6 +718,68 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { Log.d(TAG, "语言对已切换: ${newConfig.sourceLanguage} <-> ${newConfig.targetLanguage}") } + /** + * 设置是否启用音频播放 + * @param enabled true表示启用音频播放,false表示只合成不播放 + */ + fun setAudioPlaybackEnabled(enabled: Boolean) { + serviceConfig.enableAudioPlayback = enabled + Log.d(TAG, "音频播放设置更新: $enabled") + + // 重新设置语音合成器以应用新配置 + if (serviceState.isInitialized.get()) { + setupSpeechSynthesizer() + } + + eventCallback?.onStateChanged("AudioPlayback", enabled) + } + + /** + * 获取当前音频播放状态 + * @return true表示启用音频播放,false表示禁用 + */ + fun isAudioPlaybackEnabled(): Boolean { + return serviceConfig.enableAudioPlayback + } + + /** + * 单独的语音合成方法,只合成不播放,返回音频数据 + * @param text 要合成的文本 + * @return 合成的音频数据,失败时返回null + */ + suspend fun synthesizeTextToAudio(text: String): ByteArray? = withContext(Dispatchers.IO) { + try { + Log.d(TAG, "开始纯音频合成: $text") + + val ssml = generateOptimizedSsml(text) + + // 创建一个临时的合成器,输出到内存而不是扬声器 + val audioConfig = AudioConfig.fromStreamOutput(AudioOutputStream.createPullStream()) + val tempSynthesizer = SpeechSynthesizer(speechConfig, audioConfig) + + try { + val result = tempSynthesizer.SpeakSsmlAsync(ssml).get() + + if (result?.reason == ResultReason.SynthesizingAudioCompleted) { + val audioData = result.audioData + if (audioData != null && audioData.isNotEmpty()) { + Log.d(TAG, "纯音频合成成功,大小: ${audioData.size} 字节") + return@withContext audioData + } + } else { + Log.e(TAG, "纯音频合成失败: ${result?.reason}") + } + } finally { + tempSynthesizer.close() + } + + return@withContext null + } catch (e: Exception) { + Log.e(TAG, "纯音频合成异常", e) + return@withContext null + } + } + /** * 推送外部音频数据 */ @@ -774,9 +861,9 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { val audioData = audioQueue.poll(100, java.util.concurrent.TimeUnit.MILLISECONDS) audioData?.let { - Log.d(TAG, "音频处理: ${it}") + Log.d(TAG, "音频处理: ${it.size}") pushAudioStream?.write(it) - recordfile?.saveAudioDataToWav(it) + recordfile1?.saveAudioDataToWav(it) } } catch (e: InterruptedException) { Thread.currentThread().interrupt() diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt index acd098897..4263376e1 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -680,9 +680,16 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { result.error("ENABLERECORD_ERROR", e.message, null) } } - + "startContinuousTranslation" -> { + try { + FileLogger.d(tag, "开启翻译") + azureAstHelper.startContinuousTranslation() + result.success(true) + } catch (e: Exception) { + result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) + } + } "stopContinuousTranslation" -> { - val isSave = call.argument("isSave") ?: false try { FileLogger.d(tag, "停止翻译") azureAstHelper.stopContinuousTranslation() @@ -691,6 +698,23 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) } } + "dispose" -> { + try { + FileLogger.d(tag, "释放AST资源") + azureAstHelper.dispose() + result.success(true) + } catch (e: Exception) { + result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) + } + } + + "recognizeCallback" -> { + + + FileLogger.d(tag, "recognizeCallback:") // + result.success(true) + } + "path" -> { val filePath = call.argument("filePath") ?: "" try { @@ -769,6 +793,14 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } override fun onRecognized(text: String, language: String, confidence: Float) { + FileLogger.d(tag, "识别到文本: $text, 语言: $language, 置信度: $confidence") + sendAsrEvent( + mapOf( + "type" to "result1", + "text" to text, + "detectedLanguage" to language + ) + ) // sendAstEvent( // mapOf( // "type" to "recognized", @@ -823,6 +855,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } override fun onSynthesisCompleted(text: String) { + FileLogger.d(tag, "语音合成完成,文本=${text}") + + + // sendAstEvent( // mapOf( // "type" to "synthesisCompleted", @@ -830,7 +866,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { // ) // ) } - + override fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) { + FileLogger.d(tag, "语音合成音频生成,文本=${text},音频数据大小=${audioData.size}") + BleService.writeExternalAudioData(audioData) + } override fun onSynthesisFailed(text: String, error: String) { // sendAstEvent( // mapOf( diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/RecordFile.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/RecordFile.kt index 0d0cf6afc..d0b69ccb1 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/RecordFile.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/RecordFile.kt @@ -9,11 +9,15 @@ import java.io.FileInputStream import java.util.concurrent.LinkedBlockingQueue import java.util.concurrent.atomic.AtomicBoolean -object RecordFile { +/** + * 音频录制文件管理类 + * 支持单声道和双声道录制,提供WAV文件格式输出 + */ +class RecordFile { private var currentAudioFile: File? = null private var fos: FileOutputStream? = null - // 新增:用于异步写入的队列和线程 + // 用于异步写入的队列和线程 private val writeQueue = LinkedBlockingQueue() private val isWriting = AtomicBoolean(false) private var writeThread: Thread? = null @@ -26,7 +30,7 @@ object RecordFile { private var sampleRate = 16000 private var channels = 1 // 1=单声道, 2=立体声 - // 新增:用于存储最后成功保存的文件 + // 用于存储最后成功保存的文件 private var lastSavedFile: File? = null private var lastFileTimestamp: String = "" var isPause = false @@ -44,16 +48,15 @@ object RecordFile { private var leftChannelBytesWritten = 0 private var rightChannelBytesWritten = 0 - /** * 设置音频配置 * @param sampleRate 采样率 (8000, 16000, 24000, 32000, 44100, 48000) * @param channels 声道数 (1=单声道, 2=立体声) */ - internal fun setAudioConfig(sampleRate: Int, channels: Int) { + fun setAudioConfig(sampleRate: Int, channels: Int) { this.sampleRate = sampleRate this.channels = channels - Log.d("tag", "音频配置已更新: 采样率=${sampleRate}Hz, 声道数=${channels}") + Log.d("RecordFile", "音频配置已更新: 采样率=${sampleRate}Hz, 声道数=${channels}") } /** diff --git a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt index cda3594a4..af59cbd22 100644 --- a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt +++ b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt @@ -1170,10 +1170,10 @@ object BleService { // //重新编码 // opusManager?.writeEncodeStream(rightBuffer) - notifyAudioDataReceived1(leftBuffer) + //notifyAudioDataReceived1(leftBuffer) //回调 notifyAudioDataReceived(rightBuffer)//对方的 - // notifyAudioDataReceived1(leftBuffer)//自己的 + notifyAudioDataReceived1(leftBuffer)//自己的 } else if (option!!.getChannel() == 1) { notifyAudioDataReceived(data) } else {