From 906234c4686f8443e66a967ca60c28e6070afdd5 Mon Sep 17 00:00:00 2001 From: wolfplus Date: Thu, 19 Jun 2025 17:21:56 +0100 Subject: [PATCH 1/5] add --- android/settings.gradle.kts | 4 +++- .../controllers/pairing_controller.dart | 21 ++++++++++++++-- .../agent_service/android/build.gradle.kts | 1 + .../agent_service/AgentService.kt | 24 +++++++++---------- 4 files changed, 35 insertions(+), 15 deletions(-) diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts index cf7591c68..acc6518c9 100644 --- a/android/settings.gradle.kts +++ b/android/settings.gradle.kts @@ -40,6 +40,7 @@ include(":classic_bluetooth") include(":agent_service") include(":speech") include(":chat_api") +include(":bytedance_speech") // 设置azure_speech项目的路径 project(":azure_speech").projectDir = file("../local_plugins/azure_speech/android") @@ -50,4 +51,5 @@ project(":ble_service").projectDir = file("../local_plugins/ble_service/android" project(":classic_bluetooth").projectDir = file("../local_plugins/classic_bluetooth/android") project(":agent_service").projectDir = file("../local_plugins/agent_service/android") project(":speech").projectDir = file("../local_plugins/speech/android") -project(":chat_api").projectDir = file("../local_plugins/chat_api/android") \ No newline at end of file +project(":chat_api").projectDir = file("../local_plugins/chat_api/android") +project(":bytedance_speech").projectDir = file("../local_plugins/bytedance_speech/android") \ No newline at end of file diff --git a/lib/modules/pairing/controllers/pairing_controller.dart b/lib/modules/pairing/controllers/pairing_controller.dart index 34574e0f0..47e2a00b6 100644 --- a/lib/modules/pairing/controllers/pairing_controller.dart +++ b/lib/modules/pairing/controllers/pairing_controller.dart @@ -506,7 +506,15 @@ class PairingController extends GetxController { /// 导航到主页 void _navigateToHome() { - Future.delayed(const Duration(seconds: 1), () { + Future.delayed(const Duration(seconds: 1), () async { + // 启动AgentService服务 + try { + Logger.i(_tag, '启动AgentService服务'); + await _bleManager.startAgentService(); + } catch (e) { + Logger.e(_tag, '启动AgentService失败: $e'); + } + Get.offAllNamed(Routes.home); }); } @@ -563,8 +571,17 @@ class PairingController extends GetxController { } /// 用户点击跳过按钮 - void onSkipTap() { + void onSkipTap() async { Logger.i(_tag, '用户选择跳过配对'); + + // 启动AgentService服务 + try { + Logger.i(_tag, '启动AgentService服务'); + await _bleManager.startAgentService(); + } catch (e) { + Logger.e(_tag, '启动AgentService失败: $e'); + } + Get.offAllNamed(Routes.home); } diff --git a/local_plugins/agent_service/android/build.gradle.kts b/local_plugins/agent_service/android/build.gradle.kts index 9d8dea1bb..d193bfbae 100644 --- a/local_plugins/agent_service/android/build.gradle.kts +++ b/local_plugins/agent_service/android/build.gradle.kts @@ -50,4 +50,5 @@ dependencies { add("compileOnly", project(":speech")) add("compileOnly", project(":chat_api")) add("compileOnly", project(":location_service")) + add("compileOnly", project(":bytedance_speech")) } \ No newline at end of file diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt index 00055f5eb..009ccd4ba 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt @@ -19,7 +19,7 @@ import android.media.MediaPlayer import com.deep_voice.speech.tts.TtsEvent import com.deep_voice.speech.tts.TtsEventListener import com.deep_voice.speech.tts.TtsEventType -import com.yunqiinnovation.azure_speech.AzureTtsHelper +import com.deep_voice.bytedance_speech.BytedanceTTS /** * 代理服务事件监听接口 */ @@ -52,8 +52,8 @@ object AgentService : CoroutineScope { // Azure服务 private var azureAsrHelper: AzureAsrHelper? = null - // 使用AzureTtsHelper作为唯一的TTS实现 - private var ttsService: AzureTtsHelper? = null + // 使用BytedanceTTS作为唯一的TTS实现 + private var ttsService: BytedanceTTS? = null // ChatAPI服务 - 使用新的ChatApiService private lateinit var chatApiService: ChatApiService @@ -216,20 +216,20 @@ object AgentService : CoroutineScope { val ttsAppToken = config["volcanoToken"]?.toString() ?: "" val ttsLanguage = config["ttsLanguage"]?.toString() ?: "zh-CN" - // 创建并初始化AzureTtsHelper - val azureTtsHelper = AzureTtsHelper(context) - ttsService = azureTtsHelper + // 创建并初始化BytedanceTTS + val bytedanceTts = BytedanceTTS(context) + ttsService = bytedanceTts - // 初始化Azure TTS - val success = azureTtsHelper.initialize( - ttsAppId = "", // Azure TTS不需要appId - ttsAppToken = config["azureSpeechKey"]?.toString() ?: "", // Azure需要subscription key - ttsResource = config["azureSpeechRegion"]?.toString() ?: "", // Azure需要region信息 + // 初始化Bytedance TTS + val success = bytedanceTts.initialize( + ttsAppId = ttsAppId, + ttsAppToken = ttsAppToken, + ttsResource = "", // Bytedance TTS不需要resource参数 language = ttsLanguage ) if (!success) { - Log.e(TAG, "Azure TTS服务初始化失败") + Log.e(TAG, "Bytedance TTS服务初始化失败") } // 添加TTS事件监听 From 9ee7fde0b07d76fef128dd0ee039cf4088e4568c Mon Sep 17 00:00:00 2001 From: tanlongsheng <252620078@qq.com> Date: Sat, 21 Jun 2025 17:07:11 +0800 Subject: [PATCH 2/5] =?UTF-8?q?=E6=B7=BB=E5=8A=A0=E5=88=86=E4=BA=AB?= =?UTF-8?q?=E9=93=BE=E6=8E=A5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../meeting/views/share_bottom_sheet.dart | 54 +++++++++++++++---- 1 file changed, 44 insertions(+), 10 deletions(-) diff --git a/lib/modules/meeting/views/share_bottom_sheet.dart b/lib/modules/meeting/views/share_bottom_sheet.dart index 42b377255..318d40df7 100644 --- a/lib/modules/meeting/views/share_bottom_sheet.dart +++ b/lib/modules/meeting/views/share_bottom_sheet.dart @@ -72,16 +72,42 @@ class _ShareBottomSheetState extends State { child: Column( children: [ _box([ - _shareItem(Icons.link, '分享链接', isBorder: false, onTap: () { - final GetStorage storage = GetStorage(); - String token = storage.read("logintoken") ?? ''; - String userId = ''; - Map? userInfo = storage.read("user_info"); - if (userInfo != null) { - userId = userInfo['user']['uid']; - } - String id = '${userId}_${_controller.meetingData.value.id}'; - }), + _shareItem( + Icons.link, + '分享链接', + isBorder: false, + onTap: () async { + Get.back(); + final GetStorage storage = GetStorage(); + String token = storage.read("logintoken") ?? ''; + String userId = ''; + Map? userInfo = storage.read("user_info"); + if (userInfo != null) { + userId = userInfo['user']['uid']; + } + + String id = + '${userId}_${_controller.meetingData.value.id}'; + + final thumbnail = await getAssetThumbnailFile( + 'assets/images/headphone_dark.png', + 'share_thumb.png', + ); + + final result = await SharePlus.instance.share(ShareParams( + title: _controller.meetingData.value.title, + subject: + 'DeepSound WEB,由DeepSound.Ai提供支持-实时记录会议纪要,多语言翻译。', + uri: Uri.parse( + 'http://web.ideapsound.com/share/$id/$token', + ), + previewThumbnail: thumbnail, + )); + if (result.status == ShareResultStatus.success) { + EasyLoading.showToast('分享完成!'); + } + }, + ), ]), _box([ _shareItem(Icons.copy, '复制转写', onTap: () { @@ -223,4 +249,12 @@ class _ShareBottomSheetState extends State { } return text; } + + Future getAssetThumbnailFile(String assetPath, String fileName) async { + final byteData = await rootBundle.load(assetPath); + final tempDir = await getTemporaryDirectory(); + final file = File('${tempDir.path}/$fileName'); + await file.writeAsBytes(byteData.buffer.asUint8List()); + return XFile(file.path); + } } From 87d5fbaa937e9cfe509e4a10df638e04fe582060 Mon Sep 17 00:00:00 2001 From: fdp <1286779656@qq.com> Date: Sat, 21 Jun 2025 19:23:19 +0800 Subject: [PATCH 3/5] =?UTF-8?q?=E4=BC=98=E5=8C=96=E4=BC=9A=E8=AE=AE?= =?UTF-8?q?=E5=8A=A9=E6=89=8B=EF=BC=8C=E7=BF=BB=E8=AF=91=E4=B8=8D=E5=AF=BC?= =?UTF-8?q?=E5=85=A5=E6=96=87=E4=BB=B6=E5=8D=A1=E4=BD=8F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- lib/modules/FTFTranslation/README.md | 160 ++++++++++++ .../bindings/FTFTranslation_binding.dart | 10 + .../FTFTranslation_controller.dart | 241 ++++++++++++++++++ .../views/FTFTranslation_view.dart | 212 +++++++++++++++ .../controllers/meeting_controller.dart | 7 +- .../meeting_record_controller.dart | 5 +- lib/modules/meeting/views/meeting_view.dart | 88 ++++--- lib/modules/settings/views/settings_view.dart | 22 ++ .../controllers/translation_controller.dart | 8 +- .../translation/views/translation_view.dart | 4 +- lib/routes/app_pages.dart | 9 + lib/routes/app_routes.dart | 1 + .../yunqiinnovation/ble_service/BleService.kt | 129 +++++----- 13 files changed, 780 insertions(+), 116 deletions(-) create mode 100644 lib/modules/FTFTranslation/README.md create mode 100644 lib/modules/FTFTranslation/bindings/FTFTranslation_binding.dart create mode 100644 lib/modules/FTFTranslation/controllers/FTFTranslation_controller.dart create mode 100644 lib/modules/FTFTranslation/views/FTFTranslation_view.dart diff --git a/lib/modules/FTFTranslation/README.md b/lib/modules/FTFTranslation/README.md new file mode 100644 index 000000000..c41004a52 --- /dev/null +++ b/lib/modules/FTFTranslation/README.md @@ -0,0 +1,160 @@ +# Realtime 实时语音聊天模块 + +## ⚠️ 当前状态 + +**此模块为功能预览版本,提供完整的实时语音对话功能。** + +通过原生插件实现语音识别、AI对话和语音合成的完整语音交互体验。 + +## 概述 + +Realtime模块实现了类似ChatGPT app的实时语音对话功能,当前版本提供: + +- ✅ **完整语音交互** - 实时语音识别、AI对话和语音合成 +- ✅ **状态管理** - 聆听、思考、回答状态的可视化 +- ✅ **动画效果** - 基于Shader的流动云效果,支持音频振幅驱动 +- ✅ **字幕显示** - 可选择开启/关闭的字幕模式 +- ✅ **智能UI** - 根据连接状态自动调整按钮可用性 + +## 界面设计 + +### 布局结构 +- **顶部状态区域** - 显示当前状态和字幕模式切换按钮 +- **中间可视化区域** - 圆形Shader动画,根据音频振幅呈现呼吸效果 +- **底部控制区域** - 主控制按钮和状态指示器 + +### 状态指示 +- 🔵 **聆听状态** - 蓝色动画,接收用户语音输入 +- 🟠 **思考状态** - 橙色动画,AI处理中 +- 🟢 **回答状态** - 绿色动画,AI语音输出 +- ⚪ **空闲状态** - 灰色静态,等待交互 + +### 字幕模式 +- **默认关闭** - 初始状态下不显示任何字幕内容 +- **手动切换** - 通过右上角按钮手动开启/关闭字幕 +- **完整隐藏** - 关闭时字幕区域完全不显示,不占用界面空间 +- **智能布局** - 开启时圆形动画自动上移,字幕区域占据下半屏 + +### 交互控制 +- **录音按钮** - 仅在成功连接AI服务器且开始录音后可用 +- **连接状态** - 未连接时录音按钮显示不可用状态 +- **状态反馈** - 按钮颜色和可点击性根据连接状态动态调整 + +## 使用方法 + +### 路由导航 + +```dart +// 跳转到实时语音对话页面 +Get.toNamed(Routes.realtime); +``` + +### 交互流程 + +1. **自动连接** - 页面加载后自动尝试连接AI服务器 +2. **等待连接** - 录音按钮在连接成功前保持不可用状态 +3. **开始录音** - 连接成功后自动开始录音,录音按钮变为可用 +4. **语音对话** - 用户说话 → AI处理 → 语音回复 +5. **字幕控制** - 可随时通过右上角按钮切换字幕显示 + +## 技术架构 + +### 控制器 (RealtimeController) +- **连接管理**: WebSocket连接到AI服务器 +- **状态同步**: `isConnected`, `isListening`, `isSpeaking` +- **字幕控制**: `isSubtitleMode` 默认为false,用户手动控制 +- **消息管理**: 智能去重,避免流式响应重复显示 +- **音频处理**: 原生层计算RMS值,Flutter层接收并控制动画 + +### 视图组件 +- **RealtimeView** - 主视图容器,响应式布局 +- **ShaderMicFlow** - 基于GLSL Shader的流动云动画组件 +- **AutoScrollList** - 自动滚动的字幕列表 + +### 原生集成 +- **RealtimeService** - 原生音频处理和WebSocket通信 +- **音频格式** - 16kHz, 1声道, 16位PCM +- **实时传输** - 低延迟音频流处理 +- **RMS计算** - 原生层实时计算音频振幅 + +## 默认行为 + +### 字幕模式 +- **初始状态**: 关闭 (`isSubtitleMode.value = false`) +- **显示逻辑**: 仅在用户手动开启时显示字幕内容 +- **空间占用**: 关闭时字幕区域高度为0,不影响其他UI元素 +- **切换效果**: 平滑动画过渡,圆形动画位置和大小同步调整 + +### 按钮状态 +- **录音按钮**: 默认不可用,连接成功且开始录音后才可点击 +- **状态指示**: 通过颜色和图标变化反映当前可用性 +- **交互反馈**: 不可用时点击无效果,避免误操作 + +## 开发计划 + +### 已完成功能 ✅ +- [x] 完整语音交互流程 +- [x] 实时WebSocket通信 +- [x] Shader流动云动画 +- [x] RMS音频振幅计算 +- [x] 智能UI状态管理 +- [x] 字幕模式控制 +- [x] 消息去重优化 + +### 计划增强 ⏳ +- [ ] 多语言语音识别 +- [ ] 语音中断检测 +- [ ] 音频质量自适应 +- [ ] 离线降级处理 +- [ ] 语音情感识别 + +## 注意事项 + +1. **网络依赖** - 需要连接到AI服务器 (ws://192.168.1.11:8000/ws) +2. **麦克风权限** - 需要获取麦克风使用权限 +3. **音频会话** - 自动管理音频会话,支持蓝牙设备 +4. **资源管理** - 页面关闭时自动清理音频资源和网络连接 + +## 自定义配置 + +### 修改服务器地址 + +```dart +// 在RealtimeController中修改连接URL +await _realtimeService.initialize( + serverUrl: 'ws://your-server:port/ws', + // ... 其他参数 +); +``` + +### 调整UI布局 + +```dart +// 修改字幕模式下圆形动画位置 +alignment: controller.isSubtitleMode.value + ? const Alignment(0, -0.9) // 调整Y轴位置 + : const Alignment(0, -0.25), +``` + +### 音频参数调优 + +```dart +// 在初始化时调整音频参数 +sampleRate: 16000, // 采样率 +channels: 1, // 声道数 +bitsPerSample: 16, // 位深度 +``` + +## 技术栈 + +- **Flutter SDK**: >= 3.0.0 +- **GetX**: >= 4.6.5 (状态管理) +- **Flutter Shaders**: >= 0.0.6 (Shader动画) +- **原生插件**: Realtime服务 (音频处理) +- **WebSocket**: 实时通信协议 + +## 版本信息 + +- **当前版本**: v1.0.0 (Production) +- **最后更新**: 2024年 +- **状态**: 完整功能实现 \ No newline at end of file diff --git a/lib/modules/FTFTranslation/bindings/FTFTranslation_binding.dart b/lib/modules/FTFTranslation/bindings/FTFTranslation_binding.dart new file mode 100644 index 000000000..85e44791c --- /dev/null +++ b/lib/modules/FTFTranslation/bindings/FTFTranslation_binding.dart @@ -0,0 +1,10 @@ +import 'package:get/get.dart'; +import '../controllers/FTFTranslation_controller.dart'; + +class FTFTranslationBinding extends Bindings { + @override + void dependencies() { + // 注册实时语音聊天控制器 + Get.lazyPut(() => FTFTranslationController()); + } +} diff --git a/lib/modules/FTFTranslation/controllers/FTFTranslation_controller.dart b/lib/modules/FTFTranslation/controllers/FTFTranslation_controller.dart new file mode 100644 index 000000000..53cffe68d --- /dev/null +++ b/lib/modules/FTFTranslation/controllers/FTFTranslation_controller.dart @@ -0,0 +1,241 @@ +import 'dart:async'; +import 'dart:typed_data'; + +import 'package:get/get.dart'; +import '../../../core/utils/logger.dart'; +import '../../../data/services/asr_service.dart'; +import '../../../data/services/speech_impl/xunfei_asr_service.dart'; +import '../../../data/services/tts_service.dart'; +import '../../../data/services/language_manager.dart'; +import '../../../data/services/volcano_translation_service.dart'; + +class FTFTranslationController extends GetxController + with GetSingleTickerProviderStateMixin { + final messages = [].obs; + final isChineseInput = true.obs; + final leftLanguage = '中文'.obs; + final rightLanguage = 'English'.obs; + final isLeftInput = true.obs; + + final supportedLanguages = const [ + '中文', + 'English', + '日本語', + '한국어', + 'Español', + ]; + + late final AsrService _asrService; + + StreamSubscription? _recognitionSubscription; + + RxString intermediateContent = ''.obs; + RxString finalContent = ''.obs; + BytesBuilder _originaBytes = BytesBuilder(); //原始音频 + + final VolcanoTranslationService _translationService = + Get.find(); + final LanguageManager _languageManager = Get.find(); + // 当前使用的TTS服务 + late TtsService _ttsService; + final isTranslating = false.obs; + final sourceLanguageCode = 'zh-CN'.obs; + final targetLanguageCode = 'en-US'.obs; + final isTtsEnabled = true.obs; + + @override + void onInit() { + super.onInit(); + Get.put(XunfeiAsrService()); + final TtsService _ttsService = Get.find(); + _asrService = Get.find(); + } + + @override + void onClose() { + _asrService.dispose(); + _recognitionSubscription?.cancel(); + super.onClose(); + } + + //开始录音 + Future _startRecorder() async { + final recognitionStream = + await _asrService.startContinuousRecognition(false); + _recognitionSubscription = recognitionStream.listen( + _handleRecognitionEvent, + ); + } + + //结束录音 + Future _stopRecorder() async { + await _asrService.stopContinuousRecognition(); + } + + // 处理语音识别事件 + void _handleRecognitionEvent(RecognitionEvent event) { + switch (event.type) { + case RecognitionEventType.intermediateResult: + intermediateContent.value = event.text; + handleFinalResult(event.text); + break; + case RecognitionEventType.finalResult: + finalContent.value += _spliceText(event); + intermediateContent.value = ''; + _originaBytes.add(event.audio!); + handleIntermediateResult(event.text); + break; + default: + // 不做任何处理 + break; + } + } + + // 翻译文本 + Future translateText(String sourceText, + {bool isFinal = false}) async { + if (sourceText.isEmpty) return null; + + isTranslating.value = true; + String? translationResult; + + try { + // 根据检测到的语言确定源语言和目标语言 + final bool shouldSwap = + intermediateContent.value == sourceLanguageCode.value; + final detectedSourceLanguageCode = + shouldSwap ? sourceLanguageCode.value : targetLanguageCode.value; + final detectedTargetLanguageCode = + shouldSwap ? targetLanguageCode.value : sourceLanguageCode.value; + + // 调用翻译服务 + translationResult = await _translationService.translateText( + text: sourceText, + sourceLanguageCode: detectedSourceLanguageCode, + targetLanguageCode: detectedTargetLanguageCode, + ); + + if (translationResult != null) { + // 查找匹配的项目 - 优先使用timestamp查找 + int index = -1; + + // 播放TTS - 只在最终结果时播放 + if (isTtsEnabled.value && isFinal) { + // Logger.info('播放翻译文本: ${translationResult}'); + await playTranslatedText(translationResult); + } + } + } catch (e) { + Logger.error('翻译失败: ${e.toString()}'); + } finally { + isTranslating.value = false; + } + + return translationResult; + } + + // 处理中间识别结果 + void handleIntermediateResult(String text) { + if (text.isEmpty) return; + + translateText(text, isFinal: false); + } + + // 处理最终识别结果 + Future handleFinalResult(String text) async { + if (text.isEmpty) return; + + // 使用项目的时间戳进行翻译 + translateText(text, isFinal: true); + } + + // 播放翻译文本 + Future playTranslatedText(String text) async { + if (text.isEmpty) return; + + try { + final bool shouldSwap = + intermediateContent.value == targetLanguageCode.value; + final detectedTargetLanguageCode = + shouldSwap ? sourceLanguageCode.value : targetLanguageCode.value; + final voiceName = _languageManager + .getTtsVoiceNameByAsrCode(detectedTargetLanguageCode) ?? + 'en-US-AriaNeural'; + + await _ttsService.setVoice(voiceName); + await _ttsService.speakOnce(text); + } catch (e) { + Logger.error('播放翻译失败: ${e.toString()}'); + } + } + + String _spliceText(RecognitionEvent event) { + String eventText = event.text; + String text = ''; + if (finalContent.value.isNotEmpty) { + if (event.role != '') { + final punctuations = { + ',', + '。', + '!', + '?', + ';', + ':', + ',', + '.', + ':', + ';', + '!', + '?' + }; + final firstChar = eventText[0]; + if (punctuations.contains(firstChar)) { + text += '$firstChar\n'; + eventText = eventText.substring(1); + } else { + text += '\n'; + } + text += '说话人${event.role}:'; + } + } else { + text += '说话人${event.role}:'; + } + text += eventText; + return text; + } + + void addMessage(String text) { + final fromLang = + isLeftInput.value ? leftLanguage.value : rightLanguage.value; + final toLang = isLeftInput.value ? rightLanguage.value : leftLanguage.value; + + messages.add( + FTFTranslationMessage( + original: text, + translated: '[$fromLang → $toLang] $text (翻译模拟)', + isChineseSpeaker: isLeftInput.value, + ), + ); + } + + void swapLanguages() { + final temp = leftLanguage.value; + leftLanguage.value = rightLanguage.value; + rightLanguage.value = temp; + isLeftInput.value = !isLeftInput.value; + } + + void toggleInputSide(bool isLeft) => isLeftInput.value = isLeft; +} + +class FTFTranslationMessage { + final String original; + final String translated; + final bool isChineseSpeaker; + + FTFTranslationMessage({ + required this.original, + required this.translated, + required this.isChineseSpeaker, + }); +} diff --git a/lib/modules/FTFTranslation/views/FTFTranslation_view.dart b/lib/modules/FTFTranslation/views/FTFTranslation_view.dart new file mode 100644 index 000000000..84d8c1c6a --- /dev/null +++ b/lib/modules/FTFTranslation/views/FTFTranslation_view.dart @@ -0,0 +1,212 @@ +import 'package:flutter/material.dart'; +import 'package:get/get.dart'; +import '../controllers/FTFTranslation_controller.dart'; + +class FTFTranslationView extends GetView { + const FTFTranslationView({Key? key}) : super(key: key); + + @override + Widget build(BuildContext context) { + final TextEditingController inputController = TextEditingController(); + + return Scaffold( + backgroundColor: const Color(0xFF0E0F12), + appBar: AppBar( + backgroundColor: Colors.black, + title: const Text('面对面翻译'), + centerTitle: true, + ), + body: Column( + children: [ + Expanded( + child: Obx(() => ListView.builder( + padding: const EdgeInsets.all(16), + itemCount: controller.messages.length, + itemBuilder: (context, index) { + final msg = controller.messages[index]; + final isChinese = msg.isChineseSpeaker; + return Align( + alignment: isChinese + ? Alignment.centerLeft + : Alignment.centerRight, + child: Container( + margin: const EdgeInsets.symmetric(vertical: 6), + padding: const EdgeInsets.all(12), + decoration: BoxDecoration( + color: isChinese ? Colors.blue : Colors.grey[800], + borderRadius: BorderRadius.circular(12), + ), + constraints: const BoxConstraints(maxWidth: 300), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + Text( + msg.original, + style: const TextStyle( + color: Colors.white, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 4), + Text( + msg.translated, + style: TextStyle( + color: Colors.white.withOpacity(0.9), + fontSize: 14, + ), + ), + ], + ), + ), + ); + }, + )), + ), + _buildBottomInput(inputController), + ], + ), + ); + } + + Widget _buildBottomInput(TextEditingController inputController) { + return Obx(() { + final isLeftInput = controller.isLeftInput.value; + final inputLang = isLeftInput + ? controller.leftLanguage.value + : controller.rightLanguage.value; + + return Container( + color: Colors.black, + padding: const EdgeInsets.symmetric(horizontal: 12, vertical: 8), + child: Column( + children: [ + Row( + children: [ + Expanded( + child: TextField( + controller: inputController, + style: const TextStyle(color: Colors.white), + decoration: InputDecoration( + filled: true, + fillColor: Colors.grey[900], + hintText: '请输入 $inputLang 内容', + hintStyle: const TextStyle(color: Colors.grey), + border: OutlineInputBorder( + borderRadius: BorderRadius.circular(8), + borderSide: BorderSide.none, + ), + contentPadding: + const EdgeInsets.symmetric(horizontal: 12), + ), + ), + ), + IconButton( + onPressed: () { + final text = inputController.text.trim(); + if (text.isNotEmpty) { + controller.addMessage(text); + inputController.clear(); + } + }, + icon: const Icon(Icons.send, color: Colors.white), + ), + ], + ), + const SizedBox(height: 8), + Row( + children: [ + Expanded( + child: _buildLanguageDropdown( + selected: controller.leftLanguage.value, + items: controller.supportedLanguages, + onChanged: (value) { + if (value != null) controller.leftLanguage.value = value; + }, + icon: Icons.headphones, + isSelected: isLeftInput, + label: '左方语言', + onTapInputSide: () => controller.toggleInputSide(true), + ), + ), + const SizedBox(width: 12), + GestureDetector( + onTap: controller.swapLanguages, + child: Container( + padding: const EdgeInsets.all(8), + decoration: BoxDecoration( + color: Colors.white10, + shape: BoxShape.circle, + ), + child: + const Icon(Icons.compare_arrows, color: Colors.white), + ), + ), + const SizedBox(width: 12), + Expanded( + child: _buildLanguageDropdown( + selected: controller.rightLanguage.value, + items: controller.supportedLanguages, + onChanged: (value) { + if (value != null) controller.rightLanguage.value = value; + }, + icon: Icons.smartphone, + isSelected: !isLeftInput, + label: '右方语言', + onTapInputSide: () => controller.toggleInputSide(false), + ), + ), + ], + ), + ], + ), + ); + }); + } + + Widget _buildLanguageDropdown({ + required String selected, + required List items, + required void Function(String?) onChanged, + required IconData icon, + required bool isSelected, + required String label, + required VoidCallback onTapInputSide, + }) { + return GestureDetector( + onTap: onTapInputSide, + child: Container( + padding: const EdgeInsets.symmetric(vertical: 8, horizontal: 12), + decoration: BoxDecoration( + border: Border.all( + color: isSelected ? Colors.white : Colors.grey, + ), + borderRadius: BorderRadius.circular(24), + color: isSelected ? Colors.white10 : Colors.transparent, + ), + child: Row( + children: [ + Icon(icon, color: Colors.white), + const SizedBox(width: 8), + Expanded( + child: DropdownButton( + dropdownColor: Colors.grey[900], + value: selected, + isExpanded: true, + underline: const SizedBox(), + iconEnabledColor: Colors.white, + style: const TextStyle(color: Colors.white), + items: items.map((lang) { + return DropdownMenuItem( + value: lang, + child: Text(lang), + ); + }).toList(), + onChanged: onChanged, + ), + ), + ], + ), + ), + ); + } +} diff --git a/lib/modules/meeting/controllers/meeting_controller.dart b/lib/modules/meeting/controllers/meeting_controller.dart index e38cba917..0058b31d8 100644 --- a/lib/modules/meeting/controllers/meeting_controller.dart +++ b/lib/modules/meeting/controllers/meeting_controller.dart @@ -14,6 +14,7 @@ import 'package:permission_handler/permission_handler.dart'; import '../../../core/utils/logger.dart'; import '../../../core/utils/synchrodata.dart'; import '../../../core/utils/upload_oss.dart'; +import '../../../data/services/ble_manager.dart'; import '../../../data/services/db/sqflite_api.dart'; import '../../../core/utils/permission_util.dart'; @@ -37,7 +38,8 @@ class MeetingController extends GetxController { TextEditingController titleController = TextEditingController(); // 录音标题 RxList dataList = [].obs; - + // 蓝牙服务 + final bleManager = Get.find(); List dataListTask = []; bool isDisposed = false; String filePath = ''; @@ -313,6 +315,9 @@ class MeetingController extends GetxController { PlatformFile file = result.files.first; filePath = await _saveFile(file); fileName = file.name; + } else { + EasyLoading.dismiss(); + return; } } diff --git a/lib/modules/meeting/controllers/meeting_record_controller.dart b/lib/modules/meeting/controllers/meeting_record_controller.dart index 2fea38ea5..ee4640b18 100644 --- a/lib/modules/meeting/controllers/meeting_record_controller.dart +++ b/lib/modules/meeting/controllers/meeting_record_controller.dart @@ -113,12 +113,15 @@ class MeetingRecordController extends GetxController switch (audioType.value) { case 0: fileName.value = "现场录音"; + _audioSourceType = false; break; case 1: fileName.value = "音、视频录音"; + _audioSourceType = true; break; case 2: fileName.value = "通话录音"; + _audioSourceType = true; break; } newName.value = fileName.value; @@ -234,7 +237,7 @@ class MeetingRecordController extends GetxController void _startHardwareServices() { switch (audioType.value) { case 0: - _bleManager.openEncoder(); + // _bleManager.openEncoder(); break; case 1: _bleManager.openDecoder(); diff --git a/lib/modules/meeting/views/meeting_view.dart b/lib/modules/meeting/views/meeting_view.dart index 35d34e6d8..2a1c2c7e9 100644 --- a/lib/modules/meeting/views/meeting_view.dart +++ b/lib/modules/meeting/views/meeting_view.dart @@ -454,43 +454,43 @@ class MeetingView extends GetView { // TODO: 实现从文件导入逻辑 }, ), - SizedBox(height: 8.w), - _importAudioItem1( - context, - isDarkMode, - icon: Image.asset('assets/images/explore.png', - width: 32.w, height: 32.w), - title: '从翻译里导入', - onTap: () { - Navigator.of(context).pop(); - _showLocalAudioVideoPicker(context, isDarkMode); - // TODO: 实现从相册导入逻辑 - }, - ), - SizedBox(height: 8.w), - _importAudioItem1( - context, - isDarkMode, - icon: Row( - mainAxisSize: MainAxisSize.min, - children: [ - Image.asset('assets/images/explore.png', - width: 20.w, height: 20.w), - SizedBox(width: 4.w), - Image.asset('assets/images/explore.png', - width: 20.w, height: 20.w), - SizedBox(width: 4.w), - Image.asset('assets/images/explore.png', - width: 20.w, height: 20.w), - ], - ), - title: '从opus导出', - onTap: () { - Navigator.of(context).pop(); - // TODO: 实现从其他APP导入逻辑 - _showLocalOpusPicker(context, isDarkMode); - }, - ), + // SizedBox(height: 8.w), + // _importAudioItem1( + // context, + // isDarkMode, + // icon: Image.asset('assets/images/explore.png', + // width: 32.w, height: 32.w), + // title: '从翻译里导入', + // onTap: () { + // Navigator.of(context).pop(); + // _showLocalAudioVideoPicker(context, isDarkMode); + // // TODO: 实现从相册导入逻辑 + // }, + // ), + // SizedBox(height: 8.w), + // _importAudioItem1( + // context, + // isDarkMode, + // icon: Row( + // mainAxisSize: MainAxisSize.min, + // children: [ + // Image.asset('assets/images/explore.png', + // width: 20.w, height: 20.w), + // SizedBox(width: 4.w), + // Image.asset('assets/images/explore.png', + // width: 20.w, height: 20.w), + // SizedBox(width: 4.w), + // Image.asset('assets/images/explore.png', + // width: 20.w, height: 20.w), + // ], + // ), + // title: '从opus导出', + // onTap: () { + // Navigator.of(context).pop(); + // // TODO: 实现从其他APP导入逻辑 + // _showLocalOpusPicker(context, isDarkMode); + // }, + // ), ], ), ), @@ -578,13 +578,12 @@ class MeetingView extends GetView { SizedBox(height: 8.w), // 音、视频录音 - 根据控制器状态禁用 _importAudioItem( - context, - isDarkMode, + context, isDarkMode, icon: Image.asset('assets/images/explore.png', width: 32.w, height: 32.w), title: '音、视频录音', - enabled: false, // 依赖控制器状态 - onTap: false + enabled: controller.bleManager.isConnected, // 依赖控制器状态 + onTap: controller.bleManager.isConnected ? () async { // 条件启用 Navigator.of(context).pop(); @@ -599,14 +598,13 @@ class MeetingView extends GetView { SizedBox(height: 8.w), // 通话录音 - 根据控制器状态禁用 _importAudioItem( - context, - isDarkMode, + context, isDarkMode, icon: Row( // ... 图标组合保持不变 ), title: '通话录音', - enabled: false, // 依赖控制器状态 - onTap: false + enabled: controller.bleManager.isConnected, // 依赖控制器状态 + onTap: controller.bleManager.isConnected ? () async { // 条件启用 Navigator.of(context).pop(); diff --git a/lib/modules/settings/views/settings_view.dart b/lib/modules/settings/views/settings_view.dart index 35bb6c90b..777f8c48f 100644 --- a/lib/modules/settings/views/settings_view.dart +++ b/lib/modules/settings/views/settings_view.dart @@ -543,6 +543,28 @@ class SettingsView extends GetView { }, isDarkMode: isDarkMode, ), + Divider( + height: 1, + color: isDarkMode + ? Colors.white.withOpacity(0.1) + : Colors.grey[200]), + // 面对面翻译测试 + _buildSimpleNavigationSetting( + title: '面对面翻译测试', + subtitle: '测试面对面翻译', + icon: Icons.bluetooth_searching, + iconBgColor: isDarkMode + ? Colors.green[900]!.withOpacity(0.3) + : Colors.green[100]!, + iconColor: + isDarkMode ? Colors.green[300]! : Colors.green[600]!, + titleColor: isDarkMode ? Colors.white : null, + subtitleColor: isDarkMode ? Colors.white70 : null, + onTap: () { + Get.toNamed(Routes.ftftranslation); + }, + isDarkMode: isDarkMode, + ), Divider( height: 1, color: isDarkMode diff --git a/lib/modules/translation/controllers/translation_controller.dart b/lib/modules/translation/controllers/translation_controller.dart index 9a911accf..d83b04bdd 100644 --- a/lib/modules/translation/controllers/translation_controller.dart +++ b/lib/modules/translation/controllers/translation_controller.dart @@ -26,7 +26,7 @@ class TranslationController extends GetxController { final LanguageManager _languageManager = Get.find(); final GetStorage _storage = GetStorage(); // 蓝牙服务 - final _bleManager = Get.find(); + final bleManager = Get.find(); // 音频输入源 bool _audioSourceType = false; // 存储相关 @@ -308,14 +308,14 @@ class TranslationController extends GetxController { // translationHistory.add(newItem); // 开始连续语音识别 if (currentMode.value == "audioVideo") { - _bleManager.openDecoder(); + bleManager.openDecoder(); // 发送ble音乐或者通话远端声音 _audioSourceType = true; isTtsEnabled.value = false; // 禁用TTS以避免干扰 Logger.info('发送ble音乐或者通话远端声音'); } else if (currentMode.value == "call") { // 发送ble系统mic和dac(音乐或者通话远端)声音 - _bleManager.openA2DPDecoder(); + bleManager.openA2DPDecoder(); _audioSourceType = true; isTtsEnabled.value = true; Logger.info('发送ble系统mic和dac(音乐或者通话远端)声音'); @@ -563,7 +563,7 @@ class TranslationController extends GetxController { Future stopAll() async { try { if (currentMode.value == "audioVideo" || currentMode.value == "call") { - _bleManager.closeCodec(); + bleManager.closeCodec(); // 发送ble音乐或者通话远端声音 Logger.info('发送ble音乐或者通话远端声音'); } diff --git a/lib/modules/translation/views/translation_view.dart b/lib/modules/translation/views/translation_view.dart index e8eaf3d3e..048f5f463 100644 --- a/lib/modules/translation/views/translation_view.dart +++ b/lib/modules/translation/views/translation_view.dart @@ -67,12 +67,12 @@ class TranslationView extends GetView { _buildPopupMenuItem( 'audioVideo', 'audioVideoTranslation'.tr, - enabled: true, // 根据连接状态启用/禁用 + enabled: controller.bleManager.isConnected, // 根据连接状态启用/禁用 ), _buildPopupMenuItem( 'call', 'callTranslation'.tr, - enabled: true, // 根据连接状态启用/禁用 + enabled: controller.bleManager.isConnected, // 根据连接状态启用/禁用 ), ], child: Row( diff --git a/lib/routes/app_pages.dart b/lib/routes/app_pages.dart index 66c5f2516..8f3a14799 100644 --- a/lib/routes/app_pages.dart +++ b/lib/routes/app_pages.dart @@ -46,6 +46,8 @@ import '../modules/realtime/bindings/realtime_binding.dart'; import '../modules/realtime/views/realtime_view.dart'; import '../modules/music/views/music_playlist_view.dart'; import '../modules/music/bindings/music_playlist_binding.dart'; +import '../modules/FTFTranslation/bindings/FTFTranslation_binding.dart'; +import '../modules/FTFTranslation/views/FTFTranslation_view.dart'; abstract class AppPages { static final pages = [ @@ -191,6 +193,13 @@ abstract class AppPages { transition: Transition.rightToLeft, transitionDuration: Duration(milliseconds: 250), ), + GetPage( + name: Routes.ftftranslation, + page: () => const FTFTranslationView(), + binding: FTFTranslationBinding(), + transition: Transition.noTransition, + transitionDuration: Duration(milliseconds: 250), + ), GetPage( name: Routes.opusTest, page: () => const OpusTestView(), diff --git a/lib/routes/app_routes.dart b/lib/routes/app_routes.dart index 3e2e1be89..550ab16cc 100644 --- a/lib/routes/app_routes.dart +++ b/lib/routes/app_routes.dart @@ -28,6 +28,7 @@ abstract class Routes { static const translationHistory = '/translation/history'; // 翻译历史 static const speechDemo = '/speech_demo'; // 语音演示 static const translationList = '/translation/List'; // 翻译历史 + static const ftftranslation = '/ftftranslation'; // 面对面翻译 // 测试路由(生产环境建议移除) static const ttsTest = '/tts_test'; // 文字转语音测试 static const asrTest = '/asr_test'; // 语音识别测试 diff --git a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt index 55eb87911..2e9d93f87 100644 --- a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt +++ b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt @@ -22,6 +22,7 @@ import java.util.concurrent.LinkedBlockingQueue import android.os.Handler import java.util.concurrent.atomic.AtomicBoolean import android.util.Log + /** * BLE服务类:提供蓝牙低功耗设备的扫描、连接和通信功能 * 支持: @@ -67,17 +68,20 @@ object BleService { private var bluetoothLeScanner: BluetoothLeScanner? = null // GATT连接相关 - var bluetoothGatt: BluetoothGatt? = null + var bluetoothGatt: BluetoothGatt? = null private var notifyChar: BluetoothGattCharacteristic? = null private var writeChar: BluetoothGattCharacteristic? = null private var audioChar: BluetoothGattCharacteristic? = null var recordfile: RecordingFile? = null + // 扫描相关 private lateinit var scanHandler: Handler private val scanResults = ArrayList() private var isScanning = false -// 是否是ota模式 + + // 是否是ota模式 var isEnterOta = false + // 连接状态 val connectionState = MutableLiveData(BleConst.STATE_DISCONNECTED) @@ -136,7 +140,7 @@ object BleService { Log.e(TAG, "OpusManager初始化失败: ${e.message}", e) // 根据需要决定是否因为Opus初始化失败而返回false } -recordfile = RecordingFile(this.context) + recordfile = RecordingFile(this.context) isInitialized = true Log.d(TAG, "BLE服务初始化成功") return true @@ -198,7 +202,6 @@ recordfile = RecordingFile(this.context) // 获取一个局部引用,避免并发访问问题 val scanner = bluetoothLeScanner ?: return false if (isScanning) return false - Log.i(TAG, "开始主动扫描BLE设备...") scanResults.clear() @@ -341,13 +344,12 @@ recordfile = RecordingFile(this.context) Log.w(TAG, "未找到已配对的设备 MAC 地址") return } - - - if(!isEnterOta) - { - Log.i(TAG, "检测到已配对设备,MAC: $pairedMac,尝试连接") - // 调用现有的 connect 方法 - connect(pairedMac) + + + if (!isEnterOta) { + Log.i(TAG, "检测到已配对设备,MAC: $pairedMac,尝试连接") + // 调用现有的 connect 方法 + connect(pairedMac) } } catch (e: Exception) { @@ -518,7 +520,7 @@ recordfile = RecordingFile(this.context) when (c.uuid) { // 音频特征数据 BleConst.RECEIVE_AUDIO_CHAR_UUID -> { - + // Log.i(TAG, "收到音频特征数据") processAudioData(data) } @@ -597,7 +599,7 @@ recordfile = RecordingFile(this.context) private fun processAudioData(data: ByteArray) { try { if (opusManager?.isDecodeStream == true) { - recordfile?.saveAudioDataToWav(data) + recordfile?.saveAudioDataToWav(data) opusManager?.writeAudioStream(data) } else { // Log.d(TAG, "Opus解码流未启动,忽略音频数据") @@ -795,17 +797,16 @@ recordfile = RecordingFile(this.context) BleConst.CODEC_CONTROL_ENCODE_ON -> "已打开编码" else -> "未知状态($codecStatus)" } -if (codecStatus == BleConst.CODEC_CONTROL_DECODE_ON || - codecStatus == BleConst.CODEC_CONTROL_A2DP_PLAY || - codecStatus == BleConst.CODEC_CONTROL_ENCODE_ON) { - - recordfile!!.closeFile() - recordfile!!.creatingFiles() -} -else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) -{ - recordfile!!.closeFile() -} + if (codecStatus == BleConst.CODEC_CONTROL_DECODE_ON || + codecStatus == BleConst.CODEC_CONTROL_A2DP_PLAY || + codecStatus == BleConst.CODEC_CONTROL_ENCODE_ON + ) { + + recordfile!!.closeFile() + recordfile!!.creatingFiles() + } else if (codecStatus == BleConst.CODEC_CONTROL_CLOSE) { + recordfile!!.closeFile() + } // 声道模式描述 val channelDesc = when (channelMode) { BleConst.AUDIO_CHANNEL_LEFT -> "左声道" @@ -1093,7 +1094,7 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) // // The library might expect this in a different unit or derive it. // // Given JlOpusPlugin.kt, packetSize refers to Opus encoded frame duration in ms. // ) - startOpusStreamDecoding(false, 1, 16000, 40) + startOpusStreamDecoding(false, 1, 16000, 40) Log.i(TAG, "打开解码0xA2") return sendCommand( @@ -1110,7 +1111,7 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) */ fun openA2DPDecoder(): Boolean { Log.i(TAG, "打开编码 0xA1") - startOpusStreamDecoding(false, 1, 16000, 80) + startOpusStreamDecoding(false, 1, 16000, 80) // Log.i(TAG, "打开解码...") return sendCommand( BleConst.CMD_CONTROL_CODEC.toByte(), byteArrayOf( @@ -1135,7 +1136,7 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) fun openEncoder(): Boolean { Log.i(TAG, "打开编码0xB1") - startOpusStreamDecoding(false, 1, 16000, 40) + startOpusStreamDecoding(false, 1, 16000, 40) // 目前仅发送命令通知设备开始编码。 return sendCommand( BleConst.CMD_CONTROL_CODEC.toByte(), @@ -1385,9 +1386,10 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) } } - + } - class RecordingFile(private val context: Context) { + +class RecordingFile(private val context: Context) { private var currentAudioFile: File? = null private var fos: FileOutputStream? = null @@ -1408,8 +1410,9 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) // 创建新的音频文件 val dateFormat = SimpleDateFormat("yyyyMMdd_HHmmss", Locale.getDefault()) val timestamp = dateFormat.format(Date()) - - val filePath = File(context.getExternalFilesDir(null), "${fileName}_${timestamp}.opus").absolutePath + + val filePath = + File(context.getExternalFilesDir(null), "${fileName}_${timestamp}.opus").absolutePath currentAudioFile = File(filePath) currentAudioFile?.createNewFile() // 追加音频数据到文件 @@ -1417,42 +1420,42 @@ else if(codecStatus == BleConst.CODEC_CONTROL_CLOSE) startWriteThread() // 新增:启动写入线程 } - // 新增:启动写入线程 - private fun startWriteThread() { - if (isWriting.get()) return - isWriting.set(true) - writeThread = Thread { - try { - while (isWriting.get() || writeQueue.isNotEmpty()) { - val data = writeQueue.poll() ?: continue - fos?.write(data) - } - } catch (e: Exception) { - Log.e("", "异步写入音频数据失败: ${e.message}") + // 新增:启动写入线程 + private fun startWriteThread() { + if (isWriting.get()) return + isWriting.set(true) + writeThread = Thread { + try { + while (isWriting.get() || writeQueue.isNotEmpty()) { + val data = writeQueue.poll() ?: continue + fos?.write(data) } + } catch (e: Exception) { + Log.e("", "异步写入音频数据失败: ${e.message}") } - writeThread?.start() } + writeThread?.start() + } - /** - * 保存音频数据到 WAV 文件(异步) - */ - internal fun saveAudioDataToWav(buffer: ByteArray) { - if (fos == null || currentAudioFile == null) return - // 放入队列,由写线程写入 - // 1. 将字节数据转换为十六进制字符串 - // val hexData = buildString { - // buffer.forEachIndexed { index, byte -> - // append("%02X".format(byte)) - - // } - // } - - // 2. 写入十六进制字符串 - - writeQueue.offer(buffer) - totalBytesWritten += buffer.size - } + /** + * 保存音频数据到 WAV 文件(异步) + */ + internal fun saveAudioDataToWav(buffer: ByteArray) { + if (fos == null || currentAudioFile == null) return + // 放入队列,由写线程写入 + // 1. 将字节数据转换为十六进制字符串 + // val hexData = buildString { + // buffer.forEachIndexed { index, byte -> + // append("%02X".format(byte)) + + // } + // } + + // 2. 写入十六进制字符串 + + writeQueue.offer(buffer) + totalBytesWritten += buffer.size + } internal fun closeFile() { From 47432987635afe8493ee96cfe41c3f4430cb6f2e Mon Sep 17 00:00:00 2001 From: wolfplus2048 <30993207+wolfplus2048@users.noreply.github.com> Date: Sat, 21 Jun 2025 21:15:15 +0100 Subject: [PATCH 4/5] add --- .../azure_speech/AzureTtsHelper.kt | 34 ++ .../bytedance_speech/BytedanceAudioPlayer.kt | 342 +++++++----------- .../bytedance_speech/BytedanceTTS.kt | 177 ++++----- .../chat_api/ChatApiService.kt | 3 +- .../com/deep_voice/speech/ITtsService.kt | 8 + 5 files changed, 240 insertions(+), 324 deletions(-) diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 62ac83d49..9b3d8ccfe 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -169,6 +169,40 @@ class AzureTtsHelper(private val context: Context) : ITtsService { audioDataListeners.remove(listener) } + /** + * 设置是否使用内部播放器 + * + * @param useInternalPlayer true: 使用内部播放器自动播放音频 + * false: 仅通过音频数据监听器输出数据,不播放 + */ + override fun setUseInternalPlayer(useInternalPlayer: Boolean) { + // Azure TTS 通过 AudioConfig 控制播放器 + // 如果不使用内部播放器,需要重新配置 synthesizer + if (!useInternalPlayer && customAudioOutputStream == null) { + // 创建自定义音频输出流以便捕获音频数据 + customAudioOutputStream = SimpleAudioPlayer(context).getAudioOutputStream() + + // 如果已经初始化,需要重新创建 synthesizer + if (isInitialized && speechConfig != null) { + synthesizer?.close() + val audioConfig = AudioConfig.fromStreamOutput(customAudioOutputStream) + synthesizer = SpeechSynthesizer(speechConfig, audioConfig) + setupEventListeners() + } + } else if (useInternalPlayer && customAudioOutputStream != null) { + // 切换回使用内部播放器 + customAudioOutputStream = null + + // 如果已经初始化,需要重新创建 synthesizer + if (isInitialized && speechConfig != null) { + synthesizer?.close() + val audioConfig = AudioConfig.fromDefaultSpeakerOutput() + synthesizer = SpeechSynthesizer(speechConfig, audioConfig) + setupEventListeners() + } + } + } + /** * 触发事件通知 */ diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt index 9587f401f..84e73b1be 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt @@ -7,112 +7,134 @@ import android.media.AudioTrack import android.os.Build import android.util.Log import com.deep_voice.speech.tts.AudioDataListener -import kotlinx.coroutines.CoroutineScope -import kotlinx.coroutines.Dispatchers -import kotlinx.coroutines.Job -import kotlinx.coroutines.SupervisorJob -import kotlinx.coroutines.cancelChildren -import kotlinx.coroutines.delay -import kotlinx.coroutines.isActive -import kotlinx.coroutines.launch -import kotlinx.coroutines.withContext -import java.util.concurrent.ConcurrentLinkedQueue -import java.util.concurrent.atomic.AtomicBoolean -import java.util.concurrent.CancellationException + /** - * 字节跳动语音合成音频播放器 - * - * 使用AudioTrack播放PCM格式音频流,基于Kotlin协程实现异步处理 + * 简化的音频流播放器 + * 基于位置标记触发播放完成回调 */ -class BytedanceAudioPlayer : AudioDataListener, CoroutineScope { +class BytedanceAudioPlayer : AudioDataListener { companion object { private const val TAG = "BytedanceAudioPlayer" - private const val SAMPLE_RATE = 24000 // 样本率 - private const val CHANNEL_CONFIG = AudioFormat.CHANNEL_OUT_MONO // 单声道 - private const val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT // 16位PCM - private const val IDLE_TIMEOUT_MS = 800L // 无数据超时时间 + private const val SAMPLE_RATE = 24000 + private const val CHANNEL_CONFIG = AudioFormat.CHANNEL_OUT_MONO + private const val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT } - // 协程相关 - private val job = SupervisorJob() - override val coroutineContext = Dispatchers.IO + job - - // 音频处理相关 private var audioTrack: AudioTrack? = null - private val isPlaying = AtomicBoolean(false) - private val isPaused = AtomicBoolean(false) - - // 播放状态 - private val playStarted = AtomicBoolean(false) - private val completionReported = AtomicBoolean(false) + private var isFirstData = true // 是否是第一次接收数据 + private var totalBytesWritten = 0 // 总共写入的字节数 + private var sessionActive = true // 会话是否活跃 - // 使用简单的并发队列,保证线程安全 - private val audioDataQueue = ConcurrentLinkedQueue() - - // 播放状态监听 + // 回调 private var onPlayStarted: (() -> Unit)? = null private var onPlayCompleted: (() -> Unit)? = null - private var onError: ((String) -> Unit)? = null - // 处理协程 - private var processingJob: Job? = null - private var lastDataTime = 0L + // 播放位置监听器 + private val playbackListener = object : AudioTrack.OnPlaybackPositionUpdateListener { + override fun onMarkerReached(track: AudioTrack) { + Log.d(TAG, "播放到达标记位置: ${track.playbackHeadPosition}") + onPlayCompleted?.invoke() + } + + override fun onPeriodicNotification(track: AudioTrack) { + // 不使用周期性通知 + } + } - init { - Log.d(TAG, "BytedanceAudioPlayer初始化") + + /** + * 开始新的播放会话 + */ + fun startSession() { + Log.d(TAG, "开始新会话") + + // 重置状态 + isFirstData = true + totalBytesWritten = 0 + sessionActive = true + + // 重置 AudioTrack + audioTrack?.let { track -> + if (track.state == AudioTrack.STATE_INITIALIZED) { + // 停止并清空缓冲区 + track.pause() + track.flush() + // 重新开始播放 + track.play() + Log.d(TAG, "AudioTrack 已重置") + } + } ?: initAudioTrack() // 如果没有初始化,则初始化 } /** - * 开始播放 + * 标记数据流结束 */ - fun start() { - if (isPlaying.getAndSet(true)) return + fun endSession() { + if (!sessionActive) return - isPaused.set(false) - playStarted.set(false) - completionReported.set(false) - lastDataTime = System.currentTimeMillis() + Log.d(TAG, "数据流结束,总共写入字节数: $totalBytesWritten") + sessionActive = false - processingJob = launch { - try { - initAudioTrack() + audioTrack?.let { track -> + if (totalBytesWritten > 0) { + // 计算总帧数(16-bit 单声道,每帧2字节) + val totalFrames = totalBytesWritten / 2 - // 处理队列中的音频数据 - while (isActive && isPlaying.get()) { - processQueuedAudio() - - // 检查是否播放完成 - checkPlaybackCompletion() - - // 队列为空时短暂延迟 - if (audioDataQueue.isEmpty()) { - delay(10) - } + // 设置标记位置 + try { + track.setNotificationMarkerPosition(totalFrames) + Log.d(TAG, "设置播放完成标记位置: $totalFrames") + } catch (e: Exception) { + Log.e(TAG, "设置标记失败: ${e.message}") + // 设置失败时,使用延迟触发作为后备 + val durationMs = (totalFrames * 1000L) / SAMPLE_RATE + android.os.Handler(android.os.Looper.getMainLooper()).postDelayed({ + onPlayCompleted?.invoke() + }, durationMs + 500) } - } catch (e: CancellationException) { - // 协程被取消,正常行为 - } catch (e: Exception) { - Log.e(TAG, "播放错误: ${e.message}") - onError?.invoke("播放错误: ${e.message}") - stopInternal(false) + } else { + // 没有数据,直接触发完成 + onPlayCompleted?.invoke() } } } /** - * 暂停播放 + * 接收音频数据 */ - fun pause() { - if (!isPlaying.get() || isPaused.getAndSet(true)) return - audioTrack?.pause() - } - - /** - * 恢复播放 - */ - fun resume() { - if (!isPlaying.get() || !isPaused.getAndSet(false)) return - audioTrack?.play() + override fun onAudioData(data: ByteArray) { + if (data.isEmpty()) return + if (!sessionActive) { + Log.w(TAG, "收到音频数据但会话未激活,忽略数据") + return + } + if (audioTrack == null || audioTrack?.state != AudioTrack.STATE_INITIALIZED) { + Log.w(TAG, "收到音频数据但 AudioTrack 未初始化或状态不正确,忽略数据") + return + } + try { + audioTrack?.let { track -> + if (track.state == AudioTrack.STATE_INITIALIZED) { + val bytesWritten = track.write(data, 0, data.size) + + if (bytesWritten > 0) { + // 第一次写入数据时自动触发开始回调 + if (isFirstData) { + isFirstData = false + onPlayStarted?.invoke() + Log.d(TAG, "播放开始") + } + + // 累计写入字节数 + if (sessionActive) { + totalBytesWritten += bytesWritten + } + } + } + } + } catch (e: Exception) { + Log.e(TAG, "写入音频数据失败: ${e.message}") + } } /** @@ -120,44 +142,32 @@ class BytedanceAudioPlayer : AudioDataListener, CoroutineScope { */ fun stop() { Log.d(TAG, "停止播放") - stopInternal(true) - } - - /** - * 内部停止处理 - */ - private fun stopInternal(reportCompletion: Boolean) { - if (!isPlaying.getAndSet(false)) return - - isPaused.set(false) - processingJob?.cancel() - - audioTrack?.stop() - // 不在stopInternal中释放资源,只停止播放 - // 清空队列 - audioDataQueue.clear() + sessionActive = false - // 播放完成通知 - if (reportCompletion && playStarted.get() && !completionReported.getAndSet(true)) { - launch(Dispatchers.Main) { - onPlayCompleted?.invoke() + audioTrack?.let { track -> + if (track.state == AudioTrack.STATE_INITIALIZED) { + track.pause() + track.flush() } } + + // 如果已经开始播放,触发完成回调 + if (!isFirstData) { + onPlayCompleted?.invoke() + } } /** * 释放资源 */ fun release() { - stopInternal(false) + Log.d(TAG, "释放资源") + + stop() - // 在release方法中释放AudioTrack资源 audioTrack?.release() audioTrack = null - - // 释放协程资源 - job.cancel() } /** @@ -175,120 +185,32 @@ class BytedanceAudioPlayer : AudioDataListener, CoroutineScope { } /** - * 设置错误回调 + * 设置错误回调(兼容接口) */ fun setOnError(listener: (String) -> Unit) { - onError = listener - } - - /** - * 接收音频数据(实现AudioDataListener接口) - */ - override fun onAudioData(data: ByteArray) { - if (data.isEmpty()) return - - lastDataTime = System.currentTimeMillis() - - if (!isPlaying.get()) { - start() - } - - // 添加到队列 - 简单有效,保证FIFO顺序 - audioDataQueue.add(data.copyOf()) + // 不实现,仅为兼容 } /** * 检查是否正在播放 */ - fun isPlaying(): Boolean = isPlaying.get() && !isPaused.get() - - /** - * 处理队列中的音频数据 - */ - private suspend fun processQueuedAudio() { - if (isPaused.get() || !isPlaying.get() || !isActive) return - - // 获取并处理队列中的数据 - val data = audioDataQueue.poll() ?: return - - // 写入音频数据 - val result = audioTrack?.write(data, 0, data.size) ?: 0 - - if (result > 0) { - // 标记播放开始 - if (!playStarted.getAndSet(true)) { - withContext(Dispatchers.Main) { - onPlayStarted?.invoke() - } - } - } else if (result < 0) { - // 处理错误 - handleAudioTrackError(result) - } + fun isPlaying(): Boolean { + return audioTrack?.playState == AudioTrack.PLAYSTATE_PLAYING } /** - * 检查播放是否完成(超时无数据) - */ - private suspend fun checkPlaybackCompletion() { - val currentTime = System.currentTimeMillis() - - // 超时判断 - 队列为空且超过超时时间 - if (playStarted.get() && audioDataQueue.isEmpty() && - currentTime - lastDataTime > IDLE_TIMEOUT_MS && isPlaying.get()) { - - // 播放完成 - if (!completionReported.getAndSet(true)) { - withContext(Dispatchers.Main) { - onPlayCompleted?.invoke() - } - stopInternal(false) // 已报告,不需要再次报告 - } - } - } - - /** - * 处理AudioTrack错误 - */ - private fun handleAudioTrackError(result: Int) { - val errorMsg = when(result) { - AudioTrack.ERROR_INVALID_OPERATION -> "无效操作" - AudioTrack.ERROR_BAD_VALUE -> "参数错误" - AudioTrack.ERROR_DEAD_OBJECT -> "对象已销毁" - else -> "未知错误" - } - - // 对象已销毁时尝试重建 - if (result == AudioTrack.ERROR_DEAD_OBJECT) { - try { - audioTrack?.release() - initAudioTrack() - } catch (e: Exception) { - Log.e(TAG, "重建AudioTrack失败: ${e.message}") - throw e - } - } else { - Log.e(TAG, "AudioTrack错误: $errorMsg") - } - } - - /** - * 初始化AudioTrack + * 初始化 AudioTrack */ private fun initAudioTrack() { - val minBufferSize = AudioTrack.getMinBufferSize( - SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT - ) - - if (minBufferSize == AudioTrack.ERROR || minBufferSize == AudioTrack.ERROR_BAD_VALUE) { - throw IllegalStateException("无法获取有效的音频缓冲区大小") - } + // 释放旧实例 + audioTrack?.release() - // 使用较大的缓冲区提高稳定性 - val bufferSize = minBufferSize * 4 + // 计算缓冲区大小 + val minBufferSize = AudioTrack.getMinBufferSize(SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT) + val bufferSize = minBufferSize * 2 - // 创建AudioTrack - audioTrack = if (Build.VERSION.SDK_INT >= 23) { // Android M (6.0) + // 创建 AudioTrack + audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { AudioTrack.Builder() .setAudioAttributes( AudioAttributes.Builder() @@ -318,12 +240,12 @@ class BytedanceAudioPlayer : AudioDataListener, CoroutineScope { ) } - // 检查初始化状态 - if (audioTrack?.state != AudioTrack.STATE_INITIALIZED) { - throw IllegalStateException("AudioTrack初始化失败") - } + // 设置播放位置监听器 + audioTrack?.setPlaybackPositionUpdateListener(playbackListener) // 开始播放 audioTrack?.play() + + Log.d(TAG, "AudioTrack 初始化成功") } -} \ No newline at end of file +} \ No newline at end of file diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt index d202a36e4..49d83efcc 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt @@ -100,6 +100,9 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { // 内部音频播放器 private val audioPlayer = BytedanceAudioPlayer() + // 是否使用内部播放器 + private var useInternalPlayer = true + // WebSocket连接 private var webSocket: WebSocket? = null private val client: OkHttpClient @@ -119,9 +122,7 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { private var language: String = "zh-CN" // 流式处理状态 - private var isStreamMode = false private var currentStatus = STATUS_STOPPED - private var isSpeaking = false // 连接相关状态 private var connectionAttempts = 0 @@ -155,12 +156,8 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { * 初始化音频播放器回调 */ private fun initAudioPlayerCallbacks() { - // 添加播放器作为音频数据监听器 - addAudioDataListener(audioPlayer) - audioPlayer.setOnPlayStarted { Log.d(TAG, "播放开始") - isSpeaking = true updateStatus(STATUS_SPEAKING) notifyEvent(TtsEventType.SYNTHESIS_STARTED) } @@ -168,17 +165,18 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { // 设置播放完成回调 audioPlayer.setOnPlayCompleted { Log.d(TAG, "播放结束") - isSpeaking = false - updateStatus(if (isStreamMode) STATUS_READY else STATUS_STOPPED) + updateStatus(STATUS_READY) notifyEvent(TtsEventType.SYNTHESIS_COMPLETED) } // 设置错误回调 audioPlayer.setOnError { errorMsg -> - isSpeaking = false updateStatus(STATUS_ERROR) notifyEvent(TtsEventType.ERROR, mapOf("errorCode" to "PLAYER_ERROR", "errorMessage" to errorMsg)) } + + // 根据设置决定是否使用内部播放器 + updateInternalPlayerUsage() } /** @@ -208,16 +206,19 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { */ override fun stop(): Boolean { try { - audioPlayer.stop() + Log.i(TAG, ">stop()") + + if (useInternalPlayer) { + audioPlayer.stop() + } textProcessingJob?.cancel() - launch { - if (isSessionStarted) { - finishSession() - } - isSpeaking = false // 显式设置,因为这是强制停止 - updateStatus(STATUS_STOPPED) + + if (isSessionStarted) { + finishSession() } + updateStatus(STATUS_STOPPED) + return true } catch (e: Exception) { Log.e(TAG, "停止合成失败: ${e.message}") @@ -277,7 +278,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { isConnected = false isSessionStarted = false - isSpeaking = false updateStatus(STATUS_STOPPED) startTextProcessing() @@ -304,29 +304,9 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { * 单次播放(非流式) */ override fun speakOnce(text: String): Boolean { - try { - if (text.isEmpty()) { - return false - } - - isStreamMode = false - - launch { - if (!ensureSessionReady()) { - notifyEvent(TtsEventType.ERROR, mapOf("errorCode" to "SESSION_ERROR", - "errorMessage" to "无法建立会话")) - return@launch - } - - sendTextRequest(text) - } - - return true - } catch (e: Exception) { - Log.e(TAG, "语音合成失败: ${e.message}") - updateStatus(STATUS_ERROR) - return false - } + // 未实现:只支持流式模式 + Log.w(TAG, "speakOnce未实现,请使用speakStream") + return false } /** @@ -337,7 +317,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { if (text.isEmpty()) { return true } - isStreamMode = true // 确保会话已就绪 if (!isSessionStarted && !ensureSessionReady()) { @@ -365,17 +344,11 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { */ override fun flushStream(): Boolean { try { - if (!isStreamMode) return true // Log.d(TAG, "flushStream: $sessionId") - launch { - withContext(Dispatchers.IO) { - delay(300) // 确保现有文本处理完成 - } - - if (isSessionStarted) { - isSessionStarted = false - finishSession() - } + + if (isSessionStarted) { + isSessionStarted = false + finishSession() } return true @@ -440,6 +413,29 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { audioDataListeners.remove(listener) } + /** + * 设置是否使用内部播放器 + */ + override fun setUseInternalPlayer(useInternalPlayer: Boolean) { + this.useInternalPlayer = useInternalPlayer + updateInternalPlayerUsage() + } + + /** + * 更新内部播放器的使用状态 + */ + private fun updateInternalPlayerUsage() { + if (useInternalPlayer) { + // 使用内部播放器 + if (!audioDataListeners.contains(audioPlayer)) { + audioDataListeners.add(audioPlayer) + } + } else { + // 不使用内部播放器 + audioDataListeners.remove(audioPlayer) + } + } + /** * 通知事件处理 */ @@ -461,15 +457,14 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { * 通知音频数据 */ private fun notifyAudioData(data: ByteArray) { - launch(Dispatchers.Main) { - for (listener in audioDataListeners) { - try { - listener.onAudioData(data) - } catch (e: Exception) { - Log.e(TAG, "音频数据回调异常: ${e.message}") - } + for (listener in audioDataListeners) { + try { + listener.onAudioData(data) + } catch (e: Exception) { + Log.e(TAG, "音频数据回调异常: ${e.message}") } } + } /** @@ -483,7 +478,7 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { // 如果已连接但会话未开始,只需要开始会话 if (isConnected && !isSessionStarted) { - Log.d(TAG, "WebSocket已连接,正在启动新会话...") + // Log.d(TAG, "WebSocket已连接,正在启动新会话...") return startSessionOnly() } @@ -648,7 +643,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { "errorMessage" to errorMsg)) webSocket.close(1000, "Error") updateStatus(STATUS_ERROR) - isSpeaking = false // 错误情况下显式设置 isSessionStarted = false isConnected = false isConnecting.set(false) @@ -656,6 +650,10 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { EVENT_SESSION_STARTED -> { connectionAttempts = 0 + // 会话开始时,如果使用内部播放器则启动播放器会话 + if (useInternalPlayer) { + audioPlayer.startSession() + } } EVENT_TTS_RESPONSE -> { @@ -673,16 +671,18 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { EVENT_SESSION_FINISHED -> { isSessionStarted = false - - if (!isStreamMode) { - finishConnection(webSocket) + Log.i(TAG, "EVENT_SESSION_FINISHED, ${sessionId}") + // 如果使用内部播放器则结束播放器会话 + if (useInternalPlayer) { + audioPlayer.endSession() } + + // 流式模式下不自动关闭连接 } } } catch (e: Exception) { Log.e(TAG, "解析响应失败: ${e.message}") updateStatus(STATUS_ERROR) - isSpeaking = false // 错误情况下显式设置 isConnecting.set(false) } } @@ -690,7 +690,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { override fun onFailure(webSocket: WebSocket, t: Throwable, response: Response?) { isConnected = false isSessionStarted = false - isSpeaking = false isConnecting.set(false) Log.e(TAG, "WebSocket连接失败: ${t.message}") @@ -702,7 +701,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { override fun onClosed(webSocket: WebSocket, code: Int, reason: String) { isConnected = false isSessionStarted = false - isSpeaking = false isConnecting.set(false) updateStatus(STATUS_STOPPED) notifyEvent(TtsEventType.SYNTHESIS_CANCELED) @@ -798,53 +796,6 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { sendEvent(webSocket, header, optional, payload) } - /** - * 发送文本进行合成 (非流式) - */ - private fun sendTextRequest(text: String) { - webSocket?.let { ws -> - val header = Header( - protocolVersion = PROTOCOL_VERSION, - headerSize = DEFAULT_HEADER_SIZE, - messageType = FULL_CLIENT_REQUEST, - messageTypeSpecificFlags = MSG_TYPE_FLAG_WITH_EVENT, - serializationMethod = JSON, - messageCompression = COMPRESSION_NO, - reserved = 0 - ) - - val optional = Optional( - event = EVENT_TASK_REQUEST, - sessionId = sessionId - ) - - val jsonObject = JSONObject() - val user = JSONObject() - user.put("uid", "123456") - jsonObject.put("user", user) - jsonObject.put("event", EVENT_TASK_REQUEST) - jsonObject.put("namespace", "BidirectionalTTS") - - val reqParams = JSONObject() - // 将#和*替换为空格 - val processedText = text.replace("#", " ").replace("*", " ") - reqParams.put("text", processedText) - reqParams.put("speaker", speaker) - - val audioParams = JSONObject() - audioParams.put("format", format) - audioParams.put("sample_rate", sampleRate) - audioParams.put("emotion", "happy") - - reqParams.put("audio_params", audioParams) - jsonObject.put("req_params", reqParams) - - val payload = jsonObject.toString().toByteArray() - Log.d(TAG, "发送文本合成请求: ${jsonObject.toString()}") - - sendEvent(ws, header, optional, payload) - } - } /** * 发送文本进行合成 (流式) diff --git a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt index 3e3d056db..13f74a898 100644 --- a/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt +++ b/local_plugins/chat_api/android/src/main/kotlin/com/yunqiinnovation/chat_api/ChatApiService.kt @@ -157,7 +157,8 @@ class ChatApiService(private val context: android.content.Context? = null) : Cor currentStreamJob = null // 2. 通知旧会话被中止 - currSessionCallback?.onError(ChatApiException("Session aborted by new request")) + // currSessionCallback?.onError(ChatApiException("Session aborted by new request")) + currSessionCallback?.onComplete() // 直接完成当前会话 // 3. 清理状态 currSessionId = "" diff --git a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt index 91c8945d6..d43604582 100644 --- a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt +++ b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt @@ -89,4 +89,12 @@ interface ITtsService { * @param listener 要移除的音频数据监听器 */ fun removeAudioDataListener(listener: AudioDataListener) + + /** + * 设置是否使用内部播放器 + * + * @param useInternalPlayer true: 使用内部播放器自动播放音频 + * false: 仅通过音频数据监听器输出数据,不播放 + */ + fun setUseInternalPlayer(useInternalPlayer: Boolean) } \ No newline at end of file From a63217e5cad512ae813f758badad16df26f4e54f Mon Sep 17 00:00:00 2001 From: wolfplus2048 <30993207+wolfplus2048@users.noreply.github.com> Date: Sat, 21 Jun 2025 22:02:59 +0100 Subject: [PATCH 5/5] add --- .../azure_speech/AzureSpeechPlugin.kt | 8 ++ .../azure_speech/AzureTtsHelper.kt | 44 +++++++- .../bytedance_speech/BytedanceAudioPlayer.kt | 100 ++++++++++++++++-- .../bytedance_speech/BytedanceSpeechPlugin.kt | 6 ++ .../bytedance_speech/BytedanceTTS.kt | 24 ++++- .../com/deep_voice/speech/ITtsService.kt | 7 ++ .../kotlin/com/deep_voice/speech/TtsEvents.kt | 16 +++ 7 files changed, 193 insertions(+), 12 deletions(-) diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt index e19579cc3..54e960092 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -158,6 +158,14 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { ) } + TtsEventType.PLAYBACK_STARTED -> mapOf( + "type" to "playback_started" + ) + + TtsEventType.PLAYBACK_COMPLETED -> mapOf( + "type" to "playback_completed" + ) + TtsEventType.ERROR -> { val params = event.params mapOf( diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 9b3d8ccfe..0465e5e4c 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -1,14 +1,16 @@ package com.yunqiinnovation.azure_speech import android.content.Context +import android.media.AudioManager import com.microsoft.cognitiveservices.speech.* import com.microsoft.cognitiveservices.speech.audio.* import com.yunqiinnovation.azure_speech.utils.FileLogger +import com.deep_voice.speech.tts.AudioDataListener +import com.deep_voice.speech.tts.AudioOutputDevice import com.deep_voice.speech.tts.ITtsService import com.deep_voice.speech.tts.TtsEvent import com.deep_voice.speech.tts.TtsEventListener import com.deep_voice.speech.tts.TtsEventType -import com.deep_voice.speech.tts.AudioDataListener import kotlinx.coroutines.* import java.io.ByteArrayInputStream import java.io.InputStream @@ -203,6 +205,42 @@ class AzureTtsHelper(private val context: Context) : ITtsService { } } + /** + * 设置音频输出设备 + * + * @param device 音频输出设备类型 + */ + override fun setAudioOutputDevice(device: AudioOutputDevice) { + // Azure TTS使用系统默认的音频路由 + // 音频输出设备的控制需要通过Android的AudioManager实现 + val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager + audioManager?.let { manager -> + when (device) { + AudioOutputDevice.DEFAULT -> { + // 默认模式:系统自动选择 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + } + AudioOutputDevice.SPEAKER -> { + // 强制使用扬声器 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = true + } + AudioOutputDevice.HEADPHONES -> { + // 强制使用耳机(如果已连接) + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 + } + AudioOutputDevice.EARPIECE -> { + // 强制使用听筒 + manager.mode = AudioManager.MODE_IN_COMMUNICATION + manager.isSpeakerphoneOn = false + } + } + } + } + /** * 触发事件通知 */ @@ -228,6 +266,8 @@ class AzureTtsHelper(private val context: Context) : ITtsService { SynthesisStarted?.addEventListener { _, eventArgs -> FileLogger.d(TAG, "语音合成开始: resultId=${eventArgs.result.resultId}") notifyEvent(TtsEventType.SYNTHESIS_STARTED) + // Azure TTS 在使用默认音频输出时会立即开始播放 + notifyEvent(TtsEventType.PLAYBACK_STARTED) } // 合成中事件(接收音频数据) @@ -253,6 +293,8 @@ class AzureTtsHelper(private val context: Context) : ITtsService { FileLogger.d(TAG, "语音合成完成: resultId=${eventArgs.result.resultId}, 音频长度=${eventArgs.result.audioLength} 字节") isSpeaking = false notifyEvent(TtsEventType.SYNTHESIS_COMPLETED) + // Azure TTS 合成完成即播放完成 + notifyEvent(TtsEventType.PLAYBACK_COMPLETED) } // 合成取消事件 diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt index 84e73b1be..a3f2c972b 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceAudioPlayer.kt @@ -1,5 +1,6 @@ package com.deep_voice.bytedance_speech +import android.content.Context import android.media.AudioAttributes import android.media.AudioFormat import android.media.AudioManager @@ -7,12 +8,13 @@ import android.media.AudioTrack import android.os.Build import android.util.Log import com.deep_voice.speech.tts.AudioDataListener +import com.deep_voice.speech.tts.AudioOutputDevice /** * 简化的音频流播放器 * 基于位置标记触发播放完成回调 */ -class BytedanceAudioPlayer : AudioDataListener { +class BytedanceAudioPlayer(private val context: Context) : AudioDataListener { companion object { private const val TAG = "BytedanceAudioPlayer" private const val SAMPLE_RATE = 24000 @@ -24,6 +26,8 @@ class BytedanceAudioPlayer : AudioDataListener { private var isFirstData = true // 是否是第一次接收数据 private var totalBytesWritten = 0 // 总共写入的字节数 private var sessionActive = true // 会话是否活跃 + private var audioOutputDevice = AudioOutputDevice.DEFAULT // 音频输出设备 + private var audioManager: AudioManager? = null // 回调 private var onPlayStarted: (() -> Unit)? = null @@ -40,6 +44,12 @@ class BytedanceAudioPlayer : AudioDataListener { // 不使用周期性通知 } } + + init { + Log.d(TAG, "BytedanceAudioPlayer 初始化") + audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager + initAudioTrack() + } /** @@ -198,6 +208,23 @@ class BytedanceAudioPlayer : AudioDataListener { return audioTrack?.playState == AudioTrack.PLAYSTATE_PLAYING } + /** + * 设置音频输出设备 + */ + fun setAudioOutputDevice(device: AudioOutputDevice) { + if (audioOutputDevice != device) { + audioOutputDevice = device + // 如果AudioTrack已初始化,需要重新创建以应用新的输出设备设置 + if (audioTrack != null) { + val wasPlaying = isPlaying() + initAudioTrack() + if (wasPlaying) { + audioTrack?.play() + } + } + } + } + /** * 初始化 AudioTrack */ @@ -209,15 +236,36 @@ class BytedanceAudioPlayer : AudioDataListener { val minBufferSize = AudioTrack.getMinBufferSize(SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT) val bufferSize = minBufferSize * 2 - // 创建 AudioTrack - audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { - AudioTrack.Builder() - .setAudioAttributes( + // 根据输出设备配置AudioAttributes + val audioAttributes = when (audioOutputDevice) { + AudioOutputDevice.EARPIECE -> { + // 听筒模式 + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { + AudioAttributes.Builder() + .setUsage(AudioAttributes.USAGE_VOICE_COMMUNICATION) + .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH) + .build() + } else { + null + } + } + else -> { + // 默认、耳机、扬声器模式 + if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M) { AudioAttributes.Builder() .setUsage(AudioAttributes.USAGE_MEDIA) .setContentType(AudioAttributes.CONTENT_TYPE_SPEECH) .build() - ) + } else { + null + } + } + } + + // 创建 AudioTrack + audioTrack = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.M && audioAttributes != null) { + AudioTrack.Builder() + .setAudioAttributes(audioAttributes) .setAudioFormat( AudioFormat.Builder() .setEncoding(AUDIO_FORMAT) @@ -230,8 +278,12 @@ class BytedanceAudioPlayer : AudioDataListener { .build() } else { @Suppress("DEPRECATION") + val streamType = when (audioOutputDevice) { + AudioOutputDevice.EARPIECE -> AudioManager.STREAM_VOICE_CALL + else -> AudioManager.STREAM_MUSIC + } AudioTrack( - AudioManager.STREAM_MUSIC, + streamType, SAMPLE_RATE, CHANNEL_CONFIG, AUDIO_FORMAT, @@ -243,9 +295,43 @@ class BytedanceAudioPlayer : AudioDataListener { // 设置播放位置监听器 audioTrack?.setPlaybackPositionUpdateListener(playbackListener) + // 配置音频路由 + configureAudioRouting() + // 开始播放 audioTrack?.play() Log.d(TAG, "AudioTrack 初始化成功") } + + /** + * 配置音频路由 + */ + private fun configureAudioRouting() { + audioManager?.let { manager -> + when (audioOutputDevice) { + AudioOutputDevice.DEFAULT -> { + // 默认模式:系统自动选择 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + } + AudioOutputDevice.SPEAKER -> { + // 强制使用扬声器 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = true + } + AudioOutputDevice.HEADPHONES -> { + // 强制使用耳机(如果已连接) + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 + } + AudioOutputDevice.EARPIECE -> { + // 强制使用听筒 + manager.mode = AudioManager.MODE_IN_COMMUNICATION + manager.isSpeakerphoneOn = false + } + } + } + } } \ No newline at end of file diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceSpeechPlugin.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceSpeechPlugin.kt index d86faa71f..a6003e050 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceSpeechPlugin.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceSpeechPlugin.kt @@ -255,6 +255,12 @@ class BytedanceSpeechPlugin : FlutterPlugin { TtsEventType.SYNTHESIS_CANCELED -> { sendTTSEvent("canceled", null) } + TtsEventType.PLAYBACK_STARTED -> { + sendTTSEvent("playback_started", null) + } + TtsEventType.PLAYBACK_COMPLETED -> { + sendTTSEvent("playback_completed", null) + } TtsEventType.ERROR -> { val params = HashMap() params["code"] = event.params["errorCode"] ?: "UNKNOWN_ERROR" diff --git a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt index 49d83efcc..b3cd7e75f 100644 --- a/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt +++ b/local_plugins/bytedance_speech/android/src/main/kotlin/com/deep_voice/bytedance_speech/BytedanceTTS.kt @@ -3,6 +3,7 @@ package com.deep_voice.bytedance_speech import android.content.Context import android.util.Log import com.deep_voice.speech.tts.AudioDataListener +import com.deep_voice.speech.tts.AudioOutputDevice import com.deep_voice.speech.tts.ITtsService import com.deep_voice.speech.tts.TtsEvent import com.deep_voice.speech.tts.TtsEventListener @@ -98,7 +99,7 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { private val audioDataListeners = mutableListOf() // 内部音频播放器 - private val audioPlayer = BytedanceAudioPlayer() + private val audioPlayer = BytedanceAudioPlayer(context) // 是否使用内部播放器 private var useInternalPlayer = true @@ -159,14 +160,14 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { audioPlayer.setOnPlayStarted { Log.d(TAG, "播放开始") updateStatus(STATUS_SPEAKING) - notifyEvent(TtsEventType.SYNTHESIS_STARTED) + notifyEvent(TtsEventType.PLAYBACK_STARTED) } // 设置播放完成回调 audioPlayer.setOnPlayCompleted { Log.d(TAG, "播放结束") updateStatus(STATUS_READY) - notifyEvent(TtsEventType.SYNTHESIS_COMPLETED) + notifyEvent(TtsEventType.PLAYBACK_COMPLETED) } // 设置错误回调 @@ -421,6 +422,15 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { updateInternalPlayerUsage() } + /** + * 设置音频输出设备 + */ + override fun setAudioOutputDevice(device: AudioOutputDevice) { + if (useInternalPlayer) { + audioPlayer.setAudioOutputDevice(device) + } + } + /** * 更新内部播放器的使用状态 */ @@ -650,7 +660,9 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { EVENT_SESSION_STARTED -> { connectionAttempts = 0 - // 会话开始时,如果使用内部播放器则启动播放器会话 + // 会话开始时触发合成开始事件 + notifyEvent(TtsEventType.SYNTHESIS_STARTED) + // 如果使用内部播放器则启动播放器会话 if (useInternalPlayer) { audioPlayer.startSession() } @@ -672,6 +684,10 @@ class BytedanceTTS(private val context: Context) : ITtsService, CoroutineScope { EVENT_SESSION_FINISHED -> { isSessionStarted = false Log.i(TAG, "EVENT_SESSION_FINISHED, ${sessionId}") + + // 触发合成完成事件 + notifyEvent(TtsEventType.SYNTHESIS_COMPLETED) + // 如果使用内部播放器则结束播放器会话 if (useInternalPlayer) { audioPlayer.endSession() diff --git a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt index d43604582..00db2f2b3 100644 --- a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt +++ b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/ITtsService.kt @@ -97,4 +97,11 @@ interface ITtsService { * false: 仅通过音频数据监听器输出数据,不播放 */ fun setUseInternalPlayer(useInternalPlayer: Boolean) + + /** + * 设置音频输出设备 + * + * @param device 音频输出设备类型 + */ + fun setAudioOutputDevice(device: AudioOutputDevice) } \ No newline at end of file diff --git a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt index cb69a849d..6758c6f05 100644 --- a/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt +++ b/local_plugins/speech/android/src/main/kotlin/com/deep_voice/speech/TtsEvents.kt @@ -13,6 +13,12 @@ enum class TtsEventType { /** 合成取消 */ SYNTHESIS_CANCELED, + /** 播放开始 */ + PLAYBACK_STARTED, + + /** 播放结束 */ + PLAYBACK_COMPLETED, + /** 发生错误 */ ERROR } @@ -52,4 +58,14 @@ interface AudioDataListener { * @param data 音频数据字节数组 */ fun onAudioData(data: ByteArray) +} + +/** + * 音频输出设备类型 + */ +enum class AudioOutputDevice { + DEFAULT, // 默认(如果有耳机选耳机,否则使用系统扬声器) + HEADPHONES, // 强制使用耳机 + SPEAKER, // 强制使用扬声器 + EARPIECE // 强制使用听筒 } \ No newline at end of file