diff --git a/.metadata b/.metadata index b35d0bbff..7686b46e4 100644 --- a/.metadata +++ b/.metadata @@ -15,7 +15,7 @@ migration: - platform: root create_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e base_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e - - platform: macos + - platform: android create_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e base_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts index 5f55eb578..93d066c28 100644 --- a/android/settings.gradle.kts +++ b/android/settings.gradle.kts @@ -28,6 +28,7 @@ plugins { id("com.android.application") version "8.9.1" apply false id("com.android.library") version "8.9.1" apply false id("org.jetbrains.kotlin.android") version "2.1.10" apply false + id("com.google.protobuf") version "0.9.4" apply false } include(":app") diff --git a/lib/data/services/ast_service.dart b/lib/data/services/ast_service.dart index 244a7fb37..8692e2b45 100644 --- a/lib/data/services/ast_service.dart +++ b/lib/data/services/ast_service.dart @@ -7,19 +7,25 @@ abstract class AstService { List get supportedLanguages; /// 初始化语音识别服务 - Future initialize({required List supportedLanguages}); + /// [supportedLanguages] 支持的语言列表 + /// [provider] 提供商:azure / volcano / alibaba / iflytek + Future initialize( + {required List supportedLanguages, String provider = 'azure'}); + + /// 设置低音量阈值 + void setLowVolumeThreshold(int threshold); /// 开始录音 Future enableRecord(String filePath); - /// + /// 开始连续翻译 Future startContinuousTranslation(); - /// + /// 停止连续翻译 Future stopContinuousTranslation(); /// 返回一个包含识别事件的流 - Future> recognizeCallback(); + Future> recognizeCallback(); /// 开始录音 Future path(String filePath); @@ -29,13 +35,19 @@ abstract class AstService { } // 识别事件类型 -enum RecognitionEventType1 { +enum ASTEventType { /// 最终识别结果 finalResult, /// 中间识别结果(实时反馈) intermediateResult, + /// 翻译结果 + translationResult, + + /// 翻译中(实时反馈) + translationInterim, + /// 会话开始 sessionStarted, @@ -50,9 +62,15 @@ enum RecognitionEventType1 { } // 识别事件 -class RecognitionEvent1 { +class ASTEvent { /// 事件类型 - final RecognitionEventType1 type; + final ASTEventType type; + + /// 识别ID + final String utteranceId; + + /// 服务ID(A/B) + final String serviceId; /// 识别文本(仅在 finalResult 和 intermediateResult 类型中有效) final String text; @@ -60,54 +78,55 @@ class RecognitionEvent1 { /// 检测到的语言 final String detectedLanguage; - /// 角色 - final String role; - /// 原始音频 final Uint8List? audio; /// 错误信息(仅在 error 和 canceled 类型中有效) final String error; - RecognitionEvent1({ + ASTEvent({ required this.type, + this.utteranceId = '', this.text = '', this.detectedLanguage = '', - this.role = '', this.audio, this.error = '', + this.serviceId = '', }); /// 创建最终结果事件的快捷构造函数 - factory RecognitionEvent1.finalResult({ + factory ASTEvent.finalResult({ required String text, + String utteranceId = '', String detectedLanguage = '', + String serviceId = '', }) { - return RecognitionEvent1( - type: RecognitionEventType1.finalResult, + return ASTEvent( + type: ASTEventType.finalResult, + utteranceId: utteranceId, text: text, detectedLanguage: detectedLanguage, + serviceId: serviceId, ); } /// 创建错误事件的快捷构造函数 - factory RecognitionEvent1.error(String errorMessage) { - return RecognitionEvent1( - type: RecognitionEventType1.error, + factory ASTEvent.error(String errorMessage) { + return ASTEvent( + type: ASTEventType.error, error: errorMessage, ); } /// 检查是否为最终结果 - bool get isFinalResult => type == RecognitionEventType1.finalResult; + bool get isFinalResult => type == ASTEventType.finalResult; /// 检查是否为错误 bool get isError => - type == RecognitionEventType1.error || - type == RecognitionEventType1.canceled; + type == ASTEventType.error || type == ASTEventType.canceled; @override String toString() { - return 'RecognitionEvent{type: $type, text: $text, detectedLanguage: $detectedLanguage, error: $error}'; + return 'ASTEvent{type: $type, text: $text, detectedLanguage: $detectedLanguage, error: $error}'; } } diff --git a/lib/data/services/language_manager.dart b/lib/data/services/language_manager.dart index d20f3776e..b41b17166 100644 --- a/lib/data/services/language_manager.dart +++ b/lib/data/services/language_manager.dart @@ -1,5 +1,12 @@ import 'package:get/get.dart'; import '../../core/utils/logger.dart'; +import '../services/speech_factory.dart'; +import 'language_configs/azure_language_config.dart'; +import 'language_configs/google_language_config.dart'; +import 'language_configs/iflytek_language_config.dart'; +import 'language_configs/volcano_language_config.dart'; +import 'language_configs/alibaba_language_config.dart'; +import 'language_configs/ast_config.dart'; /// 语言信息类,包含语言的各种表示形式 class LanguageInfo { @@ -529,4 +536,172 @@ class LanguageManager extends GetxService { Map getShortCodeToAsrCodeMap() { return Map.from(_shortCodeToAsrCodeMap); } + + /// 返回当前所有的AST语言配置(中文名称) + List getAllAstLanguageNames() { + return getAstLanguageConfigs() + .map((e) => e['chineseName'] ?? '') + .where((e) => e.isNotEmpty) + .toList(); + } + + /// 查找同时支持源语言和目标语言的最佳提供商(通过 shortCode 匹配) + /// 优先级:豆包(Volcano) > 阿里(Alibaba) > 微软(Azure) > 讯飞(Iflytek) + BestProviderResult? findBestMatchingProvider( + String sourceShortCode, String targetShortCode) { + final providers = [ + SpeechServiceType.volcano, + SpeechServiceType.alibaba, + SpeechServiceType.azure, + SpeechServiceType.iflytek, + ]; + + for (final provider in providers) { + final transSpecs = _getTranslationSpecs(provider); + + Map? sourceTransSpec; + Map? targetTransSpec; + + for (final spec in transSpecs) { + if (spec['shortCode'] == sourceShortCode) { + sourceTransSpec ??= {}; + sourceTransSpec["translationCode"] = spec["translationCode"]!; + } + if (spec['shortCode'] == targetShortCode) { + targetTransSpec ??= {}; + targetTransSpec["translationCode"] = spec["translationCode"]!; + } + } + + if (sourceTransSpec == null || targetTransSpec == null) { + continue; + } + + // 豆包特殊限制:源语言或目标语言必须有一个是中文或英语 + if (provider == SpeechServiceType.volcano) { + final isSourceZhOrEn = + sourceShortCode == 'zh' || sourceShortCode == 'en'; + final isTargetZhOrEn = + targetShortCode == 'zh' || targetShortCode == 'en'; + if (!isSourceZhOrEn && !isTargetZhOrEn) { + continue; + } + } + + // 对于非端到端服务(Azure, Iflytek),还需要检查 ASR 和 TTS 支持 + if (provider != SpeechServiceType.volcano && + provider != SpeechServiceType.alibaba) { + final asrSpecs = _getAsrSpecs(provider); + final ttsSpecs = _getTtsSpecs(provider); + + for (final spec in asrSpecs) { + if (spec['shortCode'] == sourceShortCode) { + sourceTransSpec["asrCode"] = spec["asrCode"]!; + } + if (spec['shortCode'] == targetShortCode) { + targetTransSpec["asrCode"] = spec["asrCode"]!; + } + } + + for (final spec in ttsSpecs) { + if (spec['shortCode'] == sourceShortCode) { + sourceTransSpec["ttsCode"] = spec["ttsVoiceName"]!; + } + if (spec['shortCode'] == targetShortCode) { + targetTransSpec["ttsCode"] = spec["ttsVoiceName"]!; + } + } + } else if (provider == SpeechServiceType.alibaba) { + final ttsSpecs = _getTtsSpecs(provider); + for (final spec in ttsSpecs) { + if (spec['shortCode'] == sourceShortCode) { + sourceTransSpec["ttsCode"] = spec["ttsVoiceName"]!; + } + if (spec['shortCode'] == targetShortCode) { + targetTransSpec["ttsCode"] = spec["ttsVoiceName"]!; + } + } + } + + final providerString = _providerStringFromType(provider); + Logger.info( + 'Found best provider: $providerString for $sourceShortCode -> $targetShortCode'); + return BestProviderResult( + providerString, sourceTransSpec, targetTransSpec); + } + + return null; + } + + String _providerStringFromType(SpeechServiceType type) { + switch (type) { + case SpeechServiceType.iflytek: + return 'iflytek'; + case SpeechServiceType.azure: + return 'azure'; + case SpeechServiceType.google: + return 'google'; + case SpeechServiceType.alibaba: + return 'alibaba'; + case SpeechServiceType.volcano: + return 'volcano'; + case SpeechServiceType.flutter: + return 'flutter'; + } + } + + // ===== 私有工具方法:加载不同提供商的语言规格 ===== + + List> _getAsrSpecs(SpeechServiceType provider) { + switch (provider) { + case SpeechServiceType.azure: + return getAzureASRLanguageSpecs(); + case SpeechServiceType.google: + return getGoogleASRLanguageSpecs(); + case SpeechServiceType.iflytek: + return getIflytekASRLanguageSpecs(); + default: + return getAzureASRLanguageSpecs(); + } + } + + List> _getTranslationSpecs(SpeechServiceType provider) { + switch (provider) { + case SpeechServiceType.azure: + return getAzureTranslationLanguageSpecs(); + case SpeechServiceType.google: + return getGoogleTranslationLanguageSpecs(); + case SpeechServiceType.iflytek: + return getIflytekTranslationLanguageSpecs(); + case SpeechServiceType.volcano: + return getVolcanoTranslationLanguageSpecs(); + case SpeechServiceType.alibaba: + return getAlibabaTranslationLanguageSpecs(); + default: + return getAzureTranslationLanguageSpecs(); + } + } + + List> _getTtsSpecs(SpeechServiceType provider) { + switch (provider) { + case SpeechServiceType.azure: + return getAzureTTSLanguageSpecs(); + case SpeechServiceType.google: + return getGoogleTTSLanguageSpecs(); + case SpeechServiceType.iflytek: + return getIflytekTTSLanguageSpecs(); + case SpeechServiceType.alibaba: + return getAlibabaTranslationLanguageSpecs(); + default: + return getAzureTTSLanguageSpecs(); + } + } +} + +class BestProviderResult { + final String provider; + final Map sourceSpec; + final Map targetSpec; + + BestProviderResult(this.provider, this.sourceSpec, this.targetSpec); } diff --git a/lib/data/services/speech_factory.dart b/lib/data/services/speech_factory.dart index 0d4568023..7127431aa 100644 --- a/lib/data/services/speech_factory.dart +++ b/lib/data/services/speech_factory.dart @@ -11,6 +11,18 @@ enum SpeechServiceType { /// Azure 语音服务 azure, + /// 科大讯飞语音服务 + iflytek, + + /// Google 语音服务 + google, + + /// 阿里巴巴语音服务 + alibaba, + + /// 火山引擎(豆包)语音服务 + volcano, + /// Flutter 本地语音服务 flutter, } diff --git a/lib/data/services/speech_impl/azure_ast_service.dart b/lib/data/services/speech_impl/azure_ast_service.dart index a806adc6d..1fa59abe2 100644 --- a/lib/data/services/speech_impl/azure_ast_service.dart +++ b/lib/data/services/speech_impl/azure_ast_service.dart @@ -5,12 +5,6 @@ import '../../../core/utils/logger.dart'; import 'package:get/get.dart'; import '../ast_service.dart'; -/// 音频源类型 -enum AudioSourceType { - microphone, // 使用设备麦克风 - external // 使用外部提供的音频数据 -} - /// Azure 语音识别服务 /// /// 该服务提供了通过平台通道与原生 Microsoft Speech SDK 交互的接口 @@ -19,16 +13,31 @@ class AzureAstService extends GetxService implements AstService { static const MethodChannel _channel = MethodChannel('azure_speech/ast'); static const EventChannel _eventChannel = EventChannel('azure_speech/ast_events'); - // final GetStorage _storage = GetStorage(); + bool _isInitialized = false; late final String _subscriptionKey; late final String _serviceRegion; - late String _baseUrl; - final String _endpoint = '/'; // 修改为根路径 - late final String _accessKey; - late final String _secretKey; - late final String _region; - late final String _service; + late final String _azureTranslationKey; + late final String _azureTranslationRegion; + + late final String _volcanoTranslationAccessKey; + late final String _volcanoTranslationSecretKey; + late final String _volcanoTranslationRegion; + + late final String _xunfeiAppId; + late final String _xunfeiAccessKeyId; + late final String _xunfeiAccessKeySecret; + late final String _iflytekHost; + + /// 豆包语音识别服务相关配置 + late final String _doubaoAppKey; + late final String _doubaoAccessKey; + late final String _doubaoResourceId; + + /// 阿里巴巴语音识别服务相关配置 + late final String _alibabaAppKey; + late final String _alibabaAppId; + late final String _alibabaAppURL; final List _defaultSupportedLanguages = ['zh-CN', 'en-US']; @override @@ -36,7 +45,7 @@ class AzureAstService extends GetxService implements AstService { //连续识别相关 bool _isContinuousRecognitionActive = false; - StreamController? _eventStreamController; + StreamController? _eventStreamController; StreamSubscription? _eventSubscription; // 最新的识别结果 @@ -47,9 +56,6 @@ class AzureAstService extends GetxService implements AstService { String _latestDetectedLanguage = ''; String get latestDetectedLanguage => _latestDetectedLanguage; - // 当前音频源类型 - AudioSourceType _audioSourceType = AudioSourceType.microphone; - AzureAstService() { _loadConfig(); } @@ -71,7 +77,8 @@ class AzureAstService extends GetxService implements AstService { }, onError: _handleRecognitionError); } - /// 处理来自原生端的识别事件 + /// 处理来自原生端的识别事件(AST事件) + /// 将原生侧 AST 事件映射为统一的 ASTEvent,提供给业务层使用。 void _handleRecognitionEvent(dynamic event) { if (event is! Map || _eventStreamController == null) return; @@ -79,40 +86,78 @@ class AzureAstService extends GetxService implements AstService { final String eventType = eventMap['type'] as String? ?? ''; switch (eventType) { - case 'result': + case 'recognized': + print("Ast处理识别完成事件:${eventMap}"); + final String serviceId = eventMap['serviceId'] as String? ?? ''; + final String utteranceId = eventMap['utteranceId'] as String? ?? ''; final String text = eventMap['text'] as String? ?? ''; - final String detectedLanguage = - eventMap['detectedLanguage'] as String? ?? ''; + final String detectedLanguage = eventMap['language'] as String? ?? ''; _latestRecognizedText = text; _latestDetectedLanguage = detectedLanguage; - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.finalResult, + _eventStreamController?.add(ASTEvent( + type: ASTEventType.finalResult, + serviceId: serviceId, + utteranceId: utteranceId, text: text, detectedLanguage: detectedLanguage, )); break; case 'recognizing': + print("Ast处理识别中事件:${eventMap}"); + final String serviceId = eventMap['serviceId'] as String? ?? ''; + final String utteranceId = eventMap['utteranceId'] as String? ?? ''; final String text = eventMap['text'] as String? ?? ''; + final String detectedLanguage = eventMap['language'] as String? ?? ''; + _eventStreamController?.add(ASTEvent( + type: ASTEventType.intermediateResult, + serviceId: serviceId, + utteranceId: utteranceId, + text: text, + detectedLanguage: detectedLanguage, + )); + break; + case 'translatedInterim': + print("Ast处理翻译中事件:${eventMap}"); + final String serviceId = eventMap['serviceId'] as String? ?? ''; + final String utteranceId = eventMap['utteranceId'] as String? ?? ''; + final String text = eventMap['translatedText'] as String? ?? ''; + final String detectedLanguage = + eventMap['targetLanguage'] as String? ?? ''; + _eventStreamController?.add(ASTEvent( + type: ASTEventType.translationInterim, + serviceId: serviceId, + utteranceId: utteranceId, + text: text, + detectedLanguage: detectedLanguage, + )); + break; + case 'translated': + print("Ast处理翻译结果事件:${eventMap}"); + final String serviceId = eventMap['serviceId'] as String? ?? ''; + final String utteranceId = eventMap['utteranceId'] as String? ?? ''; + final String text = eventMap['translatedText'] as String? ?? ''; final String detectedLanguage = - eventMap['detectedLanguage'] as String? ?? ''; - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.intermediateResult, + eventMap['targetLanguage'] as String? ?? ''; + _eventStreamController?.add(ASTEvent( + type: ASTEventType.translationResult, + serviceId: serviceId, + utteranceId: utteranceId, text: text, detectedLanguage: detectedLanguage, )); break; + case 'sessionStarted': - Logger.info('AST provider: azure'); - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.sessionStarted, + _eventStreamController?.add(ASTEvent( + type: ASTEventType.sessionStarted, )); break; case 'sessionStopped': _isContinuousRecognitionActive = false; - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.sessionStopped, + _eventStreamController?.add(ASTEvent( + type: ASTEventType.sessionStopped, )); break; @@ -125,17 +170,24 @@ class AzureAstService extends GetxService implements AstService { Logger.error('识别取消: $reason - ${errorDetails.toString()}'); } - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.canceled, + _eventStreamController?.add(ASTEvent( + type: ASTEventType.canceled, error: '$reason: $errorDetails', )); break; case 'error': - final String error = eventMap['message'] as String? ?? ''; + String error = eventMap['message'] as String? ?? ''; + if (error.isEmpty) { + error = eventMap['error'] as String? ?? ''; + } + final String code = eventMap['code']?.toString() ?? ''; + if (code.isNotEmpty) { + error = '$error (Code: $code)'; + } Logger.error('识别错误: ${error.toString()}'); - _eventStreamController?.add(RecognitionEvent1( - type: RecognitionEventType1.error, + _eventStreamController?.add(ASTEvent( + type: ASTEventType.error, error: error, )); break; @@ -158,14 +210,28 @@ class AzureAstService extends GetxService implements AstService { /// 从环境变量加载配置 void _loadConfig() { - // final _env = _storage.read("ENV") as Map; _subscriptionKey = AppConfig.env('AZURE_SPEECH_KEY') ?? ''; _serviceRegion = AppConfig.env('AZURE_SPEECH_REGION') ?? ''; - _accessKey = AppConfig.env('VOLCANO_TRANSLATION_ACCESS_KEY') ?? ''; - _secretKey = AppConfig.env('VOLCANO_TRANSLATION_SECRET_KEY') ?? ''; - _region = AppConfig.env('VOLCANO_TRANSLATION_REGION') ?? 'cn-north-1'; - _service = 'translate'; - _baseUrl = 'https://translate.volcengineapi.com'; + _azureTranslationKey = AppConfig.env('AZURE_TRANSLATION_KEY') ?? ''; + _azureTranslationRegion = AppConfig.env('AZURE_TRANSLATION_REGION') ?? ''; + _volcanoTranslationAccessKey = + AppConfig.env('VOLCANO_TRANSLATION_ACCESS_KEY') ?? ''; + _volcanoTranslationSecretKey = + AppConfig.env('VOLCANO_TRANSLATION_SECRET_KEY') ?? ''; + _volcanoTranslationRegion = + AppConfig.env('VOLCANO_TRANSLATION_REGION') ?? 'cn-north-1'; + _xunfeiAppId = AppConfig.env('XUNFEI_ASR_APP_ID') ?? ''; + _xunfeiAccessKeyId = AppConfig.env('XUNFEI_ASR_ACCESS_KEY_ID') ?? ''; + _xunfeiAccessKeySecret = + AppConfig.env('XUNFEI_ASR_ACCESS_KEY_SECRET') ?? ''; + _iflytekHost = AppConfig.env('IFLYTEK_ASR_HOST') ?? ''; + _doubaoAppKey = AppConfig.env('VOLC_OPENSPEECH_APP_ID') ?? ''; + _doubaoAccessKey = AppConfig.env('VOLC_OPENSPEECH_ACCESS_TOKEN') ?? ''; + _doubaoResourceId = + AppConfig.env('VOLC_OPENSPEECH_TRANSLATION_BIGMODEL') ?? ''; + _alibabaAppKey = AppConfig.env('ALIBABA_OPENSPEECH_APP_KEY') ?? ''; + _alibabaAppId = AppConfig.env('ALIBABA_OPENSPEECH_APP_ID') ?? ''; + _alibabaAppURL = AppConfig.env('ALIBABA_OPENSPEECH_APP_URL') ?? ''; if (_subscriptionKey.isEmpty || _serviceRegion.isEmpty) { throw Exception( '未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION'); @@ -173,12 +239,12 @@ class AzureAstService extends GetxService implements AstService { } @override - Future> recognizeCallback() async { + Future> recognizeCallback() async { if (!_isInitialized) { await initialize(); } try { - _eventStreamController = StreamController.broadcast(); + _eventStreamController = StreamController.broadcast(); // 开始连续识别 final bool result = await _channel.invokeMethod('recognizeCallback'); @@ -250,13 +316,11 @@ class AzureAstService extends GetxService implements AstService { @override Future dispose() async { try { - // 取消事件订阅,关闭事件流,重置状态(补充释放,避免监听残留) await _eventSubscription?.cancel(); _eventSubscription = null; await _eventStreamController?.close(); _eventStreamController = null; - // 通知原生端释放资源,并重置初始化标记 await _channel.invokeMethod('dispose'); _isInitialized = false; Logger.info('Azure AST 资源已释放'); @@ -269,46 +333,47 @@ class AzureAstService extends GetxService implements AstService { @override Future initialize({ List? supportedLanguages, - bool useExternalAudio = false, - bool useEchoCancellation = false, + String provider = 'azure', }) async { try { final List languages = supportedLanguages ?? _defaultSupportedLanguages; -//底层会初始化前释放 - // // 检查是否需要重新初始化 - // if (_isInitialized) { - // await dispose(); - // } - - // 设置音频源类型 - _audioSourceType = useExternalAudio - ? AudioSourceType.external - : AudioSourceType.microphone; -// Future initialize({ -// required String subscriptionKey, -// required String region, -// required List supportedLanguages, -// required String audioSourceType, -// required String translationAccessKey, -// required String translationSecretKey, -// String translationRegion = 'cn-north-1', -// }); + // 底层会初始化前释放 + if (_isInitialized) { + await dispose(); + } final bool result = await _channel.invokeMethod('initialize', { + 'provider': provider, 'subscriptionKey': _subscriptionKey, 'region': _serviceRegion, 'supportedLanguages': languages, - 'audioSourceType': _audioSourceType.toString().split('.').last, - 'translationAccessKey': _accessKey, - 'translationSecretKey': _secretKey, - 'translationRegion': _region, + + 'volcanoTranslationAccessKey': _volcanoTranslationAccessKey, + 'volcanoTranslationSecretKey': _volcanoTranslationSecretKey, + 'volcanoTranslationRegion': _volcanoTranslationRegion, + + 'azureTranslationKey': _azureTranslationKey, + 'azureTranslationServiceRegion': _azureTranslationRegion, + + 'xfyunAppId': _xunfeiAppId, + 'xfyunAccessKeyId': _xunfeiAccessKeyId, + 'xfyunAccessKeySecret': _xunfeiAccessKeySecret, + 'iflytekHost': _iflytekHost, + + 'appKey': _doubaoAppKey, + 'accessKey': _doubaoAccessKey, + 'resourceId': _doubaoResourceId, + + 'alibabaAppKey': _alibabaAppKey, + 'alibabaAppId': _alibabaAppId, + 'alibabaAppURL': _alibabaAppURL, }); _isInitialized = result; _setupEventChannel(); + print('Azure 语音识别服务初始化${result ? '成功' : '失败'}'); Logger.info('Azure 语音识别服务初始化${result ? '成功' : '失败'}'); - if (result) Logger.info('AST provider: azure'); return result; } catch (e) { Logger.error('Azure 语音识别服务初始化失败: ${e.toString()}'); @@ -316,4 +381,17 @@ class AzureAstService extends GetxService implements AstService { rethrow; } } + + @override + void setLowVolumeThreshold(int threshold) { + try { + _channel.invokeMethod('setLowVolumeThreshold', { + 'threshold': threshold, + }); + } catch (e) { + Logger.error('设置低音量阈值失败: ${e.toString()}'); + rethrow; + } + } + } diff --git a/lib/modules/speech_test/views/speech_test_view.dart b/lib/modules/speech_test/views/speech_test_view.dart index a87ec9676..afea231e4 100644 --- a/lib/modules/speech_test/views/speech_test_view.dart +++ b/lib/modules/speech_test/views/speech_test_view.dart @@ -101,6 +101,14 @@ class SpeechTestView extends GetView { switch (type) { case SpeechServiceType.azure: return 'Azure语音'; + case SpeechServiceType.iflytek: + return '讯飞语音'; + case SpeechServiceType.google: + return 'Google语音'; + case SpeechServiceType.alibaba: + return '阿里语音'; + case SpeechServiceType.volcano: + return '豆包语音'; case SpeechServiceType.flutter: return 'Flutter本地'; } diff --git a/lib/modules/translation/controllers/translation_controller.dart b/lib/modules/translation/controllers/translation_controller.dart index 4cc9621fa..cffd7cf59 100644 --- a/lib/modules/translation/controllers/translation_controller.dart +++ b/lib/modules/translation/controllers/translation_controller.dart @@ -145,6 +145,7 @@ class TranslationController extends GetxController with WidgetsBindingObserver { // ==================== 流订阅 ==================== StreamSubscription? _recognitionSubscription; + StreamSubscription? _astEventSubscription; StreamSubscription? _voiceInteractionSubscription; Timer? _translationDebounceTimer; @@ -845,12 +846,35 @@ class TranslationController extends GetxController with WidgetsBindingObserver { _recognitionSubscription = recognitionStream.listen(_handleRecognitionEvent); + final sourceShort = _languageManager.getShortCodeByAsrCode(sourceLanguageCode.value) ?? 'zh'; + final targetShort = _languageManager.getShortCodeByAsrCode(targetLanguageCode.value) ?? 'en'; + final bestProvider = _languageManager.findBestMatchingProvider(sourceShort, targetShort); + final astProvider = bestProvider?.provider ?? 'azure'; + + // 构建完整的语言参数:[asrCode0, asrCode1, transCode0, transCode1, ttsVoice0, ttsVoice1] + final transCode0 = bestProvider?.sourceSpec['translationCode'] ?? sourceLanguageCode.value; + final transCode1 = bestProvider?.targetSpec['translationCode'] ?? targetLanguageCode.value; + final ttsVoice0 = bestProvider?.sourceSpec['ttsCode'] ?? + (_languageManager.getTtsVoiceNameByAsrCode(sourceLanguageCode.value) ?? 'zh-CN-XiaoxiaoNeural'); + final ttsVoice1 = bestProvider?.targetSpec['ttsCode'] ?? + (_languageManager.getTtsVoiceNameByAsrCode(targetLanguageCode.value) ?? 'en-US-AriaNeural'); final List callModeLanguages = [ - sourceLanguageCode.value, - targetLanguageCode.value + sourceLanguageCode.value, // [0] ASR source + targetLanguageCode.value, // [1] ASR target + transCode0, // [2] translation source + transCode1, // [3] translation target + ttsVoice0, // [4] TTS voice source + ttsVoice1, // [5] TTS voice target ]; - Logger.info('1初始化AST服务,支持语言: $callModeLanguages'); //ast=asr+translate - await _astService.initialize(supportedLanguages: callModeLanguages); + Logger.info('1初始化AST服务,支持语言: $callModeLanguages, provider: $astProvider'); + + await _astService.initialize(supportedLanguages: callModeLanguages, provider: astProvider); + + // 获取识别事件流 + var astStream = await _astService.recognizeCallback(); + _astEventSubscription?.cancel(); + _astEventSubscription = astStream.listen(_handleAstEvent); + Logger.info('通话模式语音翻译服务初始化完成'); } catch (e) { Logger.error('通话模式语音翻译服务初始化失败: ${e.toString()}'); @@ -862,6 +886,8 @@ class TranslationController extends GetxController with WidgetsBindingObserver { stopAll(); _recognitionSubscription?.cancel(); _recognitionSubscription = null; + _astEventSubscription?.cancel(); + _astEventSubscription = null; _voiceInteractionSubscription?.cancel(); _translationDebounceTimer?.cancel(); restoreOtherServices(); @@ -1062,12 +1088,27 @@ class TranslationController extends GetxController with WidgetsBindingObserver { /// 配置通话模式 Future _configureCallMode() async { + final sourceShort = _languageManager.getShortCodeByAsrCode(sourceLanguageCode.value) ?? 'zh'; + final targetShort = _languageManager.getShortCodeByAsrCode(targetLanguageCode.value) ?? 'en'; + final bestProvider = _languageManager.findBestMatchingProvider(sourceShort, targetShort); + final astProvider = bestProvider?.provider ?? 'azure'; + + final transCode0 = bestProvider?.sourceSpec['translationCode'] ?? sourceLanguageCode.value; + final transCode1 = bestProvider?.targetSpec['translationCode'] ?? targetLanguageCode.value; + final ttsVoice0 = bestProvider?.sourceSpec['ttsCode'] ?? + (_languageManager.getTtsVoiceNameByAsrCode(sourceLanguageCode.value) ?? 'zh-CN-XiaoxiaoNeural'); + final ttsVoice1 = bestProvider?.targetSpec['ttsCode'] ?? + (_languageManager.getTtsVoiceNameByAsrCode(targetLanguageCode.value) ?? 'en-US-AriaNeural'); final List callModeLanguages = [ sourceLanguageCode.value, - targetLanguageCode.value + targetLanguageCode.value, + transCode0, + transCode1, + ttsVoice0, + ttsVoice1, ]; - Logger.info('1初始化AST服务,支持语言: $callModeLanguages'); //ast=asr+translate - await _astService.initialize(supportedLanguages: callModeLanguages); + Logger.info('1初始化AST服务,支持语言: $callModeLanguages, provider: $astProvider'); + await _astService.initialize(supportedLanguages: callModeLanguages, provider: astProvider); isPreparing.value = true; await bleManager.openA2DPDecoder(); int attempts = 0; @@ -1190,6 +1231,79 @@ class TranslationController extends GetxController with WidgetsBindingObserver { } } + /// 处理 AST(语音识别+翻译一体)事件 + void _handleAstEvent(ASTEvent event) { + switch (event.type) { + case ASTEventType.intermediateResult: + if (event.text.isEmpty) break; + Logger.info('AST 识别中 [${event.serviceId}]: ${event.text}'); + handleIntermediateResult(event.text); + break; + + case ASTEventType.finalResult: + if (event.text.isEmpty) break; + Logger.info('AST 识别完成 [${event.serviceId}]: ${event.text}'); + // AST 已内置翻译,直接更新源文本,不再调用 translateText + if (translationHistory.isNotEmpty && + translationHistory.last.isIntermediate) { + translationHistory.last.sourceText = event.text; + translationHistory.last.isIntermediate = false; + translationHistory.refresh(); + _scrollToBottom(); + saveTranslationHistory(); + } else { + final newItem = TranslationItem( + sourceText: event.text, + translatedText: '', + sourceLanguageCode: sourceLanguageCode.value, + targetLanguageCode: targetLanguageCode.value, + timestamp: DateTime.now(), + sessionId: currentSessionId ?? + DateTime.now().millisecondsSinceEpoch.toString(), + isFirstInSession: translationHistory.isEmpty || + translationHistory.last.sessionId != currentSessionId, + isIntermediate: false, + ); + translationHistory.add(newItem); + translationHistory.refresh(); + _scrollToBottom(); + saveTranslationHistory(); + } + break; + + case ASTEventType.translationInterim: + if (event.text.isEmpty) break; + Logger.info('AST 翻译中 [${event.serviceId}]: ${event.text}'); + if (translationHistory.isNotEmpty) { + translationHistory.last.translatedText = event.text; + translationHistory.refresh(); + } + break; + + case ASTEventType.translationResult: + if (event.text.isEmpty) break; + Logger.info('AST 翻译完成 [${event.serviceId}]: ${event.text}'); + if (translationHistory.isNotEmpty) { + translationHistory.last.translatedText = event.text; + translationHistory.refresh(); + _scrollToBottom(); + saveTranslationHistory(); + } + break; + + case ASTEventType.error: + Logger.error('AST 错误: ${event.error}'); + break; + + case ASTEventType.canceled: + Logger.error('AST 取消: ${event.error}'); + break; + + default: + break; + } + } + /// 停止语音识别 Future stopRecognition() async { if (!isRecognizing.value) return; diff --git a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt index b08534a5e..2d0dde0a3 100644 --- a/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt +++ b/local_plugins/agent_service/android/src/main/kotlin/com/yunqiinnovation/agent_service/AgentService.kt @@ -298,10 +298,7 @@ object AgentService : CoroutineScope { region = config["azureSpeechRegion"]?.toString() ?: "", supportedLanguages = supportedLanguages, audioSourceType = if (isExternalActive) AzureAsrHelper.AudioSourceType.EXTERNAL - else AzureAsrHelper.AudioSourceType.MICROPHONE, - xunfeiAppId = config["xunfeiAppId"]?.toString() ?: "", - xunfeiAccessKeyId = config["xunfeiAccessKeyId"]?.toString() ?: "", - xunfeiAccessKeySecret = config["xunfeiAccessKeySecret"]?.toString() ?: "" + else AzureAsrHelper.AudioSourceType.MICROPHONE ) if (asrInitSuccess) { @@ -705,7 +702,7 @@ object AgentService : CoroutineScope { if(isExternalActive){ isKeepResult = true } - azureAsrHelper?.startContinuousRecognition(audioSourceType, isRemoveFirstPunctuation = false) + azureAsrHelper?.startContinuousRecognition(audioSourceType) // 根据模式启动相应的空闲检测 startIdleCheckForMode(mode) @@ -1485,10 +1482,7 @@ object AgentService : CoroutineScope { subscriptionKey = this.azureSpeechKey, region = this.azureSpeechRegion, supportedLanguages = languages.toTypedArray(), - audioSourceType = audioSourceType, - xunfeiAppId = this.xunfeiAppId, - xunfeiAccessKeyId = this.xunfeiAccessKeyId, - xunfeiAccessKeySecret = this.xunfeiAccessKeySecret + audioSourceType = audioSourceType ) ?: false if (asrSuccess) { recognizeCallback() @@ -1725,7 +1719,7 @@ object AgentService : CoroutineScope { private fun sendEvent(eventName: String, data: Map) { val shouldAttachAsrProvider = eventName == "recognizing" || eventName.startsWith("recognition_") val finalData: Map = if (shouldAttachAsrProvider && !data.containsKey("provider")) { - val provider = azureAsrHelper?.getAsrProvider() ?: "unknown" + val provider = "azure" data.toMutableMap().apply { put("provider", provider) } } else { data diff --git a/local_plugins/azure_speech/.metadata b/local_plugins/azure_speech/.metadata index 5c982c8d4..7686b46e4 100644 --- a/local_plugins/azure_speech/.metadata +++ b/local_plugins/azure_speech/.metadata @@ -4,7 +4,7 @@ # This file should be version controlled and should not be manually edited. version: - revision: "ea121f8859e4b13e47a8f845e4586164519588bc" + revision: "30ade5e937e6afbd2d2c38ec53175b89b556447e" channel: "stable" project_type: app @@ -13,11 +13,11 @@ project_type: app migration: platforms: - platform: root - create_revision: ea121f8859e4b13e47a8f845e4586164519588bc - base_revision: ea121f8859e4b13e47a8f845e4586164519588bc - - platform: ios - create_revision: ea121f8859e4b13e47a8f845e4586164519588bc - base_revision: ea121f8859e4b13e47a8f845e4586164519588bc + create_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e + base_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e + - platform: android + create_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e + base_revision: 30ade5e937e6afbd2d2c38ec53175b89b556447e # User provided section diff --git a/local_plugins/azure_speech/android/build.gradle.kts b/local_plugins/azure_speech/android/build.gradle.kts index d4731f14c..1f0d19d01 100644 --- a/local_plugins/azure_speech/android/build.gradle.kts +++ b/local_plugins/azure_speech/android/build.gradle.kts @@ -1,6 +1,8 @@ +import com.google.protobuf.gradle.* plugins { id("com.android.library") id("org.jetbrains.kotlin.android") + id("com.google.protobuf") } android { @@ -20,6 +22,9 @@ android { getByName("main") { manifest.srcFile("src/main/AndroidManifest.xml") java.srcDirs("src/main/kotlin") + proto { + srcDir("src/main/protos") + } } } @@ -55,6 +60,39 @@ dependencies { implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.48.1") implementation(project(":speech")) implementation("com.squareup.okhttp3:okhttp:4.12.0") + implementation("com.google.protobuf:protobuf-javalite:3.25.1") add("compileOnly", project(":ble_service")) -} \ No newline at end of file + // gRPC + val grpcVersion = "1.57.2" + implementation("io.grpc:grpc-android:$grpcVersion") + implementation("io.grpc:grpc-okhttp:$grpcVersion") + implementation("io.grpc:grpc-protobuf-lite:$grpcVersion") + implementation("io.grpc:grpc-stub:$grpcVersion") + implementation("javax.annotation:javax.annotation-api:1.3.2") +} + +protobuf { + protoc { + artifact = "com.google.protobuf:protoc:3.25.1" + } + plugins { + id("grpc") { + artifact = "io.grpc:protoc-gen-grpc-java:1.57.2" + } + } + generateProtoTasks { + all().forEach { task -> + task.builtins { + id("java") { + option("lite") + } + } + task.plugins { + id("grpc") { + option("lite") + } + } + } + } +} diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt index 9a4297cfe..d6d3cc862 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -24,940 +24,796 @@ import java.util.UUID */ class AzureAsrHelper(private val context: Context) { - private val tag = "AzureAsrHelper" - - // 核心组件 - private var speechConfig: SpeechConfig? = null - private var recognizer: SpeechRecognizer? = null - private var audioConfig: AudioConfig? = null - private var continuousCallback: ContinuousRecognizeCallback? = null - - // 添加前台服务管理标志 - private var isForegroundServiceRunning = false - - // 配置参数 - private var currentLanguage = "zh-CN" - private var supportedLanguages = arrayOf("zh-CN") - private var isAutoDetectLanguage = false - private var subscriptionKey = "" - private var region = "" - - private var useXunfei = false - private var xunFeiAsrHelper: XunFeiAsrHelper? = null - - /// 原生日志回调,由 Plugin 层设置,通过 EventChannel 回传 Dart - var nativeLogCallback: ((String, String) -> Unit)? = null - private var xunfeiAppId: String = "" - private var xunfeiAccessKeyId: String = "" - private var xunfeiAccessKeySecret: String = "" - - fun getAsrProvider(): String { - return if (useXunfei) "xunfei" else "azure" - } - - // 音频源配置 - var audioSourceType = AudioSourceType.MICROPHONE - - // 音频处理 - Initialize immediately to avoid null checks - var audioStream: SimpleAudioReceiver = SimpleAudioReceiver(context) - - // 网络状态监听 - private var networkMonitor: NetworkStateMonitor = NetworkStateMonitor(context) - private var isNetworkRecovering = false - - // 网络丢失计数器 - private var networkLostCount = 0 - - // 最大网络丢失次数(3秒检测) - private val maxNetworkLostCount = 3 - - // 防抖机制相关变量 - private var lastNetworkLostTime = 0L - private val networkLostDebounceInterval = 3000L // 2秒防抖间隔 - private var currsessionid = "" //识别回话id 关联到整个聊天过程中 - - - /** - * 音频来源类型 - */ - enum class AudioSourceType { - /** 使用设备麦克风 */ - MICROPHONE, - - /** 使用外部提供的音频数据 */ - EXTERNAL - } - - - /** - * 初始化Azure语音服务 - * - * @param subscriptionKey Azure 订阅密钥 - * @param region Azure 区域 - * @param supportedLanguages 支持的语言数组 - * @param audioSourceType 音频源类型 - * @return 初始化是否成功 - */ - fun initialize( - subscriptionKey: String, - region: String, - supportedLanguages: Array = arrayOf("zh-CN", "en-US"), - audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE, - xunfeiAppId: String = "", - xunfeiAccessKeyId: String = "", - xunfeiAccessKeySecret: String = "" - ): Boolean { - try { - // 检查配置是否为空 - if (subscriptionKey.isEmpty() || region.isEmpty()) { - Log.e(tag, "Azure 配置信息不完整") - return false - } - - // 释放之前的资源 - dispose() - - // 保存配置 - this.subscriptionKey = subscriptionKey - this.region = region - this.audioSourceType = audioSourceType - this.xunfeiAppId = xunfeiAppId - this.xunfeiAccessKeyId = xunfeiAccessKeyId - this.xunfeiAccessKeySecret = xunfeiAccessKeySecret - - // 设置语言 - if (supportedLanguages.isNotEmpty()) { - this.supportedLanguages = supportedLanguages - } - - val initMsg = "[XunFei_ASR] ASR initialize: supportedLanguages=${supportedLanguages.joinToString(",")}, audioSourceType=$audioSourceType" - Log.i(tag, initMsg) - nativeLogCallback?.invoke("INFO", initMsg) - - val xunfeiConfigReady = this.xunfeiAppId.isNotBlank() && - this.xunfeiAccessKeyId.isNotBlank() && - this.xunfeiAccessKeySecret.isNotBlank() - val onlySupportedByXunfei = supportedLanguages.all { - it == "zh-CN" || it == "en-US" || it == "zh-HK" || it == "zh-TW" - } - val chineseVariantsCount = supportedLanguages - .map { it.trim().lowercase() } - .filter { it == "zh" || it.startsWith("zh-") } - .distinct() - .size - val hasMultipleChineseVariants = chineseVariantsCount >= 2 - this.useXunfei = onlySupportedByXunfei && xunfeiConfigReady && !hasMultipleChineseVariants - - // 根据支持的语言数量决定是否启用自动语言检测 - this.isAutoDetectLanguage = supportedLanguages.size >= 2 - - // 如果只有一种语言,设置为当前语言 - if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { - this.currentLanguage = supportedLanguages[0] - } - - val derivedMsg = "[XunFei_ASR] ASR config: provider=${getAsrProvider()}, xunfeiConfigured=$xunfeiConfigReady, onlyXunfeiLangs=$onlySupportedByXunfei, multiChinese=$hasMultipleChineseVariants, autoDetect=$isAutoDetectLanguage, currentLang=$currentLanguage" - Log.i(tag, derivedMsg) - nativeLogCallback?.invoke("INFO", derivedMsg) - - // 创建语音配置 - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region).apply { - if (isAutoDetectLanguage) { - // 直接启用语言检测模式 - setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") - } else { - // 设置指定的识别语言 - speechRecognitionLanguage = currentLanguage - } + private val tag = "AzureAsrHelper" - setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", "300") - setProperty("Speech_SegmentationSilenceTimeoutMs", "300") - - // 优化:添加低延迟连接配置 - setProperty("SpeechServiceConnection_InitialSilenceTimeoutMs", "200") - setProperty("SpeechServiceConnection_RecoMode", "INTERACTIVE") - setProperty("Speech_PushStreamFormat", "PCM") - - // 设置分段策略为时间模式 - setProperty("Speech_SegmentationStrategy", "Time") - } - // 初始化网络监听 - networkMonitor.initialize(object : NetworkStateMonitor.NetworkStateListener { - override fun onNetworkAvailable() { - handleNetworkAvailable() - } + // 核心组件 + private var speechConfig: SpeechConfig? = null + private var recognizer: SpeechRecognizer? = null + private var audioConfig: AudioConfig? = null + private var continuousCallback: ContinuousRecognizeCallback? = null - override fun onNetworkLost() { - handleNetworkLost() - } - }) + // 添加前台服务管理标志 + private var isForegroundServiceRunning = false - return true - } catch (e: Exception) { - Log.e(tag, "初始化失败: ${e.message}") - return false - } - } - - // /** - // * 设置音频配置 - // * @param sampleRate 采样率 (8000, 16000, 24000, 32000, 44100, 48000) - // * @param channels 声道数 (1=单声道, 2=立体声) - // */ - // fun setAudioConfig(sampleRate: Int, channels: Int) { - // try { - // // 设置RecordFile的音频配置 - // recordfile?.setAudioConfig(sampleRate, channels) - // Log.d(tag, "音频配置已设置: 采样率=${sampleRate}Hz, 声道数=${channels}") - // } catch (e: Exception) { - // Log.e(tag, "设置音频配置失败: ${e.message}") - // } - // } - - private fun isRecognizerValid(): Boolean { - return recognizer != null - } - - /** - * 设置识别器 - */ - private fun setupRecognizer(): Boolean { - try { - // 如果正在进行连续识别,先停止 - if (audioStream.isContinuousRecognitionActive) { - // 直接停止,不等待结果 - recognizer?.stopContinuousRecognitionAsync() - audioStream.isContinuousRecognitionActive = false - } - setupMicrophoneStream() - - recognizer?.close() - recognizer = null - //创建识别器 - recognizer = if (isAutoDetectLanguage) { - val autoDetectConfig = - AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) - } else { - SpeechRecognizer(speechConfig, audioConfig) - } - - return true - } catch (e: Exception) { - Log.e(tag, "创建识别器失败: ${e.message}") - stopAudioProcessing() - return false - } - } - - /** - * 设置麦克风流 - 使用拉流方式 - */ - private fun setupMicrophoneStream() { - try { - Log.d(tag, "设置麦克风流 - 使用拉流方式: }") - - // Initialize if not already done - if (!audioStream.isInitialized()) { - audioStream.initAudioRecord() - } - audioStream.setupStream() - audioConfig = AudioConfig.fromStreamInput(audioStream.pushAudioStream) - - } catch (e: Exception) { - Log.e(tag, "设置麦克风流失败: ${e.message}") - } - } - - - /** - * 开始连续语音识别 - * @param audioSourceType 音频源类型 - * @param audioDataCallback 音频数据回调 - * @param isRemoveFirstPunctuation 是否移除第一个标点符号 - * @return 是否成功开始识别 - */ - fun startContinuousRecognition( - audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE, - audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = null, - isRemoveFirstPunctuation: Boolean = true - ): Boolean { - - Log.d(tag, "startContinuousRecognition:${audioStream.isContinuousRecognitionActive} ") - - // 检查网络状态 - if (!checkNetworkStatus()) { - Log.e(tag, "网络不可用,无法开始语音识别") - continuousCallback?.onError(currsessionid, 1000, "网络连接不可用,请检查网络设置") - return false - } + // 配置参数 + private var currentLanguage = "zh-CN" + private var supportedLanguages = arrayOf("zh-CN") + private var isAutoDetectLanguage = false + private var subscriptionKey = "" + private var region = "" - if (audioStream.isContinuousRecognitionActive) { - return true - } + // 音频源配置 + var audioSourceType = AudioSourceType.MICROPHONE + + // 音频处理 - Initialize immediately to avoid null checks + var audioStream: SimpleAudioReceiver = SimpleAudioReceiver(context) + + // 网络状态监听 + private var networkMonitor: NetworkStateMonitor = NetworkStateMonitor(context) + private var isNetworkRecovering = false + + // 网络丢失计数器 + private var networkLostCount = 0 + + // 最大网络丢失次数(3秒检测) + private val maxNetworkLostCount = 3 - // 启动前台服务(在开始音频录制前) - if (!isForegroundServiceRunning && audioSourceType == AudioSourceType.MICROPHONE) { - try { - AudioRecordingForegroundService.startService(context) - isForegroundServiceRunning = true - Log.d(tag, "前台服务已启动") - } catch (e: Exception) { - Log.e(tag, "启动前台服务失败: ${e.message}") - continuousCallback?.onError(currsessionid, 1003, "启动前台服务失败: ${e.message}") - return false - } + // 防抖机制相关变量 + private var lastNetworkLostTime = 0L + private val networkLostDebounceInterval = 3000L // 2秒防抖间隔 + private var currsessionid = "" //识别回话id 关联到整个聊天过程中 + + + /** + * 音频来源类型 + */ + enum class AudioSourceType { + /** 使用设备麦克风 */ + MICROPHONE, + + /** 使用外部提供的音频数据 */ + EXTERNAL } - if (useXunfei) { - Log.d(tag, "Using Xunfei Recognition") - val callback = continuousCallback - if (callback == null) { - Log.e(tag, "无法启动连续识别:回调为空") - continuousCallback?.onError(currsessionid, 1001, "回调为空") - return false - } - - if (xunFeiAsrHelper == null) { - xunFeiAsrHelper = XunFeiAsrHelper( - context = context, - appId = xunfeiAppId, - accessKeyId = xunfeiAccessKeyId, - accessKeySecret = xunfeiAccessKeySecret - ) - xunFeiAsrHelper?.logCallback = nativeLogCallback - nativeLogCallback?.invoke("INFO", "[XunFei_ASR] Config init (callback ready): appId=${xunfeiAppId.take(6)}***, keyId=${xunfeiAccessKeyId.take(6)}***, secret=${if (xunfeiAccessKeySecret.isNotBlank()) "SET" else "EMPTY"}") - } else { - xunFeiAsrHelper?.updateConfig( - appId = xunfeiAppId, - accessKeyId = xunfeiAccessKeyId, - accessKeySecret = xunfeiAccessKeySecret - ) - } - - try { - Log.d(tag, "startContinuousRecognition: ${audioStream.isContinuousRecognitionActive}") - xunFeiAsrHelper?.start( - callback, - currentLanguage, - supportedLanguages.toList(), - isAutoDetectLanguage, - isRemoveFirstPunctuation - ) - this.audioSourceType = audioSourceType - if (!audioStream.isContinuousRecognitionActive) { - // 确保音频流已初始化 - if (!audioStream.isInitialized()) { - audioStream.initAudioRecord() - audioStream.setupStream() - } - - audioStream.startAudioRecord( - when (audioSourceType) { - AudioSourceType.MICROPHONE -> SimpleAudioReceiver.AudioSourceType.MICROPHONE - AudioSourceType.EXTERNAL -> SimpleAudioReceiver.AudioSourceType.EXTERNAL - }, - object : SimpleAudioReceiver.AudioDataCallback { - override fun onAudio(data: ByteArray) { - audioDataCallback?.onAudio(data) - xunFeiAsrHelper?.sendAudio(data) - } + /** + * 初始化Azure语音服务 + * + * @param subscriptionKey Azure 订阅密钥 + * @param region Azure 区域 + * @param supportedLanguages 支持的语言数组 + * @param audioSourceType 音频源类型 + * @return 初始化是否成功 + */ + fun initialize( + subscriptionKey: String, + region: String, + supportedLanguages: Array = arrayOf("zh-CN", "en-US"), + audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE + ): Boolean { + try { + // 检查配置是否为空 + if (subscriptionKey.isEmpty() || region.isEmpty()) { + Log.e(tag, "Azure 配置信息不完整") + return false } - ) + + // 释放之前的资源 + dispose() + + // 保存配置 + this.subscriptionKey = subscriptionKey + this.region = region + this.audioSourceType = audioSourceType + + // 设置语言 + if (supportedLanguages.isNotEmpty()) { + this.supportedLanguages = supportedLanguages + } + + // 根据支持的语言数量决定是否启用自动语言检测 + this.isAutoDetectLanguage = supportedLanguages.size >= 2 + + // 如果只有一种语言,设置为当前语言 + if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { + this.currentLanguage = supportedLanguages[0] + } + + // 创建语音配置 + speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region).apply { + if (isAutoDetectLanguage) { + // 直接启用语言检测模式 + setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") + } else { + // 设置指定的识别语言 + speechRecognitionLanguage = currentLanguage + } + + } + // 初始化网络监听 + networkMonitor.initialize(object : NetworkStateMonitor.NetworkStateListener { + override fun onNetworkAvailable() { + handleNetworkAvailable() + } + + override fun onNetworkLost() { + handleNetworkLost() + } + }) + + return true + } catch (e: Exception) { + Log.e(tag, "初始化失败: ${e.message}") + return false } - audioStream.isContinuousRecognitionActive = true - return true - } catch (e: Exception) { - Log.e(tag, "Start Xunfei failed: ${e.message}") - xunFeiAsrHelper?.stop() - audioStream.isContinuousRecognitionActive = false - return false - } } - if (!isRecognizerValid()) { - Log.d(tag, "重新启动连续识别") - val callback = continuousCallback - if (callback == null) { - Log.e(tag, "无法重新启动连续识别:回调为空") - continuousCallback?.onError(currsessionid, 1001, "回调为空") - return false - } - setupEventListeners(callback) - } - this.audioSourceType = audioSourceType - - try { - Log.d(tag, "startContinuousRecognition: ${audioStream.isWriting()}") - // 启动音频处理(若已启动则跳过) - if (!audioStream.isContinuousRecognitionActive) { - Log.d(tag, "startAudioRecord:$audioSourceType") - audioStream.startAudioRecord( - when (audioSourceType) { - AudioSourceType.MICROPHONE -> SimpleAudioReceiver.AudioSourceType.MICROPHONE - AudioSourceType.EXTERNAL -> SimpleAudioReceiver.AudioSourceType.EXTERNAL - }, - audioDataCallback - ) - } else { - Log.d(tag, "音频已启动,跳过重复启动") - } - - recognizer?.startContinuousRecognitionAsync() - audioStream.isContinuousRecognitionActive = true - return true - } catch (e: Exception) { - Log.e(tag, "开始连续识别失败: ${e.message}") - stopContinuousRecognition() - audioStream.isContinuousRecognitionActive = false - return false - } - } - - /** - * 设置事件监听器 - */ - fun setupEventListeners(callback: ContinuousRecognizeCallback): Boolean { - Log.d(tag, "设置ssssss监听器:${speechConfig ?: "null"} ") - if (speechConfig == null) { - callback.onError(currsessionid, 1002, "语音服务未初始化") - return false + // /** + // * 设置音频配置 + // * @param sampleRate 采样率 (8000, 16000, 24000, 32000, 44100, 48000) + // * @param channels 声道数 (1=单声道, 2=立体声) + // */ + // fun setAudioConfig(sampleRate: Int, channels: Int) { + // try { + // // 设置RecordFile的音频配置 + // recordfile?.setAudioConfig(sampleRate, channels) + // Log.d(tag, "音频配置已设置: 采样率=${sampleRate}Hz, 声道数=${channels}") + // } catch (e: Exception) { + // Log.e(tag, "设置音频配置失败: ${e.message}") + // } + // } + + private fun isRecognizerValid(): Boolean { + return recognizer != null } - // 重设识别器 - if (!setupRecognizer()) { - Log.d(tag, "初始化失败: ") - return false - } - try { - continuousCallback = callback; - Log.d(tag, "设置ssssss监听器: ") - // 优化:识别中事件 - 添加文本长度检查 - recognizer?.recognizing?.addEventListener( - EventHandler { _, event -> - // 优化:只处理非空结果 - if (event.result.text.isNotEmpty()) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" + /** + * 设置识别器 + */ + private fun setupRecognizer(): Boolean { + try { + // 如果正在进行连续识别,先停止 + if (audioStream.isContinuousRecognitionActive) { + // 直接停止,不等待结果 + recognizer?.stopContinuousRecognitionAsync() + audioStream.isContinuousRecognitionActive = false + } + setupMicrophoneStream() + + recognizer?.close() + recognizer = null + //创建识别器 + recognizer = if (isAutoDetectLanguage) { + val autoDetectConfig = + AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) } else { - currentLanguage + SpeechRecognizer(speechConfig, audioConfig) } - callback.onRecognizing(currsessionid, event.result.text, detectedLanguage) - } + + return true + } catch (e: Exception) { + Log.e(tag, "创建识别器失败: ${e.message}") + stopAudioProcessing() + return false } - ) - - // 优化:识别完成事件 - 优化语言检测 - recognizer?.recognized?.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizedSpeech && event.result.text.isNotEmpty()) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language - ?: supportedLanguages[0] - } else { - currentLanguage + } + + /** + * 设置麦克风流 - 使用拉流方式 + */ + private fun setupMicrophoneStream() { + try { + Log.d(tag, "设置麦克风流 - 使用拉流方式: }") + + // Initialize if not already done + if (!audioStream.isInitialized()) { + audioStream.initAudioRecord() } - callback.onResult(currsessionid, event.result.text, detectedLanguage) - currsessionid = UUID.randomUUID().toString() - } + audioStream.setupStream() + audioConfig = AudioConfig.fromStreamInput(audioStream.pushAudioStream) + + } catch (e: Exception) { + Log.e(tag, "设置麦克风流失败: ${e.message}") } - ) - - // 会话开始事件 - recognizer?.sessionStarted?.addEventListener( - EventHandler { _, _ -> - // 直接在当前线程调用回调 - Log.d(tag, "会话开始事件") - currsessionid = UUID.randomUUID().toString() - callback.onSessionStarted(currsessionid) + } + + + /** + * 开始连续语音识别 + * @param audioSourceType 音频源类型 + * @param audioDataCallback 音频数据回调 + * @return 是否成功开始识别 + */ + fun startContinuousRecognition( + audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE, + audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = null + ): Boolean { + + Log.d(tag, "startContinuousRecognition:${audioStream.isContinuousRecognitionActive} ") + + // 检查网络状态 + if (!checkNetworkStatus()) { + Log.e(tag, "网络不可用,无法开始语音识别") + continuousCallback?.onError(currsessionid, 1000, "网络连接不可用,请检查网络设置") + return false + } + + if (audioStream.isContinuousRecognitionActive) { + return true } - ) - - // 会话结束事件 - recognizer?.sessionStopped?.addEventListener( - EventHandler { _, _ -> - // 直接在当前线程调用回调 - Log.d(tag, "会话结束事件") - // if (audioSourceType == AudioSourceType.EXTERNAL) { - callback.onSessionStopped(currsessionid) - audioStream?.isContinuousRecognitionActive = false - // } + + // 启动前台服务(在开始音频录制前) + if (!isForegroundServiceRunning && audioSourceType == AudioSourceType.MICROPHONE) { + try { + AudioRecordingForegroundService.startService(context) + isForegroundServiceRunning = true + Log.d(tag, "前台服务已启动") + } catch (e: Exception) { + Log.e(tag, "启动前台服务失败: ${e.message}") + continuousCallback?.onError(currsessionid, 1003, "启动前台服务失败: ${e.message}") + return false + } } - ) - // 取消事件 - recognizer?.canceled?.addEventListener( - EventHandler { _, event -> - val errorDetails = event.errorDetails ?: "未知错误" - val reason = event.reason.toString() + if (!isRecognizerValid()) { + Log.d(tag, "重新启动连续识别") + val callback = continuousCallback + if (callback == null) { + Log.e(tag, "无法重新启动连续识别:回调为空") + continuousCallback?.onError(currsessionid, 1001, "回调为空") + return false + } + setupEventListeners(callback) + } + this.audioSourceType = audioSourceType + + try { + Log.d(tag, "startContinuousRecognition: ${audioStream.isWriting()}") + // 启动音频处理(若已启动则跳过) + if (!audioStream.isContinuousRecognitionActive) { + Log.d(tag, "startAudioRecord:$audioSourceType") + audioStream.startAudioRecord( + when (audioSourceType) { + AudioSourceType.MICROPHONE -> SimpleAudioReceiver.AudioSourceType.MICROPHONE + AudioSourceType.EXTERNAL -> SimpleAudioReceiver.AudioSourceType.EXTERNAL + }, + audioDataCallback + ) + } else { + Log.d(tag, "音频已启动,跳过重复启动") + } - Log.d(tag, "识别被取消: reason=$reason, details=$errorDetails") - callback.onCanceled(currsessionid, reason, errorDetails) + recognizer?.startContinuousRecognitionAsync() + audioStream.isContinuousRecognitionActive = true + return true + } catch (e: Exception) { + Log.e(tag, "开始连续识别失败: ${e.message}") + stopContinuousRecognition() + audioStream.isContinuousRecognitionActive = false + return false } - ) - } catch (e: Exception) { - callback.onError(currsessionid, 1001, "启动连续识别失败: ${e.message}") - return false } - return true - } - - - /** - * 停止连续语音识别 - * - * @return 是否成功停止 - */ - fun stopContinuousRecognition(): Boolean { - if (useXunfei) { - try { - xunFeiAsrHelper?.stop() - stopAudioProcessing() - audioStream.isContinuousRecognitionActive = false - - if (isForegroundServiceRunning) { - try { - AudioRecordingForegroundService.stopService(context) - isForegroundServiceRunning = false - } catch (e: Exception) { - Log.e(tag, "停止前台服务失败: ${e.message}") - } + + /** + * 设置事件监听器 + */ + fun setupEventListeners(callback: ContinuousRecognizeCallback): Boolean { + Log.d(tag, "设置ssssss监听器:${speechConfig ?: "null"} ") + if (speechConfig == null) { + callback.onError(currsessionid, 1002, "语音服务未初始化") + return false + } + + // 重设识别器 + if (!setupRecognizer()) { + Log.d(tag, "初始化失败: ") + return false + } + try { + continuousCallback = callback; + Log.d(tag, "设置ssssss监听器: ") + // 优化:识别中事件 - 添加文本长度检查 + recognizer?.recognizing?.addEventListener( + EventHandler { _, event -> + // 优化:只处理非空结果 + if (event.result.text.isNotEmpty()) { + val detectedLanguage = if (isAutoDetectLanguage) { + AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" + } else { + currentLanguage + } + callback.onRecognizing(currsessionid, event.result.text, detectedLanguage) + } + } + ) + + // 优化:识别完成事件 - 优化语言检测 + recognizer?.recognized?.addEventListener( + EventHandler { _, event -> + if (event.result.reason == ResultReason.RecognizedSpeech && event.result.text.isNotEmpty()) { + val detectedLanguage = if (isAutoDetectLanguage) { + AutoDetectSourceLanguageResult.fromResult(event.result)?.language + ?: supportedLanguages[0] + } else { + currentLanguage + } + callback.onResult(currsessionid, event.result.text, detectedLanguage) + currsessionid = UUID.randomUUID().toString() + } + } + ) + + // 会话开始事件 + recognizer?.sessionStarted?.addEventListener( + EventHandler { _, _ -> + // 直接在当前线程调用回调 + Log.d(tag, "会话开始事件") + currsessionid = UUID.randomUUID().toString() + callback.onSessionStarted(currsessionid) + } + ) + + // 会话结束事件 + recognizer?.sessionStopped?.addEventListener( + EventHandler { _, _ -> + // 直接在当前线程调用回调 + Log.d(tag, "会话结束事件") + callback.onSessionStopped(currsessionid) + if (audioSourceType == AudioSourceType.EXTERNAL) { + + audioStream?.isContinuousRecognitionActive = false + } + } + ) + + // 取消事件 + recognizer?.canceled?.addEventListener( + EventHandler { _, event -> + val errorDetails = event.errorDetails ?: "未知错误" + val reason = event.reason.toString() + + Log.d(tag, "识别被取消: reason=$reason, details=$errorDetails") + callback.onCanceled(currsessionid, reason, errorDetails) + } + ) + } catch (e: Exception) { + callback.onError(currsessionid, 1001, "启动连续识别失败: ${e.message}") + return false } return true - } catch (e: Exception) { - Log.e(tag, "Stop Xunfei failed: ${e.message}") - return false - } } - if (speechConfig == null) { - Log.e(tag, "语音服务未初始化") - return false - } - try { - Log.d(tag, "停止连续语音识别: ") - // 停止音频处理 - audioStream.stopMicrophoneCapture() - // 直接停止连续识别(SDK内部已是异步操作) - recognizer?.stopContinuousRecognitionAsync()?.get(1000, TimeUnit.MILLISECONDS) - - recognizer?.close() - recognizer = null - audioStream.isContinuousRecognitionActive = false - - // 停止前台服务 - if (isForegroundServiceRunning) { + + /** + * 停止连续语音识别 + * + * @return 是否成功停止 + */ + fun stopContinuousRecognition(): Boolean { + if (speechConfig == null) { + Log.e(tag, "语音服务未初始化") + return false + } try { - AudioRecordingForegroundService.stopService(context) - isForegroundServiceRunning = false - Log.d(tag, "前台服务已停止") + Log.d(tag, "停止连续语音识别: ") + // 停止音频处理 + audioStream.stopMicrophoneCapture() + // 直接停止连续识别(SDK内部已是异步操作) + recognizer?.stopContinuousRecognitionAsync()?.get(1000, TimeUnit.MILLISECONDS) + + recognizer?.close() + recognizer = null + audioStream.isContinuousRecognitionActive = false + + // 停止前台服务 + if (isForegroundServiceRunning) { + try { + AudioRecordingForegroundService.stopService(context) + isForegroundServiceRunning = false + Log.d(tag, "前台服务已停止") + } catch (e: Exception) { + Log.e(tag, "停止前台服务失败: ${e.message}") + } + } + + return true } catch (e: Exception) { - Log.e(tag, "停止前台服务失败: ${e.message}") + // 强制重置状态 + audioStream.isContinuousRecognitionActive = false + Log.e(tag, "停止连续识别失败: ${e.message}") + // 停止音频处理 + audioStream.stopMicrophoneCapture() + recognizer?.stopContinuousRecognitionAsync() + recognizer?.close() + recognizer = null + audioStream.isContinuousRecognitionActive = false + + // 确保前台服务被停止 + if (isForegroundServiceRunning) { + try { + AudioRecordingForegroundService.stopService(context) + isForegroundServiceRunning = false + } catch (e2: Exception) { + Log.e(tag, "强制停止前台服务失败: ${e2.message}") + } + } + + return false } - } - - return true - } catch (e: Exception) { - // 强制重置状态 - audioStream.isContinuousRecognitionActive = false - Log.e(tag, "停止连续识别失败: ${e.message}") - // 停止音频处理 - audioStream.stopMicrophoneCapture() - recognizer?.stopContinuousRecognitionAsync() - recognizer?.close() - recognizer = null - audioStream.isContinuousRecognitionActive = false - - // 确保前台服务被停止 - if (isForegroundServiceRunning) { + } + + /** + * 检查连续识别是否活跃 + */ + fun isContinuousRecognitionActive(): Boolean = audioStream.isContinuousRecognitionActive + + /** + * 释放所有资源 + */ + fun dispose() { try { - AudioRecordingForegroundService.stopService(context) - isForegroundServiceRunning = false - } catch (e2: Exception) { - Log.e(tag, "强制停止前台服务失败: ${e2.message}") - } - } + // 如果正在进行连续识别,先停止 + if (audioStream.isContinuousRecognitionActive) { + // 直接停止,不等待结果 + recognizer?.stopContinuousRecognitionAsync() + audioStream.isContinuousRecognitionActive = false + } + // 停止音频处理 + stopAudioProcessing() + // 清理网络监听 + clearNetworkDetection() + // 停止录音 + audioStream.recordfile?.closeFile(true) + + // 释放recognizer + recognizer?.close() + recognizer = null + + // 释放speechConfig + speechConfig?.close() + speechConfig = null + + // 释放音频配置 + audioConfig?.close() + audioConfig = null + + // 确保前台服务被停止 + if (isForegroundServiceRunning) { + try { + AudioRecordingForegroundService.stopService(context) + isForegroundServiceRunning = false + Log.d(tag, "dispose时前台服务已停止") + } catch (e: Exception) { + Log.e(tag, "dispose时停止前台服务失败: ${e.message}") + } + } - return false - } - } - - /** - * 检查连续识别是否活跃 - */ - fun isContinuousRecognitionActive(): Boolean = audioStream.isContinuousRecognitionActive - - /** - * 释放所有资源 - */ - fun dispose() { - try { - // 如果正在进行连续识别,先停止 - if (audioStream.isContinuousRecognitionActive) { - if (useXunfei) { - xunFeiAsrHelper?.stop() - } else { - // 直接停止,不等待结果 - recognizer?.stopContinuousRecognitionAsync() + // 确保状态被重置 + audioStream.isContinuousRecognitionActive = false + currsessionid = "" + } catch (e: Exception) { + // 确保状态被重置 + audioStream.isContinuousRecognitionActive = false + audioConfig = null + recognizer = null + speechConfig = null + isForegroundServiceRunning = false + currsessionid = "" } - audioStream.isContinuousRecognitionActive = false - } - // 停止音频处理 - stopAudioProcessing() - // 清理网络监听 - clearNetworkDetection() - // 停止录音 - audioStream.recordfile?.closeFile(true) - - // 释放recognizer - recognizer?.close() - recognizer = null - - xunFeiAsrHelper?.stop() - xunFeiAsrHelper = null - - // 释放speechConfig - speechConfig?.close() - speechConfig = null - - // 释放音频配置 - audioConfig?.close() - audioConfig = null - - // 确保前台服务被停止 - if (isForegroundServiceRunning) { + } + + // 音频处理相关方法 + /** + * 停止音频处理 + */ + private fun stopAudioProcessing() { try { - AudioRecordingForegroundService.stopService(context) - isForegroundServiceRunning = false - Log.d(tag, "dispose时前台服务已停止") + Log.e(tag, "停止音频处理: }") + audioStream.releaseAudioResources() + audioStream.pushAudioStream?.close() } catch (e: Exception) { - Log.e(tag, "dispose时停止前台服务失败: ${e.message}") + Log.e(tag, "关闭麦克风流失败: ${e.message}") + e.printStackTrace() } - } - - // 确保状态被重置 - audioStream.isContinuousRecognitionActive = false - currsessionid = "" - } catch (e: Exception) { - // 确保状态被重置 - audioStream.isContinuousRecognitionActive = false - audioConfig = null - recognizer = null - speechConfig = null - isForegroundServiceRunning = false - currsessionid = "" } - } - - // 音频处理相关方法 - /** - * 停止音频处理 - */ - private fun stopAudioProcessing() { - try { - Log.e(tag, "停止音频处理: }") - audioStream.releaseAudioResources() - audioStream.pushAudioStream?.close() - } catch (e: Exception) { - Log.e(tag, "关闭麦克风流失败: ${e.message}") - e.printStackTrace() - } - } //录音音频文件 - /** - * 开启录音 - */ - fun enableRecord( - audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE, - filePath: String, - audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = null - ) { - Log.i(tag, "开启录音:") - this.audioSourceType = audioSourceType - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里录音") - return - } + /** + * 开启录音 + */ + fun enableRecord( + audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE, + filePath: String, + audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = null + ) { + Log.i(tag, "开启录音:") + this.audioSourceType = audioSourceType + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里录音") + return + } - // 启动前台服务(在开始录音前) - if (!isForegroundServiceRunning) { - try { - AudioRecordingForegroundService.startService(context) - isForegroundServiceRunning = true - Log.d(tag, "录音前台服务已启动") - } catch (e: Exception) { - Log.e(tag, "启动录音前台服务失败: ${e.message}") - } - } - if (audioStream.recordfile == null) { - audioStream.recordfile = RecordFile() - } + // 启动前台服务(在开始录音前) + if (!isForegroundServiceRunning) { + try { + AudioRecordingForegroundService.startService(context) + isForegroundServiceRunning = true + Log.d(tag, "录音前台服务已启动") + } catch (e: Exception) { + Log.e(tag, "启动录音前台服务失败: ${e.message}") + } + } + if (audioStream.recordfile == null) { + audioStream.recordfile = RecordFile() + } - if (!audioStream.isContinuousRecognitionActive) { - // 启动音频处理 - Log.i(tag, "开启录音:audioSourceType${audioSourceType}") - // Initialize if not already done - if (!audioStream.isInitialized()) { - audioStream.initAudioRecord() - } - audioStream.setupStream() - - - audioStream.startAudioRecord( - when (audioSourceType) { - AudioSourceType.MICROPHONE -> SimpleAudioReceiver.AudioSourceType.MICROPHONE - AudioSourceType.EXTERNAL -> SimpleAudioReceiver.AudioSourceType.EXTERNAL - }, - audioDataCallback - ) - } + if (!audioStream.isContinuousRecognitionActive) { + // 启动音频处理 + Log.i(tag, "开启录音:audioSourceType${audioSourceType}") + // Initialize if not already done + if (!audioStream.isInitialized()) { + audioStream.initAudioRecord() + } + audioStream.setupStream() - audioStream.isRecord = true - audioStream.recordfile?.closeFile(true) - audioStream.recordfile?.creatingFiles(filePath) - } - - /** - * 移动文件到新路径 - */ - fun moveFile(sourcePath: String, destPath: String): Boolean { - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里移动文件") - return false + + audioStream.startAudioRecord( + when (audioSourceType) { + AudioSourceType.MICROPHONE -> SimpleAudioReceiver.AudioSourceType.MICROPHONE + AudioSourceType.EXTERNAL -> SimpleAudioReceiver.AudioSourceType.EXTERNAL + }, + audioDataCallback + ) + } + + audioStream.isRecord = true + audioStream.recordfile?.closeFile(true) + audioStream.recordfile?.creatingFiles(filePath) } - Log.i(tag, "移动文件到新路径:") - audioStream.recordfile?.moveFile(sourcePath, destPath) ?: return false - return true - } - - /** - * 重命名指定路径的音频文件 - */ - fun renameFile(filePath: String, newName: String): Boolean { - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里重命名文件") - return false + + /** + * 移动文件到新路径 + */ + fun moveFile(sourcePath: String, destPath: String): Boolean { + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里移动文件") + return false + } + Log.i(tag, "移动文件到新路径:") + audioStream.recordfile?.moveFile(sourcePath, destPath) ?: return false + return true } - Log.i(tag, "重命名指定路径的音频文件:") - audioStream.recordfile?.renameFile(filePath, newName) ?: return false - return true - } - - /** - * 停止录音 - */ - fun pauseRecord() { - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里暂停录音") - return + + /** + * 重命名指定路径的音频文件 + */ + fun renameFile(filePath: String, newName: String): Boolean { + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里重命名文件") + return false + } + Log.i(tag, "重命名指定路径的音频文件:") + audioStream.recordfile?.renameFile(filePath, newName) ?: return false + return true } - Log.i(tag, "停止连续录音:") - audioStream.recordfile ?: return - audioStream.isRecord = false - } - - /** - * 继续录音 - */ - fun resumeRecord() { - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里继续录音") - return + + /** + * 停止录音 + */ + fun pauseRecord() { + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里暂停录音") + return + } + Log.i(tag, "停止连续录音:") + audioStream.recordfile ?: return + audioStream.isRecord = false } - Log.i(tag, "继续录音:") - audioStream.recordfile ?: return - audioStream.isRecord = true - } - - /** - * 关闭录音 - */ - fun stopRecord(isSave: Boolean) { - if (audioSourceType == AudioSourceType.EXTERNAL) { - Log.i(tag, "外部音频源不在这里关闭录音") - return + + /** + * 继续录音 + */ + fun resumeRecord() { + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里继续录音") + return + } + Log.i(tag, "继续录音:") + audioStream.recordfile ?: return + audioStream.isRecord = true } - Log.i(tag, "关闭录音:") - audioStream.recordfile ?: return - if (!audioStream.isContinuousRecognitionActive) { - audioStream.stopMicrophoneCapture() + /** + * 关闭录音 + */ + fun stopRecord(isSave: Boolean) { + if (audioSourceType == AudioSourceType.EXTERNAL) { + Log.i(tag, "外部音频源不在这里关闭录音") + return + } + Log.i(tag, "关闭录音:") + audioStream.recordfile ?: return - // 停止录音时也停止前台服务(如果没有其他音频任务) - if (isForegroundServiceRunning) { - try { - AudioRecordingForegroundService.stopService(context) - isForegroundServiceRunning = false - Log.d(tag, "录音前台服务已停止") - } catch (e: Exception) { - Log.e(tag, "停止录音前台服务失败: ${e.message}") + if (!audioStream.isContinuousRecognitionActive) { + audioStream.stopMicrophoneCapture() + + // 停止录音时也停止前台服务(如果没有其他音频任务) + if (isForegroundServiceRunning) { + try { + AudioRecordingForegroundService.stopService(context) + isForegroundServiceRunning = false + Log.d(tag, "录音前台服务已停止") + } catch (e: Exception) { + Log.e(tag, "停止录音前台服务失败: ${e.message}") + } + } } - } - } - audioStream.isRecord = false - audioStream.recordfile?.closeFile(isSave) - } + audioStream.isRecord = false + audioStream.recordfile?.closeFile(isSave) + } - /** - * 一次性识别回调接口 - */ - interface RecognizeCallback { /** - * 返回识别结果 - * - * @param text 识别的文本 - * @param detectedLanguage 检测到的语言 + * 一次性识别回调接口 */ - fun onResult(text: String, detectedLanguage: String) + interface RecognizeCallback { + /** + * 返回识别结果 + * + * @param text 识别的文本 + * @param detectedLanguage 检测到的语言 + */ + fun onResult(text: String, detectedLanguage: String) + + /** + * 识别错误时调用 + * + * @param code 错误码 + * @param error 错误信息 + */ + fun onError(code: Int, error: String) + } /** - * 识别错误时调用 - * - * @param code 错误码 - * @param error 错误信息 + * 连续识别回调接口 */ - fun onError(code: Int, error: String) - } + interface ContinuousRecognizeCallback { + /** + * 返回识别结果 + * + * @param text 识别的文本 + * @param detectedLanguage 检测到的语言 + */ + fun onResult(sessiond: String, text: String, detectedLanguage: String) + + /** + * 识别进行中调用 + * + * @param recognizing 正在识别的文本 + * @param detectedLanguage 检测到的语言 + */ + fun onRecognizing(sessiond: String, recognizing: String, detectedLanguage: String) + + /** + * 会话开始时调用 + */ + fun onSessionStarted(sessiond: String) + + /** + * 会话结束时调用 + */ + fun onSessionStopped(sessiond: String) + + + /** + * 识别取消时调用 + * + * @param reason 取消原因 + * @param errorDetails 错误详情 + */ + fun onCanceled(sessiond: String, reason: String, errorDetails: String) + + /** + * 识别出错时调用 + * + * @param error 错误信息 + */ + fun onError(sessiond: String, code: Int, error: String) + } + - /** - * 连续识别回调接口 - */ - interface ContinuousRecognizeCallback { /** - * 返回识别结果 - * - * @param text 识别的文本 - * @param detectedLanguage 检测到的语言 + * 检查当前网络状态 + * @return true表示网络可用,false表示网络不可用 */ - fun onResult(sessiond: String, text: String, detectedLanguage: String) + private fun checkNetworkStatus(): Boolean { + return networkMonitor.checkNetworkStatus() + } /** - * 识别进行中调用 - * - * @param recognizing 正在识别的文本 - * @param detectedLanguage 检测到的语言 + * 检查是否是网络相关错误 */ - fun onRecognizing(sessiond: String, recognizing: String, detectedLanguage: String) + private fun isNetworkRelatedError(reason: String, errorDetails: String): Boolean { + return networkMonitor.isNetworkRelatedError(reason, errorDetails) + } /** - * 会话开始时调用 + * 处理网络连接可用 */ - fun onSessionStarted(sessiond: String) + private fun handleNetworkAvailable() { + // 网络恢复时重置计数器 + networkLostCount = 0 + lastNetworkLostTime = 0L + if (!isNetworkRecovering) { + isNetworkRecovering = true + Log.d(tag, "网络恢复,准备重新启动识别") + } + } /** - * 会话结束时调用 + * 处理网络连接丢失(带防抖机制) */ - fun onSessionStopped(sessiond: String) + private fun handleNetworkLost() { + val currentTime = System.currentTimeMillis() + + // 防抖机制:如果距离上次触发时间小于防抖间隔,则忽略本次触发 + if (currentTime - lastNetworkLostTime < networkLostDebounceInterval) { + Log.d( + tag, + "网络丢失事件被防抖过滤,距离上次触发仅${currentTime - lastNetworkLostTime}ms" + ) + return + } + // 更新最后触发时间 + lastNetworkLostTime = currentTime - /** - * 识别取消时调用 - * - * @param reason 取消原因 - * @param errorDetails 错误详情 - */ - fun onCanceled(sessiond: String, reason: String, errorDetails: String) + + //这里启动一个3秒定时器,3秒后检查网络是否恢复 + // 启动3秒定时器,3秒后检查网络是否恢复 + Handler(Looper.getMainLooper()).postDelayed({ + Log.e(tag, "网络不可用") + val isNetworkAvailable = checkNetworkStatus() + if (isNetworkAvailable) { + // 如果网络恢复,重置计数器和防抖时间 + Log.d(tag, "网络已恢复,重置计数器") + lastNetworkLostTime = 0L + isNetworkRecovering = true + } else { + continuousCallback?.onError(currsessionid, 1000, "网络连接不可用,请检查网络设置") + Log.d(tag, "网络仍未恢复,继续检测") + networkLostCount = 0 + stopAudioProcessingImmediately() + } + }, 3000L) // 3秒延时 + } /** - * 识别出错时调用 - * - * @param error 错误信息 + * 立即停止音频处理(网络断开时使用) */ - fun onError(sessiond: String, code: Int, error: String) - } - - - /** - * 检查当前网络状态 - * @return true表示网络可用,false表示网络不可用 - */ - private fun checkNetworkStatus(): Boolean { - return networkMonitor.checkNetworkStatus() - } - - /** - * 检查是否是网络相关错误 - */ - private fun isNetworkRelatedError(reason: String, errorDetails: String): Boolean { - return networkMonitor.isNetworkRelatedError(reason, errorDetails) - } - - /** - * 处理网络连接可用 - */ - private fun handleNetworkAvailable() { - // 网络恢复时重置计数器 - networkLostCount = 0 - lastNetworkLostTime = 0L - if (!isNetworkRecovering) { - isNetworkRecovering = true - Log.d(tag, "网络恢复,准备重新启动识别") - } - } - - /** - * 处理网络连接丢失(带防抖机制) - */ - private fun handleNetworkLost() { - val currentTime = System.currentTimeMillis() - - // 防抖机制:如果距离上次触发时间小于防抖间隔,则忽略本次触发 - if (currentTime - lastNetworkLostTime < networkLostDebounceInterval) { - Log.d(tag, "网络丢失事件被防抖过滤,距离上次触发仅${currentTime - lastNetworkLostTime}ms") - return + private fun stopAudioProcessingImmediately() { + try { + Log.d(tag, "立即停止音频处理") + stopContinuousRecognition() + // 不改变isContinuousRecognitionActive状态,保持识别意图 + } catch (e: Exception) { + Log.e(tag, "立即停止音频处理失败: ${e.message}") + } } - // 更新最后触发时间 - lastNetworkLostTime = currentTime - + /** + * 清理网络检测相关资源 + */ + private fun clearNetworkDetection() { - //这里启动一个3秒定时器,3秒后检查网络是否恢复 - // 启动3秒定时器,3秒后检查网络是否恢复 - Handler(Looper.getMainLooper()).postDelayed({ - Log.e(tag, "网络不可用") - val isNetworkAvailable = checkNetworkStatus() - if (isNetworkAvailable) { - // 如果网络恢复,重置计数器和防抖时间 - Log.d(tag, "网络已恢复,重置计数器") - lastNetworkLostTime = 0L - isNetworkRecovering = true - } else { - continuousCallback?.onError(currsessionid, 1000, "网络连接不可用,请检查网络设置") - Log.d(tag, "网络仍未恢复,继续检测") networkLostCount = 0 - stopAudioProcessingImmediately() - } - }, 3000L) // 3秒延时 - } - - /** - * 立即停止音频处理(网络断开时使用) - */ - private fun stopAudioProcessingImmediately() { - try { - Log.d(tag, "立即停止音频处理") - stopContinuousRecognition() - // 不改变isContinuousRecognitionActive状态,保持识别意图 - } catch (e: Exception) { - Log.e(tag, "立即停止音频处理失败: ${e.message}") + lastNetworkLostTime = 0L } - } - - /** - * 清理网络检测相关资源 - */ - private fun clearNetworkDetection() { - - networkLostCount = 0 - lastNetworkLostTime = 0L - } } - + \ No newline at end of file diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt index b8e5e3056..da25d8224 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt @@ -25,6 +25,7 @@ import javax.crypto.Mac import javax.crypto.spec.SecretKeySpec import com.yunqiinnovation.azure_speech.tools.RecordFile import com.yunqiinnovation.azure_speech.tools.SimpleAudioPlayer + /** * 整合的语音翻译服务 * 集成ASR语音识别、翻译服务和TTS语音合成 @@ -39,7 +40,7 @@ class IntegratedSpeechTranslationService( // 音频配置常量 private const val SAMPLE_RATE = 16000 - private const val CHANNELS = 1 + private const val CHANNELS = 1 // 单声道输出 private const val BITS_PER_SAMPLE = 16 private const val BUFFER_SIZE = 4096 @@ -65,21 +66,101 @@ class IntegratedSpeechTranslationService( // 音频处理 private var audioProcessor: AudioProcessor? = null private var audioConfig: AudioConfig? = null + private var synthOutputStream: PushAudioOutputStream? = null + // 录音文件处理 var recordfile1: RecordFile? = null var filePath: String? = null + // 配置管理 private var serviceConfig = ServiceConfiguration() // 状态管理 private val serviceState = ServiceState() + // 最终翻译队列(按识别完成顺序排队执行) + private val finalTranslationQueue = LinkedBlockingQueue>() + // 事件回调 private var eventCallback: ServiceEventCallback? = null // 语音映射缓存 private val voiceCache = ConcurrentHashMap() + // 管理“识别中”翻译的协程任务,便于在识别完成时及时取消 + private var recognizingTranslationJob: Job? = null + + // ===== Utterance 关联ID管理 ===== + // 为每个“完整一句话”的识别-翻译-TTS链路生成并贯穿唯一ID + private val utteranceSeq = java.util.concurrent.atomic.AtomicLong(0) + private var currentUtteranceId: String? = null + private val isInUtterance = AtomicBoolean(false) + + // 记录当前TTS合成的utterance信息,用于事件回调携带ID + private var currentSynthesisId: String? = null + private var currentSynthesisText: String? = null + + + + private var ttsdata: ByteArray = ByteArray(0) + /** + * 将合成中的音频按1280字节对齐分片并回调发送 + * 输入的音频数据会追加到内部缓存;每次仅发送满足1280整数倍的部分; + * 余下不足部分保留到缓存,等待下一次补齐。 + */ + private fun processSynthesisAudioChunk( + utteranceId: String, + text: String, + incoming: ByteArray, + multiple: Int = 1280 + ) { + if (incoming.isNotEmpty()) { + ttsdata = ttsdata + incoming + } + + val sendLen = (ttsdata.size / multiple) * multiple + if (sendLen > 0) { + val chunk = ttsdata.copyOfRange(0, sendLen) + Log.d(TAG, "按${multiple}字节对齐发送音频: ${chunk.size} 字节,缓存剩余: ${ttsdata.size - sendLen}") + eventCallback?.onSynthesisAudioGenerated(utteranceId, text, chunk) + ttsdata = ttsdata.copyOfRange(sendLen, ttsdata.size) + } else { + Log.d(TAG, "暂未够${multiple}字节,当前缓存: ${ttsdata.size}") + } + } + /** + * 生成新的 utteranceId + * 用途:在识别的一个话段开始时生成,并贯穿该段的识别/翻译/合成事件 + */ + private fun nextUtteranceId(): String = "utt-${System.currentTimeMillis()}-${utteranceSeq.incrementAndGet()}" + + /** + * 安全取消“识别中翻译”任务,避免阻塞主线程 + * + * 功能: + * - 在后台调度器执行 cancelAndJoin,防止主线程死锁/ANR + * - 捕获取消异常并记录日志,保证流程稳定 + */ + private suspend fun cancelRecognizingTranslationJobSafely() { + val job = recognizingTranslationJob + if (job != null && job.isActive) { + withContext(Dispatchers.Default) { + try { + job.cancelAndJoin() + } catch (e: CancellationException) { + Log.d(TAG, "识别中翻译任务取消完成") + } catch (e: Exception) { + Log.w(TAG, "取消识别中翻译任务异常: ${e.message}", e) + } finally { + serviceState.isTranslating.set(false) + } + } + } else { + // 无活动任务时也确保状态为 false + serviceState.isTranslating.set(false) + } + } + /** * 服务配置类 */ @@ -87,6 +168,8 @@ class IntegratedSpeechTranslationService( var sourceLanguage: String = "zh-CN", var targetLanguage: String = "en-US", var currentVoice: String = "en-US-AriaNeural", + var translationSourceLanguage: String = "", + var translationTargetLanguage: String = "", var speechRate: String = "0%", var speechPitch: String = "0%", var speechVolume: String = "100%", @@ -113,21 +196,30 @@ class IntegratedSpeechTranslationService( */ interface ServiceEventCallback { fun onServiceInitialized() - fun onRecognizing(text: String, language: String, confidence: Float) - fun onRecognized(text: String, language: String, confidence: Float) - fun onTranslated(originalText: String, translatedText: String, targetLanguage: String) - fun onTranslationStarted(text: String) - fun onTranslationFailed(text: String, error: String) - fun onSynthesisStarted(text: String) - fun onSynthesisCompleted(text: String) - fun onSynthesisFailed(text: String, error: String) - fun onSynthesisProgress(text: String, progress: Float) + fun onRecognizing(utteranceId: String, text: String, language: String, confidence: Float) + fun onRecognized(utteranceId: String, text: String, language: String, confidence: Float) + fun onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) + /** + * 临时翻译结果回调(识别中) + * 用途:当“识别中事件”产生的临时文本完成翻译时触发,不进行合成。 + * @param originalText 原始识别中的文本 + * @param translatedText 翻译后的文本 + * @param targetLanguage 目标语言 + */ + fun onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String) + fun onTranslationStarted(utteranceId: String, text: String) + fun onTranslationFailed(utteranceId: String, text: String, error: String) + fun onSynthesisStarted(utteranceId: String, text: String) + fun onSynthesisCompleted(utteranceId: String, text: String) + fun onSynthesisFailed(utteranceId: String, text: String, error: String) + fun onSynthesisProgress(utteranceId: String, text: String, progress: Float) fun onRecognitionStarted() fun onRecognitionStopped() fun onStateChanged(component: String, isActive: Boolean) fun onError(component: String, error: String) + // 新增:返回合成的音频数据 - fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) + fun onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: ByteArray) } /** @@ -177,17 +269,20 @@ class IntegratedSpeechTranslationService( ): Boolean = withContext(Dispatchers.IO) { try { Log.d(TAG, "开始初始化服务") - + // 重新初始化前先清理 if (serviceState.isInitialized.get()) { Log.d(TAG, "检测到服务已初始化,先进行清理") dispose() } - + this@IntegratedSpeechTranslationService.eventCallback = callback serviceConfig?.let { this@IntegratedSpeechTranslationService.serviceConfig = it } - Log.d(TAG, "开始初始化整合服务,azureConfig:${azureConfig},translationConfig:${translationConfig}") + Log.d( + TAG, + "开始初始化整合服务,azureConfig:${azureConfig},translationConfig:${translationConfig}" + ) // 初始化Azure语音服务 if (!initializeAzureServices(azureConfig)) { @@ -212,13 +307,13 @@ class IntegratedSpeechTranslationService( serviceState.isInitialized.set(true) Log.d(TAG, "整合服务初始化成功") - startContinuousTranslation() + startContinuousTranslation() withContext(Dispatchers.Main) { callback.onServiceInitialized() } - - recordfile1 = RecordFile(); - return@withContext true + + recordfile1 = RecordFile(); + return@withContext true } catch (e: Exception) { Log.e(TAG, "初始化失败", e) withContext(Dispatchers.Main) { @@ -227,7 +322,7 @@ class IntegratedSpeechTranslationService( "初始化失败: ${e.message}" ) } - return@withContext false + return@withContext false } } @@ -240,22 +335,10 @@ class IntegratedSpeechTranslationService( speechConfig = SpeechConfig.fromSubscription(config.subscriptionKey, config.region).apply { speechRecognitionLanguage = serviceConfig.sourceLanguage - setSpeechSynthesisVoiceName(getVoiceForLanguage(serviceConfig.targetLanguage)) + val voice = serviceConfig.currentVoice.ifEmpty { serviceConfig.targetLanguage } + setSpeechSynthesisVoiceName(voice) Log.d(TAG, "设置语音合成器语言: ${serviceConfig.targetLanguage}") - setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff16Khz16BitMonoPcm) - - // 优化配置 - setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", END_SILENCE_TIMEOUT) - setProperty("Speech_SegmentationSilenceTimeoutMs", SEGMENTATION_SILENCE_TIMEOUT) - setProperty( - "SpeechServiceConnection_InitialSilenceTimeoutMs", - INITIAL_SILENCE_TIMEOUT - ) - setProperty("SpeechServiceConnection_RecoMode", "INTERACTIVE") - - // 启用详细结果 - setProperty("SpeechServiceResponse_RequestDetailedResultTrueFalse", "true") - + setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Raw16Khz16BitMonoPcm) // 自动语言检测 if (serviceConfig.enableAutoLanguageDetection) { setProperty("SpeechServiceConnection_LanguageIdMode", "Continuous") @@ -275,10 +358,10 @@ class IntegratedSpeechTranslationService( */ private suspend fun initializeTranslationService(config: TranslationConfiguration): Boolean { return try { - translationService = VolcanoTranslationServiceImpl().apply { + translationService = MicrosoftTranslationServiceImpl().apply { initialize(config.toConfigMap()) } - Log.d(TAG, "翻译服务初始化完成") + Log.d(TAG, "翻译服务初始化完成(Microsoft)") true } catch (e: Exception) { Log.e(TAG, "翻译服务初始化失败", e) @@ -308,12 +391,32 @@ class IntegratedSpeechTranslationService( recognizing.addEventListener { _, event -> if (event.result.text.isNotEmpty()) { val confidence = extractConfidence(event.result) - Log.d(TAG, "识别中事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") + // 开始新的话段时生成并记录 utteranceId + if (!isInUtterance.get()) { + currentUtteranceId = nextUtteranceId() + isInUtterance.set(true) + } + val uttId = currentUtteranceId ?: nextUtteranceId() + Log.d( + TAG, + "识别中事件[${uttId}]: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}" + ) eventCallback?.onRecognizing( + uttId, event.result.text, serviceConfig.sourceLanguage, confidence ) + + // 当服务处于运行状态时,为“识别中”的内容启动仅翻译流程(不合成) + if (serviceState.isRecognizing.get()) { + // 取消上一次“识别中翻译”任务,避免旧翻译结果覆盖新结果 + recognizingTranslationJob?.cancel() + recognizingTranslationJob = launch { + Log.d(TAG, "识别中翻译触发: ${event.result.text}") + processTranslationOnly(uttId, event.result.text) + } + } } } @@ -323,17 +426,27 @@ class IntegratedSpeechTranslationService( ResultReason.RecognizedSpeech -> { if (event.result.text.isNotEmpty()) { val confidence = extractConfidence(event.result) + val uttId = currentUtteranceId ?: nextUtteranceId() eventCallback?.onRecognized( + uttId, event.result.text, serviceConfig.sourceLanguage, confidence ) - Log.d(TAG, "识别完成事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") + Log.d( + TAG, + "识别完成事件[${uttId}]: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}" + ) // 只有在服务仍在运行时才触发翻译流程 if (serviceState.isRecognizing.get()) { + isInUtterance.set(false) + currentUtteranceId = null launch { - Log.d(TAG, "触发翻译流程: ${event.result.text}") - processTranslationAndSynthesis(event.result.text) + // 识别完成时,马上打断上一次“识别中”的翻译,但不要在主线程 join + cancelRecognizingTranslationJobSafely() + Log.d(TAG, "触发最终翻译流程: ${event.result.text}") + processTranslationAndSynthesis(uttId, event.result.text) + } } else { Log.d(TAG, "服务已停止,跳过翻译流程: ${event.result.text}") @@ -353,7 +466,7 @@ class IntegratedSpeechTranslationService( // 会话开始事件 sessionStarted.addEventListener { _, _ -> - Log.d(TAG, "会话开始事件") + Log.d(TAG, "会话开始事件") serviceState.isRecognizing.set(true) eventCallback?.onRecognitionStarted() eventCallback?.onStateChanged("Recognition", true) @@ -361,7 +474,7 @@ class IntegratedSpeechTranslationService( // 会话停止事件 sessionStopped.addEventListener { _, _ -> - Log.d(TAG, "会话停止事件") + Log.d(TAG, "会话停止事件") serviceState.isRecognizing.set(false) eventCallback?.onRecognitionStopped() eventCallback?.onStateChanged("Recognition", false) @@ -369,7 +482,7 @@ class IntegratedSpeechTranslationService( // 取消事件 canceled.addEventListener { _, event -> - Log.d(TAG, "取消事件") + Log.d(TAG, "取消事件") serviceState.isRecognizing.set(false) val errorDetails = event.errorDetails ?: "未知错误" eventCallback?.onError("Recognition", "识别被取消: $errorDetails") @@ -383,50 +496,81 @@ class IntegratedSpeechTranslationService( /** * 设置语音合成器 */ + /** + * 设置语音合成器 + * + * 说明:始终提供有效的 AudioConfig;当禁用播放时使用 Push 输出流以避免传入 null。 + */ private fun setupSpeechSynthesizer() { synthesizer?.close() - - // 根据配置决定音频输出方式 + + // 根据配置决定音频输出方式(禁用播放时也提供推流) val audioConfig = if (serviceConfig.enableAudioPlayback) { AudioConfig.fromDefaultSpeakerOutput() } else { - // 不输出到扬声器,只生成音频数据 - null + // 创建空消费的推流,避免使用 null AudioConfig + synthOutputStream?.close() + synthOutputStream = AudioOutputStream.createPushStream(object : PushAudioOutputStreamCallback() { + override fun write(dataBuffer: ByteArray): Int { + // 丢弃数据但保持流有效 + return dataBuffer.size + } + override fun close() { /* no-op */ } + }) + AudioConfig.fromStreamOutput(synthOutputStream) } - + synthesizer = SpeechSynthesizer(speechConfig, audioConfig).apply { - // 合成开始事件 - SynthesisStarted.addEventListener { _, _ -> - Log.d(TAG, "合成开始事件") - serviceState.isSynthesizing.set(true) - eventCallback?.onStateChanged("Synthesis", true) - } + // 合成开始事件 + SynthesisStarted.addEventListener { _, _ -> + Log.d(TAG, "合成开始事件") + ttsdata = ByteArray(0) + serviceState.isSynthesizing.set(true) + val uttId = currentSynthesisId ?: "" + val text = currentSynthesisText ?: "" + eventCallback?.onSynthesisStarted(uttId, text) + eventCallback?.onStateChanged("Synthesis", true) + } - // 合成进行中事件 - Synthesizing.addEventListener { _, event -> - Log.d(TAG, "合成进行中事件") - // 计算进度(简化版) - val progress = 0.5f // 实际应用中可以根据音频数据计算 - eventCallback?.onSynthesisProgress("", progress) - } + // 合成进行中事件 + Synthesizing.addEventListener { _, event -> + Log.d(TAG, "合成进行中事件") + // 计算进度(简化版) + val progress = 0.5f // 实际应用中可以根据音频数据计算 + val uttId = currentSynthesisId ?: "" + val text = currentSynthesisText ?: "" + eventCallback?.onSynthesisProgress(uttId, text, progress) + + val incoming = event.result.audioData ?: ByteArray(0) + processSynthesisAudioChunk(uttId, text, incoming, 1280) + } - // 合成完成事件 - SynthesisCompleted.addEventListener { _, event -> - Log.d(TAG, "合成完成事件") - // eventCallback?.onSynthesisCompleted("") - eventCallback?.onStateChanged("Synthesis", false) - } + // 合成完成事件 + SynthesisCompleted.addEventListener { _, event -> + Log.d(TAG, "合成完成事件") + val uttId = currentSynthesisId ?: "" + val text = currentSynthesisText ?: "" + eventCallback?.onSynthesisCompleted(uttId, text) + currentSynthesisId = null + currentSynthesisText = null + eventCallback?.onStateChanged("Synthesis", false) + } - // 合成取消事件 - SynthesisCanceled.addEventListener { _, event -> - Log.d(TAG, "合成取消事件") - serviceState.isSynthesizing.set(false) - val reason = event.result.reason - eventCallback?.onError("Synthesis", "语音合成取消: $reason") - eventCallback?.onStateChanged("Synthesis", false) - } + // 合成取消事件 + SynthesisCanceled.addEventListener { _, event -> + Log.d(TAG, "合成取消事件") + ttsdata = ByteArray(0) + serviceState.isSynthesizing.set(false) + val reason = event.result.reason + val uttId = currentSynthesisId ?: "" + val text = currentSynthesisText ?: "" + eventCallback?.onSynthesisFailed(uttId, text, "语音合成取消: $reason") + currentSynthesisId = null + currentSynthesisText = null + eventCallback?.onStateChanged("Synthesis", false) } + } Log.d(TAG, "语音合成器设置完成,播放模式: ${serviceConfig.enableAudioPlayback}") } @@ -435,8 +579,11 @@ class IntegratedSpeechTranslationService( * 开始连续语音翻译 */ fun startContinuousTranslation(): Boolean { - Log.d(TAG, "尝试启动连续翻译,当前状态:初始化=${serviceState.isInitialized.get()}, 识别中=${serviceState.isRecognizing.get()}") - + Log.d( + TAG, + "尝试启动连续翻译,当前状态:初始化=${serviceState.isInitialized.get()}, 识别中=${serviceState.isRecognizing.get()}" + ) + if (!serviceState.isInitialized.get()) { eventCallback?.onError("Service", "服务未初始化") return false @@ -465,6 +612,43 @@ class IntegratedSpeechTranslationService( false } } + + /** + * 初始化成功后示例合成一句目标语言话语 + * 功能:将固定中文短句翻译为目标语言并进行语音合成,便于确认语音配置与管线是否正常 + */ + fun speakDemoSentence() { + if (!serviceState.isInitialized.get()) { + return + } + + val srcLang = if (serviceConfig.translationSourceLanguage.isNotBlank()) serviceConfig.translationSourceLanguage else serviceConfig.sourceLanguage + val tgtLang = if (serviceConfig.translationTargetLanguage.isNotBlank()) serviceConfig.translationTargetLanguage else serviceConfig.targetLanguage + + launch { + try { + val sourceText = "初始化成功" + val res = translationService?.translateText(sourceText, srcLang, tgtLang) + val textToSpeak = res?.translatedText ?: run { + when { + tgtLang.lowercase().startsWith("en") -> "Initialization successful" + tgtLang.lowercase().startsWith("zh") -> "初始化成功" + tgtLang.lowercase().startsWith("ja") -> "初期化が完了しました" + tgtLang.lowercase().startsWith("ko") -> "초기화가 완료되었습니다" + tgtLang.lowercase().startsWith("fr") -> "Initialisation réussie" + tgtLang.lowercase().startsWith("es") -> "Inicialización completada" + else -> "Initialization completed" + } + } + + val uttId = nextUtteranceId() + synthesizeText(uttId, textToSpeak) + } catch (e: Exception) { + Log.w(TAG, "演示合成失败: ${e.message}") + } + } + } + fun enableRecord(filePath: String) { Log.i(TAG, "开启录音:") @@ -489,10 +673,10 @@ class IntegratedSpeechTranslationService( launch(Dispatchers.IO) { // 1) 停止识别器:等待短超时,避免卡住 try { + // 非阻塞停止:调用异步停止,不等待 Future,避免抛出超时异常 recognizer?.stopContinuousRecognitionAsync() - ?.get(1500, java.util.concurrent.TimeUnit.MILLISECONDS) } catch (t: Throwable) { - Log.w(TAG, "停止识别器超时或失败", t) + Log.w(TAG, "停止识别器调用失败", t) } // 2) 停止录音线程(打断并短等待退出) @@ -530,64 +714,86 @@ class IntegratedSpeechTranslationService( /** * 处理翻译和合成流程 */ - private suspend fun processTranslationAndSynthesis(text: String) { + /** + * 处理翻译和合成流程(最终结果) + * 参数: + * - utteranceId: 该句话的唯一ID,用于绑定识别/翻译/合成事件 + * - text: 识别完成的最终文本 + */ + private suspend fun processTranslationAndSynthesis(utteranceId: String, text: String) { // 检查服务是否仍在运行 if (!serviceState.isRecognizing.get()) { Log.d(TAG, "服务已停止,取消翻译流程: $text") return } - + Log.d(TAG, " 翻译状态:${serviceState.isTranslating.get()}") + var targetUtteranceId = utteranceId + var targetText = text if (serviceState.isTranslating.get()) { - Log.w(TAG, "翻译正在进行中,跳过当前请求") - return + finalTranslationQueue.offer(utteranceId to text) + while (serviceState.isTranslating.get()) { + delay(50) + } + finalTranslationQueue.poll()?.let { pair -> + targetUtteranceId = pair.first + targetText = pair.second + } } serviceState.isTranslating.set(true) - eventCallback?.onTranslationStarted(text) + eventCallback?.onTranslationStarted(targetUtteranceId, targetText) eventCallback?.onStateChanged("Translation", true) - Log.d(TAG, "翻译开始${text}") + Log.d(TAG, "翻译开始${targetText}") try { + val srcLang = if (serviceConfig.translationSourceLanguage.isNotEmpty()) serviceConfig.translationSourceLanguage else serviceConfig.sourceLanguage + val tgtLang = if (serviceConfig.translationTargetLanguage.isNotEmpty()) serviceConfig.translationTargetLanguage else serviceConfig.targetLanguage val translationResult = withTimeout(serviceConfig.translationTimeout) { translationService?.translateText( - text = text, - sourceLanguage = serviceConfig.sourceLanguage, - targetLanguage = serviceConfig.targetLanguage + text = targetText, + sourceLanguage = srcLang, + targetLanguage = tgtLang ) } serviceState.isTranslating.set(false) eventCallback?.onStateChanged("Translation", false) + /** + * 执行最终翻译并进行合成 + * 区别于识别中:此方法通过 onTranslated 回调返回最终翻译结果,并触发合成。 + * @param text 识别完成后的最终文本 + */ when { translationResult?.success == true && !translationResult.translatedText.isNullOrEmpty() -> { Log.d(TAG, "翻译成功${translationResult.translatedText}") eventCallback?.onTranslated( - text, + targetUtteranceId, + targetText, translationResult.translatedText, - serviceConfig.targetLanguage + tgtLang ) - synthesizeText(translationResult.translatedText) + synthesizeText(targetUtteranceId, translationResult.translatedText) } translationResult?.error != null -> { - eventCallback?.onTranslationFailed(text, translationResult.error) + eventCallback?.onTranslationFailed(targetUtteranceId, targetText, translationResult.error) } else -> { - eventCallback?.onTranslationFailed(text, "翻译服务返回空结果") + eventCallback?.onTranslationFailed(targetUtteranceId, targetText, "翻译服务返回空结果") } } } catch (e: TimeoutCancellationException) { serviceState.isTranslating.set(false) eventCallback?.onStateChanged("Translation", false) - eventCallback?.onTranslationFailed(text, "翻译超时") + eventCallback?.onTranslationFailed(targetUtteranceId, targetText, "翻译超时") } catch (e: Exception) { serviceState.isTranslating.set(false) eventCallback?.onStateChanged("Translation", false) - eventCallback?.onTranslationFailed(text, "翻译异常: ${e.message}") + eventCallback?.onTranslationFailed(targetUtteranceId, targetText, "翻译异常: ${e.message}") } } @@ -596,7 +802,14 @@ class IntegratedSpeechTranslationService( * @param text 要合成的文本 * @return 返回合成的音频数据,如果合成失败则返回null */ - private suspend fun synthesizeText(text: String): ByteArray? { + /** + * 语音合成 + * 参数: + * - utteranceId: 句子唯一ID + * - text: 合成目标文本 + * 返回:音频数据(可能为 null) + */ + private suspend fun synthesizeText(utteranceId: String, text: String): ByteArray? { if (serviceState.isSynthesizing.get()) { Log.w(TAG, "语音合成正在进行中") return null @@ -605,41 +818,128 @@ class IntegratedSpeechTranslationService( try { Log.d(TAG, "开始语音合成: $text") serviceState.isSynthesizing.set(true) - eventCallback?.onSynthesisStarted(text) - + + // 在开始时记录当前合成上下文,并抛出开始事件 + currentSynthesisId = utteranceId + currentSynthesisText = text + eventCallback?.onSynthesisStarted(utteranceId, text) + val ssml = generateOptimizedSsml(text) - + // 始终使用异步方法获取音频数据 val result = synthesizer?.SpeakSsmlAsync(ssml)?.get() - + if (result?.reason == ResultReason.SynthesizingAudioCompleted) { - val playbackStatus = if (serviceConfig.enableAudioPlayback) "音频已播放" else "音频已合成但未播放" + val playbackStatus = + if (serviceConfig.enableAudioPlayback) "音频已播放" else "音频已合成但未播放" Log.d(TAG, "语音合成成功,$playbackStatus") - - // 获取音频数据 - val audioData = result.audioData - if (audioData != null && audioData.isNotEmpty()) { - // 通过回调返回音频数据 - eventCallback?.onSynthesisAudioGenerated(text, audioData) - Log.d(TAG, "音频数据大小: ${audioData.size} 字节") - } - - return audioData + + // 获取原始音频数据 + val rawAudioData = result.audioData + + Log.d(TAG, "语音合成成功,发送音频: ${ttsdata.size} 字节,缓存剩余: ${ttsdata.size}") + eventCallback?.onSynthesisAudioGenerated(utteranceId, text, ttsdata) + ttsdata = ByteArray(0) + return rawAudioData } else { Log.e(TAG, "语音合成失败: ${result?.reason}") - eventCallback?.onSynthesisFailed(text, "合成失败: ${result?.reason}") + eventCallback?.onSynthesisFailed(utteranceId, text, "合成失败: ${result?.reason}") return null } - + } catch (e: Exception) { Log.e(TAG, "语音合成失败", e) - eventCallback?.onSynthesisFailed(text, "合成失败: ${e.message}") + eventCallback?.onSynthesisFailed(utteranceId, text, "合成失败: ${e.message}") return null } finally { serviceState.isSynthesizing.set(false) } } + + /** + * 仅执行文本翻译,不进行语音合成 + * + * 功能: + * - 在“识别中事件”产生临时文本时执行翻译 + * - 通过回调返回翻译结果,不触发合成 + * - 支持取消(识别完成时会取消该任务) + * + * 参数: + * - text: 待翻译的源文本 + */ + /** + * 仅执行文本翻译,不进行语音合成(识别中临时结果) + * 参数: + * - utteranceId: 句子唯一ID + * - text: 待翻译的源文本 + */ + private suspend fun processTranslationOnly(utteranceId: String, text: String) { + // 若服务已停止,直接跳过 + if (!serviceState.isRecognizing.get()) { + Log.d(TAG, "服务已停止,取消识别中翻译流程: $text") + return + } + + // 若已有翻译在进行,直接退出(避免并发) + if (serviceState.isTranslating.get()) { + Log.d(TAG, "已有翻译进行中,识别中翻译跳过: $text") + return + } + + serviceState.isTranslating.set(true) + eventCallback?.onTranslationStarted(utteranceId, text) + eventCallback?.onStateChanged("Translation", true) + Log.d(TAG, "识别中翻译开始: $text") + + try { + val srcLang = if (serviceConfig.translationSourceLanguage.isNotEmpty()) serviceConfig.translationSourceLanguage else serviceConfig.sourceLanguage + val tgtLang = if (serviceConfig.translationTargetLanguage.isNotEmpty()) serviceConfig.translationTargetLanguage else serviceConfig.targetLanguage + val translationResult = withTimeout(serviceConfig.translationTimeout) { + translationService?.translateText( + text = text, + sourceLanguage = srcLang, + targetLanguage = tgtLang + ) + } + + when { + translationResult?.success == true && !translationResult.translatedText.isNullOrEmpty() -> { + Log.d(TAG, "识别中翻译成功: ${translationResult.translatedText}") + // 使用“识别中”专用的临时翻译回调,区分最终翻译 + eventCallback?.onInterimTranslated( + utteranceId, + text, + translationResult.translatedText, + tgtLang + ) + } + + translationResult?.error != null -> { + Log.w(TAG, "识别中翻译失败: ${translationResult.error}") + eventCallback?.onTranslationFailed(utteranceId, text, translationResult.error) + } + + else -> { + Log.w(TAG, "识别中翻译返回空结果") + eventCallback?.onTranslationFailed(utteranceId, text, "翻译服务返回空结果") + } + } + } catch (e: CancellationException) { + // 被识别完成事件或新识别中事件取消,不做失败提示 + Log.d(TAG, "识别中翻译任务已取消: $text") + } catch (e: TimeoutCancellationException) { + Log.w(TAG, "识别中翻译超时: $text") + eventCallback?.onTranslationFailed(utteranceId, text, "翻译超时") + } catch (e: Exception) { + Log.e(TAG, "识别中翻译异常: ${e.message}", e) + eventCallback?.onTranslationFailed(utteranceId, text, "翻译异常: ${e.message}") + } finally { + serviceState.isTranslating.set(false) + eventCallback?.onStateChanged("Translation", false) + } + } + /** * 生成优化的SSML */ @@ -681,66 +981,7 @@ class IntegratedSpeechTranslationService( .replace("'", "'") } - /** - * 根据语言获取对应的语音 - */ - private fun getVoiceForLanguage(language: String): String { - return voiceCache.getOrPut(language) { - when (language) { - // 中文相关 - "zh-CN" -> "zh-CN-XiaoxiaoNeural" - "zh-TW" -> "zh-TW-HsiaoChenNeural" - "zh-HK" -> "zh-HK-HiuMaanNeural" - - // 英语相关 - "en-US" -> "en-US-AriaNeural" - - // 亚洲语言 - "ja-JP" -> "ja-JP-NanamiNeural" - "ko-KR" -> "ko-KR-SunHiNeural" - "th-TH" -> "th-TH-PremwadeeNeural" - "vi-VN" -> "vi-VN-HoaiMyNeural" - "id-ID" -> "id-ID-GadisNeural" - "hi-IN" -> "hi-IN-SwaraNeural" - "ms-MY" -> "ms-MY-YasminNeural" - "bn-IN" -> "bn-IN-TanishaaNeural" - "ta-IN" -> "ta-IN-PallaviNeural" - "te-IN" -> "te-IN-ShrutiNeural" - "mr-IN" -> "mr-IN-AarohiNeural" - "ur-IN" -> "ur-IN-GulNeural" - "tr-TR" -> "tr-TR-EmelNeural" - "fa-IR" -> "fa-IR-DilaraNeural" - - // 欧洲语言 - "fr-FR" -> "fr-FR-DeniseNeural" - "es-ES" -> "es-ES-ElviraNeural" - "pt-BR" -> "pt-BR-FranciscaNeural" - "it-IT" -> "it-IT-ElsaNeural" - "de-DE" -> "de-DE-KatjaNeural" - "ru-RU" -> "ru-RU-SvetlanaNeural" - "pl-PL" -> "pl-PL-AgnieszkaNeural" - "nl-NL" -> "nl-NL-ColetteNeural" - "sv-SE" -> "sv-SE-SofieNeural" - "cs-CZ" -> "cs-CZ-VlastaNeural" - "el-GR" -> "el-GR-AthinaNeural" - "ro-RO" -> "ro-RO-AlinaNeural" - "hu-HU" -> "hu-HU-NoemiNeural" - "uk-UA" -> "uk-UA-PolinaNeural" - "da-DK" -> "da-DK-ChristelNeural" - "fi-FI" -> "fi-FI-NooraNeural" - "nb-NO" -> "nb-NO-IselinNeural" - "hr-HR" -> "hr-HR-GabrijelaNeural" - "ca-ES" -> "ca-ES-JoanaNeural" - - // 阿拉伯语和非洲语言 - "ar-EG" -> "ar-EG-SalmaNeural" - "sw-KE" -> "sw-KE-ZuriNeural" - else -> "en-US-AriaNeural" - }.also { - serviceConfig.currentVoice = it - } - } - } + /** * 提取识别置信度 @@ -775,7 +1016,8 @@ class IntegratedSpeechTranslationService( } if (needsSynthesizerUpdate) { - speechConfig?.setSpeechSynthesisVoiceName(getVoiceForLanguage(serviceConfig.targetLanguage)) + val voice = serviceConfig.currentVoice.ifEmpty { serviceConfig.targetLanguage } + speechConfig?.setSpeechSynthesisVoiceName(voice) setupSpeechSynthesizer() } @@ -783,35 +1025,35 @@ class IntegratedSpeechTranslationService( } /** - * 设置音频输出设备 - */ -fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { - try { - val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager - audioManager?.let { manager -> - when (device) { - com.deep_voice.speech.tts.AudioOutputDevice.SPEAKER -> { - manager.mode = AudioManager.MODE_IN_COMMUNICATION - manager.isSpeakerphoneOn = true - } + * 设置音频输出设备 + */ + fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { + try { + val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager + audioManager?.let { manager -> + when (device) { + com.deep_voice.speech.tts.AudioOutputDevice.SPEAKER -> { + manager.mode = AudioManager.MODE_IN_COMMUNICATION + manager.isSpeakerphoneOn = true + } - com.deep_voice.speech.tts.AudioOutputDevice.HEADPHONES -> { - manager.mode = AudioManager.MODE_NORMAL - manager.isSpeakerphoneOn = false - } + com.deep_voice.speech.tts.AudioOutputDevice.HEADPHONES -> { + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + } - com.deep_voice.speech.tts.AudioOutputDevice.DEFAULT -> { - manager.mode = AudioManager.MODE_NORMAL - manager.isSpeakerphoneOn = false + com.deep_voice.speech.tts.AudioOutputDevice.DEFAULT -> { + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + } } } + Log.d(TAG, "音频输出设备设置为: $device") + } catch (e: Exception) { + Log.e(TAG, "设置音频输出设备失败", e) + eventCallback?.onError("Audio", "设置音频输出设备失败: ${e.message}") } - Log.d(TAG, "音频输出设备设置为: $device") - } catch (e: Exception) { - Log.e(TAG, "设置音频输出设备失败", e) - eventCallback?.onError("Audio", "设置音频输出设备失败: ${e.message}") } -} /** * 切换语言对 @@ -832,12 +1074,12 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { fun setAudioPlaybackEnabled(enabled: Boolean) { serviceConfig.enableAudioPlayback = enabled Log.d(TAG, "音频播放设置更新: $enabled") - + // 重新设置语音合成器以应用新配置 if (serviceState.isInitialized.get()) { setupSpeechSynthesizer() } - + eventCallback?.onStateChanged("AudioPlayback", enabled) } @@ -849,27 +1091,27 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { fun changeRecognitionLanguage(newLanguage: String, restartRecognition: Boolean = true) { try { val wasRecognizing = serviceState.isRecognizing.get() - + // 如果正在识别,先停止 if (wasRecognizing && restartRecognition) { stopContinuousTranslation() } - + // 更新配置 serviceConfig.sourceLanguage = newLanguage speechConfig?.speechRecognitionLanguage = newLanguage - + // 重新设置识别器 setupSpeechRecognizer() - + Log.d(TAG, "识别语言已更改为: $newLanguage") eventCallback?.onStateChanged("LanguageChanged", true) - + // 如果之前在识别且需要重启,则重新开始 if (wasRecognizing && restartRecognition) { startContinuousTranslation() } - + } catch (e: Exception) { Log.e(TAG, "更改识别语言失败", e) eventCallback?.onError("LanguageChange", "更改识别语言失败: ${e.message}") @@ -881,24 +1123,30 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { * @param languages 支持的语言列表 * @param enableAutoDetection 是否启用自动语言检测 */ - fun setupMultiLanguageRecognition(languages: List, enableAutoDetection: Boolean = true) { + fun setupMultiLanguageRecognition( + languages: List, + enableAutoDetection: Boolean = true + ) { try { serviceConfig.enableAutoLanguageDetection = enableAutoDetection - + if (enableAutoDetection && languages.isNotEmpty()) { // 设置自动语言检测的候选语言 speechConfig?.setProperty("SpeechServiceConnection_LanguageIdMode", "Continuous") - + // 构建语言候选列表 val languageList = languages.joinToString(",") - speechConfig?.setProperty("SpeechServiceConnection_ContinuousLanguageIdPriority", languageList) - + speechConfig?.setProperty( + "SpeechServiceConnection_ContinuousLanguageIdPriority", + languageList + ) + Log.d(TAG, "多语言识别已设置,支持语言: $languageList") } - + // 重新设置识别器 setupSpeechRecognizer() - + } catch (e: Exception) { Log.e(TAG, "设置多语言识别失败", e) eventCallback?.onError("MultiLanguageSetup", "设置多语言识别失败: ${e.message}") @@ -908,44 +1156,44 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { /** * 获取当前支持的语言列表 */ - -fun getSupportedLanguages(): List { - return listOf( - // 亚洲语言 - "zh-CN", "zh-TW", "zh-HK", "ja-JP", "ko-KR", - "hi-IN", "ta-IN", "te-IN", "bn-IN", "pa-IN", - "ur-PK", "th-TH", "vi-VN", "ms-MY", "id-ID", - "fil-PH", "km-KH", "my-MM", "lo-LA", "ne-NP", - - // 欧洲语言 - "en-US", "en-GB", "en-CA", "en-AU", "fr-FR", - "fr-CA", "de-DE", "es-ES", "es-MX", "it-IT", - "nl-NL", "pl-PL", "ru-RU", "tr-TR", "uk-UA", - "cs-CZ", "hu-HU", "sv-SE", "fi-FI", "da-DK", - "no-NO", "ro-RO", "el-GR", "bg-BG", "hr-HR", - "sr-RS", "sk-SK", "sl-SI", "lt-LT", "lv-LV", - "et-EE", "is-IS", "ga-IE", - - // 中东/非洲语言 - "ar-SA", "ar-EG", "he-IL", "fa-IR", "ps-AF", - "ku-TR", "sw-KE", "am-ET", "yo-NG", "zu-ZA", - "xh-ZA", "af-ZA", "st-ZA", "tn-ZA", "ha-NG", - "ig-NG", "mg-MG", "rw-RW", "so-SO", "ti-ER", - - // 美洲/大洋洲 - "pt-BR", "pt-PT", "qu-PE", "gn-PY", "ay-BO", - "mi-NZ", "haw-US", "sm-WS", "to-TO", "fj-FJ", - - // 其他重要语言 - "as-IN", "or-IN", "kn-IN", "ml-IN", "gu-IN", - "mr-IN", "sa-IN", "sd-IN", "bo-CN", "ug-CN", - "ii-CN", "mn-MN", "jv-ID", "su-ID", "ceb-PH", - "gl-ES", "eu-ES", "ca-ES", "gd-GB", "cy-GB", - "br-FR", "fy-NL", "lb-LU", "mt-MT", "sq-AL", - "hy-AM", "ka-GE", "be-BY", "kk-KZ", "uz-UZ", - "ky-KG", "tg-TJ", "tk-TM", "tt-RU", "cv-RU" - ) -} + + fun getSupportedLanguages(): List { + return listOf( + // 亚洲语言 + "zh-CN", "zh-TW", "zh-HK", "ja-JP", "ko-KR", + "hi-IN", "ta-IN", "te-IN", "bn-IN", "pa-IN", + "ur-PK", "th-TH", "vi-VN", "ms-MY", "id-ID", + "fil-PH", "km-KH", "my-MM", "lo-LA", "ne-NP", + + // 欧洲语言 + "en-US", "en-GB", "en-CA", "en-AU", "fr-FR", + "fr-CA", "de-DE", "es-ES", "es-MX", "it-IT", + "nl-NL", "pl-PL", "ru-RU", "tr-TR", "uk-UA", + "cs-CZ", "hu-HU", "sv-SE", "fi-FI", "da-DK", + "no-NO", "ro-RO", "el-GR", "bg-BG", "hr-HR", + "sr-RS", "sk-SK", "sl-SI", "lt-LT", "lv-LV", + "et-EE", "is-IS", "ga-IE", + + // 中东/非洲语言 + "ar-SA", "ar-EG", "he-IL", "fa-IR", "ps-AF", + "ku-TR", "sw-KE", "am-ET", "yo-NG", "zu-ZA", + "xh-ZA", "af-ZA", "st-ZA", "tn-ZA", "ha-NG", + "ig-NG", "mg-MG", "rw-RW", "so-SO", "ti-ER", + + // 美洲/大洋洲 + "pt-BR", "pt-PT", "qu-PE", "gn-PY", "ay-BO", + "mi-NZ", "haw-US", "sm-WS", "to-TO", "fj-FJ", + + // 其他重要语言 + "as-IN", "or-IN", "kn-IN", "ml-IN", "gu-IN", + "mr-IN", "sa-IN", "sd-IN", "bo-CN", "ug-CN", + "ii-CN", "mn-MN", "jv-ID", "su-ID", "ceb-PH", + "gl-ES", "eu-ES", "ca-ES", "gd-GB", "cy-GB", + "br-FR", "fy-NL", "lb-LU", "mt-MT", "sq-AL", + "hy-AM", "ka-GE", "be-BY", "kk-KZ", "uz-UZ", + "ky-KG", "tg-TJ", "tk-TM", "tt-RU", "cv-RU" + ) + } /** @@ -961,13 +1209,22 @@ fun getSupportedLanguages(): List { ) { try { speechConfig?.apply { - setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", endSilenceTimeout.toString()) + setProperty( + "SpeechServiceConnection_EndSilenceTimeoutMs", + endSilenceTimeout.toString() + ) setProperty("Speech_SegmentationSilenceTimeoutMs", segmentationTimeout.toString()) - setProperty("SpeechServiceConnection_InitialSilenceTimeoutMs", initialSilenceTimeout.toString()) + setProperty( + "SpeechServiceConnection_InitialSilenceTimeoutMs", + initialSilenceTimeout.toString() + ) } - - Log.d(TAG, "识别参数已更新: 结束静音=${endSilenceTimeout}ms, 分段静音=${segmentationTimeout}ms, 初始静音=${initialSilenceTimeout}ms") - + + Log.d( + TAG, + "识别参数已更新: 结束静音=${endSilenceTimeout}ms, 分段静音=${segmentationTimeout}ms, 初始静音=${initialSilenceTimeout}ms" + ) + } catch (e: Exception) { Log.e(TAG, "设置识别参数失败", e) eventCallback?.onError("ParameterSetup", "设置识别参数失败: ${e.message}") @@ -996,7 +1253,7 @@ fun getSupportedLanguages(): List { Log.e(TAG, "暂停服务失败", e) } } - + /** * 恢复服务 */ @@ -1010,7 +1267,7 @@ fun getSupportedLanguages(): List { Log.e(TAG, "恢复服务失败", e) } } - + /** * 检查服务健康状态 */ @@ -1038,16 +1295,16 @@ fun getSupportedLanguages(): List { suspend fun synthesizeTextToAudio(text: String): ByteArray? = withContext(Dispatchers.IO) { try { Log.d(TAG, "开始纯音频合成: $text") - + val ssml = generateOptimizedSsml(text) - + // 创建一个临时的合成器,输出到内存而不是扬声器 val audioConfig = AudioConfig.fromStreamOutput(AudioOutputStream.createPullStream()) val tempSynthesizer = SpeechSynthesizer(speechConfig, audioConfig) - + try { val result = tempSynthesizer.SpeakSsmlAsync(ssml).get() - + if (result?.reason == ResultReason.SynthesizingAudioCompleted) { val audioData = result.audioData if (audioData != null && audioData.isNotEmpty()) { @@ -1060,7 +1317,7 @@ fun getSupportedLanguages(): List { } finally { tempSynthesizer.close() } - + return@withContext null } catch (e: Exception) { Log.e(TAG, "纯音频合成异常", e) @@ -1087,45 +1344,66 @@ fun getSupportedLanguages(): List { ) } + /** + * 释放整合语音服务的所有资源。优先等待识别器停止后再关闭对象与流,降低并发关闭引起的原生库崩溃 + */ fun dispose() { try { Log.d(TAG, "开始清理服务资源") - - // 停止连续识别(避免调用异步 stopContinuousTranslation 后立刻 reset 导致被取消) + + // 首先标记停止,避免新任务继续入队 runCatching { serviceState.isRecognizing.set(false) } - try { - recognizer?.stopContinuousRecognitionAsync() - ?.get(1500, java.util.concurrent.TimeUnit.MILLISECONDS) - } catch (t: Throwable) { - Log.w(TAG, "清理阶段停止识别器超时或失败", t) - } - - // 停止录音线程(打断并短等待退出) + + // 先停止音频输入,减少识别器停不下来的概率 runCatching { audioProcessor?.stopRecording() } .onFailure { Log.w(TAG, "清理阶段停止录音失败", it) } - - // 关闭录音文件(若有) runCatching { recordfile1?.closeFile(true) } .onFailure { Log.w(TAG, "清理阶段关闭录音文件失败", it) } - - // 清理翻译/合成状态位(不额外等待,避免阻塞主线程) + + // 清理翻译/合成状态位 serviceState.isTranslating.set(false) serviceState.isSynthesizing.set(false) - - // 添加协程清理(放在资源停止之后,避免取消正在进行的停止协程) + + // 添加协程清理 resetCoroutineScope() - - // 清理识别器 - recognizer?.let { + + // 清理识别器:事件等待 + 超时保护 + recognizer?.let { r -> try { - it.stopContinuousRecognitionAsync() - it.close() + val latch = java.util.concurrent.CountDownLatch(1) + r.sessionStopped.addEventListener { _, _ -> latch.countDown() } + r.canceled.addEventListener { _, _ -> latch.countDown() } + + try { + // 非阻塞停止:仅触发异步停止,不等待 Future,以降低超时概率 + r.stopContinuousRecognitionAsync() + } catch (ie: java.lang.InterruptedException) { + Thread.currentThread().interrupt() + Log.w(TAG, "等待识别器停止时被中断") + } catch (t: Throwable) { + Log.w(TAG, "清理阶段停止识别器调用失败", t) + } + + // 等待会话结束事件,最多2秒 + try { + latch.await(2000, java.util.concurrent.TimeUnit.MILLISECONDS) + } catch (ie: java.lang.InterruptedException) { + Thread.currentThread().interrupt() + } + + try { + r.close() + } catch (e: IllegalStateException) { + Log.w(TAG, "识别器仍在异步运行,忽略关闭异常") + } catch (e: Exception) { + Log.w(TAG, "清理识别器失败", e) + } } catch (e: Exception) { - Log.w(TAG, "清理识别器失败", e) + Log.w(TAG, "清理识别器过程中发生异常", e) } } recognizer = null - + // 清理合成器 synthesizer?.let { try { @@ -1135,36 +1413,42 @@ fun getSupportedLanguages(): List { } } synthesizer = null - + // 清理音频处理器 audioProcessor?.dispose() audioProcessor = null - + // 清理音频配置 - audioConfig?.close() + runCatching { audioConfig?.close() } + .onFailure { Log.w(TAG, "清理音频配置失败", it) } audioConfig = null - + // 清理推流输出 + runCatching { synthOutputStream?.close() } + .onFailure { Log.w(TAG, "清理推流输出失败", it) } + synthOutputStream = null + // 清理语音配置 - speechConfig?.close() + runCatching { speechConfig?.close() } + .onFailure { Log.w(TAG, "清理语音配置失败", it) } speechConfig = null - + // 清理翻译服务 translationService?.dispose() translationService = null - + // 清理录音文件 recordfile1?.closeFile(true) recordfile1 = null - + // 重置状态 serviceState.isInitialized.set(false) serviceState.isRecognizing.set(false) serviceState.isSynthesizing.set(false) serviceState.isTranslating.set(false) - + // 清理缓存 voiceCache.clear() - + Log.d(TAG, "服务资源清理完成") } catch (e: Exception) { Log.e(TAG, "清理服务资源失败", e) @@ -1198,9 +1482,9 @@ fun getSupportedLanguages(): List { val audioData = audioQueue.poll(100, java.util.concurrent.TimeUnit.MILLISECONDS) audioData?.let { - //Log.d(TAG, "音频处理: ${it.size}") + //Log.d(TAG, "音频处理: ${it.size}") pushAudioStream?.write(it) - recordfile1?.saveAudioDataToWav(it) + recordfile1?.saveAudioDataToWav(it) } } catch (e: InterruptedException) { Thread.currentThread().interrupt() @@ -1238,51 +1522,51 @@ fun getSupportedLanguages(): List { fun dispose() { try { Log.d(TAG, "开始释放音频处理器资源...") - + // 1. 设置停止标志 if (!isRunning.compareAndSet(true, false)) { Log.d(TAG, "音频处理器已停止,跳过释放") return } - - // 1) 安全停止处理线程 - processingThread?.let { thread -> - if (thread.isAlive) { - try { - thread.interrupt() - thread.join(500) - if (thread.isAlive) { - Log.w(TAG, "音频处理线程未在500ms内结束") - // 不强制杀死,交由系统回收,但继续清理资源 - } else { - Log.d(TAG, "音频处理线程已正常结束") + + // 1) 安全停止处理线程 + processingThread?.let { thread -> + if (thread.isAlive) { + try { + thread.interrupt() + thread.join(500) + if (thread.isAlive) { + Log.w(TAG, "音频处理线程未在500ms内结束") + // 不强制杀死,交由系统回收,但继续清理资源 + } else { + Log.d(TAG, "音频处理线程已正常结束") + } + } catch (e: InterruptedException) { + Log.w(TAG, "等待线程结束时被中断", e) + Thread.currentThread().interrupt() + } catch (e: Exception) { + Log.e(TAG, "停止处理线程时发生异常", e) } - } catch (e: InterruptedException) { - Log.w(TAG, "等待线程结束时被中断", e) - Thread.currentThread().interrupt() - } catch (e: Exception) { - Log.e(TAG, "停止处理线程时发生异常", e) } } - } - processingThread = null + processingThread = null - // 2) 关闭输入流 - try { - pushAudioStream?.close() - } catch (t: Throwable) { - Log.w(TAG, "关闭 PushAudioInputStream 时异常", t) - } finally { - pushAudioStream = null - } + // 2) 关闭输入流 + try { + pushAudioStream?.close() + } catch (t: Throwable) { + Log.w(TAG, "关闭 PushAudioInputStream 时异常", t) + } finally { + pushAudioStream = null + } - // 3) 清空队列 - audioQueue.clear() + // 3) 清空队列 + audioQueue.clear() - Log.d(TAG, "音频处理器资源释放完成") - } catch (e: Exception) { - Log.e(TAG, "释放音频处理器资源失败", e) - } + Log.d(TAG, "音频处理器资源释放完成") + } catch (e: Exception) { + Log.e(TAG, "释放音频处理器资源失败", e) + } } } } @@ -1299,438 +1583,46 @@ data class AzureConfiguration( * 翻译配置类 */ data class TranslationConfiguration( - val accessKey: String, - val secretKey: String, - val region: String = "cn-north-1" + val accessKey: String = "", + val secretKey: String = "", + val region: String = "cn-north-1", + val subscriptionKey: String = "", + val location: String = "", + val endpoint: String = "", + val maxRetryAttempts: Int? = null, + val timeoutMs: Long? = null ) { + /** + * 将通用翻译配置转换为键值映射 + * 兼容微软翻译与火山翻译实现的初始化需求 + */ fun toConfigMap(): Map { - return mapOf( - "accessKey" to accessKey, - "secretKey" to secretKey, - "region" to region - ) - } -} - - - -/** - * 火山翻译服务实现 - */ -class VolcanoTranslationServiceImpl : - IntegratedSpeechTranslationService.TranslationServiceInterface { - companion object { - private const val TAG = "VolcanoTranslationService" - private const val BASE_URL = "https://translate.volcengineapi.com" - private const val ENDPOINT = "/" - private const val SERVICE = "translate" - private const val VERSION = "2020-06-01" - private const val ACTION = "TranslateText" - private const val ALGORITHM = "HMAC-SHA256" - } - - private var isInitialized = false - private var accessKey: String = "" - private var secretKey: String = "" - private var region: String = "cn-north-1" - private var maxRetryAttempts: Int = 3 - private var timeout: Long = 10000L - - // HTTP客户端 - private val httpClient = OkHttpClient.Builder() - .connectTimeout(10, TimeUnit.SECONDS) - .readTimeout(30, TimeUnit.SECONDS) - .writeTimeout(30, TimeUnit.SECONDS) - .build() - - // 语言代码映射 - private val languageCodeMap = mapOf( - // 亚洲 - "zh-CN" to "zh", // 简体中文(中国大陆) - "zh-TW" to "zh-Hant", // 繁体中文(台湾) - "zh-HK" to "zh-Hant", // 繁体中文(香港) - "ja-JP" to "ja", // 日语(日本) - "ko-KR" to "ko", // 韩语(韩国) - "hi-IN" to "hi", // 印地语(印度) - "ta-IN" to "ta", // 泰米尔语(印度) - "te-IN" to "te", // 泰卢固语(印度) - "bn-IN" to "bn", // 孟加拉语(印度) - "pa-IN" to "pa", // 旁遮普语(印度) - "ur-PK" to "ur", // 乌尔都语(巴基斯坦) - "th-TH" to "th", // 泰语(泰国) - "vi-VN" to "vi", // 越南语(越南) - "ms-MY" to "ms", // 马来语(马来西亚) - "id-ID" to "id", // 印尼语(印尼) - "fil-PH" to "tl", // 菲律宾语(菲律宾) - "sv-SE" to "sv", // 瑞典语(瑞典) - "ro-RO" to "ro", // 罗马尼亚语(罗马尼亚) - "hu-HU" to "hu", // 匈牙利语(匈牙利) - "da-DK" to "da", // 丹麦语(丹麦) - "fi-FI" to "fi", // 芬兰语(芬兰) - "nb-NO" to "nb", // 挪威语(挪威) - "hr-HR" to "hr", // 克罗地亚语(克罗地亚) - "ca-ES" to "ca", // 加泰隆语 - "mr-IN" to "mr", // 马拉地语(印度) - "ur-IN" to "ur", // 乌尔都语(印度) - // 欧洲 - "en-US" to "en", // 英语(美国) - "en-GB" to "en", // 英语(英国) - "fr-FR" to "fr", // 法语(法国) - "fr-CA" to "fr", // 法语(加拿大) - "de-DE" to "de", // 德语(德国) - "es-ES" to "es", // 西班牙语(西班牙) - "es-MX" to "es", // 西班牙语(墨西哥) - "it-IT" to "it", // 意大利语(意大利) - "nl-NL" to "nl", // 荷兰语(荷兰) - "pl-PL" to "pl", // 波兰语(波兰) - "ru-RU" to "ru", // 俄语(俄罗斯) - "tr-TR" to "tr", // 土耳其语(土耳其) - - // 中东 - "ar-SA" to "ar", // 阿拉伯语(沙特) - "ar-EG" to "ar", // 阿拉伯语(埃及) - "he-IL" to "he", // 希伯来语(以色列) - - // 非洲 - "sw-KE" to "sw", // 斯瓦希里语(肯尼亚) - "am-ET" to "am", // 阿姆哈拉语(埃塞俄比亚) - "yo-NG" to "yo", // 约鲁巴语(尼日利亚) - - // 其他 - "pt-BR" to "pt", // 葡萄牙语(巴西) - "pt-PT" to "pt", // 葡萄牙语(葡萄牙) - "fa-IR" to "fa", // 波斯语(伊朗) - "el-GR" to "el", // 希腊语(希腊) - "cs-CZ" to "cs", // 捷克语(捷克) - "uk-UA" to "uk", // 乌克兰语(乌克兰) -) - - override suspend fun initialize(config: Map): Boolean { - return try { - accessKey = config["accessKey"] ?: "" - secretKey = config["secretKey"] ?: "" - region = config["region"] ?: "cn-north-1" - maxRetryAttempts = config["maxRetryAttempts"]?.toIntOrNull() ?: 3 - timeout = config["timeout"]?.toLongOrNull() ?: 10000L - - if (accessKey.isEmpty() || secretKey.isEmpty()) { - Log.e(TAG, "缺少必要的API密钥") - return false - } - - isInitialized = true - Log.d(TAG, "火山翻译服务初始化成功") - true - } catch (e: Exception) { - Log.e(TAG, "初始化失败", e) - false + val map = mutableMapOf() + if (subscriptionKey.isNotBlank()) { + map["subscriptionKey"] = subscriptionKey + } else if (accessKey.isNotBlank()) { + map["accessKey"] = accessKey } - } - - override suspend fun translateText( - text: String, - sourceLanguage: String, - targetLanguage: String - ): IntegratedSpeechTranslationService.TranslationResult { - if (!isInitialized) { - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = "翻译服务未初始化" - ) - } - - return withContext(Dispatchers.IO) { - performTranslationWithRetry(text, sourceLanguage, targetLanguage) - } - } - - private suspend fun performTranslationWithRetry( - text: String, - sourceLanguage: String, - targetLanguage: String - ): IntegratedSpeechTranslationService.TranslationResult { - var lastException: Exception? = null - repeat(maxRetryAttempts) { attempt -> - try { - val result = performTranslation(text, sourceLanguage, targetLanguage) - if (result.success) { - return result - } - lastException = Exception(result.error) - } catch (e: Exception) { - lastException = e - Log.w(TAG, "翻译请求失败,尝试 ${attempt + 1}/$maxRetryAttempts", e) - - // 重试延迟 - if (attempt < maxRetryAttempts - 1) { - delay(500L * (attempt + 1)) - } - } - } - - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = "翻译失败: ${lastException?.message ?: "未知错误"}" - ) - } - - private suspend fun performTranslation( - text: String, - sourceLanguage: String, - targetLanguage: String - ): IntegratedSpeechTranslationService.TranslationResult { - try { - // 获取语言代码 - val sourceCode = getLanguageCode(sourceLanguage) - val targetCode = getLanguageCode(targetLanguage) - - if (sourceCode.isEmpty() || targetCode.isEmpty()) { - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = "不支持的语言代码: $sourceLanguage -> $targetLanguage" - ) - } - - // 构建请求体 - val requestBody = JSONObject().apply { - put("SourceLanguage", sourceCode) - put("TargetLanguage", targetCode) - put("TextList", JSONArray().put(text)) - } - - // 查询参数 - val queryParams = mapOf( - "Action" to ACTION, - "Version" to VERSION, - "Region" to region, - "Service" to SERVICE - ) - - // 生成签名 - val headers = generateSignature("POST", requestBody, queryParams) - - // 构建URL - val urlBuilder = BASE_URL.toHttpUrl().newBuilder() - queryParams.forEach { (key, value) -> - urlBuilder.addQueryParameter(key, value) - } - val url = urlBuilder.build() - - // 构建请求 - val requestBodyObj = RequestBody.create( - "application/json; charset=utf-8".toMediaType(), - requestBody.toString() - ) - - val request = Request.Builder() - .url(url) - .post(requestBodyObj) - .apply { - headers.forEach { (key, value) -> - addHeader(key, value) - } - } - .build() - - // 发送请求 - val response = httpClient.newCall(request).execute() - - if (response.isSuccessful) { - val responseBody = response.body?.string() - if (responseBody != null) { - val jsonResponse = JSONObject(responseBody) - val translation = extractTranslation(jsonResponse) - - if (translation != null) { - return IntegratedSpeechTranslationService.TranslationResult( - success = true, - translatedText = translation, - confidence = 0.9f - ) - } else { - val error = extractError(jsonResponse) - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = error ?: "翻译结果为空" - ) - } - } - } - - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = "HTTP错误: ${response.code}" - ) - - } catch (e: Exception) { - Log.e(TAG, "翻译请求异常", e) - return IntegratedSpeechTranslationService.TranslationResult( - success = false, - error = "请求异常: ${e.message}" - ) + if (secretKey.isNotBlank()) { + map["secretKey"] = secretKey } - } - - private fun getLanguageCode(languageCode: String): String { - return languageCodeMap[languageCode] ?: "" - } - - private fun extractTranslation(jsonResponse: JSONObject): String? { - try { - // 1. 标准响应结构 - if (jsonResponse.has("TranslationList")) { - val translationList = jsonResponse.getJSONArray("TranslationList") - if (translationList.length() > 0) { - val translationItem = translationList.getJSONObject(0) - if (translationItem.has("Translation")) { - return translationItem.getString("Translation") - } - } - } - - // 2. 其他可能的响应结构 - if (jsonResponse.has("Translation")) { - return jsonResponse.getString("Translation") - } - if (jsonResponse.has("Result")) { - val result = jsonResponse.getJSONObject("Result") - if (result.has("Translation")) { - return result.getString("Translation") - } - } - - if (jsonResponse.has("Data")) { - val data = jsonResponse.getJSONObject("Data") - if (data.has("Translation")) { - return data.getString("Translation") - } - if (data.has("TranslationList")) { - val translationList = data.getJSONArray("TranslationList") - if (translationList.length() > 0) { - val translationItem = translationList.getJSONObject(0) - if (translationItem.has("Translation")) { - return translationItem.getString("Translation") - } - } - } - } - } catch (e: Exception) { - Log.e(TAG, "提取翻译结果失败", e) + if (region.isNotBlank()) { + map["region"] = region } - - return null - } - - private fun extractError(jsonResponse: JSONObject): String? { - try { - if (jsonResponse.has("ResponseMetadata")) { - val metadata = jsonResponse.getJSONObject("ResponseMetadata") - if (metadata.has("Error")) { - val error = metadata.getJSONObject("Error") - return error.optString("Message", error.optString("Code", "未知错误")) - } - } - - if (jsonResponse.has("Error")) { - val error = jsonResponse.getJSONObject("Error") - return error.optString("Message", error.optString("Code", "未知错误")) - } - } catch (e: Exception) { - Log.e(TAG, "提取错误信息失败", e) + if (location.isNotBlank()) { + map["location"] = location } - - return null - } - - private fun generateSignature( - method: String, - requestBody: JSONObject, - queryParams: Map - ): Map { - try { - // 1. 准备时间相关参数 - val now = Date() - val dateFormat = SimpleDateFormat("yyyyMMdd", Locale.US).apply { - timeZone = TimeZone.getTimeZone("UTC") - } - val timestampFormat = SimpleDateFormat("yyyyMMdd'T'HHmmss'Z'", Locale.US).apply { - timeZone = TimeZone.getTimeZone("UTC") - } - - val date = dateFormat.format(now) - val timestamp = timestampFormat.format(now) - - // 2. 构建规范查询字符串 - val sortedParams = queryParams.toSortedMap() - val canonicalQueryString = sortedParams.map { (key, value) -> - "${URLEncoder.encode(key, "UTF-8")}=${URLEncoder.encode(value, "UTF-8")}" - }.joinToString("&") - - // 3. 创建规范请求 - val contentType = "application/json" - val payloadHash = sha256(requestBody.toString()) - val host = "translate.volcengineapi.com" - - val canonicalHeaders = "host:$host\nx-date:$timestamp\n" - val signedHeaders = "host;x-date" - - val canonicalRequest = - "$method\n$ENDPOINT\n$canonicalQueryString\n$canonicalHeaders\n$signedHeaders\n$payloadHash" - - // 4. 创建待签字符串 - val credentialScope = "$date/$region/$SERVICE/request" - val stringToSign = - "$ALGORITHM\n$timestamp\n$credentialScope\n${sha256(canonicalRequest)}" - - // 5. 计算签名 - val kSecret = secretKey.toByteArray(Charsets.UTF_8) - val kDate = hmacSha256(kSecret, date) - val kRegion = hmacSha256(kDate, region) - val kService = hmacSha256(kRegion, SERVICE) - val kSigning = hmacSha256(kService, "request") - val signature = - hmacSha256(kSigning, stringToSign).joinToString("") { "%02x".format(it) } - - // 6. 构建授权头 - val authorization = - "$ALGORITHM Credential=$accessKey/$credentialScope, SignedHeaders=$signedHeaders, Signature=$signature" - - return mapOf( - "Content-Type" to contentType, - "X-Date" to timestamp, - "Authorization" to authorization, - "Host" to host - ) - - } catch (e: Exception) { - Log.e(TAG, "生成签名失败", e) - throw e + if (endpoint.isNotBlank()) { + map["endpoint"] = endpoint } - } + maxRetryAttempts?.let { map["maxRetryAttempts"] = it.toString() } + timeoutMs?.let { map["timeout"] = it.toString() } - private fun sha256(input: String): String { - val digest = MessageDigest.getInstance("SHA-256") - val hash = digest.digest(input.toByteArray(Charsets.UTF_8)) - return hash.joinToString("") { "%02x".format(it) } + return map } +} - private fun hmacSha256(key: ByteArray, data: String): ByteArray { - val mac = Mac.getInstance("HmacSHA256") - val secretKeySpec = SecretKeySpec(key, "HmacSHA256") - mac.init(secretKeySpec) - return mac.doFinal(data.toByteArray(Charsets.UTF_8)) - } - override fun dispose() { - isInitialized = false - // 清理资源 - try { - httpClient.dispatcher.executorService.shutdown() - httpClient.connectionPool.evictAll() - } catch (e: Exception) { - Log.e(TAG, "清理资源失败", e) - } - } -} \ No newline at end of file +// VolcanoTranslationServiceImpl 已迁移至独立文件 VolcanoTranslationServiceImpl.kt diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt index 7685d59f6..72e5f43d9 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -3,9 +3,16 @@ package com.yunqiinnovation.azure_speech import android.content.Context import android.os.Handler import android.os.Looper +import android.app.Activity +import android.content.Intent +import android.util.Log +import com.yunqiinnovation.azure_speech.tools.ScreenCaptureManager +import com.yunqiinnovation.azure_speech.tools.ScreenCaptureForegroundService +import com.yunqiinnovation.azure_speech.tools.isCapturing +import io.flutter.embedding.engine.plugins.activity.ActivityAware +import io.flutter.embedding.engine.plugins.activity.ActivityPluginBinding import androidx.annotation.NonNull import com.yunqiinnovation.azure_speech.utils.FileLogger -import com.yunqiinnovation.azure_speech.tools.AudioRecordingForegroundService import com.yunqiinnovation.azure_speech.tools.SimpleAudioReceiver import com.yunqiinnovation.azure_speech.tools.SimpleAudioPlayer import io.flutter.embedding.engine.plugins.FlutterPlugin @@ -14,6 +21,7 @@ import io.flutter.plugin.common.MethodChannel import io.flutter.plugin.common.MethodChannel.MethodCallHandler import io.flutter.plugin.common.MethodChannel.Result import io.flutter.plugin.common.EventChannel +import io.flutter.plugin.common.PluginRegistry import com.deep_voice.speech.tts.TtsEvent import com.deep_voice.speech.tts.TtsEventListener import com.deep_voice.speech.tts.TtsEventType @@ -21,13 +29,30 @@ import com.yunqiinnovation.ble_service.BleService import com.deep_voice.speech.tts.AudioOutputDevice import com.deep_voice.speech.tts.AudioDataListener import kotlinx.coroutines.* +import kotlinx.coroutines.CoroutineScope +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.SupervisorJob +import kotlinx.coroutines.launch +import kotlinx.coroutines.sync.Mutex +import kotlinx.coroutines.sync.withLock import com.yunqiinnovation.azure_speech.tools.RecordFile +import com.example.astclient.DoubaoE2ETranslateHelper +import com.example.astclient.Config + /** AzureSpeechPlugin */ -class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { +class AzureSpeechPlugin : BleService.Callback, FlutterPlugin, ActivityAware, + PluginRegistry.ActivityResultListener { private val tag = "AzureSpeechPlugin" private lateinit var context: Context private val mainHandler = Handler(Looper.getMainLooper()) + // 用于 BLE 写入的单独 IO 协程作用域 + private val bleWriteScope = CoroutineScope(Dispatchers.IO + SupervisorJob()) + + // 左右声道各自的互斥锁,避免并发写入争用 + private val bleLeftMutex = Mutex() + private val bleRightMutex = Mutex() + // ASR相关 private lateinit var asrChannel: MethodChannel private lateinit var asrEventChannel: EventChannel @@ -44,21 +69,97 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { private lateinit var astChannel: MethodChannel private lateinit var astEventChannel: EventChannel private var astEventSink: EventChannel.EventSink? = null - private lateinit var azureAstHelper: IntegratedSpeechTranslationService + + // 翻译(Volcano)相关 + private lateinit var translationChannel: MethodChannel + private lateinit var translationEventChannel: EventChannel + private var translationEventSink: EventChannel.EventSink? = null + private var volcanoTranslationService: VolcanoTranslationServiceImpl? = null + private var microsoftTranslationService: MicrosoftTranslationServiceImpl? = null + private var microsoftTranslationAndTtsService: MicrosoftTranslationAndTtsService? = null + private var translationRestProvider: String = "volcano" + + // 双向 AST 服务:A 和 B + private var azureAstHelperA: IntegratedSpeechTranslationService? = null + private var azureAstHelperB: IntegratedSpeechTranslationService? = null + + private var iflytekAstHelperA: IflytekIntegratedSpeechService? = null + private var iflytekAstHelperB: IflytekIntegratedSpeechService? = null + + private var doubaoAstHelperA: DoubaoE2ETranslateHelper? = null + private var doubaoAstHelperB: DoubaoE2ETranslateHelper? = null + private var bailianAstHelperA: AliyunBailianE2EHelper? = null + private var bailianAstHelperB: AliyunBailianE2EHelper? = null + private var astWsConfig: Config? = null + + private var currentAstProvider: String = "azure" + + private var iflytekAStarted: Boolean = false + private var iflytekBStarted: Boolean = false // 音频数据相关 + private var lowVolumeThreshold: Short = 100 private lateinit var audioDataChannel: MethodChannel private lateinit var audioDataEventChannel: EventChannel private var audioDataEventSink: EventChannel.EventSink? = null private var recordfile: RecordFile? = null private var isRecord = false + // 是否已添加TTS事件监听器 private var isTtsListenerAdded = false + // 是否已注册BLE回调 private var isRegisteredToBle = false + + + private var activityBinding: ActivityPluginBinding? = null + private var activity: Activity? = null + private var screenCaptureManager: ScreenCaptureManager? = null + private var captureWidth: Int? = null + private var captureHeight: Int? = null + private var captureDpi: Int? = null + private var captureQuality: Int = 80 + private val REQUEST_SCREEN_CAPTURE = 9001 + + /** + * 推送音频到当前选择的 AST 服务 A + * + * 参数:`data` 为单声道 PCM 16kHz 16bit 音频数据 + * 返回:无 + */ + private fun pushAstAudioToA(data: ByteArray) { + if (currentAstProvider == "iflytek") { + iflytekAstHelperA?.pushAudioData(data) + } else if (currentAstProvider == "azure") { + azureAstHelperA?.pushAudioData(data) + } else if (currentAstProvider == "volcano") { + doubaoAstHelperA?.pushAudioData(data) + } else if (currentAstProvider == "alibaba") { + bailianAstHelperA?.pushAudioData(data) + } + } + + /** + * 推送音频到当前选择的 AST 服务 B + * + * 参数:`data` 为单声道 PCM 16kHz 16bit 音频数据 + * 返回:无 + */ + private fun pushAstAudioToB(data: ByteArray) { + if (currentAstProvider == "iflytek" ) { + iflytekAstHelperB?.pushAudioData(data) + } else if (currentAstProvider == "azure") { + azureAstHelperB?.pushAudioData(data) + } else if (currentAstProvider == "volcano") { + doubaoAstHelperB?.pushAudioData(data) + } else if (currentAstProvider == "alibaba") { + bailianAstHelperB?.pushAudioData(data) + } + } + // ASR 事件发送方法 private fun sendAsrEvent(event: Map) { if (asrEventSink == null) { @@ -108,73 +209,34 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } - /** - * 从WAV格式音频数据中提取纯PCM数据 - * - * @param wavData WAV格式的音频数据 - * @return 纯PCM音频数据 - */ - private fun extractPcmFromWav(wavData: ByteArray): ByteArray { - try { - // WAV文件头最小长度为44字节 - if (wavData.size < 44) { - FileLogger.e(tag, "WAV数据长度不足,无法解析文件头") - return ByteArray(0) - } - - // 检查RIFF标识符 - val riffHeader = String(wavData, 0, 4, Charsets.US_ASCII) - if (riffHeader != "RIFF") { - FileLogger.e(tag, "不是有效的RIFF WAV文件,标识符: $riffHeader") - return ByteArray(0) - } - - // 检查WAVE标识符 - val waveHeader = String(wavData, 8, 4, Charsets.US_ASCII) - if (waveHeader != "WAVE") { - FileLogger.e(tag, "不是有效的WAVE文件,标识符: $waveHeader") - return ByteArray(0) - } - - // 查找data chunk - var offset = 12 // 跳过RIFF头部 - while (offset < wavData.size - 8) { - val chunkId = String(wavData, offset, 4, Charsets.US_ASCII) - val chunkSize = (wavData[offset + 4].toInt() and 0xFF) or - ((wavData[offset + 5].toInt() and 0xFF) shl 8) or - ((wavData[offset + 6].toInt() and 0xFF) shl 16) or - ((wavData[offset + 7].toInt() and 0xFF) shl 24) - - if (chunkId == "data") { - // 找到data chunk,提取PCM数据 - val dataOffset = offset + 8 - val dataSize = minOf(chunkSize, wavData.size - dataOffset) - - if (dataSize > 0) { - val pcmData = ByteArray(dataSize) - System.arraycopy(wavData, dataOffset, pcmData, 0, dataSize) - FileLogger.d(tag, "成功提取PCM数据: ${pcmData.size}字节") - return pcmData - } - break - } - - // 移动到下一个chunk - offset += 8 + chunkSize - } - - FileLogger.e(tag, "未找到data chunk") - return ByteArray(0) - - } catch (e: Exception) { - FileLogger.e(tag, "解析WAV数据异常: ${e.message}") - return ByteArray(0) - } - } + // 翻译事件发送方法 + /** + * 发送翻译事件到 Flutter 层 + * 参数通过 `Map` 传递,包含: + * - type: 事件类型,如 `translationStarted`、`translated`、`translationFailed` + * - 其他上下文字段:`text`、`originalText`、`translatedText`、`sourceLanguage`、`targetLanguage`、`error` 等 + */ + private fun sendTranslationEvent(event: Map) { + if (translationEventSink == null) { + FileLogger.w(tag, "无法发送翻译事件:事件通道未准备好") + return + } + + mainHandler.post { + try { + translationEventSink?.success(event) + } catch (e: Exception) { + FileLogger.e(tag, "发送翻译事件失败: ${e.message}") + } + } + } + + + // 音频数据 事件发送方法 private fun sendAudioDataEvent(event: Map) { if (audioDataEventSink == null) { - FileLogger.w(tag, "无法发送音频数据事件:事件通道未准备好") + // FileLogger.w(tag, "无法发送音频数据事件:事件通道未准备好") return } @@ -186,6 +248,7 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } } + override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { context = flutterPluginBinding.applicationContext @@ -199,6 +262,12 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { // 初始化AST通道 astChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/ast") astChannel.setMethodCallHandler(AsTMethodHandler()) + + // 初始化翻译通道 + translationChannel = + MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/translation") + translationChannel.setMethodCallHandler(TranslationMethodHandler()) + // 初始化ASR事件通道 asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events") @@ -238,6 +307,18 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { astEventSink = null } }) + // 初始化翻译事件通道 + translationEventChannel = + EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/translation_events") + translationEventChannel.setStreamHandler(object : EventChannel.StreamHandler { + override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { + translationEventSink = events + } + + override fun onCancel(arguments: Any?) { + translationEventSink = null + } + }) // 初始化音频数据事件通道 audioDataEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/audio_data_events") @@ -253,10 +334,40 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { // 初始化Azure语音服务 azureTtsHelper = AzureTtsHelper(context) azureAsrHelper = AzureAsrHelper(context) - azureAsrHelper.nativeLogCallback = { level, message -> - sendAsrEvent(mapOf("type" to "nativeLog", "level" to level, "message" to message)) + + Log.d(tag, "Assigning audioDataListener") + // 设置屏幕捕获服务的音频回调 + ScreenCaptureForegroundService.audioDataListener = { data -> + Log.d(tag, "收到音频数据,大小: ${data.size}") + if (isRecord) { + recordfile?.setAudioConfig(16000, 1) + recordfile?.saveAudioDataToWav(data) + // 构建音频数据事件映射 + val audioEvent = mapOf( + "type" to "audioData", + "data" to data, + "timestamp" to System.currentTimeMillis(), + "size" to data.size + ) + // 发送音频数据事件到 Flutter 层 + sendAudioDataEvent(audioEvent) + } + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { + azureAsrHelper.audioStream?.saveAudioDataTo(data) + } } - azureAstHelper = IntegratedSpeechTranslationService(context) + + azureAstHelperA = IntegratedSpeechTranslationService(context) + azureAstHelperB = IntegratedSpeechTranslationService(context) + iflytekAstHelperA = IflytekIntegratedSpeechService(context) + iflytekAstHelperB = IflytekIntegratedSpeechService(context) + doubaoAstHelperA = DoubaoE2ETranslateHelper(context) + doubaoAstHelperB = DoubaoE2ETranslateHelper(context) + bailianAstHelperA = AliyunBailianE2EHelper(context) + bailianAstHelperB = AliyunBailianE2EHelper(context) + volcanoTranslationService = VolcanoTranslationServiceImpl() + microsoftTranslationService = MicrosoftTranslationServiceImpl() + // 2. 初始化BleService并注册回调 if (BleService.initialize(context)) { @@ -273,18 +384,62 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } + // 3. 实现ActivityPluginBinding.ActivityResultListener + override fun onAttachedToActivity(binding: ActivityPluginBinding) { + activityBinding = binding + activity = binding.activity + binding.addActivityResultListener(this) + } + + // 4. 实现ActivityPluginBinding.ActivityResultListener + override fun onDetachedFromActivityForConfigChanges() { + activityBinding?.removeActivityResultListener(this) + activityBinding = null + activity = null + } + + // 5. 实现ActivityPluginBinding.ActivityResultListener + override fun onReattachedToActivityForConfigChanges(binding: ActivityPluginBinding) { + onAttachedToActivity(binding) + } + + // 6. 实现ActivityPluginBinding.ActivityResultListener + override fun onDetachedFromActivity() { + activityBinding?.removeActivityResultListener(this) + activityBinding = null + activity = null + } + + override fun onActivityResult(requestCode: Int, resultCode: Int, data: Intent?): Boolean { + if (requestCode == REQUEST_SCREEN_CAPTURE && resultCode == Activity.RESULT_OK && data != null) { + // 使用前台服务启动屏幕捕获,满足 Android 14 的安全要求 + val ctx = activity ?: context + ScreenCaptureForegroundService.startService( + ctx, + resultCode, + data, + captureWidth, + captureHeight, + captureDpi, + captureQuality + ) + return true + } + return false + } + //设置识别器 private fun recognizeCallback(): Boolean { try { azureAsrHelper.setupEventListeners(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(sessionid:String,text: String, detectedLanguage: String) { + override fun onResult(sessionid: String, text: String, detectedLanguage: String) { sendAsrEvent( mapOf( "type" to "result", "sessionid" to sessionid, - "provider" to azureAsrHelper.getAsrProvider(), + "text" to text, "detectedLanguage" to detectedLanguage ) @@ -293,7 +448,7 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } override fun onRecognizing( - sessionid:String, + sessionid: String, recognizing: String, detectedLanguage: String ) { @@ -302,29 +457,29 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { mapOf( "type" to "recognizing", "sessionid" to sessionid, - "provider" to azureAsrHelper.getAsrProvider(), + "text" to recognizing, "detectedLanguage" to detectedLanguage ) ) } - override fun onSessionStarted(sessionid:String) { - sendAsrEvent(mapOf("type" to "sessionStarted","sessionid" to sessionid, "provider" to azureAsrHelper.getAsrProvider())) + override fun onSessionStarted(sessionid: String) { + //sendAsrEvent(mapOf("type" to "sessionStarted","sessionid" to sessionid, "provider" to azureAsrHelper.getAsrProvider())) } - override fun onSessionStopped(sessionid:String) { - sendAsrEvent(mapOf("type" to "sessionStopped","sessionid" to sessionid, "provider" to azureAsrHelper.getAsrProvider())) + override fun onSessionStopped(sessionid: String) { + //sendAsrEvent(mapOf("type" to "sessionStopped","sessionid" to sessionid, "provider" to azureAsrHelper.getAsrProvider())) } - override fun onCanceled(sessionid:String,reason: String, errorDetails: String) { + override fun onCanceled(sessionid: String, reason: String, errorDetails: String) { sendAsrEvent( mapOf( "type" to "canceled", "sessionid" to sessionid, - "provider" to azureAsrHelper.getAsrProvider(), + "reason" to reason, "errorDetails" to errorDetails ) @@ -333,12 +488,12 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } - override fun onError(sessionid:String,code: Int, error: String) { + override fun onError(sessionid: String, code: Int, error: String) { sendAsrEvent( mapOf( "type" to "error", "sessionid" to sessionid, - "provider" to azureAsrHelper.getAsrProvider(), + "code" to code, "message" to error ) @@ -407,6 +562,32 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } }) + // // 添加音频数据监听器 + // azureTtsHelper.addAudioDataListener(object : AudioDataListener { + // /** + // * 接收TTS合成的音频数据并回调到BleService + // * 处理WAV格式数据,去除文件头,只发送纯PCM数据 + // * + // * @param data TTS合成的音频数据(WAV格式) + // */ + // override fun onAudioData(data: ByteArray) { + // FileLogger.d(tag, "接收到TTS音频数据,大小=${data.size}字节") + + // // Azure TTS返回的是RIFF WAV格式,需要去除文件头 + // val pcmData = extractPcmFromWav(data) + // if (pcmData.isNotEmpty()) { + // FileLogger.d(tag, "提取PCM数据,大小=${pcmData.size}字节") + // // 将纯PCM音频数据回调到BleService + // BleService.writeExternalLeftAudioData(pcmData) + // } else { + // FileLogger.w(tag, "无法从WAV数据中提取PCM数据") + // } + // } + + + // }) + + // 标记监听器已添加,防止重复添加 isTtsListenerAdded = true } } @@ -426,13 +607,6 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { val xunfeiAccessKeySecret = call.argument("xunfeiAccessKeySecret") ?: "" try { - val isXunfeiConfigured = xunfeiAppId.isNotBlank() && - xunfeiAccessKeyId.isNotBlank() && - xunfeiAccessKeySecret.isNotBlank() - FileLogger.i( - tag, - "lxm---ASR initialize args: supportedLanguages=${supportedLanguages.joinToString(",")}, useExternalAudio=$useExternalAudio, region=$region, xunfeiConfigured=$isXunfeiConfigured" - ) FileLogger.d(tag, "选择音频源类型: ${useExternalAudio}") // // 选择音频源类型 val audioSourceType = if (useExternalAudio) { @@ -446,9 +620,9 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { region = region, supportedLanguages = supportedLanguages.toTypedArray(), audioSourceType = audioSourceType, - xunfeiAppId = xunfeiAppId, - xunfeiAccessKeyId = xunfeiAccessKeyId, - xunfeiAccessKeySecret = xunfeiAccessKeySecret, + //xunfeiAppId = xunfeiAppId, + //xunfeiAccessKeyId = xunfeiAccessKeyId, + //xunfeiAccessKeySecret = xunfeiAccessKeySecret, ) result.success(success) @@ -470,7 +644,8 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { "startContinuousRecognition" -> { val useExternalAudio = call.argument("audioSourceType") ?: false - val removeFirstPunctuation = call.argument("isRemoveFirstPunctuation") ?: true + val removeFirstPunctuation = + call.argument("isRemoveFirstPunctuation") ?: true // 确保事件通道已准备好 if (asrEventSink == null) { result.error( @@ -489,8 +664,8 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } else { AzureAsrHelper.AudioSourceType.MICROPHONE } - - val success = azureAsrHelper.startContinuousRecognition(audioSourceType, isRemoveFirstPunctuation = removeFirstPunctuation) + val success = azureAsrHelper.startContinuousRecognition(audioSourceType) + // val success = azureAsrHelper.startContinuousRecognition(audioSourceType, isRemoveFirstPunctuation = removeFirstPunctuation) result.success(true) } catch (e: Exception) { result.error("START_RECOGNITION_ERROR", e.message, null) @@ -539,10 +714,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { val newName = call.argument("newName") ?: "" try { FileLogger.d(tag, "音频文件名称为: ${filePath}") // - if(azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL){ + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { isRecord = false recordfile?.renameFile(filePath, newName) - }else{ + } else { azureAsrHelper.renameFile(filePath, newName) } result.success(true) @@ -556,10 +731,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { val destPath = call.argument("destPath") ?: "" try { FileLogger.d(tag, "音频文件名称为: ${sourcePath}") // - if(azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL){ + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { isRecord = false recordfile?.moveFile(sourcePath, destPath) - }else{ + } else { azureAsrHelper.moveFile(sourcePath, destPath) } result.success(true) @@ -568,55 +743,98 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } + "requestScreenCapture" -> { + val width = call.argument("width") + val height = call.argument("height") + val dpi = call.argument("dpi") + val quality = call.argument("quality") ?: 80 + + captureWidth = width + captureHeight = height + captureDpi = dpi + captureQuality = quality + + val act = activity + if (act == null) { + result.error("NO_ACTIVITY", "Activity 未附着,无法请求屏幕捕获权限", null) + return + } + try { + val pm = + act.getSystemService(Context.MEDIA_PROJECTION_SERVICE) as android.media.projection.MediaProjectionManager + val intent = pm.createScreenCaptureIntent() + + act.startActivityForResult(intent, REQUEST_SCREEN_CAPTURE) + result.success(true) + } catch (e: Exception) { + result.error("REQUEST_CAPTURE_ERROR", e.message, null) + } + } + + "stopScreenCapture" -> { + try { + ScreenCaptureForegroundService.stopService(activity ?: context) + result.success(true) + } catch (e: Exception) { + result.error("STOP_CAPTURE_ERROR", e.message, null) + } + } + + "isScreenCapturing" -> { + result.success(isCapturing) + } + "enableRecord" -> { val filePath = call.argument("filePath") ?: "" // 是否接受音频数据 val acceptAudioData = call.argument("acceptAudioData") ?: false - val useExternalAudio = call.argument("audioSourceType") ?: false - - + val typeInt = call.argument("audioSourceType") ?: 0 + val useExternalAudio = when (typeInt) { + 0 -> "MICROPHONE" + 1 -> "EXTERNAL" + 2 -> "EXTERNAL" + else -> "EXTERNAL" + } + try { - FileLogger.d(tag, "选择音频源类型: ${useExternalAudio}") // + FileLogger.d(tag, "选择音频源类型: ${useExternalAudio}") // // 选择音频源类型 - val audioSourceType = if (useExternalAudio) { - if(recordfile == null){ - recordfile = RecordFile() + val audioSourceType = if (useExternalAudio == "EXTERNAL") { + if (recordfile == null) { + recordfile = RecordFile() } recordfile?.closeFile(true) recordfile?.creatingFiles(filePath) - isRecord = true + isRecord = true AzureAsrHelper.AudioSourceType.EXTERNAL } else { AzureAsrHelper.AudioSourceType.MICROPHONE - // 启用录音,并根据 acceptAudioData 参数决定是否设置音频数据回调 + // 启用录音,并根据 acceptAudioData 参数决定是否设置音频数据回调 } FileLogger.d(tag, "音频文件名称为: ${filePath}") - try { - AudioRecordingForegroundService.startService(context) - } catch (e: Exception) { - FileLogger.w(tag, "启动录音前台服务失败: ${e.message}") - } - - azureAsrHelper.enableRecord(audioSourceType, filePath, if(acceptAudioData) { - // 创建音频数据回调,将音频数据发送到 Flutter 层 - object : SimpleAudioReceiver.AudioDataCallback { - override fun onAudio(audioData: ByteArray) { - // 构建音频数据事件映射 - val audioEvent = mapOf( - "type" to "audioData", - "data" to audioData, - "timestamp" to System.currentTimeMillis(), - "size" to audioData.size - ) - // 发送音频数据事件到 Flutter 层 - sendAudioDataEvent(audioEvent) + azureAsrHelper.enableRecord( + audioSourceType, + filePath, + if (acceptAudioData) { + // 创建音频数据回调,将音频数据发送到 Flutter 层 + object : SimpleAudioReceiver.AudioDataCallback { + override fun onAudio(audioData: ByteArray) { + // 构建音频数据事件映射 + val audioEvent = mapOf( + "type" to "audioData", + "data" to audioData, + "timestamp" to System.currentTimeMillis(), + "size" to audioData.size + ) + // 发送音频数据事件到 Flutter 层 + sendAudioDataEvent(audioEvent) + } } - } - } else { - null - }) + } else { + null + }) result.success(true) } catch (e: Exception) { result.error("ENABLERECORD_ERROR", e.message, null) @@ -624,44 +842,35 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } "pauseRecord" -> { - if(azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL){ + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { isRecord = false - }else{ + } else { + azureAsrHelper.pauseRecord() + + azureAsrHelper.pauseRecord() - - - azureAsrHelper.pauseRecord() } result.success(true) } + "resumeRecord" -> { - if(azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL){ - isRecord = true - }else{ + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { + isRecord = true + } else { azureAsrHelper.resumeRecord() - } + } result.success(true) } "stopRecord" -> { val isSave = call.argument("isSave") ?: false try { - if(azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL){ + if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { isRecord = false recordfile?.closeFile(isSave) - } - else - { + } else { azureAsrHelper.stopRecord(isSave) } - - if (!azureAsrHelper.isContinuousRecognitionActive()) { - try { - AudioRecordingForegroundService.stopService(context) - } catch (e: Exception) { - FileLogger.w(tag, "停止录音前台服务失败: ${e.message}") - } - } result.success(true) } catch (e: Exception) { result.error("ENABLERECORD_ERROR", e.message, null) @@ -669,10 +878,10 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } "setAudioConfig" -> { - val sampleRate = call.argument("sampleRate") ?: 16000 - val channels = call.argument("channels") ?: 1 + val sampleRate = call.argument("sampleRate")?.toInt() ?: 16000 + val channels = call.argument("channels")?.toInt() ?: 1 try { - // azureAsrHelper.audioStream?.recordfile?.setAudioConfig(sampleRate, channels) + // azureAsrHelper.audioStream?.recordfile?.setAudioConfig(sampleRate, channels) result.success(true) } catch (e: Exception) { result.error("SET_AUDIO_CONFIG_ERROR", e.message, null) @@ -686,6 +895,7 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } + // TTS方法处理器 inner class TtsMethodHandler : MethodCallHandler { override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { @@ -693,7 +903,6 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { "initialize" -> { val subscriptionKey = call.argument("subscriptionKey") ?: "" val region = call.argument("region") ?: "" - val language = call.argument("language") ?: "zh-CN" val useCustomAudioOutput = call.argument("useCustomAudioOutput") ?: false @@ -703,7 +912,7 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { ttsAppId = "", // Azure TTS不需要appId ttsAppToken = subscriptionKey, ttsResource = region, - language = language + language = "zh-CN" ) // 如果初始化成功且需要使用自定义音频输出,则设置自定义音频播放器 @@ -729,13 +938,6 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } } - "setSpeechParams" -> { - val rate = call.argument("rate") ?: 0 - val pitch = call.argument("pitch") ?: 0 - val volume = call.argument("volume") ?: 100 - result.success(azureTtsHelper.setSpeechParams(rate = rate, pitch = pitch, volume = volume)) - } - "setVoice" -> { val voiceName = call.argument("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) @@ -783,7 +985,7 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } "setAudioOutputDevice" -> { - val type = call.argument("type") + val type = call.argument("type")?.toInt() var success = false if (type == 0) { // 默认(如果有耳机选耳机,否则使用系统扬声器) @@ -809,103 +1011,133 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { // ASt方法处理器 inner class AsTMethodHandler : MethodCallHandler { + + /** + * 流式读取 WAV 文件并以音频块推送到两个 AST 服务 + * + * @param file 本地 WAV 文件 + */ private suspend fun streamAudioFile(file: java.io.File) { - try { - val inputStream = file.inputStream() - val buffer = ByteArray(1024) // 每次读取1KB - - // WAV文件参数(假设16kHz, 16bit, 单声道) - val sampleRate = 16000 // 采样率 - val bytesPerSample = 2 // 16bit = 2字节 - val channels = 1 // 单声道 - - // 计算每秒需要的字节数 - val bytesPerSecond = sampleRate * bytesPerSample * channels - - // 计算每个缓冲区对应的播放时间(毫秒) - val bufferDurationMs = (buffer.size * 1000L) / bytesPerSecond - - FileLogger.d("TAG", "开始流式读取音频文件,缓冲区大小: ${buffer.size}, 播放间隔: ${bufferDurationMs}ms") - - var bytesRead: Int - while (inputStream.read(buffer).also { bytesRead = it } != -1) { - // 只发送实际读取的字节数 - val audioChunk = if (bytesRead < buffer.size) { - buffer.copyOf(bytesRead) - } else { - buffer - } - - // 推送音频数据块 - withContext(Dispatchers.Main) { - azureAstHelper.pushAudioData(audioChunk) - FileLogger.d("TAG", "推送音频数据块: ${audioChunk.size} 字节") + try { + val bytes = file.readBytes() + val headerSize = 44 + if (bytes.size <= headerSize) { + FileLogger.e("TAG", "WAV 文件长度不足,无法去除头部: ${bytes.size} 字节") + return + } + + val pcmData = bytes.copyOfRange(headerSize, bytes.size) + + withContext(Dispatchers.Main) { + // TODO: 双端翻译时启用左右声道分离写入 + // BleService.writeExternalRightAudioData(pcmData) + FileLogger.d("TAG", "一次性推送音频数据(暂不处理): ${pcmData.size} 字节") + } + + } catch (e: Exception) { + FileLogger.e("TAG", "读取音频文件失败: ${e.message}") } } - - inputStream.close() - FileLogger.d("TAG", "音频文件流式读取完成") - - } catch (e: Exception) { - FileLogger.e("TAG", "流式读取音频文件失败: ${e.message}") - } -} + + + + + + + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { when (call.method) { - // "enableRecord" -> { - // val filePath = call.argument("filePath") ?: "" - // try { - // FileLogger.d(tag, "音频文件名称为: ${filePath}") // - - // //azureAstHelper.enableRecord(filePath) - // result.success(true) - // } catch (e: Exception) { - // result.error("ENABLERECORD_ERROR", e.message, null) - // } - // } - "startContinuousTranslation" -> { + "startContinuousTranslation" -> { try { - FileLogger.d(tag, "开启翻译") - azureAstHelper.startContinuousTranslation() + FileLogger.d(tag, "开启双向翻译") + if (currentAstProvider == "iflytek" ) { + iflytekAstHelperA?.startContinuousTranslation() + iflytekAstHelperB?.startContinuousTranslation() + } else if (currentAstProvider == "azure" ) { + azureAstHelperA?.startContinuousTranslation() + azureAstHelperB?.startContinuousTranslation() + } else if (currentAstProvider == "volcano" ) { + doubaoAstHelperA?.startContinuousTranslation() + doubaoAstHelperB?.startContinuousTranslation() + } else if (currentAstProvider == "alibaba" ) { + bailianAstHelperA?.startContinuousConversation() + bailianAstHelperB?.startContinuousConversation() + } + result.success(true) } catch (e: Exception) { - result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) + result.error("START_CONTINUOUS_TRANSLATION_ERROR", e.message, null) } } + "stopContinuousTranslation" -> { try { - FileLogger.d(tag, "停止翻译") - azureAstHelper.stopContinuousTranslation() + FileLogger.d(tag, "停止双向翻译") + if (currentAstProvider == "iflytek" ) { + iflytekAstHelperA?.stopContinuousTranslation() + iflytekAstHelperB?.stopContinuousTranslation() + } else if (currentAstProvider == "azure" ) { + azureAstHelperA?.stopContinuousTranslation() + azureAstHelperB?.stopContinuousTranslation() + } else if (currentAstProvider == "volcano" ) { + doubaoAstHelperA?.stopContinuousTranslation() + doubaoAstHelperB?.stopContinuousTranslation() + } else if (currentAstProvider == "alibaba" ) { + bailianAstHelperA?.stopContinuousConversation() + bailianAstHelperB?.stopContinuousConversation() + } + + + result.success(true) + } catch (e: Exception) { result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) } } - "dispose" -> { + + "dispose" -> { try { - FileLogger.d(tag, "释放AST资源") - azureAstHelper.dispose() - result.success(true) + FileLogger.d(tag, "释放双向 AST 资源(异步)") + GlobalScope.launch(Dispatchers.IO) { + try { + if (currentAstProvider == "iflytek" ) { + iflytekAstHelperA?.stopContinuousTranslation() + iflytekAstHelperB?.stopContinuousTranslation() + } else if (currentAstProvider == "azure" ) { + azureAstHelperA?.dispose() + azureAstHelperB?.dispose() + } else if (currentAstProvider == "volcano" ) { + doubaoAstHelperA?.dispose() + doubaoAstHelperB?.dispose() + } else if (currentAstProvider == "alibaba" ) { + bailianAstHelperA?.dispose() + bailianAstHelperB?.dispose() + } + withContext(Dispatchers.Main) { + result.success(true) + } + } catch (e: Exception) { + withContext(Dispatchers.Main) { + result.error("AST_DISPOSE_ERROR", e.message, null) + } + } + } } catch (e: Exception) { - result.error("STOP_CONTINUOUS_TRANSLATION_ERROR", e.message, null) + result.error("AST_DISPOSE_ERROR", e.message, null) } } "recognizeCallback" -> { - - - FileLogger.d(tag, "recognizeCallback:") // + FileLogger.d(tag, "recognizeCallback:") result.success(true) } - "path" -> { - val filePath = call.argument("filePath") ?: "" + "path" -> { + val filePath = call.argument("filePath") ?: "" try { - - // 读取这个wav音频文件,取里面的音频数据进行播放 val file = java.io.File(filePath) if (file.exists()) { - // 启动协程来按播放速度读取音频文件 CoroutineScope(Dispatchers.IO).launch { streamAudioFile(file) } @@ -913,233 +1145,606 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } else { result.error("FILE_NOT_FOUND", "音频文件不存在: $filePath", null) } - - } catch (e: Exception) { - result.error("READ_FILE_ERROR", "读取音频文件失败: ${e.message}", null) + } catch (e: Exception) { + result.error("READ_FILE_ERROR", "读取音频文件失败: ${e.message}", null) + } } - } - + + "setLowVolumeThreshold" -> { + val threshold = call.argument("threshold") ?: 0 + lowVolumeThreshold = threshold.toShort() + FileLogger.d(tag, "setLowVolumeThreshold: $lowVolumeThreshold") + result.success(true) + } + "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") ?: "" - val region = call.argument("region") ?: "" + val provider = call.argument("provider") ?: "" val supportedLanguages = call.argument>("supportedLanguages") ?: listOf("zh-CN") - val useExternalAudio = call.argument("useExternalAudio") ?: false + val lang0 = supportedLanguages.getOrElse(0) { "zh-CN" } + val lang1 = supportedLanguages.getOrElse(1) { "en-US" } + val translationLang0 = supportedLanguages.getOrElse(2) { "zh-Hans" } + val translationLang1 = supportedLanguages.getOrElse(3) { "en-US" } + val ttsLang0 = supportedLanguages.getOrElse(4) { "zh-CN-XiaoxiaoNeural" } + val ttsLang1 = supportedLanguages.getOrElse(5) { "en-US-AriaNeural" } + Log.d(tag, "initializeIntegrated: supportedLanguages=$supportedLanguages") + currentAstProvider = provider - // 获取翻译服务配置参数(需要从Flutter端传递) - val translationAccessKey = call.argument("translationAccessKey") ?: "" - val translationSecretKey = call.argument("translationSecretKey") ?: "" - val translationRegion = - call.argument("translationRegion") ?: "cn-north-1" + val subscriptionKey = call.argument("subscriptionKey") ?: "" + val region = call.argument("region") ?: "" - FileLogger.d(tag, "初始化AST服务") + val xfyunAppId = call.argument("xfyunAppId") ?: "" + val xfyunAccessKeyId = call.argument("xfyunAccessKeyId") ?: "" + val xfyunAccessKeySecret = call.argument("xfyunAccessKeySecret") ?: "" + val iflytekHost = call.argument("iflytekHost") ?: "ntrans.xfyun.cn" + + val volcanoTranslationAccessKey = + call.argument("volcanoTranslationAccessKey") ?: "" + val volcanoTranslationSecretKey = + call.argument("volcanoTranslationSecretKey") ?: "" + val volcanoTranslationRegion = + call.argument("volcanoTranslationRegion") ?: "cn-north-1" + + val azureTranslationKey = call.argument("azureTranslationKey") ?: "" + val azureTranslationRegion = + call.argument("azureTranslationRegion") ?: "cn-north-1" + + val wsUrl = call.argument("wsUrl") + ?: "wss://openspeech.bytedance.com/api/v4/ast/v2/translate" + val appKey = call.argument("appKey") ?: "" + val accessKey = call.argument("accessKey") ?: "" + val resourceId = + call.argument("resourceId") ?: "volc.service_type.10053" + + val alibabaAppKey = call.argument("alibabaAppKey") ?: "" + val alibabaAppId = call.argument("alibabaAppId") ?: "" + val alibabaAppURL = call.argument("alibabaAppURL") ?: "" + if (currentAstProvider == "iflytek" ) { + Log.d(tag, "initializeIntegrated: iflytek") + val azureConfig = IflytekIntegratedSpeechService.AzureConfiguration( + subscriptionKey = subscriptionKey, + region = region + ) + val translationConfig = + IflytekIntegratedSpeechService.IflytekTranslationConfiguration( + appId = xfyunAppId, + apiKey = xfyunAccessKeyId, + apiSecret = xfyunAccessKeySecret, + host = iflytekHost + ) - // 创建Azure配置 - val azureConfig = AzureConfiguration( - subscriptionKey = subscriptionKey, - region = region - ) + val serviceConfigA = IflytekIntegratedSpeechService.ServiceConfiguration( + sourceLanguage = lang0, + targetLanguage = lang1, + translationSourceLanguage = translationLang0, + translationTargetLanguage = translationLang1, + currentVoice = ttsLang1, + xfyunAppId = xfyunAppId, + xfyunAccessKeyId = xfyunAccessKeyId, + xfyunAccessKeySecret = xfyunAccessKeySecret + ) + val serviceConfigB = IflytekIntegratedSpeechService.ServiceConfiguration( + sourceLanguage = lang1, + targetLanguage = lang0, + translationSourceLanguage = translationLang1, + translationTargetLanguage = translationLang0, + currentVoice = ttsLang0, + xfyunAppId = xfyunAppId, + xfyunAccessKeyId = xfyunAccessKeyId, + xfyunAccessKeySecret = xfyunAccessKeySecret + ) - // 创建翻译配置 - val translationConfig = - TranslationConfiguration( - accessKey = translationAccessKey, - secretKey = translationSecretKey, - region = translationRegion + val callbackA = IflytekAstCallback( + "A", "$lang0->$lang1", + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleRightMutex.withLock { + // TODO: 双端翻译时启用右声道写入 + // BleService.writeExternalRightAudioData(data) + } + } + } + ) + val callbackB = IflytekAstCallback( + "B", "$lang1->$lang0", + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleLeftMutex.withLock { + // TODO: 双端翻译时启用左声道分离写入 + // BleService.writeExternalLeftAudioData(data) + BleService.writeExternalAudioData(data) + } + } + } ) - // 创建服务配置 - val serviceConfig = IntegratedSpeechTranslationService.ServiceConfiguration( - sourceLanguage = if (supportedLanguages.isNotEmpty()) supportedLanguages[0] else "zh-CN", - targetLanguage = if (supportedLanguages.size > 1) supportedLanguages[1] else "en-US" - ) + GlobalScope.launch(Dispatchers.Main) { + val initA = iflytekAstHelperA?.initialize( + azureConfig = azureConfig, + translationConfig = translationConfig, + serviceConfig = serviceConfigA, + callback = callbackA + ) ?: false - // 创建事件回调 - val callback = object : IntegratedSpeechTranslationService.ServiceEventCallback { - override fun onServiceInitialized() { - // sendAstEvent( - // mapOf( - // "type" to "serviceInitialized" - // ) - // ) - } + delay(1200L) - override fun onRecognizing(text: String, language: String, confidence: Float) { - // sendAstEvent( - // mapOf( - // "type" to "recognizing", - // "text" to text, - // "language" to language, - // "confidence" to confidence - // ) - // ) + val initB = iflytekAstHelperB?.initialize( + azureConfig = azureConfig, + translationConfig = translationConfig, + serviceConfig = serviceConfigB, + callback = callbackB + ) ?: false + + result.success(initA && initB) } + } else if (currentAstProvider == "azure" ) { + Log.d(tag, "initializeIntegrated: ttsLang1$ttsLang1, ttsLang0$ttsLang0") + val azureConfig = + AzureConfiguration(subscriptionKey = subscriptionKey, region = region) + val translationConfig = TranslationConfiguration( + subscriptionKey = azureTranslationKey, + region = azureTranslationRegion - override fun onRecognized(text: String, language: String, confidence: Float) { - FileLogger.d(tag, "识别到文本: $text, 语言: $language, 置信度: $confidence") - sendAsrEvent( - mapOf( - "type" to "result1", - "text" to text, - "detectedLanguage" to language ) - ) - // sendAstEvent( - // mapOf( - // "type" to "recognized", - // "text" to text, - // "language" to language, - // "confidence" to confidence - // ) - // ) - } - override fun onTranslated( - originalText: String, - translatedText: String, - targetLanguage: String - ) { - // sendAstEvent( - // mapOf( - // "type" to "translated", - // "originalText" to originalText, - // "translatedText" to translatedText, - // "targetLanguage" to targetLanguage - // ) - // ) - } + val serviceConfigA = + IntegratedSpeechTranslationService.ServiceConfiguration( + sourceLanguage = lang0, + targetLanguage = lang1, + currentVoice = ttsLang1, + translationSourceLanguage = translationLang0, + translationTargetLanguage = translationLang1 + ) + val serviceConfigB = + IntegratedSpeechTranslationService.ServiceConfiguration( + sourceLanguage = lang1, + targetLanguage = lang0, + currentVoice = ttsLang0, + translationSourceLanguage = translationLang1, + translationTargetLanguage = translationLang0 + ) + + val callbackA = AzureAstCallback( + "A", "$lang0->$lang1", + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleRightMutex.withLock { + // TODO: 双端翻译时启用右声道写入 + // BleService.writeExternalRightAudioData(data) + } + } + } + ) + val callbackB = AzureAstCallback( + "B", "$lang1->$lang0", + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleLeftMutex.withLock { + // TODO: 双端翻译时启用左声道分离写入 + // BleService.writeExternalLeftAudioData(data) + BleService.writeExternalAudioData(data) + } + } + } + ) - override fun onTranslationStarted(text: String) { - // sendAstEvent( - // mapOf( - // "type" to "translationStarted", - // "text" to text - // ) - // ) + GlobalScope.launch(Dispatchers.Main) { + azureAstHelperA?.initialize( + azureConfig = azureConfig, + translationConfig = translationConfig, + serviceConfig = serviceConfigA, + callback = callbackA + ) + delay(1200L) + azureAstHelperB?.initialize( + azureConfig = azureConfig, + translationConfig = translationConfig, + serviceConfig = serviceConfigB, + callback = callbackB + ) + result.success(true) } + } else if (currentAstProvider == "volcano" ) { + Log.d(tag, "initializeIntegrated: volcano") + + astWsConfig = Config( + wsUrl = wsUrl, + appKey = appKey, + accessKey = accessKey, + resourceId = resourceId, + clientWaitMs = 60_000L + ) + Log.d(tag, "initializeIntegrated: volcano $astWsConfig") + + val callbackA = DoubaoAstCallback( + "A", "$translationLang0->$translationLang1", translationLang1, + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleRightMutex.withLock { + // TODO: 双端翻译时启用右声道写入 + // BleService.writeExternalRightAudioData(data) + } + } + } + ) + val callbackB = DoubaoAstCallback( + "B", "$translationLang1->$translationLang0", translationLang0, + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleLeftMutex.withLock { + // TODO: 双端翻译时启用左声道分离写入 + // BleService.writeExternalLeftAudioData(data) + BleService.writeExternalAudioData(data) + } + } + } + ) + Log.d(tag, "initializeIntegrated:lang0= $translationLang0, lang1=$translationLang1") + // 设置会话语言(与 UI 选择一致) + val cfgA = astWsConfig!!.copy( + sourceLanguage = translationLang0, + targetLanguage = translationLang1 + ) + val cfgB = astWsConfig!!.copy( + sourceLanguage = translationLang1, + targetLanguage = translationLang0 + ) - override fun onTranslationFailed(text: String, error: String) { - // sendAstEvent( - // mapOf( - // "type" to "translationFailed", - // "text" to text, - // "error" to error - // ) - // ) + GlobalScope.launch(Dispatchers.Main) { + doubaoAstHelperA?.initialize(cfgA, callbackA) + delay(1200L) + doubaoAstHelperB?.initialize(cfgB, callbackB) + result.success(true) } + } else if (currentAstProvider == "alibaba" ) { + Log.d(tag, "initializeIntegrated: alibaba") + + val bailianConfigA = AliyunBailianE2EHelper.Config( + apiKey = alibabaAppKey, + appId = alibabaAppId, + wsUrl = alibabaAppURL, + sourceLanguage = translationLang0, + targetLanguage = translationLang1, + voice = ttsLang1 + ) + val bailianConfigB = AliyunBailianE2EHelper.Config( + apiKey = alibabaAppKey, + appId = alibabaAppId, + wsUrl = alibabaAppURL, + sourceLanguage = translationLang1, + targetLanguage = translationLang0, + voice = ttsLang0 + ) + + val callbackA = AliyunAstCallback( + "A", "$translationLang0->$translationLang1", translationLang1, + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleRightMutex.withLock { + // TODO: 双端翻译时启用右声道写入 + // BleService.writeExternalRightAudioData(data) + } + } + } + ) + val callbackB = AliyunAstCallback( + "B", "$translationLang1->$translationLang0", translationLang0, + { sendAstEvent(it) }, + { data -> + bleWriteScope.launch { + bleLeftMutex.withLock { + // TODO: 双端翻译时启用左声道分离写入 + // BleService.writeExternalLeftAudioData(data) + BleService.writeExternalAudioData(data) + } + } + } + ) - override fun onSynthesisStarted(text: String) { - // sendAstEvent( - // mapOf( - // "type" to "synthesisStarted", - // "text" to text - // ) - // ) + GlobalScope.launch(Dispatchers.Main) { + val initA = bailianAstHelperA?.initialize(bailianConfigA, callbackA) ?: false + delay(1200L) + val initB = bailianAstHelperB?.initialize(bailianConfigB, callbackB) ?: false + result.success(initA && initB) } + } + } - override fun onSynthesisCompleted(text: String) { - FileLogger.d(tag, "语音合成完成,文本=${text}") + else -> { + result.notImplemented() + } + } + } + } - + // 翻译方法处理器(Volcano/Microsoft REST) + /** + * 翻译方法通道处理器 + * 提供: + * - initialize(accessKey, secretKey, region, maxRetryAttempts, timeout) + * - translateText(text, sourceLanguage, targetLanguage) + * - dispose() + */ + inner class TranslationMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + /** + * 初始化翻译服务 + * provider: volcano/microsoft(默认 volcano) + * volcano入参:`accessKey`、`secretKey`、`region`、`maxRetryAttempts`、`timeout` + * microsoft入参:`subscriptionKey`/`accessKey`、`region`/`location`、`endpoint?`、`maxRetryAttempts?`、`timeout?` + */ + val provider = (call.argument("provider") ?: "volcano").lowercase() + translationRestProvider = provider - // sendAstEvent( - // mapOf( - // "type" to "synthesisCompleted", - // "text" to text - // ) - // ) + try { + if (provider == "microsoft") { + val subscriptionKey = call.argument("subscriptionKey") + ?: call.argument("accessKey") + ?: "" + val location = call.argument("location") + val region = call.argument("region") ?: (location ?: "") + val endpoint = call.argument("endpoint") + ?: "https://api.cognitive.microsofttranslator.com" + val maxRetryAttempts = + call.argument("maxRetryAttempts")?.toInt() ?: 3 + val timeout = call.argument("timeout")?.toLong() ?: 10000L + + val svc = microsoftTranslationService + ?: MicrosoftTranslationServiceImpl().also { + microsoftTranslationService = it + } + kotlinx.coroutines.GlobalScope.launch(kotlinx.coroutines.Dispatchers.IO) { + val success = svc.initialize( + mapOf( + "subscriptionKey" to subscriptionKey, + "region" to region, + "endpoint" to endpoint, + "maxRetryAttempts" to maxRetryAttempts.toString(), + "timeout" to timeout.toString() + ) + ) + mainHandler.post { result.success(success) } + } + } else { + val accessKey = call.argument("accessKey") ?: "" + val secretKey = call.argument("secretKey") ?: "" + val region = call.argument("region") ?: "cn-north-1" + val maxRetryAttempts = + call.argument("maxRetryAttempts")?.toInt() ?: 3 + val timeout = call.argument("timeout")?.toLong() ?: 10000L + + val svc = volcanoTranslationService + ?: VolcanoTranslationServiceImpl().also { + volcanoTranslationService = it + } + kotlinx.coroutines.GlobalScope.launch(kotlinx.coroutines.Dispatchers.IO) { + val success = svc.initialize( + mapOf( + "accessKey" to accessKey, + "secretKey" to secretKey, + "region" to region, + "maxRetryAttempts" to maxRetryAttempts.toString(), + "timeout" to timeout.toString() + ) + ) + mainHandler.post { result.success(success) } + } } - override fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) { - FileLogger.d(tag, "语音合成音频生成,文本=${text},音频数据大小=${audioData.size}") - // Azure TTS返回的是RIFF WAV格式,需要去除文件头 - val pcmData = extractPcmFromWav(audioData) - if (pcmData.isNotEmpty()) { - FileLogger.d(tag, "提取PCM数据,大小=${pcmData.size}字节") - BleService.writeExternalAudioData(pcmData) + } catch (e: Exception) { + result.error("TRANSLATION_INIT_ERROR", e.message, null) } - else { - FileLogger.e(tag, "提取PCM数据失败,音频数据可能不是WAV格式") + } + + "translateText" -> { + /** + * 文本翻译 + * 入参:`text` 源文本、`sourceLanguage` 源语言、`targetLanguage` 目标语言 + * 流程:先发送 `translationStarted` 事件,完成后发送 `translated` 或 `translationFailed` + */ + val text = call.argument("text") ?: return result.error( + "INVALID_ARGUMENTS", + "text不能为空", + null + ) + val sourceLanguage = call.argument("sourceLanguage") + ?: return result.error("INVALID_ARGUMENTS", "sourceLanguage不能为空", null) + val targetLanguage = call.argument("targetLanguage") + ?: return result.error("INVALID_ARGUMENTS", "targetLanguage不能为空", null) + Log.d( + tag, + "translateText: text=$text, sourceLanguage=$sourceLanguage, targetLanguage=$targetLanguage" + ) + if (translationEventSink == null) { + return result.error( + "EVENT_CHANNEL_NOT_READY", + "事件通道未准备好,无法执行翻译", + null + ) } + Log.d(tag, "translateText: 发送 translationStarted 事件") + val provider = + (call.argument("provider") ?: translationRestProvider).lowercase() + Log.d(tag, "translateText: provider=$provider") + val svc = microsoftTranslationService + ?: MicrosoftTranslationServiceImpl().also { + microsoftTranslationService = it } - override fun onSynthesisFailed(text: String, error: String) { - // sendAstEvent( - // mapOf( - // "type" to "synthesisFailed", - // "text" to text, - // "error" to error - // ) - // ) - } + //if (provider == "microsoft") { + microsoftTranslationService ?: MicrosoftTranslationServiceImpl().also { + microsoftTranslationService = it + } + // } else { + // volcanoTranslationService ?: VolcanoTranslationServiceImpl().also { volcanoTranslationService = it } + // } + sendTranslationEvent( + mapOf( + "type" to "translationStarted", + "text" to text, + "sourceLanguage" to sourceLanguage, + "targetLanguage" to targetLanguage + ) + ) - override fun onSynthesisProgress(text: String, progress: Float) { - // sendAstEvent( - // mapOf( - // "type" to "synthesisProgress", - // "text" to text, - // "progress" to progress - // ) - // ) + kotlinx.coroutines.GlobalScope.launch(kotlinx.coroutines.Dispatchers.IO) { + val res = svc.translateText(text, sourceLanguage, targetLanguage) + val event = if (res.success) { + mapOf( + "type" to "translated", + "originalText" to text, + "translatedText" to (res.translatedText ?: ""), + "targetLanguage" to targetLanguage + ) + } else { + mapOf( + "type" to "translationFailed", + "text" to text, + "error" to (res.error ?: "未知错误") + ) } + sendTranslationEvent(event) + mainHandler.post { result.success(res.success) } + } + } - override fun onRecognitionStarted() { - // sendAstEvent( - // mapOf( - // "type" to "recognitionStarted" - // ) - // ) - } + "initialize1" -> { + val speechSubscriptionKey = call.argument("speechSubscriptionKey") + ?: call.argument("subscriptionKey") + ?: "" + val speechRegion = call.argument("speechRegion") + ?: call.argument("region") + ?: "" + + val translationSubscriptionKey = + call.argument("translationSubscriptionKey") + ?: speechSubscriptionKey + val translationRegion = call.argument("translationRegion") + ?: speechRegion + + FileLogger.d( + tag, + "initialize1 called, speechRegion=$speechRegion, translationRegion=$translationRegion, " + + "speechSubscriptionKey=${speechSubscriptionKey}, translationSubscriptionKey=${translationSubscriptionKey}" + ) - override fun onRecognitionStopped() { - // sendAstEvent( - // mapOf( - // "type" to "recognitionStopped" - // ) - // ) + val svc = microsoftTranslationAndTtsService + ?: MicrosoftTranslationAndTtsService().also { + microsoftTranslationAndTtsService = it } - override fun onStateChanged(component: String, isActive: Boolean) { - // sendAstEvent( - // mapOf( - // "type" to "stateChanged", - // "component" to component, - // "isActive" to isActive - // ) - // ) - } + kotlinx.coroutines.GlobalScope.launch(kotlinx.coroutines.Dispatchers.IO) { + val config = MicrosoftTranslationAndTtsService.ServiceConfig( + speechSubscriptionKey = speechSubscriptionKey, + speechRegion = speechRegion, + translationSubscriptionKey = translationSubscriptionKey, + translationRegion = translationRegion + ) + val success = svc.initialize(config) + mainHandler.post { result.success(success) } + } + } - override fun onError(component: String, error: String) { - // sendAstEvent( - // mapOf( - // "type" to "error", - // "component" to component, - // "error" to error - // ) - // ) - } + "process" -> { + val text = call.argument("text") ?: return result.error( + "INVALID_ARGUMENTS", + "text不能为空", + null + ) + val sourceLanguage = call.argument("sourceLanguage") + ?: return result.error("INVALID_ARGUMENTS", "sourceLanguage不能为空", null) + val targetLanguage = call.argument("targetLanguage") + ?: return result.error("INVALID_ARGUMENTS", "targetLanguage不能为空", null) + val targetTTSLanguage = call.argument("targetTTSLanguage") + ?: return result.error( + "INVALID_ARGUMENTS", + "targetTTSLanguage不能为空", + null + ) + FileLogger.d( + tag, + "process called, textLength=${text.length}, source=$sourceLanguage, target=$targetLanguage, targetTTSLanguage=$targetTTSLanguage" + ) + + val svc = microsoftTranslationAndTtsService + if (svc == null) { + return result.error("SERVICE_NOT_INITIALIZED", "请先调用 initialize", null) + } + + // 使用应用files目录下的 tts_output 子目录 + val outputDir = java.io.File(context.filesDir, "tts_output") + + val cachedFile = java.io.File(outputDir, "$targetLanguage.wav") + if (cachedFile.exists() && cachedFile.isFile) { + FileLogger.d(tag, "process cache hit, filePath=${cachedFile.absolutePath}") + result.success(cachedFile.absolutePath) + return } - // 使用协程调用异步初始化方法 - GlobalScope.launch(Dispatchers.Main) { - azureAstHelper.initialize( - azureConfig = azureConfig, - translationConfig = translationConfig, - serviceConfig = serviceConfig, - callback = callback + kotlinx.coroutines.GlobalScope.launch(kotlinx.coroutines.Dispatchers.IO) { + val filePath = svc.processAndSaveAudio( + text, + sourceLanguage, + targetLanguage, + targetTTSLanguage, + outputDir ) + mainHandler.post { + if (filePath != null) { + FileLogger.d(tag, "process success, filePath=$filePath") + result.success(filePath) + } else { + FileLogger.e(tag, "process failed, filePath is null") + result.error("PROCESS_FAILED", "处理失败", null) + } + } } - result.success(true) } + + "dispose1" -> { + FileLogger.d( + tag, + "dispose1 called, releasing MicrosoftTranslationAndTtsService" + ) + microsoftTranslationAndTtsService?.dispose() + result.success(true) + } + + "dispose" -> { + /** + * 释放翻译服务资源 + */ + try { + volcanoTranslationService?.dispose() + microsoftTranslationService?.dispose() + result.success(true) + } catch (e: Exception) { + result.error("TRANSLATION_DISPOSE_ERROR", e.message, null) + } + } + + else -> result.notImplemented() } } } - override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { asrChannel.setMethodCallHandler(null) ttsChannel.setMethodCallHandler(null) + astChannel.setMethodCallHandler(null) + translationChannel.setMethodCallHandler(null) asrEventChannel.setStreamHandler(null) ttsEventChannel.setStreamHandler(null) + astEventChannel.setStreamHandler(null) + translationEventChannel.setStreamHandler(null) try { if (isRegisteredToBle) { @@ -1168,70 +1773,101 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { override fun onConnectionStateChanged(state: Int) { } - override fun onAudioDataReceived(data: ByteArray,channel:Int) { + override fun onAudioDataReceived(data: ByteArray, channel: Int) { + if (channel == 0) { return - } - else if (channel == 1) { - if (isRecord) { - recordfile?.setAudioConfig(16000,1) - recordfile?.saveAudioDataToWav(data) - // 构建音频数据事件映射 - val audioEvent = mapOf( - "type" to "audioData", - "data" to data, - "timestamp" to System.currentTimeMillis(), - "size" to data.size - ) - // 发送音频数据事件到 Flutter 层 - sendAudioDataEvent(audioEvent) - } + } else if (channel == 1) { + if (isRecord) { + recordfile?.setAudioConfig(16000, 1) + recordfile?.saveAudioDataToWav(data) + // 构建音频数据事件映射 + val audioEvent = mapOf( + "type" to "audioData", + "data" to data, + "timestamp" to System.currentTimeMillis(), + "size" to data.size + ) + // 发送音频数据事件到 Flutter 层 + sendAudioDataEvent(audioEvent) + } if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { - azureAsrHelper.audioStream?.saveAudioDataTo(data) - } - } - else if (channel == 2) { - if (isRecord) { - recordfile?.setAudioConfig(16000,2) - recordfile?.saveAudioDataToWav(data) - // 构建音频数据事件映射 - val audioEvent = mapOf( - "type" to "audioData", - "data" to data, - "timestamp" to System.currentTimeMillis(), - "size" to data.size - ) - // 发送音频数据事件到 Flutter 层 - sendAudioDataEvent(audioEvent) - } - val sampleCount = data.size / 4 // 每个样本4字节(左右声道各2字节) - val leftBuffer = ByteArray(sampleCount * 2) // 左声道缓冲区 - val rightBuffer = ByteArray(sampleCount * 2) // 右声道缓冲区 - - // 拆分交错的左右声道数据 - for (i in 0 until sampleCount) { - val stereoIndex = i * 4 - val monoIndex = i * 2 - - // 左声道(低位字节在前,高位字节在后) - leftBuffer[monoIndex] = data[stereoIndex] - leftBuffer[monoIndex + 1] = data[stereoIndex + 1] - - // 右声道 - rightBuffer[monoIndex] = data[stereoIndex + 2] - rightBuffer[monoIndex + 1] = data[stereoIndex + 3] - } + azureAsrHelper.audioStream?.saveAudioDataTo(data) + } + } else if (channel == 2) { + // FileLogger.d(tag, "isRecord=${isRecord}") + if (isRecord) { + + recordfile?.setAudioConfig(16000, 2) + recordfile?.saveAudioDataToWav(data) + // 构建音频数据事件映射 + val audioEvent = mapOf( + "type" to "audioData", + "data" to data, + "timestamp" to System.currentTimeMillis(), + "size" to data.size + ) + // 发送音频数据事件到 Flutter 层 + sendAudioDataEvent(audioEvent) + } + + val sampleCount = data.size / 4 // 每个样本4字节(左右声道各2字节) + val leftBuffer = ByteArray(sampleCount * 2) // 左声道缓冲区 + val rightBuffer = ByteArray(sampleCount * 2) // 右声道缓冲区 + + // 拆分交错的左右声道数据 + for (i in 0 until sampleCount) { + val stereoIndex = i * 4 + val monoIndex = i * 2 + + // 左声道(低位字节在前,高位字节在后) + leftBuffer[monoIndex] = data[stereoIndex] + leftBuffer[monoIndex + 1] = data[stereoIndex + 1] + + // 右声道 + rightBuffer[monoIndex] = data[stereoIndex + 2] + rightBuffer[monoIndex + 1] = data[stereoIndex + 3] + } + + // 过滤低音量音频 + val filteredLeftBuffer = filterLowVolumeAudio(leftBuffer, lowVolumeThreshold) + val filteredRightBuffer = filterLowVolumeAudio(rightBuffer, lowVolumeThreshold) + // 左声道是对方的,右声道是麦的 + pushAstAudioToA(filteredRightBuffer) + pushAstAudioToB(filteredLeftBuffer) - - if (azureAsrHelper.audioSourceType == AzureAsrHelper.AudioSourceType.EXTERNAL) { - azureAsrHelper.audioStream?.saveAudioDataTo(rightBuffer) - } - azureAstHelper?.pushAudioData(leftBuffer) } - + // 可选:处理音频数据 } + /** + * 过滤低音量音频(简单的噪声门算法) + * 将振幅低于阈值的样本置为静音(0),以去除背景底噪。 + * + * @param audioData PCM 16bit Little Endian 音频数据 + * @param threshold 振幅阈值(建议范围 100-1000,视具体噪音情况而定) + * @return 过滤后的音频数据 + */ + private fun filterLowVolumeAudio(audioData: ByteArray, threshold: Short): ByteArray { + // 直接在原数组上修改,减少内存分配(因为 leftBuffer/rightBuffer 已经是局部创建的副本) + for (i in 0 until audioData.size step 2) { + if (i + 1 < audioData.size) { + // 解析 16-bit Little Endian 样本 + val low = audioData[i].toInt() and 0xFF + val high = audioData[i + 1].toInt() + val sample = ((high shl 8) or low).toShort() + + // 如果绝对值小于阈值,则视为静音 + if (Math.abs(sample.toInt()) < threshold) { + audioData[i] = 0 + audioData[i + 1] = 0 + } + } + } + return audioData + } + /** * 处理唤醒信号 * 在收到唤醒信号时启动语音识别 @@ -1239,25 +1875,11 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { override fun onWakeupSignalReceived() { } - - /** - * 开始 AI 单次对话回调 - * @throws Exception 无特殊异常抛出 - * @return 无返回值 - * @param 无参数 - */ + override fun onStartSingleDialogReceived() { - // 最小实现:当前插件不处理开始单次对话事件 } - - /** - * 结束 AI 单次对话回调 - * @throws Exception 无特殊异常抛出 - * @return 无返回值 - * @param 无参数 - */ + override fun onEndSingleDialogReceived() { - // 最小实现:当前插件不处理结束单次对话事件 } override fun onDeviceInfoReceived(infoType: Int, infoData: Map) { @@ -1265,14 +1887,17 @@ class AzureSpeechPlugin : BleService.Callback, FlutterPlugin { } override fun onWakeupVoiceUpgradeProgressReceived(progress: Int) { - // 处理升级进度 } override fun onWakeupVoiceUpgradeSuccess() { } override fun onWakeupVoiceUpgradeFailed() { - } -} + } + // TODO: 双端翻译时启用副芯片回调 + // override fun onSecondaryChipConnectionStateChangedDetail(state: Int, deviceAddress: String?, reason: String?) { } + // override fun onSecondaryChipConnectionsCountChanged(count: Int, addresses: List) { } + +} diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 5b0770249..8bd155e6f 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -14,8 +14,9 @@ import com.deep_voice.speech.tts.TtsEventType import kotlinx.coroutines.* import java.io.ByteArrayInputStream import java.io.InputStream -import java.util.concurrent.atomic.AtomicInteger import com.yunqiinnovation.azure_speech.tools.SimpleAudioPlayer +import java.util.concurrent.TimeUnit + /** * Azure TTS Helper * @@ -35,7 +36,8 @@ class AzureTtsHelper(private val context: Context) : ITtsService { private var isInitialized = false private var isSpeaking = false - // 当前配置 + // 最近一次合成的原始文本,用于在合成完成时打印 + private var lastSynthesisText: String = "" private var currentVoice = DEFAULT_VOICE private var currentRate = "0%" private var currentPitch = "0%" @@ -52,9 +54,6 @@ class AzureTtsHelper(private val context: Context) : ITtsService { private val streamBuffer = StringBuilder() private var lastSpeakTime = 0L - // 待完成的合成任务计数:只有全部完成才触发 PLAYBACK_COMPLETED - private val pendingSpeakCount = AtomicInteger(0) - private var enableVoiceFlocking = false private var speakerProfileId: String = "" @@ -86,9 +85,9 @@ class AzureTtsHelper(private val context: Context) : ITtsService { // 确保使用正确的语音输出 speechConfig?.setProperty("SPEECH-AudioOutputFormat", "riff-16khz-16bit-mono-pcm") - // 优化:设置低延迟连接属性 - speechConfig?.setProperty("SpeechServiceConnection_InitialSilenceTimeoutMs", "300") - speechConfig?.setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", "300") + // // 优化:设置低延迟连接属性 + // speechConfig?.setProperty("SpeechServiceConnection_InitialSilenceTimeoutMs", "300") + // speechConfig?.setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", "300") speechConfig?.setSpeechSynthesisVoiceName(currentVoice) @@ -96,6 +95,7 @@ class AzureTtsHelper(private val context: Context) : ITtsService { FileLogger.d(TAG, "语音设置: voice=$currentVoice, format=Riff16Khz16BitMonoPcm") // 创建音频配置 + val audioConfig = if (customAudioOutputStream != null) { // 使用自定义音频输出流 AudioConfig.fromStreamOutput(customAudioOutputStream) @@ -103,7 +103,6 @@ class AzureTtsHelper(private val context: Context) : ITtsService { // 使用默认扬声器 AudioConfig.fromDefaultSpeakerOutput() } - // 创建合成器 synthesizer = SpeechSynthesizer(speechConfig, audioConfig) @@ -217,43 +216,43 @@ class AzureTtsHelper(private val context: Context) : ITtsService { * * @param device 音频输出设备类型 */ - override fun setAudioOutputDevice(device: AudioOutputDevice): Boolean{ + override fun setAudioOutputDevice(device: AudioOutputDevice): Boolean { // Azure TTS使用系统默认的音频路由 // 音频输出设备的控制需要通过Android的AudioManager实现 - try { - FileLogger.e(TAG, "音频输出设备类型device=: ${device}") - val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager - audioManager?.let { manager -> - when (device) { - AudioOutputDevice.DEFAULT -> { - // 默认模式:系统自动选择 - manager.mode = AudioManager.MODE_NORMAL - manager.isSpeakerphoneOn = false - manager.startBluetoothSco() - } + try { + FileLogger.e(TAG, "音频输出设备类型device=: ${device}") + val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as? AudioManager + audioManager?.let { manager -> + when (device) { + AudioOutputDevice.DEFAULT -> { + // 默认模式:系统自动选择 + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + manager.startBluetoothSco() + } - AudioOutputDevice.SPEAKER -> { - // 强制使用扬声器 - /* MODE_NORMAL 默认媒体模式 ❌ 不能强制扬声器(会被耳机覆盖) - MODE_IN_COMMUNICATION VoIP 通信模式 ✅ 可以强制使用扬声器或听筒 - MODE_IN_CALL 电话模式(仅限系统) ✅ 更强但风险大,不推荐使用 - */ - manager.mode = AudioManager.MODE_IN_COMMUNICATION - manager.isSpeakerphoneOn = true - manager.stopBluetoothSco() - } + AudioOutputDevice.SPEAKER -> { + // 强制使用扬声器 + /* MODE_NORMAL 默认媒体模式 ❌ 不能强制扬声器(会被耳机覆盖) + MODE_IN_COMMUNICATION VoIP 通信模式 ✅ 可以强制使用扬声器或听筒 + MODE_IN_CALL 电话模式(仅限系统) ✅ 更强但风险大,不推荐使用 + */ + manager.mode = AudioManager.MODE_IN_COMMUNICATION + manager.isSpeakerphoneOn = true + manager.stopBluetoothSco() + } - AudioOutputDevice.HEADPHONES -> { - // 强制使用耳机(如果已连接) - manager.mode = AudioManager.MODE_NORMAL - manager.isSpeakerphoneOn = false - manager.startBluetoothSco() - // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 + AudioOutputDevice.HEADPHONES -> { + // 强制使用耳机(如果已连接) + manager.mode = AudioManager.MODE_NORMAL + manager.isSpeakerphoneOn = false + manager.startBluetoothSco() + // 注意:Android不能强制路由到耳机,只能在耳机已连接时使用 + } } } - } - return true - } catch (e: Exception) { + return true + } catch (e: Exception) { FileLogger.e(TAG, "设置音频输出设备失败: ${e.message}") return false } @@ -278,6 +277,7 @@ class AzureTtsHelper(private val context: Context) : ITtsService { /** * 设置事件监听器 */ + // 方法:setupEventListeners(设置事件监听器) private fun setupEventListeners() { synthesizer?.apply { // 合成开始事件 @@ -290,35 +290,40 @@ class AzureTtsHelper(private val context: Context) : ITtsService { // 合成中事件(接收音频数据) Synthesizing?.addEventListener { _, eventArgs -> + val audioData = eventArgs.result.audioData FileLogger.d(TAG, "接收到音频数据: ${audioData.size} 字节") - // 通知音频数据监听器 - if (audioData.isNotEmpty()) { - val listeners = ArrayList(audioDataListeners) - for (listener in listeners) { - try { - listener.onAudioData(audioData) - } catch (e: Exception) { - FileLogger.e(TAG, "通知音频数据异常: ${e.message}") - } - } - } + // // 打印详细的音频数据信息 + // if (audioData.isNotEmpty()) { + // printAudioDataDetails(audioData) + // } + + } // 合成完成事件 SynthesisCompleted?.addEventListener { _, eventArgs -> FileLogger.d( TAG, - "语音合成完成: resultId=${eventArgs.result.resultId}, 音频长度=${eventArgs.result.audioLength} 字节" + "语音合成完成: resultId=${eventArgs.result.resultId}, 音频长度=${eventArgs.result.audioLength} 字节, 文本=${lastSynthesisText}" ) + // val audioData = eventArgs.result.audioData + // // 通知音频数据监听器 + // if (audioData.isNotEmpty()) { + // val listeners = ArrayList(audioDataListeners) + // for (listener in listeners) { + // try { + // listener.onAudioData(audioData) + // } catch (e: Exception) { + // FileLogger.e(TAG, "通知音频数据异常: ${e.message}") + // } + // } + // } isSpeaking = false notifyEvent(TtsEventType.SYNTHESIS_COMPLETED) - // 所有句子都播完后才触发 PLAYBACK_COMPLETED,避免多句流式合成时重复触发 - if (pendingSpeakCount.decrementAndGet() <= 0) { - pendingSpeakCount.set(0) - notifyEvent(TtsEventType.PLAYBACK_COMPLETED) - } + // Azure TTS 合成完成即播放完成 + notifyEvent(TtsEventType.PLAYBACK_COMPLETED) } // 合成取消事件 @@ -551,6 +556,7 @@ class AzureTtsHelper(private val context: Context) : ITtsService { /** * 单次播放文本(非流式) */ + // 方法:speakOnce(单次播放文本) override fun speakOnce(text: String): Boolean { if (!isInitialized) { FileLogger.e(TAG, "TTS引擎未初始化") @@ -564,13 +570,15 @@ class AzureTtsHelper(private val context: Context) : ITtsService { } try { + // 记录当前合成文本,便于在完成事件打印 + lastSynthesisText = text + // 优化:使用简化的SSML生成 val ssml = generateOptimizedSsml(text) isSpeaking = true - pendingSpeakCount.incrementAndGet() - FileLogger.d(TAG, "开始语音合成,文本长度: ${text.length},当前待完成任务: ${pendingSpeakCount.get()}") + FileLogger.d(TAG, "开始语音合成,文本长度: ${text.length}") // 直接调用SpeakSsmlAsync,SDK内部已经是异步的 synthesizer?.SpeakSsmlAsync(ssml) @@ -702,14 +710,18 @@ class AzureTtsHelper(private val context: Context) : ITtsService { * 停止当前语音合成 */ override fun stop(): Boolean { + /** + * 停止当前语音合成 + * + * 说明:避免在停止过程中立即重建合成器,降低与 SDK 异步回调的竞态风险。 + */ if (!isInitialized) return false try { synthesizer?.StopSpeakingAsync() + ?.get(1500, TimeUnit.MILLISECONDS) streamBuffer.clear() isSpeaking = false - pendingSpeakCount.set(0) - recreateSynthesizer() return true } catch (e: Exception) { FileLogger.e(TAG, "停止语音合成失败: ${e.message}") @@ -771,4 +783,169 @@ class AzureTtsHelper(private val context: Context) : ITtsService { * 是否正在播放 */ fun isSpeaking(): Boolean = isSpeaking -} + + /** + * 打印音频数据的详细信息 + * + * @param audioData 音频数据字节数组 + */ + private fun printAudioDataDetails(audioData: ByteArray) { + try { + val dataSize = audioData.size + + // 基本信息 + FileLogger.d(TAG, "=== 音频数据详情 ===") + FileLogger.d(TAG, "数据大小: $dataSize 字节") + + if (dataSize > 0) { + // 计算音频时长(基于16KHz, 16bit, 单声道) + val sampleRate = 16000 // Hz + val bitsPerSample = 16 + val channels = 1 + val bytesPerSecond = sampleRate * (bitsPerSample / 8) * channels + val durationMs = (dataSize.toDouble() / bytesPerSecond * 1000).toInt() + + FileLogger.d(TAG, "预估音频时长: ${durationMs}ms") + + // 打印前32字节的十六进制数据 + val hexPreview = audioData.take(32).joinToString(" ") { + "%02X".format(it) + } + FileLogger.d(TAG, "前32字节数据(HEX): $hexPreview") + + // 音频数据统计 + val maxValue = audioData.maxOrNull()?.toInt() ?: 0 + val minValue = audioData.minOrNull()?.toInt() ?: 0 + val avgValue = audioData.map { it.toInt() }.average().toInt() + + FileLogger.d( + TAG, + "数据统计 - 最大值: $maxValue, 最小值: $minValue, 平均值: $avgValue" + ) + + // 检查是否为有效的PCM数据(非全零) + val nonZeroCount = audioData.count { it != 0.toByte() } + val nonZeroPercentage = (nonZeroCount.toDouble() / dataSize * 100).toInt() + FileLogger.d(TAG, "非零字节占比: $nonZeroPercentage% ($nonZeroCount/$dataSize)") + + // 如果数据量较大,还可以打印尾部数据 + if (dataSize > 64) { + val hexTail = audioData.takeLast(16).joinToString(" ") { + "%02X".format(it) + } + FileLogger.d(TAG, "后16字节数据(HEX): $hexTail") + } + + // 分析音频格式特征 + analyzeAudioFormat(audioData) + } + + FileLogger.d(TAG, "=== 音频数据详情结束 ===") + + } catch (e: Exception) { + FileLogger.e(TAG, "打印音频数据详情失败: ${e.message}") + } + } + + /** + * 分析音频格式特征 + * + * @param audioData 音频数据字节数组 + */ + private fun analyzeAudioFormat(audioData: ByteArray) { + try { + // 检查是否包含WAV文件头 + if (audioData.size >= 12) { + val header = audioData.take(12).toByteArray() + val headerStr = String(header, Charsets.US_ASCII) + + if (headerStr.startsWith("RIFF") && headerStr.contains("WAVE")) { + FileLogger.d(TAG, "检测到WAV格式文件头") + + // 解析WAV文件头信息 + if (audioData.size >= 44) { + parseWavHeader(audioData) + } + } else { + FileLogger.d(TAG, "原始PCM数据(无文件头)") + + // 分析PCM数据特征 + analyzePcmData(audioData) + } + } else { + FileLogger.d(TAG, "数据量太小,无法分析格式") + } + + } catch (e: Exception) { + FileLogger.e(TAG, "分析音频格式失败: ${e.message}") + } + } + + /** + * 解析WAV文件头信息 + * + * @param audioData 包含WAV文件头的音频数据 + */ + private fun parseWavHeader(audioData: ByteArray) { + try { + // WAV文件头结构解析(小端序) + val channels = + (audioData[22].toInt() and 0xFF) or ((audioData[23].toInt() and 0xFF) shl 8) + val sampleRate = (audioData[24].toInt() and 0xFF) or + ((audioData[25].toInt() and 0xFF) shl 8) or + ((audioData[26].toInt() and 0xFF) shl 16) or + ((audioData[27].toInt() and 0xFF) shl 24) + val bitsPerSample = + (audioData[34].toInt() and 0xFF) or ((audioData[35].toInt() and 0xFF) shl 8) + + FileLogger.d(TAG, "WAV格式信息:") + FileLogger.d(TAG, " 声道数: $channels") + FileLogger.d(TAG, " 采样率: ${sampleRate}Hz") + FileLogger.d(TAG, " 位深度: ${bitsPerSample}bit") + + } catch (e: Exception) { + FileLogger.e(TAG, "解析WAV文件头失败: ${e.message}") + } + } + + /** + * 分析PCM数据特征 + * + * @param audioData PCM音频数据 + */ + private fun analyzePcmData(audioData: ByteArray) { + try { + // 分析16位PCM数据的振幅分布 + if (audioData.size >= 2) { + val samples = mutableListOf() + + // 将字节数组转换为16位采样点(小端序) + for (i in 0 until audioData.size - 1 step 2) { + val sample = ((audioData[i].toInt() and 0xFF) or + ((audioData[i + 1].toInt() and 0xFF) shl 8)).toShort() + samples.add(sample) + } + + if (samples.isNotEmpty()) { + val maxAmplitude = samples.maxOrNull() ?: 0 + val minAmplitude = samples.minOrNull() ?: 0 + val avgAmplitude = samples.map { it.toInt() }.average().toInt() + + FileLogger.d(TAG, "PCM数据分析:") + FileLogger.d(TAG, " 采样点数: ${samples.size}") + FileLogger.d(TAG, " 最大振幅: $maxAmplitude") + FileLogger.d(TAG, " 最小振幅: $minAmplitude") + FileLogger.d(TAG, " 平均振幅: $avgAmplitude") + + // 计算动态范围 + val dynamicRange = maxAmplitude - minAmplitude + FileLogger.d(TAG, " 动态范围: $dynamicRange") + } + } + + } catch (e: Exception) { + FileLogger.e(TAG, "分析PCM数据失败: ${e.message}") + } + } + +} diff --git a/local_plugins/azure_speech/ios/azure_speech/Package.swift b/local_plugins/azure_speech/ios/azure_speech/Package.swift index eaa511209..b340fbce2 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Package.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Package.swift @@ -25,7 +25,7 @@ let package = Package( "MicrosoftCognitiveServicesSpeech" ], path: "Sources", // 修改为包含整个 Sources 目录 - sources: ["azure_speech/", "tools/"] // 明确指定包含的子目录 + sources: ["azure_speech/", "tools/","protos_swift/"] // 明确指定包含的子目录 ), // 使用binaryTarget引用Azure Speech SDK的.xcframework .binaryTarget( diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift index 0029c5771..3c1dd0d29 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift @@ -93,6 +93,19 @@ import ble_service internal var astEventSink: FlutterEventSink? private let azureAstHelper = IntegratedSpeechTranslationService() + // 端到端翻译 Helpers(豆包 / 阿里百炼) + private var doubaoAstHelperA = DoubaoE2ETranslateHelper() + private var doubaoAstHelperB = DoubaoE2ETranslateHelper() + private var bailianAstHelperA = AliyunBailianE2EHelper() + private var bailianAstHelperB = AliyunBailianE2EHelper() + private var currentAstProvider: String = "azure" + private var astWsConfig: DoubaoE2ETranslateHelper.Config? + internal var doubaoFinalSourceTextCache: [String: String] = [:] + private var doubaoCallbackA: DoubaoCallbackProxy? + private var doubaoCallbackB: DoubaoCallbackProxy? + private var bailianCallbackA: AliyunCallbackProxy? + private var bailianCallbackB: AliyunCallbackProxy? + // 音频数据相关 private var audioDataEventChannel: FlutterEventChannel? internal var audioDataEventSink: FlutterEventSink? @@ -747,112 +760,190 @@ private func sendAstEvent(_ event: [String: Any]) { // } case "startContinuousTranslation": - do { - os_log("开启翻译", log: log, type: .info) - try azureAstHelper.startContinuousTranslation() + if currentAstProvider == "volcano" { + _ = doubaoAstHelperA.startContinuousTranslation() + _ = doubaoAstHelperB.startContinuousTranslation() result(true) - } catch { - result(FlutterError(code: "START_CONTINUOUS_TRANSLATION_ERROR", message: error.localizedDescription, details: nil)) + } else if currentAstProvider == "alibaba" { + _ = bailianAstHelperA.startContinuousConversation() + _ = bailianAstHelperB.startContinuousConversation() + result(true) + } else { + do { + os_log("开启翻译", log: log, type: .info) + try azureAstHelper.startContinuousTranslation() + result(true) + } catch { + result(FlutterError(code: "START_CONTINUOUS_TRANSLATION_ERROR", message: error.localizedDescription, details: nil)) + } } - + case "stopContinuousTranslation": - do { - os_log("停止翻译", log: log, type: .info) - try azureAstHelper.stopContinuousTranslation() + if currentAstProvider == "volcano" { + _ = doubaoAstHelperA.stopContinuousTranslation() + _ = doubaoAstHelperB.stopContinuousTranslation() result(true) - } catch { - result(FlutterError(code: "STOP_CONTINUOUS_TRANSLATION_ERROR", message: error.localizedDescription, details: nil)) - } - - case "dispose": - do { - os_log("释放AST资源", log: log, type: .info) - azureAstHelper.dispose() + } else if currentAstProvider == "alibaba" { + _ = bailianAstHelperA.stopContinuousConversation() + _ = bailianAstHelperB.stopContinuousConversation() result(true) - } catch { - result(FlutterError(code: "DISPOSE_ERROR", message: error.localizedDescription, details: nil)) + } else { + do { + os_log("停止翻译", log: log, type: .info) + try azureAstHelper.stopContinuousTranslation() + result(true) + } catch { + result(FlutterError(code: "STOP_CONTINUOUS_TRANSLATION_ERROR", message: error.localizedDescription, details: nil)) + } } - + + case "dispose": + os_log("释放AST资源", log: log, type: .info) + azureAstHelper.dispose() + doubaoAstHelperA.dispose() + doubaoAstHelperB.dispose() + bailianAstHelperA.dispose() + bailianAstHelperB.dispose() + result(true) + case "recognizeCallback": os_log("recognizeCallback:", log: log, type: .info) result(true) - + case "path": guard let args = call.arguments as? [String: Any], let filePath = args["filePath"] as? String else { result(FlutterError(code: "INVALID_ARGUMENTS", message: "Missing filePath", details: nil)) return } - + let fileURL = URL(fileURLWithPath: filePath) - - // if FileManager.default.fileExists(atPath: filePath) { - // // 启动异步任务来按播放速度读取音频文件 - // Task { - // // await streamAudioFile(fileURL: fileURL) - // } - // result(true) - // } else { - // result(FlutterError(code: "FILE_NOT_FOUND", message: "音频文件不存在: \(filePath)", details: nil)) - // } - + result(true) + case "initialize": guard let args = call.arguments as? [String: Any] else { result(FlutterError(code: "INVALID_ARGUMENTS", message: "Missing arguments", details: nil)) return } - - let subscriptionKey = args["subscriptionKey"] as? String ?? "" - let region = args["region"] as? String ?? "" - let supportedLanguages = args["supportedLanguages"] as? [String] ?? ["zh-CN"] - let useExternalAudio = args["useExternalAudio"] as? Bool ?? false - - // 获取翻译服务配置参数 - let translationAccessKey = args["translationAccessKey"] as? String ?? "" - let translationSecretKey = args["translationSecretKey"] as? String ?? "" - let translationRegion = args["translationRegion"] as? String ?? "cn-north-1" - - os_log("初始化AST服务", log: log, type: .info) - - // 创建Azure配置 - let azureConfig = AzureConfiguration( - subscriptionKey: subscriptionKey, - region: region - ) - - // 创建翻译配置 - let translationConfig = TranslationConfiguration( - accessKey: translationAccessKey, - secretKey: translationSecretKey, - region: translationRegion - ) - - // 创建服务配置 - let serviceConfig = IntegratedSpeechTranslationService.ServiceConfiguration( - sourceLanguage: supportedLanguages.first ?? "zh-CN", - targetLanguage: supportedLanguages.count > 1 ? supportedLanguages[1] : "en-US" - ) - - // 创建事件回调 - let callback = AstEventCallback(plugin: self) - astEventCallback = callback - - // 使用异步任务调用初始化方法 - Task { - await azureAstHelper.initialize( - azureConfig: azureConfig, - translationConfig: translationConfig, - serviceConfig: serviceConfig, - callback: callback + let provider = (args["provider"] as? String ?? "azure").lowercased() + let supported = (args["supportedLanguages"] as? [String]) ?? ["zh-CN"] + let lang0 = supported.count > 0 ? supported[0] : "zh-CN" + let lang1 = supported.count > 1 ? supported[1] : "en-US" + let translationLang0 = supported.count > 2 ? supported[2] : "zh-Hans" + let translationLang1 = supported.count > 3 ? supported[3] : "en-US" + let ttsLang0 = supported.count > 4 ? supported[4] : "zh-CN-XiaoxiaoNeural" + let ttsLang1 = supported.count > 5 ? supported[5] : "en-US-AriaNeural" + currentAstProvider = provider + + let subscriptionKeyI = (args["subscriptionKey"] as? String) ?? "" + let regionI = (args["region"] as? String) ?? "" + let volcanoTranslationAccessKey = (args["volcanoTranslationAccessKey"] as? String) ?? "" + let volcanoTranslationSecretKey = (args["volcanoTranslationSecretKey"] as? String) ?? "" + let volcanoTranslationRegion = (args["volcanoTranslationRegion"] as? String) ?? "cn-north-1" + let azureTranslationKey = (args["azureTranslationKey"] as? String) ?? "" + let azureTranslationRegion = (args["azureTranslationServiceRegion"] as? String) ?? "" + // 豆包 AST WS 配置 + let wsUrl = (args["wsUrl"] as? String) ?? "wss://openspeech.bytedance.com/api/v4/ast/v2/translate" + let appKey = (args["appKey"] as? String) ?? "" + let accessKeyD = (args["accessKey"] as? String) ?? "" + let resourceId = (args["resourceId"] as? String) ?? "volc.service_type.10053" + // 阿里百炼配置 + let alibabaAppKey = (args["alibabaAppKey"] as? String) ?? "" + let alibabaAppId = (args["alibabaAppId"] as? String) ?? "" + let alibabaAppURL = (args["alibabaAppURL"] as? String) ?? "" + + os_log("initialize provider=%@ lang0=%@ lang1=%@ trans0=%@ trans1=%@", log: log, type: .info, provider, lang0, lang1, translationLang0, translationLang1) + + if currentAstProvider == "azure" { + let svcA = IntegratedSpeechTranslationService.ServiceConfiguration( + sourceLanguage: lang0, + targetLanguage: lang1, + currentVoice: ttsLang1, + translationSourceLanguage: translationLang0, + translationTargetLanguage: translationLang1 + ) + let svcB = IntegratedSpeechTranslationService.ServiceConfiguration( + sourceLanguage: lang1, + targetLanguage: lang0, + currentVoice: ttsLang0, + translationSourceLanguage: translationLang1, + translationTargetLanguage: translationLang0 ) + let callbackA = AstEventCallback(plugin: self) + let callbackB = AstEventCallback(plugin: self) + astEventCallback = callbackA + + let azureConfigI = AzureConfiguration(subscriptionKey: subscriptionKeyI, region: regionI) + let translationConfigI = TranslationConfiguration(region: azureTranslationRegion, subscriptionKey: azureTranslationKey, location: azureTranslationRegion) + + Task { + await azureAstHelper.initialize(azureConfig: azureConfigI, translationConfig: translationConfigI, serviceConfig: svcA, callback: callbackA) + } + } else if currentAstProvider == "volcano" { + // 初始化豆包 AST(双路) + let cfg = DoubaoE2ETranslateHelper.Config(wsUrl: wsUrl, appKey: appKey, accessKey: accessKeyD, resourceId: resourceId, sourceLanguage: translationLang0, targetLanguage: translationLang1, clientWaitMs: 60_000) + astWsConfig = cfg + let cbA = DoubaoCallbackProxy(plugin: self, serviceId: "A", direction: "\(translationLang0)->\(translationLang1)", targetLanguage: translationLang1) + let cbB = DoubaoCallbackProxy(plugin: self, serviceId: "B", direction: "\(translationLang1)->\(translationLang0)", targetLanguage: translationLang0) + doubaoCallbackA = cbA + doubaoCallbackB = cbB + var cfgA = cfg + cfgA.sourceLanguage = translationLang0 + cfgA.targetLanguage = translationLang1 + var cfgB = cfg + cfgB.sourceLanguage = translationLang1 + cfgB.targetLanguage = translationLang0 + + Task { + _ = await doubaoAstHelperA.initialize(config: cfgA, cb: cbA) + try? await Task.sleep(nanoseconds: 500_000_000) + _ = await doubaoAstHelperB.initialize(config: cfgB, cb: cbB) + } + } else if currentAstProvider == "alibaba" { + // 初始化阿里百炼(双路) + var cfgA = AliyunBailianE2EHelper.Config() + cfgA.wsUrl = alibabaAppURL + cfgA.apiKey = alibabaAppKey + cfgA.appId = "qwen3-livetranslate-flash-realtime" + cfgA.sampleRate = 16000 + cfgA.sourceLanguage = translationLang0 + cfgA.targetLanguage = translationLang1 + cfgA.voice = ttsLang1 + + var cfgB = AliyunBailianE2EHelper.Config() + cfgB.wsUrl = alibabaAppURL + cfgB.apiKey = alibabaAppKey + cfgB.appId = "qwen3-livetranslate-flash-realtime" + cfgB.sampleRate = 16000 + cfgB.sourceLanguage = translationLang1 + cfgB.targetLanguage = translationLang0 + cfgB.voice = ttsLang0 + + let cbA = AliyunCallbackProxy(plugin: self, serviceId: "A", direction: "\(translationLang0)->\(translationLang1)", targetLanguage: translationLang1) + let cbB = AliyunCallbackProxy(plugin: self, serviceId: "B", direction: "\(translationLang1)->\(translationLang0)", targetLanguage: translationLang0) + bailianCallbackA = cbA + bailianCallbackB = cbB + + Task { + _ = await bailianAstHelperA.initialize(config: cfgA, cb: cbA) + _ = await bailianAstHelperB.initialize(config: cfgB, cb: cbB) + } } result(true) - + default: result(FlutterMethodNotImplemented) } } - + + /// 发送AST事件到Flutter层 + internal func sendAstEvent(_ event: [String: Any]) { + if astEventSink == nil { return } + DispatchQueue.main.async { [weak self] in + self?.astEventSink?(event) + } + } + } // MARK: - AST Event Callback /** @@ -1141,13 +1232,14 @@ extension AzureSpeechPlugin: BleService.Callback { azureAsrHelper.audioStream?.saveAudioDataTo(data: rightBuffer) - // // 安全解包版本 - // guard let audioProcessor = azureAstHelper.audioProcessor else { - // os_log("音频流未初始化", type: .error) - // return - // } - // print("liwei--------------接收到音频数据回调1") - azureAstHelper.pushAudioData(audioData: leftBuffer) + // 根据当前 AST 提供商路由音频数据 + if currentAstProvider == "volcano" { + doubaoAstHelperA.pushAudioData(leftBuffer) + } else if currentAstProvider == "alibaba" { + bailianAstHelperA.pushAudioData(leftBuffer) + } else { + azureAstHelper.pushAudioData(audioData: leftBuffer) + } } @@ -1233,6 +1325,203 @@ private class BaseEventStreamHandler: NSObject, FlutterStreamHandler { +// MARK: - Doubao AST 回调桥接 +private class DoubaoCallbackProxy: DoubaoE2ETranslateHelper.Callback { + private weak var plugin: AzureSpeechPlugin? + private let serviceId: String + private let direction: String + private let targetLanguage: String + + init(plugin: AzureSpeechPlugin, serviceId: String, direction: String, targetLanguage: String) { + self.plugin = plugin + self.serviceId = serviceId + self.direction = direction + self.targetLanguage = targetLanguage + } + + func onSessionStarted(sessionId: String) { + plugin?.sendAstEvent([ + "type": "serviceInitialized", + "serviceId": serviceId, + "direction": direction + ]) + } + + func onPartialSourceText(sessionId: String, text: String) { + plugin?.sendAstEvent([ + "type": "recognizing", + "serviceId": serviceId, + "text": text, + "language": direction, + "utteranceId": sessionId + ]) + } + + func onFinalSourceText(sessionId: String, finalText: String) { + if let p = plugin { p.doubaoFinalSourceTextCache["\(serviceId):\(sessionId)"] = finalText } + plugin?.sendAstEvent([ + "type": "recognized", + "serviceId": serviceId, + "text": finalText, + "language": direction, + "utteranceId": sessionId + ]) + } + + func onPartialText(sessionId: String, text: String) { + plugin?.sendAstEvent([ + "type": "translatedInterim", + "serviceId": serviceId, + "translatedText": text, + "targetLanguage": targetLanguage, + "utteranceId": sessionId + ]) + } + + func onPartialAudio(sessionId: String, data: Data) { + let isRightChannel = (serviceId == "A") + if data.count > 0 { + if isRightChannel { + BleService.shared.writeExternalRightAudioData(data) + } else { + BleService.shared.writeExternalLeftAudioData(data) + } + } + } + + func onSessionFinished(sessionId: String, finalText: String, finalAudio: Data) { + let key = "\(serviceId):\(sessionId)" + let original = plugin?.doubaoFinalSourceTextCache.removeValue(forKey: key) ?? "" + plugin?.sendAstEvent([ + "type": "translated", + "serviceId": serviceId, + "translatedText": finalText, + "targetLanguage": targetLanguage, + "utteranceId": sessionId, + "originalText": original + ]) + } + + func onSessionError(sessionId: String, code: Int, message: String) { + plugin?.sendAstEvent([ + "type": "error", + "serviceId": serviceId, + "direction": direction, + "component": "doubaoAst", + "error": message, + "code": code + ]) + } + + func onFinalTranslatedText(sessionId: String, finalText: String) { + let key = "\(serviceId):\(sessionId)" + let original = plugin?.doubaoFinalSourceTextCache.removeValue(forKey: key) ?? "" + plugin?.sendAstEvent([ + "type": "translated", + "serviceId": serviceId, + "translatedText": finalText, + "targetLanguage": targetLanguage, + "utteranceId": sessionId, + "originalText": original + ]) + } +} + +// MARK: - Aliyun Bailian AST 回调桥接 +private class AliyunCallbackProxy: AliyunBailianE2EHelper.Callback { + private weak var plugin: AzureSpeechPlugin? + private let serviceId: String + private let direction: String + private let targetLanguage: String + + init(plugin: AzureSpeechPlugin, serviceId: String, direction: String, targetLanguage: String) { + self.plugin = plugin + self.serviceId = serviceId + self.direction = direction + self.targetLanguage = targetLanguage + } + + func onSessionStarted(sessionId: String) { + plugin?.sendAstEvent([ + "type": "serviceInitialized", + "serviceId": serviceId, + "direction": direction + ]) + } + + func onPartialSourceText(sessionId: String, text: String) { + plugin?.sendAstEvent([ + "type": "recognizing", + "serviceId": serviceId, + "text": text, + "language": direction, + "utteranceId": sessionId + ]) + } + + func onFinalSourceText(sessionId: String, finalText: String) { + plugin?.sendAstEvent([ + "type": "recognized", + "serviceId": serviceId, + "text": finalText, + "language": direction, + "utteranceId": sessionId + ]) + } + + func onPartialText(sessionId: String, text: String) { + plugin?.sendAstEvent([ + "type": "translatedInterim", + "serviceId": serviceId, + "translatedText": text, + "targetLanguage": targetLanguage, + "utteranceId": sessionId + ]) + } + + func onPartialAudio(sessionId: String, data: Data) { + let isRightChannel = (serviceId == "A") + if data.count > 0 { + if isRightChannel { + BleService.shared.writeExternalRightAudioData(data) + } else { + BleService.shared.writeExternalLeftAudioData(data) + } + } + } + + func onSessionFinished(sessionId: String, finalText: String, finalAudio: Data) { + plugin?.sendAstEvent([ + "type": "translated", + "serviceId": serviceId, + "translatedText": finalText, + "targetLanguage": targetLanguage, + "utteranceId": sessionId + ]) + } + + func onSessionError(sessionId: String, code: Int, message: String) { + plugin?.sendAstEvent([ + "type": "error", + "serviceId": serviceId, + "direction": direction, + "component": "aliyunAst", + "error": message, + "code": code + ]) + } + + func onFinalTranslatedText(sessionId: String, finalText: String) { + plugin?.sendAstEvent([ + "type": "translated", + "serviceId": serviceId, + "translatedText": finalText, + "targetLanguage": targetLanguage, + "utteranceId": sessionId + ]) + } +} + // MARK: - TTS 事件监听实现 extension AzureSpeechPlugin: TtsEventListener { // 使用@objc特性为方法提供一个不同的Objective-C选择器名称 diff --git a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt index faabb3e61..857131e9b 100644 --- a/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt +++ b/local_plugins/ble_service/android/src/main/kotlin/com/yunqiinnovation/ble_service/BleService.kt @@ -953,7 +953,7 @@ object BleService { override fun onCharacteristicChanged(g: BluetoothGatt, c: BluetoothGattCharacteristic) { val data = c.value ?: return - Log.i(TAG, "ble指令----------- 收到通知特征数据(命令和控制 uuid=${c.uuid}") + // Log.i(TAG, "0000abc2-0001-1111-2222-123456789abc(命令和控制 uuid=${c.uuid}") // 根据特征UUID区分处理 when (c.uuid) { // 音频特征数据 diff --git a/local_plugins/ble_service/ios/ble_service/Sources/ble_service/BleConst.swift b/local_plugins/ble_service/ios/ble_service/Sources/ble_service/BleConst.swift index d60db3e39..e4fdc1f2d 100644 --- a/local_plugins/ble_service/ios/ble_service/Sources/ble_service/BleConst.swift +++ b/local_plugins/ble_service/ios/ble_service/Sources/ble_service/BleConst.swift @@ -26,7 +26,7 @@ class BleConst { /** 通话写入音频特征UUID - 文档中定义为0000ABC1-0001-1111-2222-123456789ABC */ static let CALL_WRITE_AUDIO_CHAR_UUID = CBUUID(string: "0000abc1-0001-1111-2222-123456789abc") - /** 通话接收音频特征UUID - 文档中定义为0000ABC2-0001-1111-2222-123456789ABC */ + /** 通话接收音频特征UUID - 文档中定义为0000ABC2-0001-1111-2222-123456789ABC */ static let CALL_RECEIVE_AUDIO_CHAR_UUID = CBUUID(string: "0000abc2-0001-1111-2222-123456789abc") /** 客户端特征配置描述符UUID */ static let CLIENT_CHAR_CONFIG_UUID = CBUUID(string: "00002902-0000-1000-8000-00805f9b34fb")