From f8f1e52e82d1922e9c438921874ba8682b0bca75 Mon Sep 17 00:00:00 2001 From: wolfplus Date: Thu, 27 Feb 2025 09:19:12 +0000 Subject: [PATCH] volcano tts --- android/app/build.gradle.kts | 13 +- .../com/example/deep_voice/MainActivity.kt | 105 +-- .../deep_voice/SpeechRecognitionHelper.kt | 569 ++++++++++---- .../example/deep_voice/TextToSpeechHelper.kt | 165 ---- .../example/deep_voice/VolcanoTtsHelper.kt | 412 ++++++++++ android/settings.gradle.kts | 3 - flutter_01.log | 84 -- lib/core/bindings/initial_binding.dart | 12 +- lib/core/routes/app_pages.dart | 20 +- lib/core/routes/app_routes.dart | 2 + lib/core/utils/logger.dart | 62 ++ lib/data/services/azure_tts_service.dart | 354 --------- .../services/background_agent_service.dart | 37 +- lib/data/services/microsoft_tts_service.dart | 346 --------- lib/data/services/volcano_tts_service.dart | 727 +++++++++++------- .../volcano_voice_recognition_service.dart | 381 +++++++++ .../controllers/voice_input_controller.dart | 26 +- .../microsoft_tts_continuous_example.dart | 308 -------- lib/modules/microsoft_tts_example.dart | 202 ----- lib/modules/profile/views/profile_view.dart | 19 + lib/modules/speech_demo/speech_demo_page.dart | 184 ----- lib/modules/test/bindings/test_binding.dart | 21 + .../test/controllers/asr_test_controller.dart | 103 +++ .../test/controllers/tts_test_controller.dart | 123 +++ lib/modules/test/views/asr_test_view.dart | 232 ++++++ lib/modules/test/views/tts_test_view.dart | 267 +++++++ 26 files changed, 2623 insertions(+), 2154 deletions(-) delete mode 100644 android/app/src/main/kotlin/com/example/deep_voice/TextToSpeechHelper.kt create mode 100644 android/app/src/main/kotlin/com/example/deep_voice/VolcanoTtsHelper.kt delete mode 100644 flutter_01.log create mode 100644 lib/core/utils/logger.dart delete mode 100644 lib/data/services/azure_tts_service.dart delete mode 100644 lib/data/services/microsoft_tts_service.dart create mode 100644 lib/data/services/volcano_voice_recognition_service.dart delete mode 100644 lib/modules/microsoft_tts_continuous_example.dart delete mode 100644 lib/modules/microsoft_tts_example.dart delete mode 100644 lib/modules/speech_demo/speech_demo_page.dart create mode 100644 lib/modules/test/bindings/test_binding.dart create mode 100644 lib/modules/test/controllers/asr_test_controller.dart create mode 100644 lib/modules/test/controllers/tts_test_controller.dart create mode 100644 lib/modules/test/views/asr_test_view.dart create mode 100644 lib/modules/test/views/tts_test_view.dart diff --git a/android/app/build.gradle.kts b/android/app/build.gradle.kts index 4c990a7d4..819591ade 100644 --- a/android/app/build.gradle.kts +++ b/android/app/build.gradle.kts @@ -1,6 +1,9 @@ repositories { google() mavenCentral() + maven { + url = uri("https://artifact.bytedance.com/repository/Volcengine/") + } } plugins { @@ -49,8 +52,14 @@ android { dependencies { coreLibraryDesugaring("com.android.tools:desugar_jdk_libs:2.0.4") - implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.42.0") - + // Microsoft 语音识别SDK(可以移除,因为我们现在使用火山语音识别SDK) + // implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.42.0") + + // // 添加火山语音合成SDK依赖 + // implementation("com.bytedance.speechengine:speechengine_tts_tob:5.4.8") + + // 添加火山语音识别SDK依赖 + implementation("com.bytedance.speechengine:speechengine_tob:0.0.5") } flutter { diff --git a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt b/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt index 20c61e635..8fbca00db 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt +++ b/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt @@ -15,12 +15,22 @@ import io.flutter.plugin.common.EventChannel class MainActivity: AudioServiceActivity() { private val SPEECH_RECOGNITION_CHANNEL = "com.example.deep_voice/speech_recognition" private val SPEECH_RECOGNITION_EVENT_CHANNEL = "com.example.deep_voice/speech_recognition_events" - private val TTS_CHANNEL = "com.example.deep_voice/text_to_speech" + private val VOLCANO_TTS_CHANNEL = "com.example.deep_voice/volcano_tts" private val TAG = "MainActivity" private val speechHelper = SpeechRecognitionHelper() - private val ttsHelper = TextToSpeechHelper() + private val volcanoTtsHelper = VolcanoTtsHelper() private var eventSink: EventChannel.EventSink? = null + override fun onCreate(savedInstanceState: Bundle?) { + super.onCreate(savedInstanceState) + + // 设置火山语音合成Helper的上下文 + volcanoTtsHelper.setContext(applicationContext) + + // 设置火山语音识别Helper的上下文 + speechHelper.setContext(applicationContext) + } + override fun configureFlutterEngine(flutterEngine: FlutterEngine) { super.configureFlutterEngine(flutterEngine) @@ -37,9 +47,12 @@ class MainActivity: AudioServiceActivity() { } try { - speechHelper.initialize(subscriptionKey, serviceRegion) - result.success(true) + Log.d(TAG, "初始化语音识别服务,APP_ID: ${subscriptionKey.take(3)}***,APP_KEY: ${serviceRegion.take(5)}...") + val success = speechHelper.initialize(subscriptionKey, serviceRegion) + Log.d(TAG, "语音识别服务初始化${if (success) "成功" else "失败"}") + result.success(success) } catch (e: Exception) { + Log.e(TAG, "初始化语音识别服务失败", e) result.error("INITIALIZATION_ERROR", e.message, null) } } @@ -65,7 +78,7 @@ class MainActivity: AudioServiceActivity() { return@setMethodCallHandler } - speechHelper.startContinuousRecognition(object : SpeechRecognitionHelper.ContinuousRecognizeCallback { + val success = speechHelper.startContinuousRecognition(object : SpeechRecognitionHelper.ContinuousRecognizeCallback { override fun onResult(text: String) { sendEvent(mapOf( "eventType" to "finalResult", @@ -107,7 +120,7 @@ class MainActivity: AudioServiceActivity() { )) } }) - result.success(true) + result.success(success) } catch (e: Exception) { result.error("RECOGNITION_ERROR", e.message, null) } @@ -119,7 +132,7 @@ class MainActivity: AudioServiceActivity() { return@setMethodCallHandler } - speechHelper.stopContinuousRecognition(object : SpeechRecognitionHelper.ContinuousRecognizeCallback { + val success = speechHelper.stopContinuousRecognition(object : SpeechRecognitionHelper.ContinuousRecognizeCallback { override fun onResult(text: String) {} override fun onRecognizing(recognizing: String) {} override fun onSessionStarted() {} @@ -133,7 +146,7 @@ class MainActivity: AudioServiceActivity() { result.error("STOP_ERROR", error, null) } }) - result.success(true) + result.success(success) } catch (e: Exception) { result.error("STOP_ERROR", e.message, null) } @@ -152,51 +165,39 @@ class MainActivity: AudioServiceActivity() { } } - // 设置 TTS 方法通道 - MethodChannel(flutterEngine.dartExecutor.binaryMessenger, TTS_CHANNEL).setMethodCallHandler { call, result -> + + // 设置火山语音合成方法通道 + MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOLCANO_TTS_CHANNEL).setMethodCallHandler { call, result -> when (call.method) { "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") - val serviceRegion = call.argument("serviceRegion") + val appId = call.argument("appId") + val token = call.argument("token") + val cluster = call.argument("cluster") - if (subscriptionKey == null || serviceRegion == null) { - result.error("INVALID_ARGUMENTS", "subscriptionKey and serviceRegion are required", null) + if (appId == null || token == null || cluster == null) { + result.error("INVALID_ARGUMENTS", "appId, token and cluster are required", null) return@setMethodCallHandler } try { - ttsHelper.initialize(subscriptionKey, serviceRegion) - result.success(true) + val success = volcanoTtsHelper.initialize(appId, token, cluster) + result.success(success) } catch (e: Exception) { result.error("INITIALIZATION_ERROR", e.message, null) } } - "setVoice" -> { - val voiceName = call.argument("voiceName") - - if (voiceName == null) { - result.error("INVALID_ARGUMENTS", "voiceName is required", null) - return@setMethodCallHandler - } - - try { - ttsHelper.setVoice(voiceName) - result.success(true) - } catch (e: Exception) { - result.error("SET_VOICE_ERROR", e.message, null) - } - } - "speakText" -> { + "synthesize" -> { val text = call.argument("text") + val voiceType = call.argument("voiceType") - if (text == null) { - result.error("INVALID_ARGUMENTS", "text is required", null) + if (text == null || voiceType == null) { + result.error("INVALID_ARGUMENTS", "text and voiceType are required", null) return@setMethodCallHandler } - ttsHelper.speakText(text, object : TextToSpeechHelper.TTSCallback { - override fun onSuccess(message: String) { - result.success(message) + volcanoTtsHelper.synthesize(text, voiceType, object : VolcanoTtsHelper.VolcanoTtsCallback { + override fun onSuccess(audioData: ByteArray) { + result.success(audioData) } override fun onError(error: String) { @@ -204,27 +205,29 @@ class MainActivity: AudioServiceActivity() { } }) } - "speakSsml" -> { - val ssml = call.argument("ssml") + "synthesizeSync" -> { + val text = call.argument("text") + val voiceType = call.argument("voiceType") - if (ssml == null) { - result.error("INVALID_ARGUMENTS", "ssml is required", null) + if (text == null || voiceType == null) { + result.error("INVALID_ARGUMENTS", "text and voiceType are required", null) return@setMethodCallHandler } - ttsHelper.speakSsml(ssml, object : TextToSpeechHelper.TTSCallback { - override fun onSuccess(message: String) { - result.success(message) - } - - override fun onError(error: String) { - result.error("TTS_ERROR", error, null) + try { + val audioData = volcanoTtsHelper.synthesizeSync(text, voiceType) + if (audioData != null) { + result.success(audioData) + } else { + result.error("TTS_ERROR", "Failed to synthesize text", null) } - }) + } catch (e: Exception) { + result.error("TTS_ERROR", e.message, null) + } } "dispose" -> { try { - ttsHelper.dispose() + volcanoTtsHelper.dispose() result.success(true) } catch (e: Exception) { result.error("DISPOSE_ERROR", e.message, null) @@ -258,7 +261,7 @@ class MainActivity: AudioServiceActivity() { override fun onDestroy() { speechHelper.dispose() - ttsHelper.dispose() + volcanoTtsHelper.dispose() super.onDestroy() } } \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/SpeechRecognitionHelper.kt b/android/app/src/main/kotlin/com/example/deep_voice/SpeechRecognitionHelper.kt index f5f4e7ee7..f737c680e 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/SpeechRecognitionHelper.kt +++ b/android/app/src/main/kotlin/com/example/deep_voice/SpeechRecognitionHelper.kt @@ -1,192 +1,477 @@ package com.example.deep_voice +import android.content.Context import android.util.Log -import com.microsoft.cognitiveservices.speech.* -import com.microsoft.cognitiveservices.speech.audio.* -import com.microsoft.cognitiveservices.speech.util.EventHandler -import java.util.concurrent.ExecutionException -import java.util.function.Consumer +import com.bytedance.speech.speechengine.SpeechEngine +import com.bytedance.speech.speechengine.SpeechEngineDefines +import com.bytedance.speech.speechengine.SpeechEngineGenerator +import org.json.JSONException +import org.json.JSONObject +import java.util.concurrent.CountDownLatch +import java.util.concurrent.TimeUnit -class SpeechRecognitionHelper { - private var recognizer: SpeechRecognizer? = null +/** + * 火山语音识别Helper类 + * + * 该类封装了火山语音SDK的语音识别功能,提供简单的接口供Flutter调用 + */ +class SpeechRecognitionHelper : SpeechEngine.SpeechListener { private val TAG = "SpeechRecognitionHelper" - private var isContinuousRecognitionActive = false - - // 初始化 SDK - fun initialize(subscriptionKey: String, serviceRegion: String) { + private var mSpeechEngine: SpeechEngine? = null + private var mSpeechEngineHandler: Long = -1 + private var isInitialized = false + private var applicationContext: Context? = null + + // 存储应用ID和密钥,以便在错误处理中使用 + private var appId: String = "" + private var appKey: String = "" + + // 同步识别相关变量 + private var mRecognitionLatch: CountDownLatch? = null + private var mRecognitionResult: String? = null + private var mRecognitionError: String? = null + + // 当前回调 + private var mCurrentCallback: RecognizeCallback? = null + private var mCurrentContinuousCallback: ContinuousRecognizeCallback? = null + + // 是否正在进行连续识别 + private var mIsContinuousRecognitionActive = false + + /** + * 初始化语音识别引擎 + * + * @param subscriptionKey 订阅密钥 + * @param serviceRegion 服务区域 + */ + fun initialize(subscriptionKey: String, serviceRegion: String): Boolean { try { - val config = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) - config.speechRecognitionLanguage = "zh-CN" - // 直接使用默认麦克风输入,不传递自定义音频处理选项 - val audioConfig = AudioConfig.fromDefaultMicrophoneInput() - recognizer = SpeechRecognizer(config, audioConfig) - Log.d(TAG, "Speech SDK initialized successfully") -} catch (e: Exception) { - Log.e(TAG, "初始化失败: ${e.message}") -} + // 保存应用ID和密钥,以便在错误处理中使用 + this.appId = subscriptionKey + this.appKey = serviceRegion + + // 确保应用上下文已设置 + if (applicationContext == null) { + Log.e(TAG, "应用上下文未设置,请先调用setContext方法") + return false + } + + // 准备环境 + SpeechEngineGenerator.PrepareEnvironment(applicationContext, null) + + // 获取语音引擎实例 + mSpeechEngine = SpeechEngineGenerator.getInstance() + mSpeechEngineHandler = mSpeechEngine?.createEngine() ?: -1 + + if (mSpeechEngineHandler == -1L) { + Log.e(TAG, "创建语音引擎失败") + return false + } + + // 设置上下文 + mSpeechEngine?.setContext(applicationContext) + + // 设置引擎类型为ASR + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, + SpeechEngineDefines.ASR_ENGINE + ) + + // 设置日志级别 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, + SpeechEngineDefines.LOG_LEVEL_WARN + ) + + // 设置用户ID (必需) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_UID_STRING, + "deep_voice_user" + ) + + // 设置设备ID (可选) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, + "deep_voice_device" + ) + + // 设置授权信息 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, + subscriptionKey + ) + + // 修改Token格式,不再使用"Bearer;"前缀 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, + serviceRegion // 直接使用serviceRegion作为token + ) + + // 设置网络配置 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_ADDRESS_STRING, + "wss://openspeech.bytedance.com" + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_URI_STRING, + "/api/v2/asr" + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_CLUSTER_STRING, + serviceRegion + ) + + // 设置音频来源为内置录音机 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_RECORDER_TYPE_STRING, + SpeechEngineDefines.RECORDER_TYPE_RECORDER + ) + + // 启用音量获取 + mSpeechEngine?.setOptionBoolean( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ENABLE_GET_VOLUME_BOOL, + true + ) + + // 设置最大录音时长为60秒 + mSpeechEngine?.setOptionInt( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_VAD_MAX_SPEECH_DURATION_INT, + 60000 + ) + + // 控制识别效果 + mSpeechEngine?.setOptionBoolean( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_ENABLE_DDC_BOOL, + true + ) + mSpeechEngine?.setOptionBoolean( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_SHOW_NLU_PUNC_BOOL, + true + ) + + // 设置识别结果形式为全量返回(适合一句话识别) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_RESULT_TYPE_STRING, + SpeechEngineDefines.ASR_RESULT_TYPE_FULL + ) + + // 初始化引擎 + val ret = mSpeechEngine?.initEngine(mSpeechEngineHandler) ?: -1 + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + Log.e(TAG, "初始化引擎失败: $ret") + return false + } + + // 设置监听器 + mSpeechEngine?.setListener(this) + + isInitialized = true + Log.d(TAG, "火山语音识别引擎初始化成功") + return true + } catch (e: Exception) { + Log.e(TAG, "火山语音识别引擎初始化失败: ${e.message}") + e.printStackTrace() + return false + } + } + + /** + * 设置应用上下文 + */ + fun setContext(context: Context) { + applicationContext = context.applicationContext } - // 开始一次性语音识别 + /** + * 一次性识别 + * + * @param callback 回调接口,用于返回结果或错误 + */ fun recognizeOnce(callback: RecognizeCallback) { - if (recognizer == null) { - callback.onError("SpeechRecognizer 未初始化") + if (!isInitialized) { + callback.onError("语音识别引擎尚未初始化") return } try { - // 使用同步方式调用,避免 CompletableFuture 的兼容性问题 - val result = recognizer?.recognizeOnceAsync()?.get() + Log.d(TAG, "开始一次性识别") - if (result != null) { - when (result.reason) { - ResultReason.RecognizedSpeech -> { - callback.onResult(result.text) - } - else -> { - callback.onError("识别失败,原因: ${result.reason}") - } - } - } else { - callback.onError("识别结果为空") + // 保存回调以便在onMessage中使用 + mCurrentCallback = callback + + // 设置识别结果形式为全量返回(适合一句话识别) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_RESULT_TYPE_STRING, + SpeechEngineDefines.ASR_RESULT_TYPE_FULL + ) + + // 先停止引擎,避免SDK内部异步线程带来的问题 + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎 + val ret = mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + callback.onError("启动引擎失败: $ret") } } catch (e: Exception) { - when (e) { - is InterruptedException, is ExecutionException -> { - Log.e(TAG, "识别异常: ${e.message}") - callback.onError("识别异常: ${e.message}") - } - else -> { - Log.e(TAG, "未知异常: ${e.message}") - callback.onError("未知异常: ${e.message}") - } - } + Log.e(TAG, "语音识别异常: ${e.message}") + callback.onError("语音识别异常: ${e.message}") } } - // 开始连续语音识别 - fun startContinuousRecognition(callback: ContinuousRecognizeCallback) { - if (recognizer == null) { - callback.onError("SpeechRecognizer 未初始化") - return + /** + * 开始连续识别 + * + * @param callback 回调接口,用于返回结果或错误 + */ + fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (!isInitialized) { + callback.onError("语音识别引擎尚未初始化") + return false } - - if (isContinuousRecognitionActive) { + + if (mIsContinuousRecognitionActive) { callback.onError("连续识别已经在进行中") - return + return false } - + try { - // 设置识别事件处理 - recognizer?.let { recognizer -> - // 设置识别事件处理 - recognizer.recognized.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizedSpeech) { - callback.onResult(event.result.text) - } - } - ) - - // 设置识别中事件处理(实时反馈) - recognizer.recognizing.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizingSpeech) { - callback.onRecognizing(event.result.text) - } - } - ) - - // 设置会话开始事件处理 - recognizer.sessionStarted.addEventListener( - EventHandler { _, _ -> - callback.onSessionStarted() - } - ) - - // 设置会话结束事件处理 - recognizer.sessionStopped.addEventListener( - EventHandler { _, _ -> - isContinuousRecognitionActive = false - callback.onSessionStopped() - } - ) - - // 设置取消事件处理 - recognizer.canceled.addEventListener( - EventHandler { _, event -> - val reason = event.reason - val errorDetails = if (reason == CancellationReason.Error) event.errorDetails else "" - callback.onCanceled(reason.toString(), errorDetails) - isContinuousRecognitionActive = false - } - ) - - // 开始连续识别 - recognizer.startContinuousRecognitionAsync().get() - isContinuousRecognitionActive = true - Log.d(TAG, "连续识别已开始") + Log.d(TAG, "开始连续识别") + + // 保存回调以便在onMessage中使用 + mCurrentContinuousCallback = callback + + // 设置识别结果形式为增量返回(适合连续识别) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ASR_RESULT_TYPE_STRING, + SpeechEngineDefines.ASR_RESULT_TYPE_SINGLE + ) + + // 先停止引擎,避免SDK内部异步线程带来的问题 + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎 + val ret = mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + callback.onError("启动引擎失败: $ret") + return false } - + + mIsContinuousRecognitionActive = true + return true } catch (e: Exception) { - Log.e(TAG, "开始连续识别失败: ${e.message}") - callback.onError("开始连续识别失败: ${e.message}") - isContinuousRecognitionActive = false + Log.e(TAG, "开始连续识别异常: ${e.message}") + callback.onError("开始连续识别异常: ${e.message}") + return false } } - - // 停止连续语音识别 - fun stopContinuousRecognition(callback: ContinuousRecognizeCallback) { - if (recognizer == null) { - callback.onError("SpeechRecognizer 未初始化") - return + + /** + * 停止连续识别 + * + * @param callback 回调接口,用于返回结果或错误 + */ + fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (!isInitialized) { + callback.onError("语音识别引擎尚未初始化") + return false } - - if (!isContinuousRecognitionActive) { + + if (!mIsContinuousRecognitionActive) { callback.onError("连续识别未在进行中") - return + return false } - + try { - // 停止连续识别 - recognizer?.stopContinuousRecognitionAsync()?.get() - isContinuousRecognitionActive = false - Log.d(TAG, "连续识别已停止") - callback.onSessionStopped() + Log.d(TAG, "停止连续识别") + + // 音频输入完成 + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_FINISH_TALKING, "") + + // 停止引擎 + val ret = mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + callback.onError("停止引擎失败: $ret") + return false + } + + mIsContinuousRecognitionActive = false + return true } catch (e: Exception) { - Log.e(TAG, "停止连续识别失败: ${e.message}") - callback.onError("停止连续识别失败: ${e.message}") + Log.e(TAG, "停止连续识别异常: ${e.message}") + callback.onError("停止连续识别异常: ${e.message}") + return false } } - - // 检查连续识别是否活跃 + + /** + * 检查连续识别是否活跃 + */ fun isContinuousRecognitionActive(): Boolean { - return isContinuousRecognitionActive + return mIsContinuousRecognitionActive } - // 释放资源 + /** + * 释放资源 + */ fun dispose() { try { - if (isContinuousRecognitionActive) { - recognizer?.stopContinuousRecognitionAsync()?.get() - isContinuousRecognitionActive = false + if (mSpeechEngineHandler != -1L) { + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + mSpeechEngine?.destroyEngine(mSpeechEngineHandler) + mSpeechEngineHandler = -1 } - recognizer?.close() - recognizer = null - Log.d(TAG, "语音识别资源已释放") + mSpeechEngine = null + isInitialized = false + mIsContinuousRecognitionActive = false + Log.d(TAG, "火山语音识别引擎已释放") } catch (e: Exception) { - Log.e(TAG, "释放资源失败: ${e.message}") + Log.e(TAG, "释放火山语音识别引擎失败: ${e.message}") } } - - // 一次性识别回调接口 + + // 实现SpeechListener接口 + override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { + val stdData = String(data) + + when (type) { + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { + Log.d(TAG, "引擎启动成功: $stdData") + mCurrentContinuousCallback?.onSessionStarted() + } + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { + Log.d(TAG, "引擎已停止: $stdData") + mIsContinuousRecognitionActive = false + mCurrentContinuousCallback?.onSessionStopped() + } + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { + Log.e(TAG, "引擎错误: $stdData") + try { + val errorJson = JSONObject(stdData) + val errorCode = errorJson.optInt("err_code", -1) + val errorMsg = errorJson.optString("err_msg", "未知错误") + + // 检查是否是资源授权错误 + if (errorCode == 1001 && (errorMsg.contains("requested resource not granted") || + errorMsg.contains("requested grant not found"))) { + Log.e(TAG, "资源授权错误: 应用可能未开通语音识别服务或密钥无效") + + // 记录更多调试信息,但隐藏敏感信息 + Log.d(TAG, "APP_ID长度: ${appId.length}") + Log.d(TAG, "APP_KEY长度: ${appKey.length}") + + val errorMessage = "资源授权错误: 请确保应用已开通语音识别服务并且密钥有效 (错误码: $errorCode)" + + if (mIsContinuousRecognitionActive) { + mCurrentContinuousCallback?.onError(errorMessage) + } else { + mCurrentCallback?.onError(errorMessage) + } + } else { + // 其他错误 + if (mIsContinuousRecognitionActive) { + mCurrentContinuousCallback?.onError("引擎错误: $errorMsg (错误码: $errorCode)") + } else { + mCurrentCallback?.onError("引擎错误: $errorMsg (错误码: $errorCode)") + } + } + + mRecognitionError = "引擎错误: $stdData" + mRecognitionLatch?.countDown() + } catch (e: JSONException) { + Log.e(TAG, "解析错误信息失败: ${e.message}") + if (mIsContinuousRecognitionActive) { + mCurrentContinuousCallback?.onError("引擎错误: $stdData") + } else { + mCurrentCallback?.onError("引擎错误: $stdData") + } + } + } + SpeechEngineDefines.MESSAGE_TYPE_PARTIAL_RESULT -> { + Log.d(TAG, "中间识别结果: $stdData") + processRecognitionResult(stdData, false) + } + SpeechEngineDefines.MESSAGE_TYPE_FINAL_RESULT -> { + Log.d(TAG, "最终识别结果: $stdData") + processRecognitionResult(stdData, true) + } + SpeechEngineDefines.MESSAGE_TYPE_VOLUME_LEVEL -> { + // 音量级别,可用于显示波形 + // Log.d(TAG, "音量级别: $stdData") + } + } + } + + /** + * 处理识别结果 + */ + private fun processRecognitionResult(resultData: String, isFinal: Boolean) { + try { + val resultJson = JSONObject(resultData) + if (!resultJson.has("result")) { + return + } + + val resultArray = resultJson.getJSONArray("result") + if (resultArray.length() == 0) { + return + } + + val resultObj = resultArray.getJSONObject(0) + val text = resultObj.optString("text", "") + + if (text.isEmpty()) { + return + } + + if (mIsContinuousRecognitionActive) { + if (isFinal) { + mCurrentContinuousCallback?.onResult(text) + } else { + mCurrentContinuousCallback?.onRecognizing(text) + } + } else { + if (isFinal) { + mCurrentCallback?.onResult(text) + mRecognitionResult = text + mRecognitionLatch?.countDown() + } + } + } catch (e: JSONException) { + Log.e(TAG, "解析识别结果失败: ${e.message}") + } + } + + /** + * 一次性识别回调接口 + */ interface RecognizeCallback { - fun onResult(result: String) + fun onResult(text: String) fun onError(error: String) } - - // 连续识别回调接口 + + /** + * 连续识别回调接口 + */ interface ContinuousRecognizeCallback { - fun onResult(result: String) + fun onResult(text: String) fun onRecognizing(recognizing: String) fun onSessionStarted() fun onSessionStopped() diff --git a/android/app/src/main/kotlin/com/example/deep_voice/TextToSpeechHelper.kt b/android/app/src/main/kotlin/com/example/deep_voice/TextToSpeechHelper.kt deleted file mode 100644 index 433fb1c1d..000000000 --- a/android/app/src/main/kotlin/com/example/deep_voice/TextToSpeechHelper.kt +++ /dev/null @@ -1,165 +0,0 @@ -package com.example.deep_voice - -import android.util.Log -import com.microsoft.cognitiveservices.speech.* -import java.util.concurrent.Future - -/** - * Microsoft Text-to-Speech Helper - * - * 该类封装了微软语音 SDK 的 TTS 功能,提供简单的接口供 Flutter 调用 - */ -class TextToSpeechHelper { - private val TAG = "TextToSpeechHelper" - private var speechConfig: SpeechConfig? = null - private var synthesizer: SpeechSynthesizer? = null - private var isInitialized = false - - /** - * 初始化 TTS 引擎 - * - * @param subscriptionKey Azure 语音服务订阅密钥 - * @param serviceRegion Azure 语音服务区域 - */ - fun initialize(subscriptionKey: String, serviceRegion: String) { - try { - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) - // 默认设置中文女声 - speechConfig?.setSpeechSynthesisVoiceName("zh-CN-XiaoxiaoNeural") - synthesizer = SpeechSynthesizer(speechConfig) - isInitialized = true - Log.d(TAG, "TTS 引擎初始化成功") - } catch (e: Exception) { - Log.e(TAG, "TTS 引擎初始化失败: ${e.message}") - throw e - } - } - - /** - * 设置语音 - * - * @param voiceName 语音名称,例如 "zh-CN-XiaoxiaoNeural" - */ - fun setVoice(voiceName: String) { - if (!isInitialized) { - throw Exception("TTS 引擎尚未初始化") - } - - try { - speechConfig?.setSpeechSynthesisVoiceName(voiceName) - // 重新创建合成器以应用新的语音设置 - synthesizer?.close() - synthesizer = SpeechSynthesizer(speechConfig) - Log.d(TAG, "已设置语音: $voiceName") - } catch (e: Exception) { - Log.e(TAG, "设置语音失败: ${e.message}") - throw e - } - } - - /** - * 合成文本为语音并播放 - * - * @param text 要合成的文本 - * @param callback 回调接口,用于返回结果或错误 - */ - fun speakText(text: String, callback: TTSCallback) { - if (!isInitialized) { - callback.onError("TTS 引擎尚未初始化") - return - } - - try { - Log.d(TAG, "开始合成文本: $text") - val task: Future = synthesizer!!.SpeakTextAsync(text) - - // 异步获取结果 - val result = task.get() - - when (result.reason) { - ResultReason.SynthesizingAudioCompleted -> { - Log.d(TAG, "语音合成完成") - callback.onSuccess("语音合成完成") - } - ResultReason.Canceled -> { - val cancellation = SpeechSynthesisCancellationDetails.fromResult(result) - Log.e(TAG, "语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") - callback.onError("语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") - } - else -> { - Log.e(TAG, "语音合成失败: ${result.reason}") - callback.onError("语音合成失败: ${result.reason}") - } - } - - result.close() - } catch (e: Exception) { - Log.e(TAG, "语音合成异常: ${e.message}") - callback.onError("语音合成异常: ${e.message}") - } - } - - /** - * 合成 SSML 为语音并播放 - * - * @param ssml SSML 格式的文本 - * @param callback 回调接口,用于返回结果或错误 - */ - fun speakSsml(ssml: String, callback: TTSCallback) { - if (!isInitialized) { - callback.onError("TTS 引擎尚未初始化") - return - } - - try { - Log.d(TAG, "开始合成 SSML") - val task: Future = synthesizer!!.SpeakSsmlAsync(ssml) - - // 异步获取结果 - val result = task.get() - - when (result.reason) { - ResultReason.SynthesizingAudioCompleted -> { - Log.d(TAG, "语音合成完成") - callback.onSuccess("语音合成完成") - } - ResultReason.Canceled -> { - val cancellation = SpeechSynthesisCancellationDetails.fromResult(result) - Log.e(TAG, "语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") - callback.onError("语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") - } - else -> { - Log.e(TAG, "语音合成失败: ${result.reason}") - callback.onError("语音合成失败: ${result.reason}") - } - } - - result.close() - } catch (e: Exception) { - Log.e(TAG, "语音合成异常: ${e.message}") - callback.onError("语音合成异常: ${e.message}") - } - } - - /** - * 释放资源 - */ - fun dispose() { - try { - synthesizer?.close() - speechConfig?.close() - isInitialized = false - Log.d(TAG, "TTS 引擎已释放") - } catch (e: Exception) { - Log.e(TAG, "释放 TTS 引擎失败: ${e.message}") - } - } - - /** - * TTS 回调接口 - */ - interface TTSCallback { - fun onSuccess(message: String) - fun onError(error: String) - } -} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoTtsHelper.kt b/android/app/src/main/kotlin/com/example/deep_voice/VolcanoTtsHelper.kt new file mode 100644 index 000000000..f66ea7447 --- /dev/null +++ b/android/app/src/main/kotlin/com/example/deep_voice/VolcanoTtsHelper.kt @@ -0,0 +1,412 @@ +package com.example.deep_voice + +import android.content.Context +import android.util.Log +import com.bytedance.speech.speechengine.SpeechEngine +import com.bytedance.speech.speechengine.SpeechEngineDefines +import com.bytedance.speech.speechengine.SpeechEngineGenerator +import java.util.concurrent.CountDownLatch +import java.util.concurrent.TimeUnit + +/** + * 火山语音合成Helper类 (SDK版本) + * + * 该类封装了火山语音SDK的TTS功能,提供简单的接口供Flutter调用 + */ +class VolcanoTtsHelper : SpeechEngine.SpeechListener { + private val TAG = "VolcanoTtsHelper" + private var mSpeechEngine: SpeechEngine? = null + private var mSpeechEngineHandler: Long = -1 + private var isInitialized = false + private var applicationContext: Context? = null + + // 同步合成相关变量 + private var mSynthesisLatch: CountDownLatch? = null + private var mSynthesisAudioData: ByteArray? = null + private var mSynthesisError: String? = null + + // 当前回调 + private var mCurrentCallback: VolcanoTtsCallback? = null + + /** + * 初始化TTS引擎 + * + * @param appId 火山语音服务AppID + * @param token 火山语音服务Token + * @param cluster 火山语音服务集群 + */ + fun initialize(appId: String, token: String, cluster: String): Boolean { + try { + // 确保应用上下文已设置 + if (applicationContext == null) { + Log.e(TAG, "应用上下文未设置,请先调用setContext方法") + return false + } + + // 准备环境 + SpeechEngineGenerator.PrepareEnvironment(applicationContext, null) + + // 获取语音引擎实例 + mSpeechEngine = SpeechEngineGenerator.getInstance() + mSpeechEngineHandler = mSpeechEngine?.createEngine() ?: -1 + + if (mSpeechEngineHandler == -1L) { + Log.e(TAG, "创建语音引擎失败") + return false + } + + // 设置上下文 + mSpeechEngine?.setContext(applicationContext) + + // 设置引擎类型为TTS + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, + SpeechEngineDefines.TTS_ENGINE + ) + + // 设置日志级别 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, + SpeechEngineDefines.LOG_LEVEL_WARN + ) + + // 设置用户ID (必需) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_UID_STRING, + "deep_voice_user" + ) + + // 设置设备ID (可选) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, + "deep_voice_device" + ) + + // 设置授权信息 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, + appId + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, + "Bearer;$token" + ) + + // 设置集群 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_CLUSTER_STRING, + cluster + ) + + // 设置合成场景为单次合成 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_SCENARIO_STRING, + SpeechEngineDefines.TTS_SCENARIO_TYPE_NORMAL + ) + + // 设置合成策略为在线合成 + mSpeechEngine?.setOptionInt( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_WORK_MODE_INT, + SpeechEngineDefines.TTS_WORK_MODE_ONLINE + ) + + // 设置在线请求资源配置 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_ADDRESS_STRING, + "wss://openspeech.bytedance.com" + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_URI_STRING, + "/api/v1/tts/ws_binary" + ) + + // 启用播放器 + mSpeechEngine?.setOptionBoolean( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_ENABLE_PLAYER_BOOL, + true + ) + + // 设置音频流类型为媒体 + mSpeechEngine?.setOptionInt( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_AUDIO_STREAM_TYPE_INT, + SpeechEngineDefines.AUDIO_STREAM_TYPE_MEDIA + ) + + // 设置音频格式为PCM + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + "tts_audio_format", + "pcm" + ) + + // 初始化引擎 + val ret = mSpeechEngine?.initEngine(mSpeechEngineHandler) ?: -1 + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + Log.e(TAG, "初始化引擎失败: $ret") + return false + } + + // 设置监听器 + mSpeechEngine?.setListener(this) + + isInitialized = true + Log.d(TAG, "火山语音TTS引擎初始化成功") + return true + } catch (e: Exception) { + Log.e(TAG, "火山语音TTS引擎初始化失败: ${e.message}") + e.printStackTrace() + return false + } + } + + /** + * 设置应用上下文 + */ + fun setContext(context: Context) { + applicationContext = context.applicationContext + } + + /** + * 合成文本为语音 + * + * @param text 要合成的文本 + * @param voiceType 语音类型 + * @param callback 回调接口,用于返回结果或错误 + */ + fun synthesize(text: String, voiceType: String, callback: VolcanoTtsCallback) { + if (!isInitialized) { + callback.onError("TTS引擎尚未初始化") + return + } + + try { + Log.d(TAG, "开始合成文本: $text") + + // 保存回调以便在onMessage中使用 + mCurrentCallback = callback + + // 设置发音人 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, + voiceType + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_TYPE_ONLINE_STRING, + "common" + ) + + // 设置文本 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, + text + ) + + // 设置音频格式为PCM + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + "tts_audio_format", + "pcm" + ) + + // 启用音频数据回调 + mSpeechEngine?.setOptionInt( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_DATA_CALLBACK_MODE_INT, + 2 + ) + + // 先停止引擎,避免SDK内部异步线程带来的问题 + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎,在单次合成场景下,这会自动开始合成 + val ret = mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + callback.onError("启动引擎失败: $ret") + } + } catch (e: Exception) { + Log.e(TAG, "语音合成异常: ${e.message}") + callback.onError("语音合成异常: ${e.message}") + } + } + + /** + * 同步合成文本为语音 + * + * @param text 要合成的文本 + * @param voiceType 语音类型 + * @return 合成的音频数据,如果失败则返回null + */ + fun synthesizeSync(text: String, voiceType: String): ByteArray? { + if (!isInitialized) { + Log.e(TAG, "TTS引擎尚未初始化") + return null + } + + try { + Log.d(TAG, "开始同步合成文本: $text") + + // 重置同步变量 + mSynthesisLatch = CountDownLatch(1) + mSynthesisAudioData = null + mSynthesisError = null + + // 设置发音人 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, + voiceType + ) + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_TYPE_ONLINE_STRING, + "common" + ) + + // 设置文本 + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, + text + ) + + // 设置音频格式为PCM + mSpeechEngine?.setOptionString( + mSpeechEngineHandler, + "tts_audio_format", + "pcm" + ) + + // 启用音频数据回调 + mSpeechEngine?.setOptionInt( + mSpeechEngineHandler, + SpeechEngineDefines.PARAMS_KEY_TTS_DATA_CALLBACK_MODE_INT, + 2 + ) + + // 先停止引擎,避免SDK内部异步线程带来的问题 + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎,在单次合成场景下,这会自动开始合成 + val ret = mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + Log.e(TAG, "启动引擎失败: $ret") + return null + } + + // 等待合成完成,最多等待10秒 + mSynthesisLatch?.await(10, TimeUnit.SECONDS) + + if (mSynthesisError != null) { + Log.e(TAG, "同步语音合成失败: ${mSynthesisError}") + return null + } + + if (mSynthesisAudioData == null) { + Log.e(TAG, "同步语音合成超时或返回空数据") + return null + } + + Log.d(TAG, "同步语音合成成功,音频数据大小: ${mSynthesisAudioData?.size} bytes") + return mSynthesisAudioData + } catch (e: Exception) { + Log.e(TAG, "同步语音合成异常: ${e.message}") + return null + } + } + + /** + * 释放资源 + */ + fun dispose() { + try { + if (mSpeechEngineHandler != -1L) { + mSpeechEngine?.sendDirective(mSpeechEngineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + mSpeechEngine?.destroyEngine(mSpeechEngineHandler) + mSpeechEngineHandler = -1 + } + mSpeechEngine = null + isInitialized = false + Log.d(TAG, "火山语音TTS引擎已释放") + } catch (e: Exception) { + Log.e(TAG, "释放火山语音TTS引擎失败: ${e.message}") + } + } + + // 实现SpeechListener接口 + override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { + val stdData = String(data) + + when (type) { + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { + Log.d(TAG, "引擎启动成功: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { + Log.d(TAG, "引擎已停止: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { + Log.e(TAG, "引擎错误: $stdData") + mCurrentCallback?.onError("引擎错误: $stdData") + mSynthesisError = "引擎错误: $stdData" + mSynthesisLatch?.countDown() + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_SYNTHESIS_BEGIN -> { + Log.d(TAG, "合成开始: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_SYNTHESIS_END -> { + Log.d(TAG, "合成结束: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_START_PLAYING -> { + Log.d(TAG, "开始播放: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_FINISH_PLAYING -> { + Log.d(TAG, "播放结束: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_AUDIO_DATA -> { + Log.d(TAG, "收到音频数据") + // 处理音频数据 + if (data.isNotEmpty()) { + try { + mCurrentCallback?.onSuccess(data) + mSynthesisAudioData = data + mSynthesisLatch?.countDown() + } catch (e: Exception) { + Log.e(TAG, "处理音频数据失败: ${e.message}") + mCurrentCallback?.onError("处理音频数据失败: ${e.message}") + mSynthesisError = "处理音频数据失败: ${e.message}" + mSynthesisLatch?.countDown() + } + } + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_PLAYBACK_PROGRESS -> { + Log.d(TAG, "播放进度: $stdData") + } + SpeechEngineDefines.MESSAGE_TYPE_TTS_SYNTHESIS_MODE_TOGGLE -> { + Log.d(TAG, "合成模式切换: $stdData") + } + } + } + + /** + * 火山语音TTS回调接口 + */ + interface VolcanoTtsCallback { + fun onSuccess(audioData: ByteArray) + fun onError(error: String) + } +} \ No newline at end of file diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts index 20cc3a2da..a439442c2 100644 --- a/android/settings.gradle.kts +++ b/android/settings.gradle.kts @@ -12,9 +12,6 @@ pluginManagement { repositories { google() mavenCentral() - maven { - url = uri("https://artifact.bytedance.com/repository/Volcengine/") - } gradlePluginPortal() } } diff --git a/flutter_01.log b/flutter_01.log deleted file mode 100644 index 064d75574..000000000 --- a/flutter_01.log +++ /dev/null @@ -1,84 +0,0 @@ -Flutter crash report. -Please report a bug at https://github.com/flutter/flutter/issues. - -## command - -flutter run --machine --start-paused -d 3063725814005V5 --devtools-server-address http://127.0.0.1:9100/ --target /Users/wolfplus/Developer/deep_voice/lib/main.dart - -## exception - -RPCError: setIsolatePauseMode: (112) Service has disappeared - -``` -#0 new _OutstandingRequest (package:vm_service/src/vm_service.dart:265:34) -#1 VmService._call. (package:vm_service/src/vm_service.dart:1921:25) -#2 VmService._call (package:vm_service/src/vm_service.dart:1933:8) -#3 VmService.setIsolatePauseMode (package:vm_service/src/vm_service.dart:1685:7) -#4 HotRunner._restartFromSources. (package:flutter_tools/src/run_hot.dart:662:43) -#5 _rootRunUnary (dart:async/zone.dart:1538:47) -#6 Future._propagateToListeners.handleValueCallback (dart:async/future_impl.dart:932:45) -#7 Future._propagateToListeners (dart:async/future_impl.dart:961:13) -#8 Future._completeWithValue (dart:async/future_impl.dart:712:5) -#9 Future._asyncCompleteWithValue. (dart:async/future_impl.dart:792:7) -#10 _rootRun (dart:async/zone.dart:1525:13) -#11 _CustomZone.run (dart:async/zone.dart:1422:19) -#12 _CustomZone.runGuarded (dart:async/zone.dart:1321:7) -#13 _CustomZone.bindCallbackGuarded. (dart:async/zone.dart:1362:23) -#14 _microtaskLoop (dart:async/schedule_microtask.dart:40:21) -#15 _startMicrotaskLoop (dart:async/schedule_microtask.dart:49:5) -#16 _runPendingImmediateCallback (dart:isolate-patch/isolate_patch.dart:128:13) -#17 _RawReceivePort._handleMessage (dart:isolate-patch/isolate_patch.dart:195:5) -``` - -## flutter doctor - -``` -[!] Flutter (Channel stable, 3.29.0, on macOS 15.3.1 24D70 darwin-arm64, locale en-US) [179ms] - • Flutter version 3.29.0 on channel stable at /Users/wolfplus/Developer/SDKs/flutter - ! The flutter binary is not on your path. Consider adding /Users/wolfplus/Developer/SDKs/flutter/bin to your path. - ! The dart binary is not on your path. Consider adding /Users/wolfplus/Developer/SDKs/flutter/bin to your path. - • Upstream repository https://github.com/flutter/flutter.git - • Framework revision 35c388afb5 (2 weeks ago), 2025-02-10 12:48:41 -0800 - • Engine revision f73bfc4522 - • Dart version 3.7.0 - • DevTools version 2.42.2 - • If those were intentional, you can disregard the above warnings; however it is recommended to use "git" directly to perform update checks and upgrades. - -[✓] Android toolchain - develop for Android devices (Android SDK version 35.0.1) [1,851ms] - • Android SDK at /Users/wolfplus/Library/Android/sdk - • Platform android-35, build-tools 35.0.1 - • ANDROID_HOME = /Users/wolfplus/Library/Android/sdk - • Java binary at: /Applications/Android Studio.app/Contents/jbr/Contents/Home/bin/java - This is the JDK bundled with the latest Android Studio installation on this machine. - To manually set the JDK path, use: `flutter config --jdk-dir="path/to/jdk"`. - • Java version OpenJDK Runtime Environment (build 21.0.4+-12422083-b607.1) - • All Android licenses accepted. - -[✓] Xcode - develop for iOS and macOS (Xcode 16.2) [663ms] - • Xcode at /Applications/Xcode.app/Contents/Developer - • Build 16C5032a - • CocoaPods version 1.16.2 - -[✓] Chrome - develop for the web [2ms] - • Chrome at /Applications/Google Chrome.app/Contents/MacOS/Google Chrome - -[✓] Android Studio (version 2024.2) [2ms] - • Android Studio at /Applications/Android Studio.app/Contents - • Flutter plugin can be installed from: - 🔨 https://plugins.jetbrains.com/plugin/9212-flutter - • Dart plugin can be installed from: - 🔨 https://plugins.jetbrains.com/plugin/6351-dart - • Java version OpenJDK Runtime Environment (build 21.0.4+-12422083-b607.1) - -[✓] Connected device (5 available) [5.6s] - • V2049A (mobile) • 3063725814005V5 • android-arm64 • Android 14 (API 34) - • Chris’s iPhone16 (wireless) (mobile) • 00008140-0006402C3CF3001C • ios • iOS 18.2.1 22C161 - • macOS (desktop) • macos • darwin-arm64 • macOS 15.3.1 24D70 darwin-arm64 - • Mac Designed for iPad (desktop) • mac-designed-for-ipad • darwin • macOS 15.3.1 24D70 darwin-arm64 - • Chrome (web) • chrome • web-javascript • Google Chrome 133.0.6943.127 - -[✓] Network resources [550ms] - • All expected network resources are available. - -! Doctor found issues in 1 category. -``` diff --git a/lib/core/bindings/initial_binding.dart b/lib/core/bindings/initial_binding.dart index ce9b781a9..0ea83b235 100644 --- a/lib/core/bindings/initial_binding.dart +++ b/lib/core/bindings/initial_binding.dart @@ -3,8 +3,9 @@ import '../../core/controllers/permission_controller.dart'; import '../../data/services/audio_service.dart'; import '../../data/services/notification_service.dart'; import '../../data/services/volcano_ai_service.dart'; -import '../../data/services/voice_recognition_service.dart'; +import '../../data/services/volcano_voice_recognition_service.dart'; import '../../data/services/volcano_tts_service.dart'; +import '../../core/utils/logger.dart'; /// 初始绑定,用于管理全局依赖 class InitialBinding extends Bindings { @@ -14,7 +15,7 @@ class InitialBinding extends Bindings { final notificationService = Get.put(NotificationService(), permanent: true); final audioService = Get.put(AudioServiceManager(), permanent: true); final volcanoAiService = Get.put(VolcanoAIService(), permanent: true); - final voiceRecognitionService = Get.put(VoiceRecognitionService(), permanent: true); + final voiceRecognitionService = Get.put(VolcanoVoiceRecognitionService(), permanent: true); final volcanoTtsService = Get.put(VolcanoTtsService(), permanent: true); // 只注册全局控制器 @@ -29,7 +30,7 @@ class InitialBinding extends Bindings { try { // 首先初始化通知服务 await notificationService.init().catchError((error) { - print('通知服务初始化失败: $error'); + Logger.error('通知服务初始化失败', error); return null; }); @@ -38,11 +39,10 @@ class InitialBinding extends Bindings { // 然后初始化音频服务 await audioService.init().then((_) { - print('AudioServiceManager 初始化完成,可以接收蓝牙耳机按键事件'); - + Logger.info('AudioServiceManager 初始化完成,可以接收蓝牙耳机按键事件'); }); } catch (e) { - print('服务初始化过程中发生错误: $e'); + Logger.error('服务初始化过程中发生错误', e); } // 其他服务初始化可以在这里添加 diff --git a/lib/core/routes/app_pages.dart b/lib/core/routes/app_pages.dart index 43c181749..784c85da8 100644 --- a/lib/core/routes/app_pages.dart +++ b/lib/core/routes/app_pages.dart @@ -7,7 +7,10 @@ import '../../modules/profile/views/profile_view.dart'; import '../../modules/profile/bindings/profile_binding.dart'; import '../../modules/chat/bindings/chat_binding.dart'; import '../../modules/chat/views/chat_view.dart'; -import '../../modules/speech_demo/speech_demo_page.dart'; +import '../../modules/test/views/tts_test_view.dart'; +import '../../modules/test/views/asr_test_view.dart'; +import '../../modules/test/bindings/test_binding.dart'; +// import '../../modules/speech_demo/speech_demo_page.dart'; // This file doesn't exist import './app_routes.dart'; abstract class AppPages { @@ -32,9 +35,20 @@ abstract class AppPages { page: () => const ProfileView(), binding: ProfileBinding(), ), + // Commented out because SpeechDemoPage doesn't exist + // GetPage( + // name: Routes.speechDemo, + // page: () => const SpeechDemoPage(), + // ), GetPage( - name: Routes.speechDemo, - page: () => const SpeechDemoPage(), + name: Routes.TTS_TEST, + page: () => TtsTestView(), + binding: TtsTestBinding(), + ), + GetPage( + name: Routes.ASR_TEST, + page: () => AsrTestView(), + binding: AsrTestBinding(), ), ]; } \ No newline at end of file diff --git a/lib/core/routes/app_routes.dart b/lib/core/routes/app_routes.dart index c224a2cbe..a1f31d92b 100644 --- a/lib/core/routes/app_routes.dart +++ b/lib/core/routes/app_routes.dart @@ -5,4 +5,6 @@ abstract class Routes { static const chat = '/chat'; static const profile = '/profile'; static const speechDemo = '/speech_demo'; + static const TTS_TEST = '/tts_test'; + static const ASR_TEST = '/asr_test'; } \ No newline at end of file diff --git a/lib/core/utils/logger.dart b/lib/core/utils/logger.dart new file mode 100644 index 000000000..d4e6a6145 --- /dev/null +++ b/lib/core/utils/logger.dart @@ -0,0 +1,62 @@ +import 'dart:developer' as developer; + +/// 日志级别 +enum LogLevel { + debug, + info, + warning, + error, +} + +/// 日志工具类 +/// +/// 提供统一的日志记录接口,方便后续扩展和管理 +class Logger { + /// 当前日志级别,低于此级别的日志不会被输出 + static LogLevel _currentLevel = LogLevel.debug; + + /// 设置日志级别 + static void setLevel(LogLevel level) { + _currentLevel = level; + } + + /// 输出调试日志 + static void debug(String message) { + if (_currentLevel.index <= LogLevel.debug.index) { + _log('DEBUG', message); + } + } + + /// 输出信息日志 + static void info(String message) { + if (_currentLevel.index <= LogLevel.info.index) { + _log('INFO', message); + } + } + + /// 输出警告日志 + static void warning(String message) { + if (_currentLevel.index <= LogLevel.warning.index) { + _log('WARNING', message); + } + } + + /// 输出错误日志 + static void error(String message, [dynamic error, StackTrace? stackTrace]) { + if (_currentLevel.index <= LogLevel.error.index) { + _log('ERROR', message); + if (error != null) { + _log('ERROR', 'Original error: $error'); + } + if (stackTrace != null) { + _log('ERROR', 'Stack trace: $stackTrace'); + } + } + } + + /// 内部日志输出方法 + static void _log(String level, String message) { + final timestamp = DateTime.now().toString(); + developer.log('[$timestamp] $level: $message'); + } +} diff --git a/lib/data/services/azure_tts_service.dart b/lib/data/services/azure_tts_service.dart deleted file mode 100644 index cc0508e42..000000000 --- a/lib/data/services/azure_tts_service.dart +++ /dev/null @@ -1,354 +0,0 @@ -import 'dart:async'; -import 'dart:typed_data'; -import 'package:get/get.dart'; -import 'package:just_audio/just_audio.dart'; -import 'package:flutter_azure_tts/flutter_azure_tts.dart'; -import 'package:audio_session/audio_session.dart'; -import 'package:flutter_dotenv/flutter_dotenv.dart'; - -/// A custom AudioSource that reads audio data from a byte array -class BytesAudioSource extends StreamAudioSource { - final Uint8List _bytes; - static const int _recommendedBufferSize = 8192; // 使用推荐的缓冲区大小 - - BytesAudioSource(this._bytes); - - @override - Future request([int? start, int? end]) async { - start = start ?? 0; - end = end ?? _bytes.length; - - try { - // 确保缓冲区大小至少为推荐值 - final bufferSize = (end - start) < _recommendedBufferSize - ? _recommendedBufferSize - : end - start; - - final subData = _bytes.sublist(start, end); - - return StreamAudioResponse( - sourceLength: _bytes.length, - contentLength: subData.length, - offset: start, - stream: Stream.value(subData), - contentType: 'audio/mpeg', - ); - } catch (e) { - print('BytesAudioSource request error: $e'); - rethrow; - } - } - - @override - Future get length => Future.value(_bytes.length); -} - -/// Azure TTS Service,支持将文本转换为语音并以流式方式边下载边播放 -class AzureTtsService extends GetxService { - static bool _isInitialized = false; - final _audioPlayer = AudioPlayer(); - final ConcatenatingAudioSource _playlist = ConcatenatingAudioSource(children: []); - final isEnabled = false.obs; - - // 播放状态变化通知流 - final _playingStateController = StreamController.broadcast(); - Stream get playingStateStream => _playingStateController.stream; - - StreamSubscription? _playbackEventSubscription; - StreamSubscription? _playerStateSubscription; - AudioSession? _audioSession; - - // Azure 配置 - final String _azureKey; - final String _azureRegion; - - // 句子管理 - String _pendingText = ''; - final List _sentenceQueue = []; - static final _sentenceBreaks = RegExp(r'[。!?.!?]'); - - // 状态 - bool _isFetching = false; - late Voice _selectedVoice; - bool _isDisposed = false; - - AzureTtsService() : - _azureKey = dotenv.env['AZURE_TTS_KEY'] ?? '', - _azureRegion = dotenv.env['AZURE_TTS_REGION'] ?? '' { - if (_azureKey.isEmpty || _azureRegion.isEmpty) { - throw Exception('Azure TTS配置信息不完整,请检查环境变量'); - } - } - - @override - Future onInit() async { - super.onInit(); - try { - // 初始化音频会话 - _audioSession = await AudioSession.instance; - await _audioSession?.configure(AudioSessionConfiguration( - avAudioSessionCategory: AVAudioSessionCategory.playback, - androidAudioAttributes: const AndroidAudioAttributes( - contentType: AndroidAudioContentType.speech, - usage: AndroidAudioUsage.media, - flags: AndroidAudioFlags.audibilityEnforced, - ), - androidAudioFocusGainType: AndroidAudioFocusGainType.gain, - )); - - // 设置音频播放器 - if (!_isDisposed) { - await _audioPlayer.setVolume(1.0); - await _audioPlayer.setLoopMode(LoopMode.off); - await _audioPlayer.setAudioSource( - _playlist, - initialPosition: Duration.zero, - preload: false, - ); - - _playbackEventSubscription = _audioPlayer.playbackEventStream.listen( - (event) { - if (_isDisposed) return; - print('播放事件: $event'); - if (event.processingState == ProcessingState.completed) { - _tryPlayNext(); - } - }, - onError: (error) { - if (_isDisposed) return; - print('播放错误: $error'); - _handlePlaybackError(error); - }, - ); - - _playerStateSubscription = _audioPlayer.playerStateStream.listen((state) { - if (_isDisposed) return; - print('播放器状态变化: playing=${state.playing},processingState=${state.processingState}'); - - // 通知播放状态变化 - _notifyPlayingStateChanged(state.playing && _playlist.length > 0); - - if (state.processingState == ProcessingState.completed) { - _tryPlayNext(); - } - }); - } - - // 初始化 Azure TTS (只初始化一次) - if (!_isInitialized && !_isDisposed) { - FlutterAzureTts.init( - subscriptionKey: _azureKey, - region: _azureRegion, - withLogs: true, - ); - _isInitialized = true; - } - - // 获取可用语音列表并设置默认语音 - if (!_isDisposed) { - final voicesResponse = await FlutterAzureTts.getAvailableVoices(); - _selectedVoice = voicesResponse.voices.firstWhere( - (v) => v.shortName == 'zh-CN-YunxiNeural', - orElse: () => voicesResponse.voices.first, - ); - print('已选择语音: ${_selectedVoice.shortName}'); - } - } catch (e, stackTrace) { - print('初始化TTS服务失败: $e'); - print('Stack trace: $stackTrace'); - } - } - - Future _cleanupResources() async { - _isDisposed = true; - await _playbackEventSubscription?.cancel(); - await _playerStateSubscription?.cancel(); - await _audioSession?.setActive(false); - await stop(); - await _audioPlayer.dispose(); - await _playingStateController.close(); - } - - @override - void onClose() async { - print('关闭TTS服务'); - await _cleanupResources(); - super.onClose(); - } - - void _handlePlaybackError(dynamic error) { - print('处理播放错误: $error'); - // 尝试重新初始化播放器 - _reinitializePlayer(); - } - - Future _reinitializePlayer() async { - try { - await _audioPlayer.stop(); - await _playlist.clear(); - await _audioPlayer.setAudioSource(_playlist); - - // 通知播放状态变化 - _notifyPlayingStateChanged(false); - - print('播放器重新初始化成功'); - } catch (e) { - print('播放器重新初始化失败: $e'); - } - } - - void _tryPlayNext() async { - if (_playlist.length > 0 && !_audioPlayer.playing) { - try { - await _audioPlayer.seek(Duration.zero, index: 0); - await _audioPlayer.play(); - } catch (e) { - print('尝试播放下一段失败: $e'); - } - } - } - - Future speak(String text) async { - if (!isEnabled.value) { - print('TTS 未启用,跳过播放。'); - return; - } - - try { - print('准备播放文本: $text'); - _pendingText += text; - _extractFullSentences(); - await _startPreloadIfNeeded(); - - if (!_audioPlayer.playing && _playlist.length > 0) { - print('开始播放音频'); - await _audioPlayer.play(); - } - } catch (e) { - print('播放文本失败: $e'); - } - } - - // 检查音频播放器是否真正在播放 - bool isActuallyPlaying() { - try { - // 检查播放器状态 - final isPlayerPlaying = _audioPlayer.playing; - - // 检查播放列表是否为空 - final hasAudioSource = _playlist.length > 0; - - // 只有当播放器正在播放且播放列表不为空时,才认为真正在播放 - return isPlayerPlaying && hasAudioSource; - } catch (e) { - print('检查实际播放状态失败: $e'); - return false; - } - } - - // 通知播放状态变化 - void _notifyPlayingStateChanged(bool isPlaying) { - try { - _playingStateController.add(isPlaying); - } catch (e) { - print('通知播放状态变化失败: $e'); - } - } - - void _extractFullSentences() { - if (_pendingText.isEmpty) return; - - final matches = _sentenceBreaks.allMatches(_pendingText).toList(); - if (matches.isEmpty) return; - - int lastPos = 0; - for (final match in matches) { - final end = match.end; - final sentence = _pendingText.substring(lastPos, end).trim(); - if (sentence.isNotEmpty) { - _sentenceQueue.add(sentence); - } - lastPos = end; - } - - if (lastPos > 0) { - _pendingText = _pendingText.substring(lastPos); - } - } - - Future _startPreloadIfNeeded() async { - if (_isFetching) return; - if (_sentenceQueue.isEmpty) return; - - _isFetching = true; - await _fetchNextSegment(); - } - - Future _fetchNextSegment() async { - if (_sentenceQueue.isEmpty) { - _isFetching = false; - return; - } - - final sentence = _sentenceQueue.removeAt(0); - print('获取音频片段: $sentence'); - - try { - final params = TtsParams( - voice: _selectedVoice, - audioFormat: AudioOutputFormat.audio16khz32kBitrateMonoMp3, - rate: 1.0, - text: sentence, - ); - - final ttsResponse = await FlutterAzureTts.getTts(params); - if (ttsResponse.audio != null && ttsResponse.audio.isNotEmpty) { - print('成功获取音频数据: ${ttsResponse.audio.length} bytes'); - - try { - final audioSource = BytesAudioSource(ttsResponse.audio); - await _playlist.add(audioSource); - print('音频片段已添加到播放列表'); - - if (!_audioPlayer.playing && _playlist.length == 1) { - await _audioPlayer.setVolume(1.0); - await _audioPlayer.seek(Duration.zero, index: 0); - await Future.delayed(const Duration(milliseconds: 100)); - await _audioPlayer.play(); - } - } catch (e) { - print('添加音频源失败: $e'); - } - } else { - print('获取音频数据失败: 返回为空或长度为0'); - } - } catch (e, stackTrace) { - print('获取音频失败: $e'); - print('Stack trace: $stackTrace'); - } - - await _fetchNextSegment(); - } - - Future stop() async { - try { - await _audioPlayer.stop(); - await _playlist.clear(); - _pendingText = ''; - _sentenceQueue.clear(); - _isFetching = false; - - // 通知播放状态变化 - _notifyPlayingStateChanged(false); - } catch (e) { - print('停止播放失败: $e'); - } - } - - void toggleEnabled() { - isEnabled.toggle(); - if (!isEnabled.value) { - stop(); - } - } -} \ No newline at end of file diff --git a/lib/data/services/background_agent_service.dart b/lib/data/services/background_agent_service.dart index 1fe9c625e..6c3e119ed 100644 --- a/lib/data/services/background_agent_service.dart +++ b/lib/data/services/background_agent_service.dart @@ -2,15 +2,16 @@ import 'package:get/get.dart'; import 'dart:async'; import 'volcano_ai_service.dart'; import 'volcano_tts_service.dart'; -import 'voice_recognition_service.dart'; +import 'volcano_voice_recognition_service.dart'; import '../../modules/chat/models/message_model.dart'; +import '../../core/utils/logger.dart'; class BackgroundAgentService extends GetxService { static BackgroundAgentService get to => Get.find(); final VolcanoAIService _aiService; final VolcanoTtsService _ttsService; - final VoiceRecognitionService _voiceRecognitionService; + final VolcanoVoiceRecognitionService _voiceRecognitionService; final List _messageHistory = []; String _pendingTtsText = ''; static const int _minTtsLength = 20; @@ -41,7 +42,7 @@ class BackgroundAgentService extends GetxService { Timer? _noSpeechTimer; // 最后一次识别到语音的时间 - DateTime? _lastSpeechTime; + // DateTime? _lastSpeechTime; // Removing unused field // 添加一个变量来跟踪当前的AI响应流订阅 StreamSubscription? _aiResponseSubscription; @@ -50,12 +51,12 @@ class BackgroundAgentService extends GetxService { bool _shouldCancelAiResponse = false; // 添加计数器,用于跟踪连续无输入的次数 - int _noSpeechCount = 0; + // int _noSpeechCount = 0; // Removing unused field BackgroundAgentService() : _aiService = VolcanoAIService(), _ttsService = Get.find(), - _voiceRecognitionService = Get.find() { + _voiceRecognitionService = Get.find() { // 简化构造函数,不需要定期检查TTS状态 } @@ -71,17 +72,17 @@ class BackgroundAgentService extends GetxService { try { isTtsPlaying = _ttsService.isActuallyPlaying(); } catch (e) { - print('检查TTS播放状态失败: $e'); + Logger.error('检查TTS播放状态失败', e); isTtsPlaying = false; } if (isTtsPlaying) { // 如果TTS正在播放,重置定时器 - print('TTS正在播放,重置15秒超时计时器'); + Logger.info('TTS正在播放,重置15秒超时计时器'); _startNoSpeechTimer(); } else { // 如果TTS不在播放,且15秒内没有检测到用户输入,退出交互 - print('15秒内没有检测到用户输入,退出交互'); + Logger.info('15秒内没有检测到用户输入,退出交互'); _exitInteraction(); } }); @@ -350,22 +351,27 @@ class BackgroundAgentService extends GetxService { } // 开始连续识别 - final recognitionStream = await _voiceRecognitionService.startContinuousRecognition(); + final success = await _voiceRecognitionService.startContinuousRecognition(); + if (!success) { + if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) { + _recognitionCompleter!.completeError(Exception('无法启动语音识别')); + } + return; + } + _isListening = true; isListening.value = true; recognizedText.value = ''; _hasRecognizedSpeech = false; _hasFinalResult = false; - _lastSpeechTime = null; - // 监听识别结果 - _recognitionSubscription = recognitionStream.listen((event) { + // 使用语音识别服务的recognitionStream而不是方法的返回值 + _recognitionSubscription = _voiceRecognitionService.recognitionStream?.listen((event) { if (event.type == RecognitionEventType.finalResult) { recognizedText.value = event.text; if (event.text.isNotEmpty) { _hasRecognizedSpeech = true; _hasFinalResult = true; - _lastSpeechTime = DateTime.now(); // 收到最终结果,立即完成识别过程 if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) { @@ -382,11 +388,10 @@ class BackgroundAgentService extends GetxService { _startNoSpeechTimer(); } } - } else if (event.type == RecognitionEventType.intermediateResult) { + } else if (event.type == RecognitionEventType.recognizing) { recognizedText.value = event.text; if (event.text.isNotEmpty) { _hasRecognizedSpeech = true; - _lastSpeechTime = DateTime.now(); // 检测到用户说话,重置无语音计时器 _noSpeechTimer?.cancel(); @@ -471,7 +476,7 @@ class BackgroundAgentService extends GetxService { _recognitionSubscription = null; // 停止语音识别 - await _voiceRecognitionService.stopContinuousRecognition(); + await _voiceRecognitionService.stopRecognition(); // 获取最终识别结果 final result = recognizedText.value; diff --git a/lib/data/services/microsoft_tts_service.dart b/lib/data/services/microsoft_tts_service.dart deleted file mode 100644 index 9a39b34c5..000000000 --- a/lib/data/services/microsoft_tts_service.dart +++ /dev/null @@ -1,346 +0,0 @@ -import 'dart:async'; -import 'package:flutter/services.dart'; -import 'package:get/get.dart'; -import 'package:flutter_dotenv/flutter_dotenv.dart'; - -/// 微软 Text-to-Speech 服务异常 -class MicrosoftTtsException implements Exception { - final String message; - MicrosoftTtsException(this.message); - - @override - String toString() => message; -} - -/// 微软 Text-to-Speech 服务 -/// -/// 该服务通过平台通道与 Android 上的 Microsoft Speech SDK 交互, -/// 提供文本转语音功能。 -class MicrosoftTtsService extends GetxService { - static const MethodChannel _channel = MethodChannel('com.example.deep_voice/text_to_speech'); - - bool _isInitialized = false; - late final String _subscriptionKey; - late final String _serviceRegion; - - // 当前使用的语音 - String _currentVoice = 'zh-CN-XiaoxiaoNeural'; - String get currentVoice => _currentVoice; - - // 语音合成队列 - final List _textQueue = []; - bool _isProcessingQueue = false; - bool _isSpeaking = false; - - // 可观察状态 - final isEnabled = true.obs; - final isSpeaking = false.obs; - - MicrosoftTtsService() { - _loadConfig(); - } - - /// 从环境变量加载配置 - void _loadConfig() { - _subscriptionKey = dotenv.env['AZURE_SPEECH_KEY'] ?? ''; - _serviceRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? ''; - - if (_subscriptionKey.isEmpty || _serviceRegion.isEmpty) { - throw MicrosoftTtsException('未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION'); - } - } - - /// 初始化微软 TTS SDK - /// - /// 返回 true 表示初始化成功,否则抛出 PlatformException - Future initialize() async { - if (_isInitialized) return true; - - try { - final bool result = await _channel.invokeMethod('initialize', { - 'subscriptionKey': _subscriptionKey, - 'serviceRegion': _serviceRegion, - }); - - _isInitialized = result; - return result; - } on PlatformException catch (e) { - throw MicrosoftTtsException('初始化失败: ${e.message}'); - } - } - - /// 设置语音 - /// - /// [voiceName] 语音名称,例如 "zh-CN-XiaoxiaoNeural" - /// - /// 返回 true 表示设置成功,否则抛出 PlatformException - Future setVoice(String voiceName) async { - if (!_isInitialized) { - await initialize(); - } - - try { - final bool result = await _channel.invokeMethod('setVoice', { - 'voiceName': voiceName, - }); - - if (result) { - _currentVoice = voiceName; - } - - return result; - } on PlatformException catch (e) { - throw MicrosoftTtsException('设置语音失败: ${e.message}'); - } - } - - /// 将文本转换为语音并播放 - /// - /// [text] 要转换的文本 - /// - /// 返回合成结果消息,否则抛出 PlatformException - Future speakText(String text) async { - if (!_isInitialized) { - await initialize(); - } - - if (!isEnabled.value) { - return "TTS 服务已禁用"; - } - - try { - _isSpeaking = true; - isSpeaking.value = true; - - final String result = await _channel.invokeMethod('speakText', { - 'text': text, - }); - - _isSpeaking = false; - isSpeaking.value = false; - - return result; - } on PlatformException catch (e) { - _isSpeaking = false; - isSpeaking.value = false; - throw MicrosoftTtsException('语音合成失败: ${e.message}'); - } - } - - /// 将 SSML 转换为语音并播放 - /// - /// [ssml] SSML 格式的文本 - /// - /// 返回合成结果消息,否则抛出 PlatformException - Future speakSsml(String ssml) async { - if (!_isInitialized) { - await initialize(); - } - - if (!isEnabled.value) { - return "TTS 服务已禁用"; - } - - try { - _isSpeaking = true; - isSpeaking.value = true; - - final String result = await _channel.invokeMethod('speakSsml', { - 'ssml': ssml, - }); - - _isSpeaking = false; - isSpeaking.value = false; - - return result; - } on PlatformException catch (e) { - _isSpeaking = false; - isSpeaking.value = false; - throw MicrosoftTtsException('SSML 语音合成失败: ${e.message}'); - } - } - - /// 添加文本到队列并开始处理 - /// - /// [text] 要添加到队列的文本 - /// [rate] 可选,语速,范围 -100 到 100,默认为 0 - /// [pitch] 可选,音调,范围 -100 到 100,默认为 0 - /// - /// 返回 true 表示成功添加到队列 - Future speak(String text, {int rate = 0, int pitch = 0}) async { - if (!isEnabled.value) { - return false; - } - - if (text.isEmpty) { - return false; - } - - // 生成 SSML - final ssml = generateSsml( - text: text, - rate: rate, - pitch: pitch, - ); - - // 添加到队列 - _textQueue.add(ssml); - - // 如果队列未在处理中,开始处理 - if (!_isProcessingQueue) { - _processQueue(); - } - - return true; - } - - /// 连续播放多段文本 - /// - /// [texts] 要连续播放的文本列表 - /// [rate] 可选,语速,范围 -100 到 100,默认为 0 - /// [pitch] 可选,音调,范围 -100 到 100,默认为 0 - /// - /// 返回 true 表示成功添加到队列 - Future speakMultiple(List texts, {int rate = 0, int pitch = 0}) async { - if (!isEnabled.value) { - return false; - } - - if (texts.isEmpty) { - return false; - } - - // 将所有文本添加到队列 - for (final text in texts) { - if (text.isNotEmpty) { - final ssml = generateSsml( - text: text, - rate: rate, - pitch: pitch, - ); - _textQueue.add(ssml); - } - } - - // 如果队列未在处理中,开始处理 - if (!_isProcessingQueue) { - _processQueue(); - } - - return true; - } - - /// 处理语音合成队列 - Future _processQueue() async { - if (_textQueue.isEmpty || _isProcessingQueue) { - return; - } - - _isProcessingQueue = true; - - try { - while (_textQueue.isNotEmpty) { - // 如果服务被禁用,清空队列并退出 - if (!isEnabled.value) { - _textQueue.clear(); - break; - } - - // 获取队列中的下一个 SSML - final ssml = _textQueue.removeAt(0); - - // 播放 SSML - await speakSsml(ssml); - } - } catch (e) { - print('处理语音队列时出错: $e'); - } finally { - _isProcessingQueue = false; - } - } - - /// 停止当前语音合成并清空队列 - Future stop() async { - // 清空队列 - _textQueue.clear(); - - // 如果当前正在播放,尝试停止 - if (_isSpeaking) { - try { - await _channel.invokeMethod('dispose'); - await initialize(); // 重新初始化以确保资源正确释放和重建 - _isSpeaking = false; - isSpeaking.value = false; - } catch (e) { - print('停止语音合成时出错: $e'); - } - } - } - - /// 生成 SSML 文本 - /// - /// [text] 要转换的文本 - /// [voiceName] 可选,语音名称,默认使用当前设置的语音 - /// [rate] 可选,语速,范围 -100 到 100,默认为 0 - /// [pitch] 可选,音调,范围 -100 到 100,默认为 0 - /// - /// 返回 SSML 格式的文本 - String generateSsml({ - required String text, - String? voiceName, - int rate = 0, - int pitch = 0, - }) { - final voice = voiceName ?? _currentVoice; - final rateValue = rate.clamp(-100, 100); - final pitchValue = pitch.clamp(-100, 100); - - // 将 rate 和 pitch 转换为 SSML 格式的值 - final String rateStr = _convertRateToSsml(rateValue); - final String pitchStr = _convertPitchToSsml(pitchValue); - - return ''' - - - - $text - - - - '''; - } - - /// 将 rate 值转换为 SSML 格式 - String _convertRateToSsml(int rate) { - if (rate == 0) return '0%'; - - // 将 -100 到 100 的范围映射到 -90% 到 100% - if (rate < 0) { - // 负值映射到 -90% 到 0% - return '${(rate * 0.9).round()}%'; - } else { - // 正值映射到 0% 到 100% - return '${rate}%'; - } - } - - /// 将 pitch 值转换为 SSML 格式 - String _convertPitchToSsml(int pitch) { - if (pitch == 0) return '0%'; - - // 将 -100 到 100 的范围映射到 -50% 到 50% - return '${(pitch * 0.5).round()}%'; - } - - /// 释放资源 - Future dispose() async { - if (!_isInitialized) return; - - try { - await _channel.invokeMethod('dispose'); - _isInitialized = false; - } on PlatformException catch (e) { - throw MicrosoftTtsException('释放资源失败: ${e.message}'); - } - } -} \ No newline at end of file diff --git a/lib/data/services/volcano_tts_service.dart b/lib/data/services/volcano_tts_service.dart index a1dd3ff62..5e473c757 100644 --- a/lib/data/services/volcano_tts_service.dart +++ b/lib/data/services/volcano_tts_service.dart @@ -1,21 +1,24 @@ import 'dart:async'; -import 'dart:convert'; import 'dart:typed_data'; import 'dart:io'; import 'dart:math' as math; import 'package:get/get.dart'; import 'package:just_audio/just_audio.dart'; -import 'package:web_socket_channel/web_socket_channel.dart'; import 'package:audio_session/audio_session.dart'; import 'package:flutter_dotenv/flutter_dotenv.dart'; +import 'package:flutter/services.dart'; /// 火山语音服务异常 class VolcanoTtsException implements Exception { final String message; - VolcanoTtsException(this.message); + final dynamic originalError; + + VolcanoTtsException(this.message, [this.originalError]); @override - String toString() => message; + String toString() => originalError != null + ? '$message (原始错误: $originalError)' + : message; } /// 自定义音频源,用于从字节数组读取音频数据 @@ -25,13 +28,89 @@ class BytesAudioSource extends StreamAudioSource { BytesAudioSource(this._bytes); + // 添加 PCM 头信息,转换为 WAV 格式 + Uint8List _addWavHeader(Uint8List pcmData) { + // PCM 参数 + const int sampleRate = 16000; // 采样率 + const int numChannels = 1; // 单声道 + const int bitsPerSample = 16; // 16位采样 + + // 计算数据大小 + final int dataSize = pcmData.length; + final int fileSize = 36 + dataSize; + + // 创建 WAV 头 + final ByteData header = ByteData(44); + + // RIFF 头 + header.setUint8(0, 'R'.codeUnitAt(0)); + header.setUint8(1, 'I'.codeUnitAt(0)); + header.setUint8(2, 'F'.codeUnitAt(0)); + header.setUint8(3, 'F'.codeUnitAt(0)); + + // 文件大小 + header.setUint32(4, fileSize, Endian.little); + + // WAVE 标识 + header.setUint8(8, 'W'.codeUnitAt(0)); + header.setUint8(9, 'A'.codeUnitAt(0)); + header.setUint8(10, 'V'.codeUnitAt(0)); + header.setUint8(11, 'E'.codeUnitAt(0)); + + // fmt 子块 + header.setUint8(12, 'f'.codeUnitAt(0)); + header.setUint8(13, 'm'.codeUnitAt(0)); + header.setUint8(14, 't'.codeUnitAt(0)); + header.setUint8(15, ' '.codeUnitAt(0)); + + // 子块大小 + header.setUint32(16, 16, Endian.little); + + // 音频格式 (PCM = 1) + header.setUint16(20, 1, Endian.little); + + // 声道数 + header.setUint16(22, numChannels, Endian.little); + + // 采样率 + header.setUint32(24, sampleRate, Endian.little); + + // 字节率 = 采样率 * 声道数 * 采样位数 / 8 + header.setUint32(28, sampleRate * numChannels * bitsPerSample ~/ 8, Endian.little); + + // 块对齐 = 声道数 * 采样位数 / 8 + header.setUint16(32, numChannels * bitsPerSample ~/ 8, Endian.little); + + // 采样位数 + header.setUint16(34, bitsPerSample, Endian.little); + + // data 子块 + header.setUint8(36, 'd'.codeUnitAt(0)); + header.setUint8(37, 'a'.codeUnitAt(0)); + header.setUint8(38, 't'.codeUnitAt(0)); + header.setUint8(39, 'a'.codeUnitAt(0)); + + // 数据大小 + header.setUint32(40, dataSize, Endian.little); + + // 创建完整的 WAV 数据 + final Uint8List wavData = Uint8List(44 + dataSize); + wavData.setRange(0, 44, header.buffer.asUint8List()); + wavData.setRange(44, 44 + dataSize, pcmData); + + return wavData; + } + @override Future request([int? start, int? end]) async { + // 添加 WAV 头信息 + final Uint8List wavData = _addWavHeader(_bytes); + start = start ?? 0; - end = end ?? _bytes.length; + end = end ?? wavData.length; try { - final subData = _bytes.sublist(start, end); + final subData = wavData.sublist(start, end); // 使用固定大小的块进行流式传输 final chunks = >[]; @@ -43,11 +122,11 @@ class BytesAudioSource extends StreamAudioSource { } return StreamAudioResponse( - sourceLength: _bytes.length, + sourceLength: wavData.length, contentLength: subData.length, offset: start, stream: Stream.fromIterable(chunks), - contentType: 'audio/mpeg', + contentType: 'audio/wav', ); } catch (e) { print('BytesAudioSource request error: $e'); @@ -56,13 +135,13 @@ class BytesAudioSource extends StreamAudioSource { } @override - Future get length => Future.value(_bytes.length); + Future get length => Future.value(_bytes.length + 44); // PCM 数据长度 + WAV 头长度 } -/// 火山语音合成服务 +/// 火山语音合成服务 (SDK版本) class VolcanoTtsService extends GetxService { - static const String _host = 'openspeech.bytedance.com'; - static const String _apiUrl = 'wss://$_host/api/v1/tts/ws_binary'; + // 平台通道 + static const MethodChannel _channel = MethodChannel('com.example.deep_voice/volcano_tts'); // 配置参数 final String _appId; @@ -83,65 +162,42 @@ class VolcanoTtsService extends GetxService { StreamSubscription? _playerStateSubscription; AudioSession? _audioSession; - // WebSocket 相关 - WebSocketChannel? _channel; - bool _isDisposed = false; - // 句子管理 String _pendingText = ''; final List _sentenceQueue = []; - static final _sentenceBreaks = RegExp(r'[。!?.!?,,、]'); // 添加更多标点符号 + + // 优化的句子分割正则表达式,包含更多中英文标点 + static final _sentenceBreaks = RegExp(r'[。!?.!?;;::,,、]'); + + // 优化的句子分割参数 + static const int _maxSegmentLength = 150; // 最大分段长度 + static const int _optimalSegmentLength = 80; // 最佳分段长度 + static const int _minSegmentLength = 5; // 最小分段长度 + bool _isFetching = false; - static const int _maxSegmentLength = 100; // 最大分段长度 - static const int _optimalSegmentLength = 50; // 最佳分段长度 - - // WebSocket连接配置 - static final Map _wsHeaders = { - 'User-Agent': 'DeepVoice/1.0', - 'Accept': '*/*', - 'Accept-Encoding': 'gzip, deflate, br', - 'Connection': 'Upgrade', - 'Upgrade': 'websocket', - 'Sec-WebSocket-Version': '13', - 'Sec-WebSocket-Extensions': 'permessage-deflate', - 'Origin': 'https://openspeech.bytedance.com', - }; - - // 二进制协议头部字段定义 - static const int PROTOCOL_VERSION = 0x1; // 0b0001 - 版本1 - static const int HEADER_SIZE = 0x1; // 0b0001 - 4字节 - static const int MESSAGE_TYPE = 0x1; // 0b0001 - full client request - static const int MESSAGE_FLAGS = 0x0; // 0b0000 - 无特殊标记 - static const int SERIALIZATION = 0x1; // 0b0001 - JSON - static const int COMPRESSION = 0x0; // 0b0000 - 无压缩 - static const int RESERVED = 0x0; // 0b0000 - 保留字段 - - // 构建二进制协议头 - static Uint8List _buildHeader() { - final header = ByteData(4); // 4字节的头部 - - // 第一个字节: [协议版本(4位) | 报头大小(4位)] - header.setUint8(0, (PROTOCOL_VERSION << 4) | HEADER_SIZE); - - // 第二个字节: [消息类型(4位) | 消息标记(4位)] - header.setUint8(1, (MESSAGE_TYPE << 4) | MESSAGE_FLAGS); - - // 第三个字节: [序列化方法(4位) | 压缩方法(4位)] - header.setUint8(2, (SERIALIZATION << 4) | COMPRESSION); - - // 第四个字节: [保留字段(8位)] - header.setUint8(3, RESERVED); - - return header.buffer.asUint8List(); - } + int _consecutiveErrorCount = 0; + static const int _maxConsecutiveErrors = 3; + + // 初始化状态 + bool _isInitialized = false; + bool _isDisposed = false; + + // 性能优化参数 + final int _preloadCount = 2; // 预加载片段数量 + final int _maxRetryAttempts = 3; // 最大重试次数 + + // 音频质量参数 + final double _defaultVolume = 1.0; + final double _defaultSpeed = 1.0; VolcanoTtsService() : - _appId = dotenv.env['VOLCANO_TTS_APP_ID'] ?? '', - _token = dotenv.env['VOLCANO_TTS_TOKEN'] ?? '', - _cluster = dotenv.env['VOLCANO_TTS_CLUSTER'] ?? '', - _voiceType = dotenv.env['VOLCANO_TTS_VOICE_TYPE'] ?? 'zh_male_M392_conversation_wvae_bigtts' { + // 只使用统一的环境变量 + _appId = dotenv.env['VOLCANO_APP_ID'] ?? '', + _token = dotenv.env['VOLCANO_APP_KEY'] ?? '', + _cluster = dotenv.env['VOLCANO_CLUSTER'] ?? '', + _voiceType = dotenv.env['VOLCANO_VOICE_TYPE'] ?? '' { if (_appId.isEmpty || _token.isEmpty || _cluster.isEmpty) { - throw Exception('火山语音配置信息不完整,请检查环境变量'); + throw VolcanoTtsException('火山语音配置信息不完整,请检查环境变量 VOLCANO_APP_ID, VOLCANO_APP_KEY 和 VOLCANO_CLUSTER'); } } @@ -151,24 +207,61 @@ class VolcanoTtsService extends GetxService { try { await _initAudioSession(); await _initAudioPlayer(); + await _initializeTtsEngine(); } catch (e, stackTrace) { print('初始化TTS服务失败: $e'); print('Stack trace: $stackTrace'); } } + /// 初始化TTS引擎 + Future _initializeTtsEngine() async { + try { + print('开始初始化火山语音TTS引擎,APP_ID: ${_appId.substring(0, math.min(3, _appId.length))}***,APP_KEY: ${_token.length > 10 ? "${_token.substring(0, 5)}..." : _token}'); + + final result = await _channel.invokeMethod('initialize', { + 'appId': _appId, + 'token': _token, + 'cluster': _cluster, + }); + + _isInitialized = result ?? false; + print('火山语音TTS引擎初始化${_isInitialized ? '成功' : '失败'}'); + + if (!_isInitialized) { + throw VolcanoTtsException('TTS引擎初始化失败'); + } + } catch (e) { + print('初始化火山语音TTS引擎失败: $e'); + _isInitialized = false; + throw VolcanoTtsException('初始化失败', e); + } + } + Future _initAudioSession() async { _audioSession = await AudioSession.instance; - await _audioSession?.configure(AudioSessionConfiguration( - avAudioSessionCategory: AVAudioSessionCategory.playback, - androidAudioAttributes: const AndroidAudioAttributes( - contentType: AndroidAudioContentType.speech, - usage: AndroidAudioUsage.media, - flags: AndroidAudioFlags.audibilityEnforced, - ), - androidAudioFocusGainType: AndroidAudioFocusGainType.gain, - androidWillPauseWhenDucked: true, - )); + + // 针对不同平台优化音频会话配置 + if (Platform.isAndroid) { + await _audioSession?.configure(AudioSessionConfiguration( + avAudioSessionCategory: AVAudioSessionCategory.playback, + androidAudioAttributes: const AndroidAudioAttributes( + contentType: AndroidAudioContentType.speech, + usage: AndroidAudioUsage.media, + flags: AndroidAudioFlags.audibilityEnforced, + ), + androidAudioFocusGainType: AndroidAudioFocusGainType.gain, + androidWillPauseWhenDucked: true, + )); + } else if (Platform.isIOS) { + await _audioSession?.configure(AudioSessionConfiguration( + avAudioSessionCategory: AVAudioSessionCategory.playback, + avAudioSessionCategoryOptions: AVAudioSessionCategoryOptions.duckOthers, + avAudioSessionMode: AVAudioSessionMode.spokenAudio, + avAudioSessionRouteSharingPolicy: AVAudioSessionRouteSharingPolicy.defaultPolicy, + avAudioSessionSetActiveOptions: AVAudioSessionSetActiveOptions.notifyOthersOnDeactivation, + )); + } // 确保音频会话激活并设置正确的音频模式 if (Platform.isAndroid) { @@ -193,9 +286,9 @@ class VolcanoTtsService extends GetxService { } // 基本播放器配置 - await _audioPlayer.setVolume(1.0); + await _audioPlayer.setVolume(_defaultVolume); await _audioPlayer.setLoopMode(LoopMode.off); - await _audioPlayer.setSpeed(1.0); + await _audioPlayer.setSpeed(_defaultSpeed); // 设置音频缓冲配置,启用预加载以减少延迟 await _audioPlayer.setAudioSource( @@ -207,23 +300,42 @@ class VolcanoTtsService extends GetxService { _setupEventListeners(); } - Future speak(String text) async { + /// 合成并播放文本 + /// + /// [text] 要合成的文本 + /// [voiceType] 可选的语音类型,如果提供,将覆盖默认语音类型 + Future speak(String text, {String? voiceType}) async { if (!isEnabled.value || text.trim().isEmpty) return; try { + // 如果提供了语音类型,记录日志 + if (voiceType != null && voiceType != _voiceType) { + print('使用临时语音类型: $voiceType (默认: $_voiceType)'); + } + _pendingText += text; - _extractFullSentences(); - await _startPreloadIfNeeded(); + _extractSentences(); + await _startPreloadIfNeeded(voiceType: voiceType); } catch (e, stack) { print('语音合成失败: $e'); - await _reinitializePlayer(); + print('Stack trace: $stack'); + + _consecutiveErrorCount++; + if (_consecutiveErrorCount >= _maxConsecutiveErrors) { + print('连续错误次数过多,尝试重新初始化播放器'); + await _reinitializePlayer(); + _consecutiveErrorCount = 0; + } } } - void _extractFullSentences() { + /// 优化的文本分段算法 + void _extractSentences() { if (_pendingText.isEmpty) return; + // 查找所有句子分隔符 final matches = _sentenceBreaks.allMatches(_pendingText).toList(); + if (matches.isEmpty) { // 如果没有找到句子分隔符,按长度分割 _splitByLength(_pendingText); @@ -247,11 +359,11 @@ class VolcanoTtsService extends GetxService { if (currentBatch.isEmpty) { currentBatch = sentence; } - // 如果当前批次加上新句子不超过30字符,则合并 - else if ((currentBatch + sentence).length <= 30) { + // 如果当前批次加上新句子不超过最佳长度,则合并 + else if ((currentBatch + sentence).length <= _optimalSegmentLength) { currentBatch += sentence; } - // 如果超过30字符,将当前批次加入队列,开始新批次 + // 如果超过最佳长度,将当前批次加入队列,开始新批次 else { if (currentBatch.isNotEmpty) { _sentenceQueue.add(currentBatch); @@ -271,10 +383,10 @@ class VolcanoTtsService extends GetxService { if (lastPos < _pendingText.length) { final remaining = _pendingText.substring(lastPos).trim(); if (remaining.isNotEmpty) { - if (remaining.length <= 30) { + if (remaining.length <= _optimalSegmentLength) { // 尝试与最后一个批次合并 if (_sentenceQueue.isNotEmpty && - (_sentenceQueue.last + remaining).length <= 30) { + (_sentenceQueue.last + remaining).length <= _optimalSegmentLength) { _sentenceQueue[_sentenceQueue.length - 1] += remaining; } else { _sentenceQueue.add(remaining); @@ -288,24 +400,32 @@ class VolcanoTtsService extends GetxService { _pendingText = ''; } + /// 按长度分割文本,优先在空格或标点处分割 void _splitByLength(String text) { if (text.isEmpty) return; int start = 0; while (start < text.length) { - int end = math.min(start + 30, text.length); + // 计算理想的结束位置 + int end = math.min(start + _optimalSegmentLength, text.length); - // 尝试在标点或空格处分割 + // 如果没有到达文本末尾,尝试在标点或空格处分割 if (end < text.length) { - final spacePos = text.lastIndexOf(' ', end); - final breakPos = _sentenceBreaks.firstMatch( - text.substring(start, end) - )?.end; + // 在理想长度附近查找标点 + int searchStart = math.max(start + _minSegmentLength, end - 20); + int searchEnd = math.min(text.length, end + 20); + + String searchText = text.substring(searchStart, searchEnd); + final breakMatch = _sentenceBreaks.firstMatch(searchText); - if (breakPos != null && breakPos > start) { - end = start + breakPos; - } else if (spacePos > start) { - end = spacePos; + if (breakMatch != null) { + end = searchStart + breakMatch.end; + } else { + // 如果没有找到标点,尝试在空格处分割 + final spacePos = text.lastIndexOf(' ', end); + if (spacePos > start && spacePos > end - 20) { + end = spacePos + 1; // 包含空格 + } } } @@ -313,7 +433,7 @@ class VolcanoTtsService extends GetxService { if (segment.isNotEmpty) { // 尝试与前一个批次合并 if (_sentenceQueue.isNotEmpty && - (_sentenceQueue.last + segment).length <= 30) { + (_sentenceQueue.last + segment).length <= _optimalSegmentLength) { _sentenceQueue[_sentenceQueue.length - 1] += segment; } else { _sentenceQueue.add(segment); @@ -323,13 +443,13 @@ class VolcanoTtsService extends GetxService { } } - Future _startPreloadIfNeeded() async { + Future _startPreloadIfNeeded({String? voiceType}) async { if (_isFetching || _sentenceQueue.isEmpty) return; _isFetching = true; - await _fetchNextSegment(); + await _fetchNextSegment(voiceType: voiceType); } - Future _fetchNextSegment() async { + Future _fetchNextSegment({String? voiceType}) async { if (_sentenceQueue.isEmpty) { _isFetching = false; return; @@ -338,13 +458,35 @@ class VolcanoTtsService extends GetxService { try { // 预加载多个片段以实现流畅播放 final segments = []; - while (_sentenceQueue.isNotEmpty && segments.length < 2) { // 减少预加载数量以降低内存压力 + while (_sentenceQueue.isNotEmpty && segments.length < _preloadCount) { segments.add(_sentenceQueue.removeAt(0)); } for (final segment in segments) { print('获取音频片段: $segment'); - final audioData = await _synthesize(segment); + + // 添加重试逻辑 + Uint8List? audioData; + int retryCount = 0; + + while (audioData == null && retryCount < _maxRetryAttempts) { + try { + // 使用提供的语音类型或默认语音类型 + audioData = await _synthesize(segment, voiceType: voiceType); + if (audioData == null && retryCount < _maxRetryAttempts - 1) { + print('合成返回空数据,尝试重试 (${retryCount + 1}/$_maxRetryAttempts)'); + await Future.delayed(Duration(milliseconds: 200 * (retryCount + 1))); + } + } catch (e) { + print('合成失败,尝试重试 (${retryCount + 1}/$_maxRetryAttempts): $e'); + if (retryCount < _maxRetryAttempts - 1) { + await Future.delayed(Duration(milliseconds: 200 * (retryCount + 1))); + } else { + rethrow; + } + } + retryCount++; + } if (audioData != null) { print('成功获取音频数据: ${audioData.length} bytes'); @@ -359,198 +501,126 @@ class VolcanoTtsService extends GetxService { await _audioSession?.setActive(true); // 设置播放参数 - await _audioPlayer.setVolume(1.0); + await _audioPlayer.setVolume(_defaultVolume); await _audioPlayer.seek(Duration.zero, index: 0); // 开始播放 await _audioPlayer.play(); + + // 重置连续错误计数 + _consecutiveErrorCount = 0; } } catch (e) { print('添加音频源失败: $e'); + throw VolcanoTtsException('添加音频源失败', e); } + } else { + print('无法获取音频数据,跳过此片段'); } } // 如果队列还有内容,继续获取下一批 if (_sentenceQueue.isNotEmpty) { - await Future.delayed(const Duration(milliseconds: 100)); // 添加延迟以避免过快请求 - await _fetchNextSegment(); + await Future.delayed(const Duration(milliseconds: 50)); // 减少延迟以提高响应速度 + await _fetchNextSegment(voiceType: voiceType); } else { _isFetching = false; // 检查是否还有未处理的文本 - _extractFullSentences(); + _extractSentences(); if (_sentenceQueue.isNotEmpty) { - await _startPreloadIfNeeded(); + await _startPreloadIfNeeded(voiceType: voiceType); } } } catch (e, stackTrace) { print('获取音频失败: $e'); print('Stack trace: $stackTrace'); _isFetching = false; - } - } - - Future _synthesize(String text) async { - WebSocket? socket; - StreamSubscription? subscription; - - try { - final requestJson = { - 'app': { - 'appid': _appId, - 'token': _token, - 'cluster': _cluster, - }, - 'user': { - 'uid': '${DateTime.now().millisecondsSinceEpoch}', - }, - 'audio': { - 'voice_type': _voiceType, - 'encoding': 'mp3', - 'speed_ratio': 1.0, - 'volume_ratio': 1.0, - 'pitch_ratio': 1.0, - 'language': 'zh', - }, - 'request': { - 'reqid': '${DateTime.now().millisecondsSinceEpoch}', - 'text': text, - 'text_type': 'plain', - 'operation': 'query', - }, - }; - - final jsonBytes = utf8.encode(json.encode(requestJson)); - final header = _buildHeader(); - final fullRequest = Uint8List(header.length + 4 + jsonBytes.length); - - fullRequest.setAll(0, header); - final lengthBytes = ByteData(4)..setInt32(0, jsonBytes.length, Endian.big); - fullRequest.setAll(header.length, lengthBytes.buffer.asUint8List()); - fullRequest.setAll(header.length + 4, jsonBytes); - - final wsUrl = Uri.parse(_apiUrl); - final completer = Completer(); - - socket = await WebSocket.connect( - wsUrl.toString(), - headers: { - ..._wsHeaders, - 'Authorization': 'Bearer;$_token', - }, - ).timeout( - const Duration(seconds: 5), - onTimeout: () { - throw TimeoutException('WebSocket连接超时'); - }, - ); - - subscription = socket.listen( - (response) { - if (!completer.isCompleted && response is Uint8List) { - if (response.length < 2) return; - - final messageType = response[1] >> 4; - - if (messageType == 0x0B) { - final headerSize = response[0] & 0x0F; - final payloadStart = headerSize * 4; - - if (response.length > payloadStart) { - final audioData = response.sublist(payloadStart); - completer.complete(audioData); - } - } else if (messageType == 0x0F) { - try { - final headerSize = response[0] & 0x0F; - final payloadStart = headerSize * 4; - - if (response.length < payloadStart + 4) { - throw VolcanoTtsException('错误响应数据不完整'); - } - - final jsonData = response.sublist(payloadStart + 4); - final jsonString = utf8.decode(jsonData, allowMalformed: true); - - if (jsonString.trim().startsWith('{')) { - final errorMap = json.decode(jsonString); - throw VolcanoTtsException(errorMap['error'] as String? ?? '未知错误'); - } - } catch (e) { - if (e is VolcanoTtsException) rethrow; - throw VolcanoTtsException('服务返回错误: $e'); - } - } - } - }, - onError: (error) { - if (!completer.isCompleted) { - completer.completeError(error); - } - }, - onDone: () { - if (!completer.isCompleted) { - completer.complete(null); - } - }, - cancelOnError: false, - ); - - socket.add(fullRequest); - return await completer.future.timeout( - const Duration(seconds: 10), - onTimeout: () => null, - ); - } catch (e) { - print('合成请求失败: $e'); - return null; - } finally { - await subscription?.cancel(); - await socket?.close(); + // 增加连续错误计数 + _consecutiveErrorCount++; + if (_consecutiveErrorCount >= _maxConsecutiveErrors) { + print('连续错误次数过多,尝试重新初始化播放器'); + await _reinitializePlayer(); + _consecutiveErrorCount = 0; + } } } - Future _playAudio(Uint8List audioData) async { - if (_isDisposed) { - print('TTS服务已销毁,跳过播放'); - return; + /// 使用原生SDK合成文本为语音 + Future _synthesize(String text, {String? voiceType}) async { + if (!_isInitialized) { + try { + await _initializeTtsEngine(); + } catch (e) { + print('重新初始化TTS引擎失败: $e'); + throw VolcanoTtsException('重新初始化TTS引擎失败', e); + } } - try { - print('准备播放音频数据: ${audioData.length} bytes'); - final audioSource = BytesAudioSource(audioData); - - print('清除播放列表'); - await _playlist.clear(); - - print('添加新的音频源到播放列表'); - await _playlist.add(audioSource); - - if (!_audioPlayer.playing) { - print('设置音量和播放位置'); - await _audioPlayer.setVolume(1.0); - await _audioPlayer.seek(Duration.zero, index: 0); - - // 确保音频会话处于激活状态 - await _audioSession?.setActive(true); + // 定义基础语音类型列表,按优先级排序 + final fallbackVoiceTypes = [ + 'zh_male_qingse_common', + 'zh_female_qingse_common', + 'zh_male_M392_conversation_wvae_bigtts', + 'zh_female_F392_conversation_wvae_bigtts' + ]; + + // 确定要使用的语音类型 + final effectiveVoiceType = voiceType ?? _voiceType; + + // 如果当前语音类型不在基础列表中,将其添加到首位 + if (!fallbackVoiceTypes.contains(effectiveVoiceType)) { + fallbackVoiceTypes.insert(0, effectiveVoiceType); + } + + // 尝试使用不同的语音类型 + VolcanoTtsException? lastException; + + for (final vType in fallbackVoiceTypes) { + try { + // 调用原生方法合成语音 + final result = await _channel.invokeMethod('synthesizeSync', { + 'text': text, + 'voiceType': vType, + }); - print('等待50ms确保音频准备就绪'); - await Future.delayed(const Duration(milliseconds: 50)); + if (result != null && result.isNotEmpty) { + // 如果使用的是回退语音类型,记录日志 + if (vType != effectiveVoiceType) { + print('使用回退语音类型成功: $vType (原始类型: $effectiveVoiceType)'); + } + return result; + } + } catch (e) { + // 检查是否是资源授权错误 + final errorMsg = e.toString().toLowerCase(); + final isAuthError = errorMsg.contains('resource not granted') || + errorMsg.contains('403') || + errorMsg.contains('授权') || + errorMsg.contains('3001'); - print('开始播放'); - await _audioPlayer.play(); - print('播放命令已发送'); - } else { - print('播放器已在播放中'); + if (isAuthError) { + print('语音类型 $vType 授权错误,尝试下一个语音类型'); + lastException = VolcanoTtsException('语音类型 $vType 授权错误', e); + continue; // 尝试下一个语音类型 + } else { + // 如果不是授权错误,直接抛出 + print('调用原生合成方法失败: $e'); + throw VolcanoTtsException('调用原生合成方法失败', e); + } } - } catch (e, stack) { - print('播放音频失败: $e'); - print('错误堆栈: $stack'); - await _reinitializePlayer(); } + + // 如果所有语音类型都失败,抛出最后一个异常 + if (lastException != null) { + throw lastException; + } + + // 如果没有异常但也没有结果,返回null + return null; } + /// 停止播放并清空队列 Future stop() async { try { print('停止播放'); @@ -565,10 +635,6 @@ class VolcanoTtsService extends GetxService { // 确保清空所有待处理的音频数据 try { - // 取消所有正在进行的WebSocket连接 - await _channel?.sink.close(); - _channel = null; - // 重置播放器状态 await _audioPlayer.pause(); await _audioPlayer.seek(Duration.zero); @@ -582,7 +648,7 @@ class VolcanoTtsService extends GetxService { } } - // 检查是否实际正在播放 + /// 检查是否实际正在播放 bool isActuallyPlaying() { try { // 检查是否有活跃的播放任务 @@ -605,8 +671,6 @@ class VolcanoTtsService extends GetxService { final isAtEnd = currentDuration.inMilliseconds > 0 && currentPosition.inMilliseconds >= currentDuration.inMilliseconds - 100; - print('【TTS实际状态】检查TTS实际播放状态: hasActiveTask=$hasActiveTask, isPlayerPlaying=$isPlayerPlaying, hasPendingText=$hasPendingText, hasAudioSource=$hasAudioSource, isAtEnd=$isAtEnd'); - // 如果播放器显示正在播放,但已经到达音频末尾,则认为实际上没有播放 if (isPlayerPlaying && isAtEnd) { // 如果检测到这种情况,尝试自动修复播放器状态 @@ -617,12 +681,12 @@ class VolcanoTtsService extends GetxService { // 如果有活跃任务、播放器正在播放且未到末尾、有待处理文本,则认为正在播放 return hasActiveTask || (isPlayerPlaying && !isAtEnd) || hasPendingText; } catch (e) { - print('【TTS实际状态】检查TTS实际播放状态出错: $e'); + print('检查TTS实际播放状态出错: $e'); return false; } } - // 修复播放器状态 + /// 修复播放器状态 void _fixPlayerState() { try { // 如果播放器显示正在播放,但实际上已经到达音频末尾,尝试修复状态 @@ -652,7 +716,7 @@ class VolcanoTtsService extends GetxService { } } - // 通知播放状态变化 + /// 通知播放状态变化 void _notifyPlayingStateChanged(bool isPlaying) { try { _playingStateController.add(isPlaying); @@ -661,6 +725,7 @@ class VolcanoTtsService extends GetxService { } } + /// 重新初始化播放器 Future _reinitializePlayer() async { try { await stop(); @@ -670,15 +735,18 @@ class VolcanoTtsService extends GetxService { await _initAudioSession(); _audioPlayer = AudioPlayer(); - await _audioPlayer.setVolume(1.0); + await _audioPlayer.setVolume(_defaultVolume); await _audioPlayer.setLoopMode(LoopMode.off); + await _audioPlayer.setSpeed(_defaultSpeed); // 重新配置Android特定的播放参数 - await _audioPlayer.setAndroidAudioAttributes(const AndroidAudioAttributes( - contentType: AndroidAudioContentType.speech, - usage: AndroidAudioUsage.media, - flags: AndroidAudioFlags.audibilityEnforced, - )); + if (Platform.isAndroid) { + await _audioPlayer.setAndroidAudioAttributes(const AndroidAudioAttributes( + contentType: AndroidAudioContentType.speech, + usage: AndroidAudioUsage.media, + flags: AndroidAudioFlags.audibilityEnforced, + )); + } await _audioPlayer.setAudioSource( _playlist, @@ -693,6 +761,7 @@ class VolcanoTtsService extends GetxService { } } + /// 设置事件监听器 void _setupEventListeners() { _playbackEventSubscription?.cancel(); _playerStateSubscription?.cancel(); @@ -706,6 +775,7 @@ class VolcanoTtsService extends GetxService { }, onError: (error) { if (_isDisposed) return; + print('播放事件流错误: $error'); _reinitializePlayer(); }, ); @@ -724,11 +794,13 @@ class VolcanoTtsService extends GetxService { }, onError: (error) { if (_isDisposed) return; + print('播放状态流错误: $error'); _reinitializePlayer(); }, ); } + /// 尝试播放下一个音频片段 Future _tryPlayNext() async { if (_isDisposed || _playlist.length <= 0) return; @@ -738,7 +810,7 @@ class VolcanoTtsService extends GetxService { await _audioSession?.setActive(true); // 减少延迟时间 - await Future.delayed(const Duration(milliseconds: 30)); + await Future.delayed(const Duration(milliseconds: 20)); if (!_isDisposed && !_audioPlayer.playing) { await _audioPlayer.seek(Duration.zero, index: 0); @@ -753,6 +825,7 @@ class VolcanoTtsService extends GetxService { } } + /// 切换TTS启用状态 void toggleEnabled() { isEnabled.toggle(); if (!isEnabled.value) { @@ -760,15 +833,48 @@ class VolcanoTtsService extends GetxService { } } + /// 设置播放速度 + Future setSpeed(double speed) async { + if (speed < 0.5 || speed > 2.0) { + throw VolcanoTtsException('播放速度必须在0.5到2.0之间'); + } + + try { + await _audioPlayer.setSpeed(speed); + } catch (e) { + print('设置播放速度失败: $e'); + } + } + + /// 设置音量 + Future setVolume(double volume) async { + if (volume < 0.0 || volume > 1.0) { + throw VolcanoTtsException('音量必须在0.0到1.0之间'); + } + + try { + await _audioPlayer.setVolume(volume); + } catch (e) { + print('设置音量失败: $e'); + } + } + + /// 清理资源 Future _cleanupResources() async { _isDisposed = true; await _playbackEventSubscription?.cancel(); await _playerStateSubscription?.cancel(); await _audioSession?.setActive(false); - await _channel?.sink.close(); await stop(); await _audioPlayer.dispose(); await _playingStateController.close(); + + // 释放原生资源 + try { + await _channel.invokeMethod('dispose'); + } catch (e) { + print('释放原生TTS资源失败: $e'); + } } @override @@ -777,4 +883,71 @@ class VolcanoTtsService extends GetxService { await _cleanupResources(); super.onClose(); } + + /// 获取当前语音类型 + String getCurrentVoiceType() { + return _voiceType; + } + + /// 获取可用的语音类型列表 + List getAvailableVoiceTypes() { + return [ + // 趣味方言 + 'zh_female_wanqudashu_moon_bigtts', // 湾区大叔 + 'zh_female_daimengchuanmei_moon_bigtts', // 呆萌川妹 + 'zh_male_guozhoudege_moon_bigtts', // 广州德哥 + 'zh_male_beijingxiaoye_moon_bigtts', // 北京小爷 + 'zh_male_haoyuxiaoge_moon_bigtts', // 浩宇小哥 + + // 通用场景 + 'zh_male_shaonianzixin_moon_bigtts', // 少年梓辛/Brayan + + // 角色扮演 + 'zh_female_meilinvyou_moon_bigtts', // 魅力女友 + 'zh_male_shenyeboke_moon_bigtts', // 深夜播客 + 'zh_female_sajiaonvyou_moon_bigtts', // 柔美女友 + 'zh_female_yuanqinvyou_moon_bigtts', // 撒娇学妹 + + // 基础语音类型 + 'zh_male_qingse_common', // 基础男声 + 'zh_female_qingse_common', // 基础女声 + 'zh_male_M392_conversation_wvae_bigtts', // 高级男声 + 'zh_female_F392_conversation_wvae_bigtts', // 高级女声 + ]; + } + + /// 检查语音类型是否可能可用 + /// 注意:此方法只是基于已知的基础语音类型进行判断,不保证实际可用性 + bool isVoiceTypeLikelyAvailable(String voiceType) { + // 基础语音类型,这些通常是免费可用的 + final basicVoiceTypes = [ + 'zh_male_qingse_common', + 'zh_female_qingse_common', + ]; + + // 如果是基础语音类型,则很可能可用 + if (basicVoiceTypes.contains(voiceType)) { + return true; + } + + // 其他语音类型可能需要授权 + final availableTypes = getAvailableVoiceTypes(); + return availableTypes.contains(voiceType); + } + + /// 设置语音类型 + /// 注意:由于_voiceType是final的,此方法不会实际修改成员变量 + /// 但可以在speak方法中使用传入的voiceType参数 + void setVoiceType(String voiceType) { + // 记录请求的语音类型变更 + print('请求设置语音类型: $voiceType (当前: $_voiceType)'); + + // 检查请求的语音类型是否可能可用 + if (!isVoiceTypeLikelyAvailable(voiceType)) { + print('警告: 请求的语音类型 $voiceType 可能不可用,建议使用基础语音类型'); + } + + // 这里不修改_voiceType成员变量,因为它是final的 + // 实际的语音类型设置是在synthesize方法中使用的 + } } \ No newline at end of file diff --git a/lib/data/services/volcano_voice_recognition_service.dart b/lib/data/services/volcano_voice_recognition_service.dart new file mode 100644 index 000000000..0fd414f52 --- /dev/null +++ b/lib/data/services/volcano_voice_recognition_service.dart @@ -0,0 +1,381 @@ +import 'dart:async'; +import 'dart:math' as math; +import 'package:flutter/services.dart'; +import 'package:flutter_dotenv/flutter_dotenv.dart'; +import 'package:get/get.dart'; +import '../../core/utils/logger.dart'; + +/// 识别事件类型 +enum RecognitionEventType { + started, + recognizing, + finalResult, + error, + completed, +} + +/// 识别事件 +class RecognitionEvent { + final RecognitionEventType type; + final String text; + final String? error; + + RecognitionEvent({ + required this.type, + this.text = '', + this.error, + }); +} + +/// 火山语音识别服务 +/// +/// 该服务提供了通过平台通道与 Android 上的火山语音识别 SDK 交互的接口 +class VolcanoVoiceRecognitionService extends GetxService { + static const MethodChannel _channel = MethodChannel('com.example.deep_voice/speech_recognition'); + static const EventChannel _eventChannel = EventChannel('com.example.deep_voice/speech_recognition_events'); + + bool _isInitialized = false; + late final String _subscriptionKey; + late final String _serviceRegion; + + // 连续识别相关 + bool _isContinuousRecognitionActive = false; + StreamController? _eventStreamController; + StreamSubscription? _eventSubscription; + + // 公开的事件流 + Stream? _recognitionStream; + Stream? get recognitionStream => _recognitionStream; + + // 最新的识别结果 + final _latestRecognizedText = ''.obs; + String get latestRecognizedText => _latestRecognizedText.value; + + // 识别状态 + final isListening = false.obs; + + // 识别结果列表 + final RxList _recognitionResults = [].obs; + List get recognitionResults => _recognitionResults; + + // 错误信息 + final RxString _errorMessage = ''.obs; + String get errorMessage => _errorMessage.value; + + VolcanoVoiceRecognitionService() { + _loadConfig(); + } + + /// 从环境变量加载配置 + void _loadConfig() { + // 只使用统一的APP_ID和APP_KEY + _subscriptionKey = dotenv.env['VOLCANO_APP_ID'] ?? ''; + _serviceRegion = dotenv.env['VOLCANO_APP_KEY'] ?? ''; + + Logger.info('火山语音识别配置: APP_ID=${_subscriptionKey.isNotEmpty ? "已设置" : "未设置"}, APP_KEY=${_serviceRegion.isNotEmpty ? "已设置" : "未设置"}'); + + if (_subscriptionKey.isEmpty || _serviceRegion.isEmpty) { + throw Exception('未找到火山语音服务配置。请在 .env 文件中设置 VOLCANO_APP_ID 和 VOLCANO_APP_KEY'); + } + } + + @override + void onInit() { + super.onInit(); + _setupMethodCallHandler(); + } + + /// 设置方法通道处理器 + void _setupMethodCallHandler() { + _channel.setMethodCallHandler((call) async { + switch (call.method) { + case 'onRecognitionResult': + final String result = call.arguments as String; + _handleRecognitionResult(result); + break; + case 'onRecognitionError': + final String error = call.arguments as String; + _handleRecognitionError(error); + break; + case 'onRecognitionComplete': + _handleRecognitionComplete(); + break; + } + }); + } + + /// 处理识别结果 + void _handleRecognitionResult(String result) { + Logger.debug('Recognition result: $result'); + _recognitionResults.add(result); + _latestRecognizedText.value = result; + + if (_eventStreamController != null) { + _eventStreamController!.add(RecognitionEvent( + type: RecognitionEventType.finalResult, + text: result, + )); + } + } + + /// 处理识别错误 + void _handleRecognitionError(String error) { + Logger.error('Recognition error: $error'); + _errorMessage.value = error; + isListening.value = false; + + if (_eventStreamController != null) { + _eventStreamController!.add(RecognitionEvent( + type: RecognitionEventType.error, + text: '', + error: error, + )); + } + } + + /// 处理识别完成 + void _handleRecognitionComplete() { + Logger.debug('Recognition complete'); + isListening.value = false; + _isContinuousRecognitionActive = false; + + if (_eventStreamController != null) { + _eventStreamController!.add(RecognitionEvent( + type: RecognitionEventType.completed, + text: '', + )); + } + } + + /// 初始化语音识别引擎 + Future initialize() async { + if (_isInitialized) return true; + + try { + Logger.info('开始初始化火山语音识别服务,APP_ID: ${_subscriptionKey.substring(0, math.min(3, _subscriptionKey.length))}***,APP_KEY: ${_serviceRegion.length > 10 ? "${_serviceRegion.substring(0, 5)}..." : _serviceRegion}'); + + final bool result = await _channel.invokeMethod('initialize', { + 'subscriptionKey': _subscriptionKey, + 'serviceRegion': _serviceRegion, + }); + + _isInitialized = result; + Logger.info('火山语音识别服务初始化${result ? '成功' : '失败'}'); + return result; + } on PlatformException catch (e) { + // 检查是否是资源授权错误 + if (e.code == 'INITIALIZATION_ERROR' && + (e.message?.contains('资源授权错误') == true || + e.message?.contains('requested resource not granted') == true || + e.message?.contains('requested grant not found') == true)) { + Logger.error('火山语音识别初始化失败: 资源授权错误', e, StackTrace.current); + Logger.info('请检查以下几点:'); + Logger.info('1. 确保您的火山引擎账户已开通语音识别服务'); + Logger.info('2. 确保您的应用ID和密钥正确且有效'); + Logger.info('3. 确保您的应用已被授权使用语音识别服务'); + _isInitialized = false; + throw PlatformException( + code: 'RESOURCE_AUTHORIZATION_ERROR', + message: '语音识别服务授权失败: 请确保应用已开通语音识别服务并且密钥有效', + details: e.message + ); + } else { + Logger.error('火山语音识别初始化失败: ${e.message}', e, StackTrace.current); + _isInitialized = false; + throw e; + } + } catch (e) { + Logger.error('火山语音识别初始化发生未知错误', e, StackTrace.current); + _isInitialized = false; + throw Exception('初始化火山语音识别服务失败: $e'); + } + } + + /// 开始一次性识别 + Future startOneTimeRecognition() async { + if (isListening.value) { + Logger.warning('已经在进行语音识别,请先停止当前识别'); + return false; + } + + if (!_isInitialized) { + try { + final bool initialized = await initialize(); + if (!initialized) { + Logger.error('语音识别服务未初始化,无法开始识别'); + _errorMessage.value = '语音识别服务未初始化,无法开始识别'; + return false; + } + } catch (e) { + Logger.error('初始化语音识别服务失败: $e'); + _errorMessage.value = '初始化语音识别服务失败: $e'; + return false; + } + } + + try { + _errorMessage.value = ''; + _recognitionResults.clear(); + + final bool result = await _channel.invokeMethod('startOneTimeRecognition'); + isListening.value = result; + + if (result) { + _eventStreamController?.add(RecognitionEvent( + type: RecognitionEventType.started, + text: '', + )); + } + + return result; + } catch (e) { + Logger.error('开始一次性识别失败: $e'); + _errorMessage.value = e.toString(); + return false; + } + } + + /// 开始连续识别 + Future startContinuousRecognition() async { + if (isListening.value) { + Logger.warning('已经在进行语音识别,请先停止当前识别'); + return false; + } + + if (!_isInitialized) { + try { + final bool initialized = await initialize(); + if (!initialized) { + Logger.error('语音识别服务未初始化,无法开始识别'); + _errorMessage.value = '语音识别服务未初始化,无法开始识别'; + return false; + } + } catch (e) { + Logger.error('初始化语音识别服务失败: $e'); + _errorMessage.value = '初始化语音识别服务失败: $e'; + return false; + } + } + + try { + _errorMessage.value = ''; + _recognitionResults.clear(); + + // 创建事件流控制器 + _eventStreamController = StreamController.broadcast(); + _recognitionStream = _eventStreamController?.stream; + + // 设置事件监听 + _eventSubscription = _eventChannel + .receiveBroadcastStream() + .listen(_handleNativeEvent, onError: (error) { + _handleRecognitionError(error.toString()); + }); + + final bool result = await _channel.invokeMethod('startContinuousRecognition'); + isListening.value = result; + _isContinuousRecognitionActive = result; + + if (result) { + _eventStreamController?.add(RecognitionEvent( + type: RecognitionEventType.started, + text: '', + )); + } + + return result; + } catch (e) { + Logger.error('开始连续识别失败: $e'); + _errorMessage.value = e.toString(); + _cleanupEventStream(); + return false; + } + } + + /// 处理来自原生端的事件 + void _handleNativeEvent(dynamic event) { + if (event is! Map) return; + + final Map eventMap = event; + final String eventType = eventMap['eventType'] as String? ?? ''; + + switch (eventType) { + case 'recognizing': + final String text = eventMap['text'] as String? ?? ''; + _eventStreamController?.add(RecognitionEvent( + type: RecognitionEventType.recognizing, + text: text, + )); + break; + case 'finalResult': + final String text = eventMap['text'] as String? ?? ''; + _latestRecognizedText.value = text; + _recognitionResults.add(text); + _eventStreamController?.add(RecognitionEvent( + type: RecognitionEventType.finalResult, + text: text, + )); + break; + case 'error': + final String error = eventMap['error'] as String? ?? '未知错误'; + _handleRecognitionError(error); + break; + } + } + + /// 清理事件流 + void _cleanupEventStream() { + _eventSubscription?.cancel(); + _eventSubscription = null; + _eventStreamController?.close(); + _eventStreamController = null; + _recognitionStream = null; + } + + /// 停止识别 + Future stopRecognition() async { + if (!isListening.value) { + Logger.warning('当前没有进行语音识别'); + return false; + } + + try { + final bool result = await _channel.invokeMethod('stopRecognition'); + isListening.value = !result; + _isContinuousRecognitionActive = !result; + + if (result) { + _cleanupEventStream(); + } + + return result; + } catch (e) { + Logger.error('停止识别失败: $e'); + _errorMessage.value = e.toString(); + return false; + } + } + + /// 检查连续识别是否活跃 + bool isContinuousRecognitionActive() { + return _isContinuousRecognitionActive; + } + + /// 清理资源 + Future dispose() async { + try { + if (isListening.value) { + await stopRecognition(); + } + _cleanupEventStream(); + } catch (e) { + Logger.error('清理语音识别资源失败: $e'); + } + } + + @override + void onClose() { + dispose(); + super.onClose(); + } +} \ No newline at end of file diff --git a/lib/modules/chat/controllers/voice_input_controller.dart b/lib/modules/chat/controllers/voice_input_controller.dart index d04732dee..47ab0a165 100644 --- a/lib/modules/chat/controllers/voice_input_controller.dart +++ b/lib/modules/chat/controllers/voice_input_controller.dart @@ -2,8 +2,9 @@ import 'dart:async'; import 'dart:math'; import 'package:get/get.dart'; import 'package:flutter/foundation.dart'; -import '../../../data/services/voice_recognition_service.dart'; +import '../../../data/services/volcano_voice_recognition_service.dart'; import '../../../data/services/volcano_tts_service.dart'; +import '../../../core/utils/logger.dart'; class VoiceInputController extends GetxController { // Observable states @@ -16,7 +17,7 @@ class VoiceInputController extends GetxController { final isUserSpeaking = false.obs; // 语音识别服务 - late final VoiceRecognitionService _voiceService; + late final VolcanoVoiceRecognitionService _voiceService; late final VolcanoTtsService _ttsService; // 连续识别相关 @@ -67,8 +68,8 @@ class VoiceInputController extends GetxController { Future _initializeVoiceService() async { try { - // 创建语音识别服务实例 - _voiceService = VoiceRecognitionService(); + // 获取语音识别服务实例 + _voiceService = Get.find(); // 初始化语音识别服务 await _voiceService.initialize(); @@ -79,7 +80,7 @@ class VoiceInputController extends GetxController { // 自动开始连续语音识别 await startContinuousRecognition(); } catch (e) { - print('初始化语音识别服务失败: $e'); + Logger.error('初始化语音识别服务失败', e); Get.snackbar( 'Error', '初始化语音识别服务失败: $e', @@ -103,12 +104,16 @@ class VoiceInputController extends GetxController { _recognitionCompleter = Completer(); // 开始连续语音识别 - final stream = await _voiceService.startContinuousRecognition(); + final success = await _voiceService.startContinuousRecognition(); + if (!success) { + isRecording.value = false; + throw Exception('无法启动语音识别'); + } // 订阅识别事件流 - _recognitionSubscription = stream.listen((event) { + _recognitionSubscription = _voiceService.recognitionStream?.listen((event) { switch (event.type) { - case RecognitionEventType.intermediateResult: + case RecognitionEventType.recognizing: // 不再更新面板中的识别文本,而是通过回调传递给ChatController if (event.text.isNotEmpty) { _hasRecognizedSpeech = true; @@ -196,7 +201,6 @@ class VoiceInputController extends GetxController { break; case RecognitionEventType.error: - case RecognitionEventType.canceled: // 处理错误 print('语音识别错误: ${event.error}'); Get.snackbar( @@ -208,7 +212,7 @@ class VoiceInputController extends GetxController { restartRecognition(); break; - case RecognitionEventType.sessionStopped: + case RecognitionEventType.completed: // 会话结束,尝试重新启动 isRecording.value = false; restartRecognition(); @@ -276,7 +280,7 @@ class VoiceInputController extends GetxController { // 停止连续识别 if (_voiceService.isContinuousRecognitionActive()) { - await _voiceService.stopContinuousRecognition(); + await _voiceService.stopRecognition(); } isRecording.value = false; diff --git a/lib/modules/microsoft_tts_continuous_example.dart b/lib/modules/microsoft_tts_continuous_example.dart deleted file mode 100644 index 24c5e0990..000000000 --- a/lib/modules/microsoft_tts_continuous_example.dart +++ /dev/null @@ -1,308 +0,0 @@ -import 'package:flutter/material.dart'; -import 'package:get/get.dart'; -import '../data/services/microsoft_tts_service.dart'; - -/// 微软 TTS 连续语音输出示例页面 -class MicrosoftTtsContinuousExample extends StatefulWidget { - const MicrosoftTtsContinuousExample({Key? key}) : super(key: key); - - @override - State createState() => _MicrosoftTtsContinuousExampleState(); -} - -class _MicrosoftTtsContinuousExampleState extends State { - final MicrosoftTtsService _ttsService = Get.find(); - - // 语音列表 - final List> _voices = [ - {'name': '晓晓(女声)', 'value': 'zh-CN-XiaoxiaoNeural'}, - {'name': '云扬(男声)', 'value': 'zh-CN-YunyangNeural'}, - {'name': '晓双(女声)', 'value': 'zh-CN-XiaoshuangNeural'}, - {'name': '云皓(男声)', 'value': 'zh-CN-YunhaoNeural'}, - {'name': '晓墨(女声)', 'value': 'zh-CN-XiaomoNeural'}, - {'name': '云泽(男声)', 'value': 'zh-CN-YunzeNeural'}, - ]; - - String _selectedVoice = 'zh-CN-XiaoxiaoNeural'; - double _rate = 0; - double _pitch = 0; - bool _isLoading = false; - String _statusMessage = ''; - - // 预设的连续语音文本 - final List _presetTexts = [ - '欢迎使用微软语音合成服务,这是连续语音输出的第一段文本。', - '这是第二段文本,用于测试连续语音输出功能。', - '现在是第三段文本,我们正在测试微软语音合成服务的连续合成能力。', - '最后一段测试文本,感谢您的收听。', - ]; - - // 自定义文本列表 - final List _textControllers = []; - - @override - void initState() { - super.initState(); - // 初始化文本控制器 - for (final text in _presetTexts) { - _textControllers.add(TextEditingController(text: text)); - } - } - - @override - void dispose() { - // 释放文本控制器 - for (final controller in _textControllers) { - controller.dispose(); - } - super.dispose(); - } - - /// 播放连续文本 - Future _speakContinuous() async { - final texts = _textControllers.map((controller) => controller.text).toList(); - - if (texts.every((text) => text.isEmpty)) { - _showSnackBar('请至少输入一段文本'); - return; - } - - setState(() { - _isLoading = true; - _statusMessage = '正在合成语音...'; - }); - - try { - // 设置语音 - await _ttsService.setVoice(_selectedVoice); - - // 停止之前的播放 - await _ttsService.stop(); - - // 连续播放多段文本 - final result = await _ttsService.speakMultiple( - texts.where((text) => text.isNotEmpty).toList(), - rate: _rate.round(), - pitch: _pitch.round(), - ); - - setState(() { - _statusMessage = result ? '语音合成已加入队列' : '语音合成失败'; - }); - } catch (e) { - _showSnackBar('语音合成失败: $e'); - } finally { - setState(() { - _isLoading = false; - }); - } - } - - /// 停止播放 - Future _stopSpeaking() async { - try { - await _ttsService.stop(); - setState(() { - _statusMessage = '语音合成已停止'; - }); - } catch (e) { - _showSnackBar('停止语音合成失败: $e'); - } - } - - /// 添加文本输入框 - void _addTextInput() { - setState(() { - _textControllers.add(TextEditingController()); - }); - } - - /// 删除文本输入框 - void _removeTextInput(int index) { - if (_textControllers.length <= 1) { - _showSnackBar('至少需要保留一个文本输入框'); - return; - } - - setState(() { - _textControllers[index].dispose(); - _textControllers.removeAt(index); - }); - } - - /// 显示提示信息 - void _showSnackBar(String message) { - ScaffoldMessenger.of(context).showSnackBar( - SnackBar(content: Text(message)), - ); - } - - @override - Widget build(BuildContext context) { - return Scaffold( - appBar: AppBar( - title: const Text('微软连续语音合成示例'), - ), - body: Padding( - padding: const EdgeInsets.all(16.0), - child: ListView( - children: [ - // 语音选择 - DropdownButtonFormField( - value: _selectedVoice, - decoration: const InputDecoration( - labelText: '选择语音', - border: OutlineInputBorder(), - ), - items: _voices.map((voice) { - return DropdownMenuItem( - value: voice['value'], - child: Text(voice['name']!), - ); - }).toList(), - onChanged: (value) { - if (value != null) { - setState(() { - _selectedVoice = value; - }); - } - }, - ), - const SizedBox(height: 16), - - // 语速调节 - Row( - children: [ - const Text('语速:'), - Expanded( - child: Slider( - min: -100, - max: 100, - divisions: 20, - value: _rate, - label: _rate.round().toString(), - onChanged: (value) { - setState(() { - _rate = value; - }); - }, - ), - ), - Text('${_rate.round()}%'), - ], - ), - - // 音调调节 - Row( - children: [ - const Text('音调:'), - Expanded( - child: Slider( - min: -100, - max: 100, - divisions: 20, - value: _pitch, - label: _pitch.round().toString(), - onChanged: (value) { - setState(() { - _pitch = value; - }); - }, - ), - ), - Text('${_pitch.round()}%'), - ], - ), - - const SizedBox(height: 16), - - // 文本输入列表标题 - Row( - mainAxisAlignment: MainAxisAlignment.spaceBetween, - children: [ - const Text( - '连续语音文本', - style: TextStyle( - fontSize: 16, - fontWeight: FontWeight.bold, - ), - ), - ElevatedButton.icon( - onPressed: _addTextInput, - icon: const Icon(Icons.add), - label: const Text('添加文本'), - ), - ], - ), - - const SizedBox(height: 8), - - // 文本输入列表 - ...List.generate(_textControllers.length, (index) { - return Padding( - padding: const EdgeInsets.only(bottom: 8.0), - child: Row( - crossAxisAlignment: CrossAxisAlignment.start, - children: [ - Expanded( - child: TextField( - controller: _textControllers[index], - maxLines: 3, - decoration: InputDecoration( - labelText: '文本 ${index + 1}', - border: const OutlineInputBorder(), - ), - ), - ), - IconButton( - icon: const Icon(Icons.delete), - onPressed: () => _removeTextInput(index), - ), - ], - ), - ); - }), - - const SizedBox(height: 16), - - // 操作按钮 - Row( - mainAxisAlignment: MainAxisAlignment.spaceEvenly, - children: [ - Expanded( - child: ElevatedButton.icon( - onPressed: _isLoading ? null : _speakContinuous, - icon: const Icon(Icons.play_arrow), - label: const Text('播放连续语音'), - ), - ), - const SizedBox(width: 8), - Expanded( - child: ElevatedButton.icon( - onPressed: _stopSpeaking, - icon: const Icon(Icons.stop), - label: const Text('停止'), - style: ElevatedButton.styleFrom( - backgroundColor: Colors.red, - ), - ), - ), - ], - ), - - const SizedBox(height: 16), - - // 状态信息 - Obx(() => Text( - _ttsService.isSpeaking.value - ? '正在播放语音...' - : _statusMessage, - style: const TextStyle(fontStyle: FontStyle.italic), - textAlign: TextAlign.center, - )), - ], - ), - ), - ); - } -} \ No newline at end of file diff --git a/lib/modules/microsoft_tts_example.dart b/lib/modules/microsoft_tts_example.dart deleted file mode 100644 index 51ebeff45..000000000 --- a/lib/modules/microsoft_tts_example.dart +++ /dev/null @@ -1,202 +0,0 @@ -import 'package:flutter/material.dart'; -import 'package:get/get.dart'; -import '../data/services/microsoft_tts_service.dart'; - -/// 微软 TTS 示例页面 -class MicrosoftTtsExample extends StatefulWidget { - const MicrosoftTtsExample({Key? key}) : super(key: key); - - @override - State createState() => _MicrosoftTtsExampleState(); -} - -class _MicrosoftTtsExampleState extends State { - final TextEditingController _textController = TextEditingController(); - final MicrosoftTtsService _ttsService = Get.find(); - - // 语音列表 - final List> _voices = [ - {'name': '晓晓(女声)', 'value': 'zh-CN-XiaoxiaoNeural'}, - {'name': '云扬(男声)', 'value': 'zh-CN-YunyangNeural'}, - {'name': '晓双(女声)', 'value': 'zh-CN-XiaoshuangNeural'}, - {'name': '云皓(男声)', 'value': 'zh-CN-YunhaoNeural'}, - {'name': '晓墨(女声)', 'value': 'zh-CN-XiaomoNeural'}, - {'name': '云泽(男声)', 'value': 'zh-CN-YunzeNeural'}, - ]; - - String _selectedVoice = 'zh-CN-XiaoxiaoNeural'; - double _rate = 0; - double _pitch = 0; - bool _isLoading = false; - String _statusMessage = ''; - - @override - void initState() { - super.initState(); - _textController.text = '欢迎使用微软语音合成服务,这是一个示例文本。'; - } - - @override - void dispose() { - _textController.dispose(); - super.dispose(); - } - - /// 播放文本 - Future _speakText() async { - if (_textController.text.isEmpty) { - _showSnackBar('请输入要合成的文本'); - return; - } - - setState(() { - _isLoading = true; - _statusMessage = '正在合成语音...'; - }); - - try { - // 设置语音 - await _ttsService.setVoice(_selectedVoice); - - // 生成 SSML - final ssml = _ttsService.generateSsml( - text: _textController.text, - rate: _rate.round(), - pitch: _pitch.round(), - ); - - // 播放 SSML - final result = await _ttsService.speakSsml(ssml); - - setState(() { - _statusMessage = result; - }); - } catch (e) { - _showSnackBar('语音合成失败: $e'); - } finally { - setState(() { - _isLoading = false; - }); - } - } - - /// 显示提示信息 - void _showSnackBar(String message) { - ScaffoldMessenger.of(context).showSnackBar( - SnackBar(content: Text(message)), - ); - } - - @override - Widget build(BuildContext context) { - return Scaffold( - appBar: AppBar( - title: const Text('微软语音合成示例'), - ), - body: Padding( - padding: const EdgeInsets.all(16.0), - child: Column( - crossAxisAlignment: CrossAxisAlignment.stretch, - children: [ - // 文本输入框 - TextField( - controller: _textController, - maxLines: 5, - decoration: const InputDecoration( - labelText: '输入要合成的文本', - border: OutlineInputBorder(), - ), - ), - const SizedBox(height: 16), - - // 语音选择 - DropdownButtonFormField( - value: _selectedVoice, - decoration: const InputDecoration( - labelText: '选择语音', - border: OutlineInputBorder(), - ), - items: _voices.map((voice) { - return DropdownMenuItem( - value: voice['value'], - child: Text(voice['name']!), - ); - }).toList(), - onChanged: (value) { - if (value != null) { - setState(() { - _selectedVoice = value; - }); - } - }, - ), - const SizedBox(height: 16), - - // 语速调节 - Row( - children: [ - const Text('语速:'), - Expanded( - child: Slider( - min: -100, - max: 100, - divisions: 20, - value: _rate, - label: _rate.round().toString(), - onChanged: (value) { - setState(() { - _rate = value; - }); - }, - ), - ), - Text('${_rate.round()}%'), - ], - ), - - // 音调调节 - Row( - children: [ - const Text('音调:'), - Expanded( - child: Slider( - min: -100, - max: 100, - divisions: 20, - value: _pitch, - label: _pitch.round().toString(), - onChanged: (value) { - setState(() { - _pitch = value; - }); - }, - ), - ), - Text('${_pitch.round()}%'), - ], - ), - - const SizedBox(height: 16), - - // 播放按钮 - ElevatedButton( - onPressed: _isLoading ? null : _speakText, - child: _isLoading - ? const CircularProgressIndicator() - : const Text('播放'), - ), - - const SizedBox(height: 16), - - // 状态信息 - Text( - _statusMessage, - style: const TextStyle(fontStyle: FontStyle.italic), - textAlign: TextAlign.center, - ), - ], - ), - ), - ); - } -} \ No newline at end of file diff --git a/lib/modules/profile/views/profile_view.dart b/lib/modules/profile/views/profile_view.dart index 90083b154..b4d6f264d 100644 --- a/lib/modules/profile/views/profile_view.dart +++ b/lib/modules/profile/views/profile_view.dart @@ -4,6 +4,7 @@ import '../controllers/profile_controller.dart'; import 'package:flutter_screenutil/flutter_screenutil.dart'; import '../../../core/widgets/common_bottom_nav.dart'; import '../../../data/services/audio_service.dart'; +import '../../../core/routes/app_routes.dart'; class ProfileView extends GetView { const ProfileView({Key? key}) : super(key: key); @@ -116,6 +117,24 @@ class ProfileView extends GetView { } }, ), + const Divider(), + _buildMenuItem( + title: '火山语音合成测试'.tr, + icon: Icons.record_voice_over, + subtitle: '测试火山语音TTS功能', + onTap: () { + Get.toNamed(Routes.TTS_TEST); + }, + ), + const Divider(), + _buildMenuItem( + title: '火山语音识别测试'.tr, + icon: Icons.mic, + subtitle: '测试火山语音ASR功能', + onTap: () { + Get.toNamed(Routes.ASR_TEST); + }, + ), ], ), bottomNavigationBar: const CommonBottomNav(currentIndex: 2), diff --git a/lib/modules/speech_demo/speech_demo_page.dart b/lib/modules/speech_demo/speech_demo_page.dart deleted file mode 100644 index 0070ca9b6..000000000 --- a/lib/modules/speech_demo/speech_demo_page.dart +++ /dev/null @@ -1,184 +0,0 @@ -import 'dart:async'; -import 'dart:developer' as developer; -import 'dart:io'; - -import 'package:flutter/material.dart'; -import 'package:flutter/services.dart'; -import 'package:get/get.dart'; -import 'package:permission_handler/permission_handler.dart'; -import 'package:device_info_plus/device_info_plus.dart'; -import 'package:path_provider/path_provider.dart'; -import 'package:flutter_dotenv/flutter_dotenv.dart'; - -import '../../data/services/voice_recognition_service.dart'; - -class SpeechDemoPage extends StatefulWidget { - const SpeechDemoPage({Key? key}) : super(key: key); - - @override - State createState() => _SpeechDemoPageState(); -} - -class _SpeechDemoPageState extends State { - late final VoiceRecognitionService _voiceService; - bool _isInitialized = false; - bool _isRecognizing = false; - String _recognizedText = ''; - String _statusMessage = '准备就绪'; - - @override - void initState() { - super.initState(); - _initializeService(); - } - - Future _initializeService() async { - setState(() { - _statusMessage = '正在初始化语音服务...'; - }); - - try { - // 创建服务实例(这一步会自动从环境变量加载配置) - _voiceService = VoiceRecognitionService(); - - // 初始化语音识别服务 - final result = await _voiceService.initialize(); - - setState(() { - _isInitialized = result; - _statusMessage = '语音服务初始化成功,可以开始识别'; - }); - } catch (e) { - setState(() { - _isInitialized = false; - _statusMessage = '初始化失败: $e'; - }); - } - } - - Future _startVoiceRecognition() async { - if (!_isInitialized) { - setState(() { - _statusMessage = '语音服务未初始化,请先初始化'; - }); - return; - } - - setState(() { - _isRecognizing = true; - _statusMessage = '正在聆听...'; - }); - - try { - final result = await _voiceService.recognizeSpeech(); - setState(() { - _recognizedText = result; - _isRecognizing = false; - _statusMessage = '识别完成'; - }); - } catch (e) { - setState(() { - _isRecognizing = false; - _statusMessage = '识别失败: $e'; - }); - } - } - - @override - Widget build(BuildContext context) { - return Scaffold( - appBar: AppBar( - title: const Text('语音识别演示'), - ), - body: Padding( - padding: const EdgeInsets.all(16.0), - child: Column( - crossAxisAlignment: CrossAxisAlignment.stretch, - children: [ - // 状态信息 - Container( - padding: const EdgeInsets.all(12), - decoration: BoxDecoration( - color: Colors.grey[200], - borderRadius: BorderRadius.circular(8), - ), - child: Text( - _statusMessage, - style: TextStyle( - color: _isInitialized ? Colors.green[700] : Colors.red[700], - fontWeight: FontWeight.bold, - ), - ), - ), - - const SizedBox(height: 24), - - // 识别结果显示区域 - Expanded( - child: Container( - padding: const EdgeInsets.all(16), - decoration: BoxDecoration( - border: Border.all(color: Colors.grey[300]!), - borderRadius: BorderRadius.circular(8), - ), - child: _recognizedText.isEmpty - ? const Center( - child: Text( - '识别结果将显示在这里', - style: TextStyle(color: Colors.grey), - ), - ) - : SingleChildScrollView( - child: Text( - _recognizedText, - style: const TextStyle( - fontSize: 18, - height: 1.5, - ), - ), - ), - ), - ), - - const SizedBox(height: 24), - - // 操作按钮 - Row( - mainAxisAlignment: MainAxisAlignment.spaceEvenly, - children: [ - ElevatedButton( - onPressed: _isRecognizing ? null : _initializeService, - style: ElevatedButton.styleFrom( - padding: const EdgeInsets.symmetric(horizontal: 24, vertical: 12), - ), - child: const Text('重新初始化'), - ), - ElevatedButton( - onPressed: _isRecognizing || !_isInitialized ? null : _startVoiceRecognition, - style: ElevatedButton.styleFrom( - padding: const EdgeInsets.symmetric(horizontal: 24, vertical: 12), - backgroundColor: Colors.blue[700], - ), - child: Text(_isRecognizing ? '正在识别...' : '开始语音识别'), - ), - ], - ), - - const SizedBox(height: 16), - - // 提示信息 - const Text( - '提示:请在安静的环境中使用,并确保已授予应用录音权限。', - style: TextStyle( - fontSize: 12, - fontStyle: FontStyle.italic, - color: Colors.grey, - ), - textAlign: TextAlign.center, - ), - ], - ), - ), - ); - } -} \ No newline at end of file diff --git a/lib/modules/test/bindings/test_binding.dart b/lib/modules/test/bindings/test_binding.dart new file mode 100644 index 000000000..1f38e629d --- /dev/null +++ b/lib/modules/test/bindings/test_binding.dart @@ -0,0 +1,21 @@ +import 'package:get/get.dart'; +import '../controllers/tts_test_controller.dart'; +import '../controllers/asr_test_controller.dart'; + +class TtsTestBinding extends Bindings { + @override + void dependencies() { + Get.lazyPut( + () => TtsTestController(), + ); + } +} + +class AsrTestBinding extends Bindings { + @override + void dependencies() { + Get.lazyPut( + () => AsrTestController(), + ); + } +} \ No newline at end of file diff --git a/lib/modules/test/controllers/asr_test_controller.dart b/lib/modules/test/controllers/asr_test_controller.dart new file mode 100644 index 000000000..609a5b0ac --- /dev/null +++ b/lib/modules/test/controllers/asr_test_controller.dart @@ -0,0 +1,103 @@ +import 'dart:async'; +import 'package:get/get.dart'; +import '../../../data/services/volcano_voice_recognition_service.dart'; + +class AsrTestController extends GetxController { + final VolcanoVoiceRecognitionService _voiceService = Get.find(); + + // 可观察状态 + final isListening = false.obs; + final errorMessage = ''.obs; + final isContinuous = true.obs; + final recognitionResults = [].obs; + + // 连续识别相关 + StreamSubscription? _recognitionSubscription; + + @override + void onInit() { + super.onInit(); + + // 监听语音识别服务的状态 + _voiceService.isListening.listen((listening) { + isListening.value = listening; + if (!listening) { + errorMessage.value = ''; + } + }); + } + + /// 开始录音 + void startListening() async { + try { + errorMessage.value = ''; + + if (isContinuous.value) { + // 连续识别模式 + final success = await _voiceService.startContinuousRecognition(); + if (!success) { + errorMessage.value = '无法启动语音识别'; + return; + } + + _recognitionSubscription = _voiceService.recognitionStream?.listen( + (event) { + switch (event.type) { + case RecognitionEventType.finalResult: + if (event.text.isNotEmpty) { + recognitionResults.add(event.text); + } + break; + case RecognitionEventType.error: + errorMessage.value = event.error ?? '未知错误'; + break; + default: + break; + } + }, + onError: (error) { + errorMessage.value = error.toString(); + isListening.value = false; + }, + ); + } else { + // 一次性识别模式 + final success = await _voiceService.startOneTimeRecognition(); + if (!success) { + errorMessage.value = '无法启动语音识别'; + return; + } + + // 一次性识别模式下,结果会通过服务的recognitionResults获取 + // 在onInit中我们已经监听了isListening状态,当识别完成时会自动更新UI + } + } catch (e) { + errorMessage.value = e.toString(); + isListening.value = false; + } + } + + /// 停止录音 + void stopListening() async { + try { + if (isContinuous.value) { + await _voiceService.stopRecognition(); + await _recognitionSubscription?.cancel(); + _recognitionSubscription = null; + } + } catch (e) { + errorMessage.value = e.toString(); + } + } + + /// 清空结果 + void clearResults() { + recognitionResults.clear(); + } + + @override + void onClose() { + _recognitionSubscription?.cancel(); + super.onClose(); + } +} \ No newline at end of file diff --git a/lib/modules/test/controllers/tts_test_controller.dart b/lib/modules/test/controllers/tts_test_controller.dart new file mode 100644 index 000000000..15813f0ac --- /dev/null +++ b/lib/modules/test/controllers/tts_test_controller.dart @@ -0,0 +1,123 @@ +import 'package:flutter/material.dart'; +import 'package:get/get.dart'; +import '../../../data/services/volcano_tts_service.dart'; + +class TtsTestController extends GetxController { + final VolcanoTtsService _ttsService = Get.find(); + + // 文本控制器 + final TextEditingController textController = TextEditingController(); + + // 可观察状态 + final isPlaying = false.obs; + final errorMessage = ''.obs; + final selectedVoiceType = ''.obs; + final voiceTypes = [].obs; + final isLoading = false.obs; + + @override + void onInit() { + super.onInit(); + + // 监听TTS播放状态 + _ttsService.playingStateStream.listen((playing) { + isPlaying.value = playing; + if (!playing) { + errorMessage.value = ''; + } + }); + + // 设置默认文本 + textController.text = '这是一个火山语音合成测试,请点击播放按钮听取合成效果。'; + + // 设置可用的语音类型 + voiceTypes.value = [ + // 趣味方言 + 'zh_female_wanqudashu_moon_bigtts', // 湾区大叔 + 'zh_female_daimengchuanmei_moon_bigtts', // 呆萌川妹 + 'zh_male_guozhoudege_moon_bigtts', // 广州德哥 + 'zh_male_beijingxiaoye_moon_bigtts', // 北京小爷 + 'zh_male_haoyuxiaoge_moon_bigtts', // 浩宇小哥 + + // 通用场景 + 'zh_male_shaonianzixin_moon_bigtts', // 少年梓辛/Brayan + + // 角色扮演 + 'zh_female_meilinvyou_moon_bigtts', // 魅力女友 + 'zh_male_shenyeboke_moon_bigtts', // 深夜播客 + 'zh_female_sajiaonvyou_moon_bigtts', // 柔美女友 + 'zh_female_yuanqinvyou_moon_bigtts', // 撒娇学妹 + + ]; + + // 设置当前选中的语音类型 + selectedVoiceType.value = _ttsService.getCurrentVoiceType(); + + // 如果当前语音类型不在列表中,添加到列表 + if (!voiceTypes.contains(selectedVoiceType.value)) { + voiceTypes.add(selectedVoiceType.value); + } + } + + /// 播放文本 + void speakText() async { + final text = textController.text.trim(); + if (text.isEmpty) { + errorMessage.value = '请输入要合成的文本'; + return; + } + + errorMessage.value = ''; + isLoading.value = true; + + try { + // 获取当前语音类型 + final currentVoiceType = selectedVoiceType.value; + + // 检查语音类型是否可能可用 + if (!_ttsService.isVoiceTypeLikelyAvailable(currentVoiceType)) { + print('警告: 选择的语音类型 $currentVoiceType 可能不可用,但仍将尝试使用'); + } + + // 播放文本,传递选定的语音类型 + await _ttsService.speak(text, voiceType: currentVoiceType); + } catch (e) { + errorMessage.value = e.toString(); + isPlaying.value = false; + + // 如果错误包含资源授权相关信息,提供更具体的提示 + final errorMsg = e.toString().toLowerCase(); + if (errorMsg.contains('resource not granted') || + errorMsg.contains('403') || + errorMsg.contains('授权') || + errorMsg.contains('3001')) { + errorMessage.value = '语音类型授权错误: 您可能没有权限使用当前选择的语音类型,请尝试使用基础语音类型'; + } + } finally { + isLoading.value = false; + } + } + + /// 停止播放 + void stopSpeaking() { + try { + _ttsService.stop(); + } catch (e) { + errorMessage.value = e.toString(); + } + } + + /// 切换语音类型 + void changeVoiceType(String voiceType) { + if (voiceType != selectedVoiceType.value) { + selectedVoiceType.value = voiceType; + errorMessage.value = ''; + } + } + + @override + void onClose() { + textController.dispose(); + super.onClose(); + } +} \ No newline at end of file diff --git a/lib/modules/test/views/asr_test_view.dart b/lib/modules/test/views/asr_test_view.dart new file mode 100644 index 000000000..f93ae2879 --- /dev/null +++ b/lib/modules/test/views/asr_test_view.dart @@ -0,0 +1,232 @@ +import 'package:flutter/material.dart'; +import 'package:get/get.dart'; +import '../controllers/asr_test_controller.dart'; + +class AsrTestView extends GetView { + const AsrTestView({Key? key}) : super(key: key); + + @override + Widget build(BuildContext context) { + return Scaffold( + appBar: AppBar( + title: const Text('火山语音识别测试'), + centerTitle: true, + ), + body: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.stretch, + children: [ + // 识别模式选择 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '识别模式', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + Obx(() => Row( + children: [ + Expanded( + child: RadioListTile( + title: const Text('一次性识别'), + value: false, + groupValue: controller.isContinuous.value, + onChanged: (value) { + if (value != null) { + controller.isContinuous.value = value; + } + }, + ), + ), + Expanded( + child: RadioListTile( + title: const Text('连续识别'), + value: true, + groupValue: controller.isContinuous.value, + onChanged: (value) { + if (value != null) { + controller.isContinuous.value = value; + } + }, + ), + ), + ], + )), + ], + ), + ), + ), + + const SizedBox(height: 16), + + // 录音控制 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '录音控制', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 16), + Center( + child: Obx(() => controller.isListening.value + ? ElevatedButton.icon( + onPressed: () => controller.stopListening(), + icon: const Icon(Icons.stop), + label: const Text('停止录音'), + style: ElevatedButton.styleFrom( + backgroundColor: Colors.red, + foregroundColor: Colors.white, + padding: const EdgeInsets.symmetric( + horizontal: 24, + vertical: 12, + ), + ), + ) + : ElevatedButton.icon( + onPressed: () => controller.startListening(), + icon: const Icon(Icons.mic), + label: const Text('开始录音'), + style: ElevatedButton.styleFrom( + backgroundColor: Colors.blue, + foregroundColor: Colors.white, + padding: const EdgeInsets.symmetric( + horizontal: 24, + vertical: 12, + ), + ), + ), + ), + ), + ], + ), + ), + ), + + const SizedBox(height: 16), + + // 识别结果 + Expanded( + child: Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + Row( + mainAxisAlignment: MainAxisAlignment.spaceBetween, + children: [ + const Text( + '识别结果', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + IconButton( + icon: const Icon(Icons.clear), + onPressed: () => controller.clearResults(), + tooltip: '清空结果', + ), + ], + ), + const SizedBox(height: 8), + Expanded( + child: Container( + padding: const EdgeInsets.all(8.0), + decoration: BoxDecoration( + border: Border.all(color: Colors.grey), + borderRadius: BorderRadius.circular(4.0), + ), + child: Obx(() => ListView.builder( + itemCount: controller.recognitionResults.length, + itemBuilder: (context, index) { + final result = controller.recognitionResults[index]; + return Padding( + padding: const EdgeInsets.symmetric(vertical: 4.0), + child: Row( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + Text( + '${index + 1}. ', + style: const TextStyle( + fontWeight: FontWeight.bold, + ), + ), + Expanded( + child: Text(result), + ), + ], + ), + ); + }, + )), + ), + ), + ], + ), + ), + ), + ), + + const SizedBox(height: 16), + + // 状态显示 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '状态', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + Obx(() => Text( + controller.isListening.value + ? '正在录音...' + : '就绪', + style: TextStyle( + color: controller.isListening.value + ? Colors.green + : Colors.grey, + fontWeight: FontWeight.bold, + ), + )), + const SizedBox(height: 8), + Obx(() => controller.errorMessage.value.isNotEmpty + ? Text( + '错误: ${controller.errorMessage.value}', + style: const TextStyle( + color: Colors.red, + ), + ) + : const SizedBox.shrink()), + ], + ), + ), + ), + ], + ), + ), + ); + } +} \ No newline at end of file diff --git a/lib/modules/test/views/tts_test_view.dart b/lib/modules/test/views/tts_test_view.dart new file mode 100644 index 000000000..83152399c --- /dev/null +++ b/lib/modules/test/views/tts_test_view.dart @@ -0,0 +1,267 @@ +import 'package:flutter/material.dart'; +import 'package:get/get.dart'; +import '../controllers/tts_test_controller.dart'; + +class TtsTestView extends GetView { + const TtsTestView({Key? key}) : super(key: key); + + @override + Widget build(BuildContext context) { + return Scaffold( + appBar: AppBar( + title: const Text('火山语音合成测试'), + centerTitle: true, + ), + body: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.stretch, + children: [ + // 语音类型选择 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '语音类型', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + Obx(() => DropdownButton( + isExpanded: true, + value: controller.selectedVoiceType.value, + onChanged: (String? newValue) { + if (newValue != null) { + controller.selectedVoiceType.value = newValue; + } + }, + items: controller.voiceTypes.map>((String value) { + // 根据语音类型ID获取友好名称 + String displayName = _getVoiceTypeDisplayName(value); + return DropdownMenuItem( + value: value, + child: Text(displayName), + ); + }).toList(), + )), + ], + ), + ), + ), + + const SizedBox(height: 16), + + // 文本输入 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '输入要合成的文本', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + TextField( + controller: controller.textController, + maxLines: 5, + decoration: const InputDecoration( + hintText: '请输入要转换为语音的文本...', + border: OutlineInputBorder(), + ), + ), + ], + ), + ), + ), + + const SizedBox(height: 16), + + // 播放控制 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '播放控制', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + Row( + mainAxisAlignment: MainAxisAlignment.spaceEvenly, + children: [ + Obx(() => ElevatedButton.icon( + onPressed: controller.isPlaying.value || controller.isLoading.value + ? null + : () => controller.speakText(), + icon: controller.isLoading.value + ? const SizedBox( + width: 20, + height: 20, + child: CircularProgressIndicator(strokeWidth: 2) + ) + : const Icon(Icons.play_arrow), + label: Text(controller.isLoading.value ? '准备中...' : '播放'), + )), + ElevatedButton.icon( + onPressed: controller.isPlaying.value + ? () => controller.stopSpeaking() + : null, + icon: const Icon(Icons.stop), + label: const Text('停止'), + style: ElevatedButton.styleFrom( + backgroundColor: Colors.red, + foregroundColor: Colors.white, + ), + ), + ], + ), + ], + ), + ), + ), + + const SizedBox(height: 16), + + // 状态显示 + Card( + child: Padding( + padding: const EdgeInsets.all(16.0), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + const Text( + '状态', + style: TextStyle( + fontSize: 16, + fontWeight: FontWeight.bold, + ), + ), + const SizedBox(height: 8), + Obx(() => Text( + controller.isLoading.value + ? '准备中...' + : controller.isPlaying.value + ? '正在播放...' + : '就绪', + style: TextStyle( + color: controller.isLoading.value + ? Colors.orange + : controller.isPlaying.value + ? Colors.green + : Colors.grey, + fontWeight: FontWeight.bold, + ), + )), + const SizedBox(height: 8), + Obx(() => controller.errorMessage.value.isNotEmpty + ? Container( + padding: const EdgeInsets.all(8), + decoration: BoxDecoration( + color: Colors.red.shade50, + borderRadius: BorderRadius.circular(4), + border: Border.all(color: Colors.red.shade200), + ), + child: Column( + crossAxisAlignment: CrossAxisAlignment.start, + children: [ + Row( + children: [ + Icon(Icons.error_outline, color: Colors.red, size: 16), + SizedBox(width: 8), + Text( + '错误', + style: TextStyle( + color: Colors.red, + fontWeight: FontWeight.bold, + ), + ), + ], + ), + SizedBox(height: 4), + Text( + controller.errorMessage.value, + style: TextStyle(color: Colors.red.shade800), + ), + if (controller.errorMessage.value.contains('授权错误')) + Padding( + padding: const EdgeInsets.only(top: 8.0), + child: Text( + '提示: 请尝试选择基础语音类型,如"zh_male_qingse_common"或"zh_female_qingse_common"', + style: TextStyle( + fontStyle: FontStyle.italic, + color: Colors.red.shade700, + ), + ), + ), + ], + ), + ) + : const SizedBox.shrink()), + ], + ), + ), + ), + ], + ), + ), + ); + } + + // 根据语音类型ID获取友好名称 + String _getVoiceTypeDisplayName(String voiceTypeId) { + switch (voiceTypeId) { + // 趣味方言 + case 'zh_female_wanqudashu_moon_bigtts': + return '湾区大叔 (趣味方言)'; + case 'zh_female_daimengchuanmei_moon_bigtts': + return '呆萌川妹 (趣味方言)'; + case 'zh_male_guozhoudege_moon_bigtts': + return '广州德哥 (趣味方言)'; + case 'zh_male_beijingxiaoye_moon_bigtts': + return '北京小爷 (趣味方言)'; + case 'zh_male_haoyuxiaoge_moon_bigtts': + return '浩宇小哥 (趣味方言)'; + + // 通用场景 + case 'zh_male_shaonianzixin_moon_bigtts': + return '少年梓辛/Brayan (通用场景)'; + + // 角色扮演 + case 'zh_female_meilinvyou_moon_bigtts': + return '魅力女友 (角色扮演)'; + case 'zh_male_shenyeboke_moon_bigtts': + return '深夜播客 (角色扮演)'; + case 'zh_female_sajiaonvyou_moon_bigtts': + return '柔美女友 (角色扮演)'; + case 'zh_female_yuanqinvyou_moon_bigtts': + return '撒娇学妹 (角色扮演)'; + + // 基础语音类型 + case 'zh_male_qingse_common': + return '基础男声'; + case 'zh_female_qingse_common': + return '基础女声'; + case 'zh_male_M392_conversation_wvae_bigtts': + return '高级男声'; + case 'zh_female_F392_conversation_wvae_bigtts': + return '高级女声'; + default: + return voiceTypeId; + } + } +} \ No newline at end of file