diff --git a/android/app/build.gradle.kts b/android/app/build.gradle.kts
index 542cf24d8..8c371408e 100644
--- a/android/app/build.gradle.kts
+++ b/android/app/build.gradle.kts
@@ -34,6 +34,25 @@ android {
jvmTarget = JavaVersion.VERSION_11.toString()
}
+ // 添加lint选项
+ lintOptions {
+ isCheckReleaseBuilds = false
+ }
+
+ // 添加packaging配置,排除冲突文件
+ packaging {
+ resources {
+ excludes.add("META-INF/DEPENDENCIES")
+ excludes.add("META-INF/LICENSE")
+ excludes.add("META-INF/LICENSE.txt")
+ excludes.add("META-INF/license.txt")
+ excludes.add("META-INF/NOTICE")
+ excludes.add("META-INF/NOTICE.txt")
+ excludes.add("META-INF/notice.txt")
+ excludes.add("META-INF/*.kotlin_module")
+ }
+ }
+
defaultConfig {
// TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html).
applicationId = "com.yunqiinnovation.deepsound"
@@ -82,14 +101,19 @@ android {
dependencies {
+ implementation("io.modelcontextprotocol:kotlin-sdk:0.4.0")
+
+
+ // 添加本地插件模块依赖
+ implementation(project(":azure_speech"))
+ implementation(project(":open_ai_service"))
+ implementation(project(":volcano_speech"))
// 添加OkHttp依赖
implementation("com.squareup.okhttp3:okhttp:4.9.3")
// 添加核心库反糖化
coreLibraryDesugaring("com.android.tools:desugar_jdk_libs:2.0.3")
- // Microsoft 语音识别SDK
- implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.42.0")
// 添加androidx.media依赖
implementation("androidx.media:media:1.6.0")
@@ -100,11 +124,7 @@ dependencies {
// 添加 AndroidX Security 加密 SharedPreferences 依赖
implementation("androidx.security:security-crypto:1.1.0-alpha06")
- // // 添加火山语音合成SDK依赖
- // implementation("com.bytedance.speechengine:speechengine_tts_tob:5.4.8")
-
- // 添加火山语音识别SDK依赖
- // implementation("com.bytedance.speechengine:speechengine_tob:0.0.5")
+
}
flutter {
diff --git a/android/app/src/main/AndroidManifest.xml b/android/app/src/main/AndroidManifest.xml
index f120cc1f7..23c380fb0 100644
--- a/android/app/src/main/AndroidManifest.xml
+++ b/android/app/src/main/AndroidManifest.xml
@@ -27,6 +27,15 @@
+
+
+
+
+
+
+
+
= arrayOf("zh-CN", "en-US")): Boolean {
- try {
- FileLogger.d(TAG, "初始化 Azure 语音服务")
-
- // 检查配置是否为空
- if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) {
- FileLogger.e(TAG, "Azure 配置信息不完整")
- return false
- }
-
- // 释放之前的资源
- dispose()
-
- this.subscriptionKey = subscriptionKey
- this.serviceRegion = serviceRegion
-
- // 设置语言
- if (supportedLanguages.isNotEmpty()) {
- this.supportedLanguages = supportedLanguages
- }
-
- // 根据支持的语言数量决定是否启用自动语言检测
- this.isAutoDetectLanguage = supportedLanguages.size >= 2
-
- // 如果只有一种语言,设置为当前语言
- if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) {
- this.currentLanguage = supportedLanguages[0]
- }
-
- // 创建语音配置
- speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
-
- // 设置语言配置
- if (isAutoDetectLanguage) {
- // 设置自动语言检测
- speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous")
- } else {
- // 设置指定的识别语言
- speechConfig?.speechRecognitionLanguage = currentLanguage
- }
-
- // 创建识别器
- try {
- if (useEchoCancellation) {
- // 如果使用回音消除,创建自定义音频输入流
- setupCustomAudioProcessing()
-
- if (isAutoDetectLanguage) {
- val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
- recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
- } else {
- recognizer = SpeechRecognizer(speechConfig, audioConfig)
- }
- } else {
- // 使用默认麦克风输入
- if (isAutoDetectLanguage) {
- val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
- recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
- } else {
- recognizer = SpeechRecognizer(speechConfig)
- }
- }
-
- FileLogger.d(TAG, "Azure 语音服务初始化成功")
- return true
- } catch (e: Exception) {
- FileLogger.e(TAG, "创建识别器失败: ${e.message}")
- stopCustomAudioProcessing()
- return false
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "初始化失败: ${e.message}")
- return false
- }
- }
-
- // 重置 recognizer
- private fun resetRecognizer(): Boolean {
- try {
- // 释放之前的 recognizer
- recognizer?.close()
- recognizer = null
-
- // 停止当前的音频处理
- stopCustomAudioProcessing()
-
- // 使用现有配置重新创建 recognizer
- if (speechConfig != null) {
- if (useEchoCancellation) {
- // 如果使用回音消除,创建自定义音频输入流
- setupCustomAudioProcessing()
-
- if (isAutoDetectLanguage) {
- val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
- recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
- } else {
- recognizer = SpeechRecognizer(speechConfig, audioConfig)
- }
- } else {
- // 使用默认麦克风输入
- if (isAutoDetectLanguage) {
- val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
- recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
- } else {
- recognizer = SpeechRecognizer(speechConfig)
- }
- }
- return true
- } else {
- FileLogger.e(TAG, "语音配置未初始化")
- return false
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "重置识别器失败: ${e.message}")
- return false
- }
- }
-
- // 开始一次性语音识别
- fun recognizeOnce(callback: RecognizeCallback) {
- if (speechConfig == null) {
- callback.onError("语音服务未初始化")
- return
- }
-
- // 重置 recognizer
- if (!resetRecognizer()) {
- callback.onError("重置识别器失败")
- return
- }
-
- try {
- // 启动音频处理
- startCustomAudioProcessing()
-
- // 执行识别
- val result = recognizer?.recognizeOnceAsync()?.get()
-
- // 停止音频处理
- stopCustomAudioProcessing()
-
- if (result != null && result.reason == ResultReason.RecognizedSpeech) {
- val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language
- callback.onResult(result.text, detectedLanguage ?: "")
- } else {
- callback.onError("未能识别语音")
- }
- } catch (e: Exception) {
- // 停止音频处理
- stopCustomAudioProcessing()
- callback.onError("识别异常: ${e.message}")
- }
- }
-
- // 开始连续语音识别
- fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
- if (speechConfig == null) {
- callback.onError("语音服务未初始化")
- return false
- }
-
- // 如果已经在进行连续识别,先停止
- if (isContinuousRecognitionActive) {
- stopContinuousRecognition(callback)
- }
-
- // 重置 recognizer
- if (!resetRecognizer()) {
- callback.onError("重置识别器失败")
- return false
- }
-
- try {
- // 启动音频处理
- startCustomAudioProcessing()
-
- // 设置识别事件处理
- // 最终识别结果
- recognizer?.recognized?.addEventListener(
- EventHandler { _, event ->
- if (event.result.reason == ResultReason.RecognizedSpeech) {
- val detectedLanguage = if (isAutoDetectLanguage) {
- AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: ""
- } else {
- currentLanguage
- }
- // FileLogger.d(TAG, "最终识别结果: ${event.result.text}")
- callback.onResult(event.result.text, detectedLanguage)
- }
- }
- )
-
- // 识别中事件
- recognizer?.recognizing?.addEventListener(
- EventHandler { _, event ->
- if (event.result.reason == ResultReason.RecognizingSpeech) {
- val detectedLanguage = if (isAutoDetectLanguage) {
- AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: ""
- } else {
- currentLanguage
- }
- // FileLogger.d(TAG, "识别中结果: ${event.result.text}")
- callback.onRecognizing(event.result.text, detectedLanguage)
- }
- }
- )
-
- // 会话事件
- recognizer?.sessionStarted?.addEventListener(
- EventHandler { _, _ ->
- isContinuousRecognitionActive = true
- callback.onSessionStarted()
- }
- )
-
- recognizer?.sessionStopped?.addEventListener(
- EventHandler { _, _ ->
- isContinuousRecognitionActive = false
- callback.onSessionStopped()
- }
- )
-
- // 取消事件
- recognizer?.canceled?.addEventListener(
- EventHandler { _, event ->
- val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else ""
- callback.onCanceled(event.reason.toString(), errorDetails)
- isContinuousRecognitionActive = false
- }
- )
-
- // 开始连续识别
- recognizer?.startContinuousRecognitionAsync()?.get()
- isContinuousRecognitionActive = true
-
- return true
- } catch (e: Exception) {
- callback.onError("开始连续识别失败: ${e.message}")
- isContinuousRecognitionActive = false
- stopCustomAudioProcessing()
- return false
- }
- }
-
- // 停止连续语音识别
- fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
- if (!isContinuousRecognitionActive || recognizer == null) {
- return true
- }
-
- try {
- recognizer?.stopContinuousRecognitionAsync()
- isContinuousRecognitionActive = false
- callback.onSessionStopped()
-
- // 停止音频处理
- stopCustomAudioProcessing()
-
- return true
- } catch (e: Exception) {
- callback.onError("停止连续识别失败: ${e.message}")
- isContinuousRecognitionActive = false
- stopCustomAudioProcessing()
- return false
- }
- }
-
- // 检查连续识别是否活跃
- fun isContinuousRecognitionActive(): Boolean {
- return isContinuousRecognitionActive
- }
-
- // 设置自定义音频处理
- private fun setupCustomAudioProcessing() {
- if (!useEchoCancellation) {
- return
- }
-
- try {
- // 1. 创建PushAudioInputStream
- pushStream = PushAudioInputStream.create()
-
- // 2. 创建AudioConfig
- audioConfig = AudioConfig.fromStreamInput(pushStream)
-
- // 3. 创建自定义音频处理器
- customAudioProcessor = CustomAudioProcessor(pushStream)
-
- FileLogger.d(TAG, "自定义音频处理设置完成")
- } catch (e: Exception) {
- FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}")
- releaseCustomAudioProcessing()
-
- // 降级处理:如果自定义处理设置失败,尝试使用默认麦克风
- try {
- FileLogger.d(TAG, "尝试降级到默认麦克风输入")
- audioConfig = AudioConfig.fromDefaultMicrophoneInput()
- } catch (e2: Exception) {
- FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}")
- audioConfig = null
- }
- }
- }
-
- // 启动自定义音频处理
- private fun startCustomAudioProcessing() {
- if (!useEchoCancellation || customAudioProcessor == null) {
- return
- }
-
- try {
- customAudioProcessor?.startRecording()
- FileLogger.d(TAG, "自定义音频处理已启动")
- } catch (e: Exception) {
- FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}")
- }
- }
-
- // 停止自定义音频处理
- private fun stopCustomAudioProcessing() {
- if (!useEchoCancellation || customAudioProcessor == null) {
- return
- }
-
- try {
- customAudioProcessor?.stopRecording()
- FileLogger.d(TAG, "自定义音频处理已停止")
- } catch (e: Exception) {
- FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}")
- }
- }
-
- // 释放自定义音频处理资源
- private fun releaseCustomAudioProcessing() {
- stopCustomAudioProcessing()
-
- try {
- customAudioProcessor = null
- pushStream?.close()
- pushStream = null
- audioConfig?.close()
- audioConfig = null
-
- FileLogger.d(TAG, "自定义音频处理资源已释放")
- } catch (e: Exception) {
- FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}")
- }
- }
-
- // 释放所有资源
- fun dispose() {
- try {
- // 停止和释放音频处理
- releaseCustomAudioProcessing()
-
- recognizer?.close()
- recognizer = null
-
- speechConfig?.close()
- speechConfig = null
-
- isContinuousRecognitionActive = false
-
- FileLogger.d(TAG, "语音识别资源已释放")
- } catch (e: Exception) {
- FileLogger.e(TAG, "释放资源出错: ${e.message}")
- }
- }
-
- // 自定义音频处理器 - 使用Android原生回音消除
- private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) {
- private val SAMPLE_RATE = 16000
- private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO
- private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT
- private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定
-
- private var audioRecord: AudioRecord? = null
- private var echoCanceler: AcousticEchoCanceler? = null
- private val isRecording = AtomicBoolean(false)
- private var recordingThread: Thread? = null
-
- // 启动录音并处理音频数据
- fun startRecording() {
- if (isRecording.get() || pushStream == null) {
- return
- }
-
- try {
- // 使用Builder模式构建AudioFormat
- val audioFormat = AudioFormat.Builder()
- .setSampleRate(SAMPLE_RATE)
- .setEncoding(AUDIO_FORMAT)
- .setChannelMask(CHANNEL_CONFIG)
- .build()
-
- // 使用Builder模式创建AudioRecord实例
- audioRecord = AudioRecord.Builder()
- .setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION)
- .setAudioFormat(audioFormat)
- .setBufferSizeInBytes(BUFFER_SIZE)
- .build()
-
- // 检查AudioRecord初始化状态
- if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) {
- FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}")
- // 尝试使用DEFAULT音频源重试一次
- audioRecord?.release()
- audioRecord = AudioRecord.Builder()
- .setAudioSource(MediaRecorder.AudioSource.DEFAULT)
- .setAudioFormat(audioFormat)
- .setBufferSizeInBytes(BUFFER_SIZE)
- .build()
-
- if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) {
- FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃")
- releaseAudioResources()
- return
- } else {
- FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord")
- }
- }
-
- // 启用音频效果(回音消除、噪声抑制等)
- enableAudioEffects()
-
- // 启动录音
- audioRecord?.startRecording()
- isRecording.set(true)
-
- // 创建录音线程
- recordingThread = Thread({
- val buffer = ByteArray(BUFFER_SIZE)
-
- while (isRecording.get()) {
- try {
- val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0
-
- if (readSize > 0) {
- try {
- // 将处理后的音频数据推送到流
- if (readSize == buffer.size) {
- // 如果读取的大小等于buffer的大小,直接写入整个buffer
- pushStream.write(buffer)
- } else {
- // 如果只读取了部分数据,创建新的数组只包含有效数据
- val validData = buffer.copyOfRange(0, readSize)
- pushStream.write(validData)
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "写入音频数据失败: ${e.message}")
- break
- }
- } else if (readSize == 0) {
- // 读取为0,可能是临时的,等待一下继续尝试
- Thread.sleep(10)
- } else {
- // 负值表示错误
- FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize")
- break
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "录音线程异常: ${e.message}")
- break
- }
- }
- }, "AudioRecordingThread")
-
- // 设置线程优先级并启动
- recordingThread?.priority = Thread.MAX_PRIORITY
- recordingThread?.start()
-
- FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else ""))
- } catch (e: Exception) {
- FileLogger.e(TAG, "启动音频录制失败: ${e.message}")
- releaseAudioResources()
- }
- }
-
- // 启用音频效果(回音消除、噪声抑制等)
- private fun enableAudioEffects() {
- try {
- val audioSessionId = audioRecord?.audioSessionId ?: -1
-
- if (audioSessionId != -1) {
- // 启用回音消除
- if (AcousticEchoCanceler.isAvailable()) {
- try {
- echoCanceler = AcousticEchoCanceler.create(audioSessionId)
- if (echoCanceler != null) {
- echoCanceler?.enabled = true
- FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId")
- } else {
- FileLogger.w(TAG, "回音消除器创建返回null")
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}")
- }
- } else {
- FileLogger.d(TAG, "设备不支持回音消除")
- }
-
- // 以下功能暂时不启用,可根据需要取消注释
- /*
- // 启用噪声抑制
- if (NoiseSuppressor.isAvailable()) {
- try {
- val ns = NoiseSuppressor.create(audioSessionId)
- ns?.enabled = true
- FileLogger.d(TAG, "噪声抑制已启用")
- } catch (e: Exception) {
- FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}")
- }
- }
-
- // 启用自动增益控制
- if (AutomaticGainControl.isAvailable()) {
- try {
- val agc = AutomaticGainControl.create(audioSessionId)
- agc?.enabled = true
- FileLogger.d(TAG, "自动增益控制已启用")
- } catch (e: Exception) {
- FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}")
- }
- }
- */
- } else {
- FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果")
- }
- } catch (e: Exception) {
- FileLogger.e(TAG, "启用音频效果时出错: ${e.message}")
- }
- }
-
- // 停止录音
- fun stopRecording() {
- if (!isRecording.get()) {
- return
- }
-
- isRecording.set(false)
-
- try {
- // 等待录音线程结束
- recordingThread?.join(1000)
-
- // 释放资源
- releaseAudioResources()
-
- FileLogger.d(TAG, "音频录制已停止")
- } catch (e: Exception) {
- FileLogger.e(TAG, "停止音频录制失败: ${e.message}")
- }
- }
-
- // 释放音频资源
- private fun releaseAudioResources() {
- try {
- // 停止录音
- try {
- if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) {
- audioRecord?.stop()
- }
- } catch (e: Exception) {
- // 忽略可能的IllegalStateException
- FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}")
- }
-
- // 释放回音消除器
- try {
- if (echoCanceler != null) {
- echoCanceler?.enabled = false
- echoCanceler?.release()
- echoCanceler = null
- }
- } catch (e: Exception) {
- FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}")
- } finally {
- echoCanceler = null
- }
-
- // 释放音频记录器
- try {
- audioRecord?.release()
- } catch (e: Exception) {
- FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}")
- } finally {
- audioRecord = null
- }
-
- // 重置线程
- recordingThread = null
-
- } catch (e: Exception) {
- FileLogger.e(TAG, "释放音频资源失败: ${e.message}")
- }
- }
- }
-
- // 一次性识别回调接口
- interface RecognizeCallback {
- fun onResult(result: String, detectedLanguage: String = "")
- fun onError(error: String)
- }
-
- // 连续识别回调接口
- interface ContinuousRecognizeCallback {
- fun onResult(result: String, detectedLanguage: String = "")
- fun onRecognizing(recognizing: String, detectedLanguage: String = "")
- fun onSessionStarted()
- fun onSessionStopped()
- fun onCanceled(reason: String, errorDetails: String)
- fun onError(error: String)
- }
-}
\ No newline at end of file
diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt b/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt
deleted file mode 100644
index 2fbcd8da8..000000000
--- a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt
+++ /dev/null
@@ -1,344 +0,0 @@
-package com.yunqiinnovation.deepsound
-
-import android.util.Log
-import okhttp3.*
-import okhttp3.MediaType.Companion.toMediaTypeOrNull
-import okhttp3.RequestBody.Companion.toRequestBody
-import org.json.JSONArray
-import org.json.JSONObject
-import java.io.IOException
-import java.util.concurrent.CountDownLatch
-import java.util.concurrent.TimeUnit
-
-/**
- * 火山AI服务的原生实现
- *
- * 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式
- */
-class VolcanoAIService() {
- private val TAG = "VolcanoAIService"
- private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3"
- private val chatEndpoint = "/chat/completions"
- private val client = OkHttpClient.Builder()
- .connectTimeout(30, TimeUnit.SECONDS)
- .readTimeout(30, TimeUnit.SECONDS)
- .writeTimeout(30, TimeUnit.SECONDS)
- .build()
-
- private var apiKey: String = ""
- private var isInitialized = false
-
- /**
- * 初始化火山AI服务
- *
- * @param apiKey 火山AI API密钥
- * @return 初始化是否成功
- */
- fun initialize(apiKey: String): Boolean {
- this.apiKey = apiKey
- isInitialized = apiKey.isNotEmpty()
-
- if (!isInitialized) {
- Log.e(TAG, "初始化失败:API key 不能为空")
- } else {
- Log.d(TAG, "火山AI服务初始化成功")
- }
-
- return isInitialized
- }
-
- /**
- * 生成个性化问候语
- *
- * @param agentName 代理名称
- * @param systemPrompt 系统提示词
- * @param callback 回调函数,返回生成的问候语
- */
- fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) {
- val messages = JSONArray().apply {
- put(JSONObject().apply {
- put("role", "system")
- put("content", systemPrompt)
- })
- put(JSONObject().apply {
- put("role", "user")
- put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。")
- })
- }
-
- sendMessageStream(messages, systemPrompt, object : StreamCallback {
- val stringBuilder = StringBuilder()
-
- override fun onToken(token: String) {
- stringBuilder.append(token)
- }
-
- override fun onComplete() {
- callback(stringBuilder.toString(), null)
- }
-
- override fun onError(e: Exception) {
- callback(null, e)
- }
- })
- }
-
- /**
- * 发送消息(非流式输出)
- *
- * @param messages 消息列表
- * @param systemPrompt 系统提示词
- * @return 返回AI的回复
- * @throws VolcanoAIException 如果API调用失败
- */
- @Throws(VolcanoAIException::class)
- fun sendMessage(messages: JSONArray, systemPrompt: String): String {
- // 检查是否已初始化
- if (!isInitialized || apiKey.isEmpty()) {
- throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")
- }
-
- val fullMessages = JSONArray().apply {
- put(JSONObject().apply {
- put("role", "system")
- put("content", systemPrompt)
- })
- for (i in 0 until messages.length()) {
- put(messages.getJSONObject(i))
- }
- }
-
- val requestBody = JSONObject().apply {
- put("model", "doubao-1-5-lite-32k-250115")
- put("messages", fullMessages)
- put("temperature", 0.7)
- put("max_tokens", 2000)
- put("stream", false)
- }
-
- val mediaType = "application/json".toMediaTypeOrNull()
- val request = Request.Builder()
- .url("$baseUrl$chatEndpoint")
- .addHeader("Content-Type", "application/json")
- .addHeader("Authorization", "Bearer $apiKey")
- .post(requestBody.toString().toRequestBody(mediaType))
- .build()
-
- try {
- client.newCall(request).execute().use { response ->
- if (!response.isSuccessful) {
- val errorBody = response.body?.string() ?: ""
- val errorMessage = try {
- JSONObject(errorBody).getJSONObject("error").getString("message")
- } catch (e: Exception) {
- "Unknown error occurred"
- }
- throw VolcanoAIException(errorMessage)
- }
-
- val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response")
- val jsonResponse = JSONObject(responseBody)
-
- if (jsonResponse.has("choices") &&
- jsonResponse.getJSONArray("choices").length() > 0 &&
- jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) {
- return jsonResponse.getJSONArray("choices")
- .getJSONObject(0)
- .getJSONObject("message")
- .getString("content")
- }
-
- throw VolcanoAIException("Invalid response format")
- }
- } catch (e: Exception) {
- if (e is VolcanoAIException) throw e
- throw VolcanoAIException("Failed to communicate with AI service: ${e.message}")
- }
- }
-
- /**
- * 发送消息(流式输出)
- *
- * @param messages 消息列表
- * @param systemPrompt 系统提示词
- * @param callback 回调函数,用于接收流式输出的结果
- */
- fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) {
- // 检查是否已初始化
- if (!isInitialized || apiKey.isEmpty()) {
- callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法"))
- return
- }
-
- val fullMessages = JSONArray().apply {
- put(JSONObject().apply {
- put("role", "system")
- put("content", systemPrompt)
- })
- for (i in 0 until messages.length()) {
- put(messages.getJSONObject(i))
- }
- }
-
- val requestBody = JSONObject().apply {
- put("model", "doubao-1-5-lite-32k-250115")
- put("messages", fullMessages)
- put("temperature", 0.7)
- put("max_tokens", 2000)
- put("stream", true)
- }
-
- val mediaType = "application/json".toMediaTypeOrNull()
- val request = Request.Builder()
- .url("$baseUrl$chatEndpoint")
- .addHeader("Content-Type", "application/json")
- .addHeader("Authorization", "Bearer $apiKey")
- .addHeader("Accept", "text/event-stream")
- .post(requestBody.toString().toRequestBody(mediaType))
- .build()
-
- client.newCall(request).enqueue(object : Callback {
- override fun onFailure(call: Call, e: IOException) {
- callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}"))
- }
-
- override fun onResponse(call: Call, response: Response) {
- if (!response.isSuccessful) {
- val errorBody = response.body?.string() ?: ""
- val errorMessage = try {
- JSONObject(errorBody).getJSONObject("error").getString("message")
- } catch (e: Exception) {
- "Unknown error occurred"
- }
- callback.onError(VolcanoAIException(errorMessage))
- return
- }
-
- val responseBody = response.body ?: return
- val source = responseBody.source()
- val bufferedSource = source.buffer
-
- try {
- while (!bufferedSource.exhausted()) {
- val line = bufferedSource.readUtf8Line() ?: continue
-
- if (line.isEmpty()) continue
- if (line.startsWith("data: ")) {
- val data = line.substring(6)
- if (data == "[DONE]") {
- callback.onComplete()
- break
- }
-
- try {
- val jsonData = JSONObject(data)
- if (jsonData.has("choices") &&
- jsonData.getJSONArray("choices").length() > 0 &&
- jsonData.getJSONArray("choices").getJSONObject(0).has("delta") &&
- jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) {
- val content = jsonData.getJSONArray("choices")
- .getJSONObject(0)
- .getJSONObject("delta")
- .getString("content")
- callback.onToken(content)
- }
- } catch (e: Exception) {
- // 忽略无效的JSON数据
- continue
- }
- }
- }
- } catch (e: Exception) {
- callback.onError(VolcanoAIException("Error processing stream: ${e.message}"))
- } finally {
- response.close()
- }
- }
- })
- }
-
- /**
- * 同步方式发送消息(流式输出)
- *
- * 注意:此方法会阻塞当前线程,请在后台线程中调用
- *
- * @param messages 消息列表
- * @param systemPrompt 系统提示词
- * @return 返回完整的AI回复
- * @throws VolcanoAIException 如果API调用失败
- */
- @Throws(VolcanoAIException::class)
- fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String {
- val result = StringBuilder()
- val latch = CountDownLatch(1)
- var exception: Exception? = null
-
- sendMessageStream(messages, systemPrompt, object : StreamCallback {
- override fun onToken(token: String) {
- result.append(token)
- }
-
- override fun onComplete() {
- latch.countDown()
- }
-
- override fun onError(e: Exception) {
- exception = e
- latch.countDown()
- }
- })
-
- // 等待流式输出完成或出错
- latch.await(60, TimeUnit.SECONDS)
-
- if (exception != null) {
- throw exception as VolcanoAIException
- }
-
- return result.toString()
- }
-
- /**
- * 创建用户消息
- */
- fun createUserMessage(content: String): JSONObject {
- return JSONObject().apply {
- put("role", "user")
- put("content", content)
- }
- }
-
- /**
- * 创建系统消息
- */
- fun createSystemMessage(content: String): JSONObject {
- return JSONObject().apply {
- put("role", "system")
- put("content", content)
- }
- }
-
- /**
- * 创建助手消息
- */
- fun createAssistantMessage(content: String): JSONObject {
- return JSONObject().apply {
- put("role", "assistant")
- put("content", content)
- }
- }
-
- /**
- * 流式输出回调接口
- */
- interface StreamCallback {
- fun onToken(token: String)
- fun onComplete()
- fun onError(e: Exception)
- }
-}
-
-/**
- * 火山AI异常
- */
-class VolcanoAIException(message: String) : Exception(message)
\ No newline at end of file
diff --git a/android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt
similarity index 100%
rename from android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt
rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt
diff --git a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt
similarity index 67%
rename from android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt
rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt
index 49fd8fdf6..1c4717f11 100644
--- a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt
+++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt
@@ -1,5 +1,6 @@
package com.yunqiinnovation.deepsound
+import android.Manifest
import android.content.Intent
import android.os.Build
import android.os.Bundle
@@ -20,33 +21,54 @@ import android.bluetooth.BluetoothDevice
import androidx.security.crypto.EncryptedSharedPreferences
import androidx.security.crypto.MasterKey
import com.yunqiinnovation.deepsound.core.utils.FileLogger
+import android.content.pm.PackageManager
+import androidx.core.app.ActivityCompat
+import androidx.core.content.ContextCompat
class MainActivity: FlutterActivity() {
- private val AZURE_ASR_CHANNEL = "com.deep_voice.azure_asr"
- private val AZURE_ASR_EVENT_CHANNEL = "com.deep_voice.azure_asr_events"
- private val AZURE_TTS_CHANNEL = "com.deep_voice.azure_tts"
private val VOICE_INTERACTION_CHANNEL = "com.deep_voice.voice_interaction"
private val VOICE_INTERACTION_EVENT_CHANNEL = "com.deep_voice.voice_interaction_events"
private val CLASSIC_BLUETOOTH_CHANNEL = "com.deep_voice.classic_bluetooth"
private val CLASSIC_BLUETOOTH_EVENT_CHANNEL = "com.deep_voice.classic_bluetooth_events"
private val TAG = "MainActivity"
- private lateinit var azureAsrHelper: AzureAsrHelper
- private lateinit var azureTtsHelper: AzureTtsHelper
private lateinit var classicBluetoothHelper: ClassicBluetoothHelper
- private var azureAsrEventSink: EventChannel.EventSink? = null
private var voiceInteractionEventSink: EventChannel.EventSink? = null
private var bluetoothEventSink: EventChannel.EventSink? = null
+ // 添加权限请求相关常量
+ private val PERMISSION_REQUEST_CODE = 100
+ private val REQUIRED_PERMISSIONS = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) {
+ arrayOf(
+ Manifest.permission.RECORD_AUDIO,
+ Manifest.permission.BLUETOOTH_CONNECT,
+ Manifest.permission.BLUETOOTH_SCAN,
+ Manifest.permission.ACCESS_FINE_LOCATION,
+ Manifest.permission.SEND_SMS,
+ Manifest.permission.READ_CONTACTS,
+ Manifest.permission.CALL_PHONE,
+ Manifest.permission.POST_NOTIFICATIONS
+ )
+ } else {
+ arrayOf(
+ Manifest.permission.RECORD_AUDIO,
+ Manifest.permission.BLUETOOTH,
+ Manifest.permission.BLUETOOTH_ADMIN,
+ Manifest.permission.ACCESS_FINE_LOCATION,
+ Manifest.permission.SEND_SMS,
+ Manifest.permission.READ_CONTACTS,
+ Manifest.permission.CALL_PHONE
+ )
+ }
+
// 广播接收器
private val voiceInteractionReceiver = object : BroadcastReceiver() {
override fun onReceive(context: Context?, intent: Intent?) {
- Log.d(TAG, "收到广播: ${intent?.action}")
+ FileLogger.d(TAG, "收到广播: ${intent?.action}")
when (intent?.action) {
VoiceInteractionService.ACTION_RECOGNITION_STARTED -> {
val timestamp = intent.getLongExtra("timestamp", 0)
- Log.d(TAG, "收到语音识别启动广播: timestamp=$timestamp")
sendVoiceInteractionEvent(mapOf(
"type" to "recognition_started",
"timestamp" to timestamp
@@ -58,12 +80,14 @@ class MainActivity: FlutterActivity() {
val assistantMessage = intent.getStringExtra("assistantMessage") ?: ""
val timestamp = intent.getLongExtra("timestamp", System.currentTimeMillis())
- Log.d(TAG, "收到聊天记录更新广播: agentId=$agentId, timestamp=$timestamp")
- Log.d(TAG, "用户消息: ${userMessage.take(50)}...")
- Log.d(TAG, "助手回复: ${assistantMessage.take(50)}...")
-
sendChatHistoryEvent(agentId, userMessage, assistantMessage, timestamp)
}
+ VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE -> {
+ sendVoiceInteractionEvent(mapOf(
+ "type" to "enter_translation_mode",
+ "timestamp" to System.currentTimeMillis()
+ ))
+ }
}
}
}
@@ -143,13 +167,24 @@ class MainActivity: FlutterActivity() {
var azureSpeechKey: String = ""
var azureSpeechRegion: String = ""
var volcanoAiApiKey: String = ""
+ var openaiApiKey: String = ""
+ var openaiBaseUrl: String? = null
+ var openaiModel: String? = null
+ var volcanoSpeechAppId: String = ""
+ var volcanoSpeechAppToken: String = ""
// 安全存储相关常量
private const val SECURE_PREFS_FILENAME = "deep_voice_secure_prefs"
private const val KEY_AZURE_SPEECH_KEY = "azure_speech_key"
private const val KEY_AZURE_SPEECH_REGION = "azure_speech_region"
private const val KEY_VOLCANO_AI_API_KEY = "volcano_ai_api_key"
-
+ private const val KEY_OPENAI_API_KEY = "openai_api_key"
+ private const val KEY_OPENAI_BASE_URL = "openai_base_url"
+ private const val KEY_OPENAI_MODEL = "openai_model"
+ private const val KEY_MCP_SERVER_ENDPOINT = "mcp_server_endpoint"
+ private const val KEY_VOLCANO_SPEECH_APP_ID = "volcano_speech_app_id"
+ private const val KEY_VOLCANO_SPEECH_APP_TOKEN = "volcano_speech_app_token"
+
// 会话管理
private const val KEY_SESSION_ID = "session_id"
private var currentSessionId = ""
@@ -183,7 +218,12 @@ class MainActivity: FlutterActivity() {
.putString(KEY_AZURE_SPEECH_KEY, azureSpeechKey)
.putString(KEY_AZURE_SPEECH_REGION, azureSpeechRegion)
.putString(KEY_VOLCANO_AI_API_KEY, volcanoAiApiKey)
+ .putString(KEY_OPENAI_API_KEY, openaiApiKey)
+ .putString(KEY_OPENAI_BASE_URL, openaiBaseUrl)
.putString(KEY_SESSION_ID, currentSessionId)
+ .putString(KEY_OPENAI_MODEL, openaiModel)
+ .putString(KEY_VOLCANO_SPEECH_APP_ID, volcanoSpeechAppId)
+ .putString(KEY_VOLCANO_SPEECH_APP_TOKEN, volcanoSpeechAppToken)
.apply()
FileLogger.d("MainActivity", "密钥已安全保存到加密存储中")
@@ -227,11 +267,16 @@ class MainActivity: FlutterActivity() {
azureSpeechKey = sharedPreferences.getString(KEY_AZURE_SPEECH_KEY, "") ?: ""
azureSpeechRegion = sharedPreferences.getString(KEY_AZURE_SPEECH_REGION, "") ?: ""
volcanoAiApiKey = sharedPreferences.getString(KEY_VOLCANO_AI_API_KEY, "") ?: ""
-
+ openaiApiKey = sharedPreferences.getString(KEY_OPENAI_API_KEY, "") ?: ""
+ openaiBaseUrl = sharedPreferences.getString(KEY_OPENAI_BASE_URL, null)
+ openaiModel = sharedPreferences.getString(KEY_OPENAI_MODEL, null)
+ volcanoSpeechAppId = sharedPreferences.getString(KEY_VOLCANO_SPEECH_APP_ID, "") ?: ""
+ volcanoSpeechAppToken = sharedPreferences.getString(KEY_VOLCANO_SPEECH_APP_TOKEN, "") ?: ""
FileLogger.d("MainActivity", "已从加密存储加载密钥")
- // 检查是否成功获取所有密钥
- return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && volcanoAiApiKey.isNotEmpty()
+ // 检查是否成功获取所有必要密钥
+ return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() &&
+ (openaiApiKey.isNotEmpty() || volcanoAiApiKey.isNotEmpty())
} catch (e: Exception) {
FileLogger.e("MainActivity", "从加密存储加载密钥失败: ${e.message}")
e.printStackTrace()
@@ -246,9 +291,10 @@ class MainActivity: FlutterActivity() {
// 初始化 FileLogger
FileLogger.init(applicationContext)
+ // 请求必要权限
+ requestRequiredPermissions()
+
// 初始化 Azure 语音服务
- azureAsrHelper = AzureAsrHelper(applicationContext)
- azureTtsHelper = AzureTtsHelper(applicationContext)
classicBluetoothHelper = ClassicBluetoothHelper(applicationContext)
// 注册广播接收器
@@ -266,7 +312,14 @@ class MainActivity: FlutterActivity() {
val voiceInteractionFilter = IntentFilter().apply {
addAction(VoiceInteractionService.ACTION_RECOGNITION_STARTED)
addAction(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED)
+ addAction(VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE)
}
+
+ // 打印已注册的广播
+ FileLogger.d(TAG, "已注册语音交互广播:${VoiceInteractionService.ACTION_RECOGNITION_STARTED}, " +
+ "${VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED}, " +
+ "${VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE}")
+
if (android.os.Build.VERSION.SDK_INT >= android.os.Build.VERSION_CODES.UPSIDE_DOWN_CAKE) {
registerReceiver(voiceInteractionReceiver, voiceInteractionFilter, Context.RECEIVER_NOT_EXPORTED)
} else {
@@ -300,8 +353,7 @@ class MainActivity: FlutterActivity() {
setupMethodChannels(flutterEngine)
// 初始化 Azure 语音服务
- azureAsrHelper = AzureAsrHelper(this)
- azureTtsHelper = AzureTtsHelper(this)
+ classicBluetoothHelper = ClassicBluetoothHelper(this)
Log.d(TAG, "Flutter 引擎配置完成")
}
@@ -327,20 +379,6 @@ class MainActivity: FlutterActivity() {
}
)
- // Azure ASR 事件通道
- EventChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_EVENT_CHANNEL).setStreamHandler(
- object : EventChannel.StreamHandler {
- override fun onListen(arguments: Any?, events: EventChannel.EventSink?) {
- Log.d(TAG, "ASR事件通道开始监听")
- azureAsrEventSink = events
- }
-
- override fun onCancel(arguments: Any?) {
- azureAsrEventSink = null
- }
- }
- )
-
// 设置蓝牙事件通道
EventChannel(flutterEngine.dartExecutor.binaryMessenger, CLASSIC_BLUETOOTH_EVENT_CHANNEL).setStreamHandler(
object : EventChannel.StreamHandler {
@@ -367,195 +405,6 @@ class MainActivity: FlutterActivity() {
private fun setupMethodChannels(flutterEngine: FlutterEngine) {
Log.d(TAG, "开始设置方法通道")
- // 设置 Azure ASR 方法通道
- MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_CHANNEL).setMethodCallHandler { call, result ->
- when (call.method) {
- "initialize" -> {
- val subscriptionKey = call.argument("subscriptionKey")
- val region = call.argument("region")
- val supportedLanguages = call.argument>("supportedLanguages")?.toTypedArray() ?: arrayOf("zh-CN", "en-US")
-
- if (subscriptionKey == null || region == null) {
- result.error("INVALID_ARGUMENTS", "订阅密钥和区域不能为空", null)
- return@setMethodCallHandler
- }
-
- try {
- val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages)
- result.success(success)
- } catch (e: Exception) {
- result.error("INITIALIZATION_ERROR", e.message, null)
- }
- }
- "recognizeOnce" -> {
-
- azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback {
- override fun onResult(text: String, detectedLanguage: String) {
- result.success(mapOf(
- "text" to text,
- "detectedLanguage" to detectedLanguage
- ))
- }
-
- override fun onError(error: String) {
- result.error("RECOGNITION_ERROR", error, null)
- }
- })
- }
- "startContinuousRecognition" -> {
- // 确保事件通道已准备好
- if (azureAsrEventSink == null) {
- result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null)
- return@setMethodCallHandler
- }
-
- val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
- override fun onResult(text: String, detectedLanguage: String) {
- sendAsrEvent(mapOf(
- "type" to "result",
- "text" to text,
- "detectedLanguage" to detectedLanguage
- ))
- }
-
- override fun onRecognizing(recognizing: String, detectedLanguage: String) {
- sendAsrEvent(mapOf(
- "type" to "recognizing",
- "text" to recognizing,
- "detectedLanguage" to detectedLanguage
- ))
- }
-
- override fun onSessionStarted() {
- sendAsrEvent(mapOf("type" to "sessionStarted"))
- }
-
- override fun onSessionStopped() {
- sendAsrEvent(mapOf("type" to "sessionStopped"))
- }
-
- override fun onCanceled(reason: String, errorDetails: String) {
- sendAsrEvent(mapOf(
- "type" to "canceled",
- "reason" to reason,
- "errorDetails" to errorDetails
- ))
- }
-
- override fun onError(error: String) {
- sendAsrEvent(mapOf("type" to "error", "message" to error))
- }
- })
- result.success(success)
- }
- "stopContinuousRecognition" -> {
- try {
- if (!azureAsrHelper.isContinuousRecognitionActive()) {
- result.success(true)
- return@setMethodCallHandler
- }
-
- val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
- override fun onResult(text: String, detectedLanguage: String) {}
- override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
- override fun onSessionStarted() {}
- override fun onSessionStopped() {}
- override fun onCanceled(reason: String, errorDetails: String) {}
- override fun onError(error: String) {
- result.error("STOP_ERROR", error, null)
- }
- })
- result.success(success)
- } catch (e: Exception) {
- result.error("STOP_ERROR", e.message, null)
- }
- }
- "isContinuousRecognitionActive" -> {
- result.success(azureAsrHelper.isContinuousRecognitionActive())
- }
- "dispose" -> {
- azureAsrHelper.dispose()
- result.success(true)
- }
- else -> {
- result.notImplemented()
- }
- }
- }
-
- // 设置 Azure TTS 方法通道
- MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_TTS_CHANNEL).setMethodCallHandler { call, result ->
- when (call.method) {
- "initialize" -> {
- val subscriptionKey = call.argument("subscriptionKey") ?: ""
- val region = call.argument("region") ?: ""
- val language = call.argument("language") ?: "zh-CN"
-
- val success = azureTtsHelper.initialize(subscriptionKey, region, language)
- result.success(success)
- }
- "setVoice" -> {
- val voiceName = call.argument("voiceName") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "语音名称不能为空", null)
- result.success(azureTtsHelper.setVoice(voiceName))
- }
- "setSpeechParams" -> {
- val rate = call.argument("rate") ?: 0
- val pitch = call.argument("pitch") ?: 0
- val volume = call.argument("volume") ?: 100
- result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume))
- }
- "setAudioOutputType" -> {
- val outputTypeStr = call.argument("outputType") ?: "speaker"
- val outputType = when (outputTypeStr.lowercase()) {
- "speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER
- "earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE
- "auto" -> AzureTtsHelper.AudioOutputType.AUTO
- else -> AzureTtsHelper.AudioOutputType.SPEAKER
- }
- result.success(azureTtsHelper.setAudioOutputType(outputType))
- }
- "speakText" -> {
- val text = call.argument("text") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "文本不能为空", null)
-
- azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback {
- override fun onSuccess(message: String) {
- runOnUiThread { result.success(message) }
- }
-
- override fun onError(error: String) {
- runOnUiThread { result.error("SPEAK_ERROR", error, null) }
- }
- })
- }
- "speakSsml" -> {
- val ssml = call.argument("ssml") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "SSML不能为空", null)
-
- azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback {
- override fun onSuccess(message: String) {
- runOnUiThread { result.success(message) }
- }
-
- override fun onError(error: String) {
- runOnUiThread { result.error("SPEAK_ERROR", error, null) }
- }
- })
- }
- "stopSpeaking" -> {
- result.success(azureTtsHelper.stopSpeaking())
- }
- "isSpeaking" -> {
- result.success(azureTtsHelper.isSpeaking())
- }
- "dispose" -> {
- azureTtsHelper.dispose()
- result.success(true)
- }
- else -> {
- result.notImplemented()
- }
- }
- }
-
// 设置语音交互方法通道
MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOICE_INTERACTION_CHANNEL).setMethodCallHandler { call, result ->
when (call.method) {
@@ -657,15 +506,6 @@ class MainActivity: FlutterActivity() {
}
}
- // ASR 事件发送方法
- private fun sendAsrEvent(event: Map) {
- if (azureAsrEventSink == null) return
-
- runOnUiThread {
- azureAsrEventSink?.success(event)
- }
- }
-
/**
* 启动语音交互服务(带配置参数)
*/
@@ -673,14 +513,14 @@ class MainActivity: FlutterActivity() {
FileLogger.d(TAG, "启动语音交互服务")
// 获取配置参数
- val key = call.argument("azure_speech_key") ?: ""
- val region = call.argument("azure_speech_region") ?: ""
- val aiKey = call.argument("volcano_ai_api_key") ?: ""
-
- // 设置 Azure Speech 配置
- azureSpeechKey = key
- azureSpeechRegion = region
- volcanoAiApiKey = aiKey
+ azureSpeechKey = call.argument("azure_speech_key") ?: ""
+ azureSpeechRegion = call.argument("azure_speech_region") ?: ""
+ openaiApiKey = call.argument("openai_api_key") ?: ""
+ openaiBaseUrl = call.argument("openai_base_url")
+ openaiModel = call.argument("openai_model")
+ volcanoSpeechAppId = call.argument("volcano_speech_app_id") ?: ""
+ volcanoSpeechAppToken = call.argument("volcano_speech_app_token") ?: ""
+
// 保存密钥到安全存储
saveKeysToSecureStorage(applicationContext)
@@ -755,14 +595,19 @@ class MainActivity: FlutterActivity() {
private fun sendVoiceInteractionEvent(event: Map) {
if (voiceInteractionEventSink == null) {
Log.e(TAG, "无法发送语音交互事件:事件通道未准备好")
+ FileLogger.e(TAG, "无法发送语音交互事件:事件通道未准备好")
return
}
+ FileLogger.d(TAG, "准备发送事件到Flutter: ${event["type"]}")
+
runOnUiThread {
try {
voiceInteractionEventSink?.success(event)
+ FileLogger.d(TAG, "成功发送事件到Flutter: ${event["type"]}")
} catch (e: Exception) {
- Log.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}", e)
+ Log.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}")
+ FileLogger.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}")
}
}
}
@@ -799,11 +644,58 @@ class MainActivity: FlutterActivity() {
}
// 释放资源
- azureAsrHelper.dispose()
- azureTtsHelper.dispose()
classicBluetoothHelper.dispose()
super.onDestroy()
}
+
+ /**
+ * 请求必要权限
+ */
+ private fun requestRequiredPermissions() {
+ val permissionsToRequest = ArrayList()
+
+ for (permission in REQUIRED_PERMISSIONS) {
+ if (ContextCompat.checkSelfPermission(this, permission) != PackageManager.PERMISSION_GRANTED) {
+ permissionsToRequest.add(permission)
+ }
+ }
+
+ if (permissionsToRequest.isNotEmpty()) {
+ ActivityCompat.requestPermissions(
+ this,
+ permissionsToRequest.toTypedArray(),
+ PERMISSION_REQUEST_CODE
+ )
+ }
+ }
+
+ /**
+ * 处理权限请求结果
+ */
+ override fun onRequestPermissionsResult(
+ requestCode: Int,
+ permissions: Array,
+ grantResults: IntArray
+ ) {
+ super.onRequestPermissionsResult(requestCode, permissions, grantResults)
+
+ if (requestCode == PERMISSION_REQUEST_CODE) {
+ val deniedPermissions = ArrayList()
+
+ for (i in permissions.indices) {
+ if (grantResults[i] != PackageManager.PERMISSION_GRANTED) {
+ deniedPermissions.add(permissions[i])
+ }
+ }
+
+ if (deniedPermissions.isNotEmpty()) {
+ // 记录未授权的权限
+ FileLogger.w(TAG, "未授权的权限: ${deniedPermissions.joinToString()}")
+ } else {
+ FileLogger.d(TAG, "所有必要权限已授权")
+ }
+ }
+ }
}
diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak
new file mode 100644
index 000000000..1a9e2df13
--- /dev/null
+++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak
@@ -0,0 +1,456 @@
+package com.yunqiinnovation.deepsound
+
+import android.content.Context
+import org.json.JSONArray
+import org.json.JSONObject
+import android.util.Log
+import com.yunqiinnovation.deepsound.core.utils.FileLogger
+import com.yunqiinnovation.azure_speech.AzureAsrHelper
+import com.yunqiinnovation.volcano_speech.VolcanoTtsHelper
+import com.yunqiinnovation.open_ai_service.OpenAIService
+
+
+/**
+ * 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑
+ */
+class VoiceInteractionHandler(
+ private val context: Context,
+ private val azureSpeechKey: String,
+ private val azureSpeechRegion: String,
+ private val openaiApiKey: String,
+ private val openaiBaseUrl: String = "",
+ private val openaiModel: String = "",
+ private val volcanoSpeechAppId: String,
+ private val volcanoSpeechAppToken: String
+) {
+ private val TAG = "VoiceInteractionHandler"
+
+ // Azure服务
+ private var azureAsrHelper: AzureAsrHelper? = null
+ private var volcanoTtsHelper: VolcanoTtsHelper? = null
+
+ // OpenAI服务
+ private val openAIService = OpenAIService()
+
+
+ // 语音功能处理
+ private val voiceFunctionHandler = VoiceFunctionHandler(openAIService, context)
+
+ // 当前用户输入
+ private var currentUserInput = ""
+
+ // 状态
+ private var isInitialized = false
+ var isRecognitionActive = false
+ private set
+ var isTtsSpeaking = false
+ private set
+ var hasSpeechDetected = false
+ private set
+
+ // 回调
+ private var callback: InteractionCallback? = null
+
+ /**
+ * 初始化
+ */
+ fun initialize(): Boolean {
+ if (isInitialized) return true
+
+ try {
+ // 初始化Azure ASR
+ azureAsrHelper = AzureAsrHelper(context).apply {
+ initialize(azureSpeechKey, azureSpeechRegion)
+ }
+
+ // 初始化Volcano TTS (使用大模型TTS)
+ volcanoTtsHelper = VolcanoTtsHelper(context).apply {
+ // 这里需要替换为实际的Volcano SDK初始化参数
+ // 暂时使用假参数,实际使用时需要替换为真实值
+ initialize(volcanoSpeechAppId, volcanoSpeechAppToken, "volc.bigasr.sauc.duration")
+
+ }
+
+ // 初始化OpenAI服务
+ openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel)
+
+ // 初始化语音功能处理器
+ voiceFunctionHandler.initialize()
+
+ isInitialized = true
+ return true
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "初始化失败: ${e.message}", e)
+ return false
+ }
+ }
+
+ /**
+ * 设置回调
+ */
+ fun setCallback(callback: InteractionCallback) {
+ this.callback = callback
+ }
+
+ /**
+ * 开始语音识别
+ */
+ fun startRecognition() {
+ if (isRecognitionActive) return
+
+ // 检查录音权限
+ if (!checkRecordAudioPermission()) {
+ callback?.onError("需要录音权限,请在设置中授予权限")
+ return
+ }
+
+ isRecognitionActive = true
+ hasSpeechDetected = false
+ notifyStateChanged()
+
+ try {
+ azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
+ override fun onRecognizing(recognizing: String, detectedLanguage: String) {
+ if (recognizing.isNotEmpty()) {
+ hasSpeechDetected = true
+ stopTts()
+ notifyStateChanged()
+ }
+ }
+
+ override fun onResult(result: String, detectedLanguage: String) {
+ if (result.isNotEmpty()) {
+ notifyStateChanged()
+
+ processWithOpenAI(result)
+
+ }
+
+ // 重置状态,继续识别
+ hasSpeechDetected = false
+ }
+
+ override fun onSessionStarted() {
+ notifyStateChanged()
+ }
+
+ override fun onSessionStopped() {
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+
+ override fun onCanceled(reason: String, errorDetails: String) {
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+
+ override fun onError(error: String) {
+ isRecognitionActive = false
+ callback?.onError("语音识别出错")
+ notifyStateChanged()
+ }
+
+ override fun onSuccess(message: String) {
+ // 处理成功事件
+ }
+ })
+ } catch (e: Exception) {
+ isRecognitionActive = false
+ FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e)
+ callback?.onError("启动语音识别失败")
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 停止语音识别
+ */
+ fun stopRecognition() {
+ if (!isRecognitionActive) return
+
+ FileLogger.d(TAG, "停止语音识别")
+
+ try {
+ azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
+ override fun onResult(result: String, detectedLanguage: String) {}
+ override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
+ override fun onSessionStarted() {}
+ override fun onSessionStopped() {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别会话已停止")
+ notifyStateChanged()
+ }
+ override fun onCanceled(reason: String, errorDetails: String) {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别已取消: $reason")
+ notifyStateChanged()
+ }
+ override fun onError(error: String) {
+ isRecognitionActive = false
+ FileLogger.e(TAG, "停止语音识别时出错: $error")
+ notifyStateChanged()
+ }
+ override fun onSuccess(message: String) {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别已停止: $message")
+ notifyStateChanged()
+ }
+ })
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e)
+ // 确保状态一致性
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 使用OpenAI处理语音识别结果
+ */
+ private fun processWithOpenAI(text: String) {
+ // 保存当前用户输入,用于后续同步聊天记录
+ currentUserInput = text
+
+ Thread {
+ try {
+ val messages = JSONArray().apply {
+ put(openAIService.createUserMessage(text))
+ }
+
+ // 创建响应构建器
+ val responseBuilder = StringBuilder()
+
+ openAIService.sendMessageStream(
+ messages = messages,
+ callback = object : OpenAIService.StreamCallback {
+ override fun onToken(token: String) {
+ // 累加响应内容
+ responseBuilder.append(token)
+ }
+
+ override fun onComplete() {
+ // 处理完整响应
+ val response = responseBuilder.toString()
+ if (response.isNotEmpty()) {
+ // 播放AI回复
+ Log.d(TAG, "AI 回复: $response")
+
+ speakAIResponse(response)
+
+ // 同步聊天记录到Flutter端
+ sendChatHistoryUpdate("personal_assistant", text, response)
+ }
+ }
+
+ override fun onError(e: Exception) {
+ FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e)
+ callback?.onError("AI处理出错")
+ }
+
+ override fun onFunctionCall(call: JSONObject) {
+ FileLogger.d(TAG, "收到函数调用请求: ${call.getString("name")}")
+
+ // 使用函数处理器处理函数调用
+ val handled = voiceFunctionHandler.handleFunctionCall(
+ functionCall = call,
+ messages = messages,
+ callback = object : VoiceFunctionHandler.FunctionCallCallback {
+ override fun onTokenReceived(token: String) {
+ responseBuilder.append(token)
+ }
+
+ override fun onComplete() {
+ val response = responseBuilder.toString()
+ if (response.isNotEmpty()) {
+ // 播放AI回复
+ Log.d(TAG, "AI Function Call 回复: $response")
+ speakAIResponse(response)
+
+ // 同步聊天记录到Flutter端
+ sendChatHistoryUpdate("personal_assistant", text, response)
+ }
+ notifyStateChanged()
+ }
+
+ override fun onError(message: String) {
+ FileLogger.e(TAG, "函数处理出错: $message")
+ callback?.onError(message)
+ }
+
+ override fun onFunctionCall(nestedCall: JSONObject) {
+ FileLogger.d(TAG, "收到嵌套函数调用: ${nestedCall.getString("name")}")
+ // 不处理嵌套函数调用,直接返回错误提示
+ // speakAIResponse("抱歉,暂不支持嵌套函数调用")
+ }
+
+ override fun onExitWithMessage(farewell: String) {
+ // 播放退出消息
+ speakAIResponse(farewell)
+
+ // 同步聊天记录
+ sendChatHistoryUpdate("personal_assistant", text, farewell)
+
+ // 停止语音识别
+ stopRecognition()
+ }
+ }
+ )
+
+ if (!handled) {
+ // 如果函数没有被处理,作为普通文本处理
+ FileLogger.d(TAG, "函数未处理,作为普通文本处理")
+ speakAIResponse("我无法处理这个请求")
+ sendChatHistoryUpdate("personal_assistant", text, "我无法处理这个请求")
+ }
+ }
+ }
+ )
+
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "AI处理出错: ${e.message}", e)
+ callback?.onError("AI处理出错")
+ }
+ }.start()
+ }
+
+ /**
+ * 播放TTS
+ */
+ fun playTts(text: String, callback: TtsCallback? = null) {
+ isTtsSpeaking = true
+ notifyStateChanged()
+
+ volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback {
+ override fun onStart(reqId: String) {
+ // TTS开始播放事件
+ }
+
+ override fun onProgress(reqId: String, progress: Double) {
+ // TTS播放进度事件
+ }
+
+ override fun onComplete(reqId: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ callback?.onComplete()
+ }
+
+ override fun onError(reqId: String, errorCode: Int, errorMsg: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ callback?.onError(errorMsg)
+ }
+ })
+ }
+
+ /**
+ * 停止TTS播放
+ */
+ fun stopTts() {
+ if (isTtsSpeaking) {
+ volcanoTtsHelper?.stop()
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 播放AI回复
+ */
+ private fun speakAIResponse(text: String) {
+ isTtsSpeaking = true
+ notifyStateChanged()
+
+ volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback {
+ override fun onStart(reqId: String) {
+ // TTS开始播放事件
+ }
+
+ override fun onProgress(reqId: String, progress: Double) {
+ // TTS播放进度事件
+ }
+
+ override fun onComplete(reqId: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+
+ override fun onError(reqId: String, errorCode: Int, errorMsg: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+ })
+ }
+
+ /**
+ * 释放资源
+ */
+ fun dispose() {
+ // 停止语音识别
+ stopRecognition()
+
+ // 停止TTS播放
+ stopTts()
+
+ // 释放Azure资源
+ azureAsrHelper?.let {
+ FileLogger.d(TAG, "关闭Azure ASR服务")
+ it.dispose()
+ }
+
+ volcanoTtsHelper?.let {
+ FileLogger.d(TAG, "关闭Volcano TTS服务")
+ it.release()
+ }
+
+ FileLogger.d(TAG, "语音交互处理器资源已释放")
+ }
+
+ /**
+ * 检查录音权限
+ */
+ private fun checkRecordAudioPermission(): Boolean {
+ val permission = android.Manifest.permission.RECORD_AUDIO
+ val result = context.checkCallingOrSelfPermission(permission)
+ return result == android.content.pm.PackageManager.PERMISSION_GRANTED
+ }
+
+ /**
+ * 通知状态变化
+ */
+ private fun notifyStateChanged() {
+ callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected)
+ }
+
+ /**
+ * 发送聊天历史更新
+ */
+ private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) {
+ val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply {
+ putExtra("agentId", agentId)
+ putExtra("userMessage", userMessage)
+ putExtra("assistantMessage", assistantMessage)
+ putExtra("timestamp", System.currentTimeMillis())
+ }
+
+ // 发送广播
+ context.sendBroadcast(intent)
+ }
+
+ /**
+ * 交互回调接口
+ */
+ interface InteractionCallback {
+ fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean)
+ fun onError(message: String)
+ fun onPromptRequest(message: String)
+ }
+
+ /**
+ * TTS回调接口
+ */
+ interface TtsCallback {
+ fun onComplete()
+ fun onError(error: String)
+ }
+}
\ No newline at end of file
diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt
new file mode 100644
index 000000000..b885117c0
--- /dev/null
+++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt
@@ -0,0 +1,416 @@
+package com.yunqiinnovation.deepsound
+
+import android.content.BroadcastReceiver
+import android.content.Context
+import android.content.Intent
+import android.content.IntentFilter
+import org.json.JSONArray
+import org.json.JSONObject
+import android.util.Log
+import com.yunqiinnovation.deepsound.core.utils.FileLogger
+import com.yunqiinnovation.azure_speech.AzureAsrHelper
+import com.yunqiinnovation.azure_speech.AzureTtsHelper
+import com.yunqiinnovation.open_ai_service.OpenAIService
+import com.yunqiinnovation.open_ai_service.SystemFunctionHandler
+
+
+/**
+ * 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑
+ */
+class VoiceInteractionHandler(
+ private val context: Context,
+ private val azureSpeechKey: String,
+ private val azureSpeechRegion: String,
+ private val openaiApiKey: String,
+ private val openaiBaseUrl: String = "",
+ private val openaiModel: String = "",
+ private val volcanoSpeechAppId: String,
+ private val volcanoSpeechAppToken: String
+) {
+ private val TAG = "VoiceInteractionHandler"
+
+ // Azure服务
+ private var azureAsrHelper: AzureAsrHelper? = null
+ private var azureTtsHelper: AzureTtsHelper? = null
+
+ // OpenAI服务
+ private val openAIService = OpenAIService(context.applicationContext)
+
+ // 当前用户输入
+ private var currentUserInput = ""
+
+ // 状态
+ private var isInitialized = false
+ var isRecognitionActive = false
+ private set
+ var isTtsSpeaking = false
+ private set
+ var hasSpeechDetected = false
+ private set
+
+ // 回调
+ private var callback: InteractionCallback? = null
+
+ // 广播接收器
+ private val exitInteractionReceiver = object : BroadcastReceiver() {
+ override fun onReceive(context: Context, intent: Intent) {
+ if (intent.action == SystemFunctionHandler.ACTION_EXIT_INTERACTION) {
+ Log.d(TAG, "收到退出交互广播")
+ stopRecognition()
+
+ }
+ }
+ }
+
+ /**
+ * 初始化
+ */
+ fun initialize(): Boolean {
+ if (isInitialized) return true
+
+ try {
+ // 初始化Azure ASR
+ azureAsrHelper = AzureAsrHelper(context).apply {
+ initialize(azureSpeechKey, azureSpeechRegion)
+ }
+
+ // 初始化Azure TTS
+ azureTtsHelper = AzureTtsHelper(context).apply {
+ initialize(azureSpeechKey, azureSpeechRegion)
+ }
+
+ // 初始化OpenAI服务
+ openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel)
+
+ // 注册广播接收器
+ try {
+ Log.d(TAG, "注册退出交互广播接收器,包名=${context.packageName}, action=${SystemFunctionHandler.ACTION_EXIT_INTERACTION}")
+ context.registerReceiver(
+ exitInteractionReceiver,
+ IntentFilter(SystemFunctionHandler.ACTION_EXIT_INTERACTION),
+ Context.RECEIVER_NOT_EXPORTED
+ )
+ Log.d(TAG, "退出交互广播接收器注册成功")
+ } catch (e: Exception) {
+ // 广播注册失败不应该影响整个应用初始化
+ Log.e(TAG, "注册退出交互广播接收器失败: ${e.message}", e)
+ }
+
+ isInitialized = true
+ return true
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "初始化失败: ${e.message}", e)
+ return false
+ }
+ }
+
+ /**
+ * 设置回调
+ */
+ fun setCallback(callback: InteractionCallback) {
+ this.callback = callback
+ }
+
+ /**
+ * 开始语音识别
+ */
+ fun startRecognition() {
+ if (isRecognitionActive) return
+
+ // 检查录音权限
+ if (!checkRecordAudioPermission()) {
+ callback?.onError("需要录音权限,请在设置中授予权限")
+ return
+ }
+
+ isRecognitionActive = true
+ hasSpeechDetected = false
+ notifyStateChanged()
+
+ try {
+ azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
+ override fun onRecognizing(recognizing: String, detectedLanguage: String) {
+ if (recognizing.isNotEmpty()) {
+ hasSpeechDetected = true
+ stopTts()
+ notifyStateChanged()
+ }
+ }
+
+ override fun onResult(result: String, detectedLanguage: String) {
+ if (result.isNotEmpty()) {
+ notifyStateChanged()
+
+ processWithOpenAI(result)
+
+ }
+
+ // 重置状态,继续识别
+ hasSpeechDetected = false
+ }
+
+ override fun onSessionStarted() {
+ notifyStateChanged()
+ }
+
+ override fun onSessionStopped() {
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+
+ override fun onCanceled(reason: String, errorDetails: String) {
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+
+ override fun onError(error: String) {
+ isRecognitionActive = false
+ callback?.onError("语音识别出错")
+ notifyStateChanged()
+ }
+
+ override fun onSuccess(message: String) {
+ // 处理成功事件
+ }
+ })
+ } catch (e: Exception) {
+ isRecognitionActive = false
+ FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e)
+ callback?.onError("启动语音识别失败")
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 停止语音识别
+ */
+ fun stopRecognition() {
+ if (!isRecognitionActive) return
+
+ FileLogger.d(TAG, "停止语音识别")
+
+ try {
+ azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
+ override fun onResult(result: String, detectedLanguage: String) {}
+ override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
+ override fun onSessionStarted() {}
+ override fun onSessionStopped() {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别会话已停止")
+ notifyStateChanged()
+ }
+ override fun onCanceled(reason: String, errorDetails: String) {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别已取消: $reason")
+ notifyStateChanged()
+ }
+ override fun onError(error: String) {
+ isRecognitionActive = false
+ FileLogger.e(TAG, "停止语音识别时出错: $error")
+ notifyStateChanged()
+ }
+ override fun onSuccess(message: String) {
+ isRecognitionActive = false
+ FileLogger.d(TAG, "语音识别已停止: $message")
+ notifyStateChanged()
+ }
+ })
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e)
+ // 确保状态一致性
+ isRecognitionActive = false
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 使用OpenAI处理语音识别结果
+ */
+ private fun processWithOpenAI(text: String) {
+ // 保存当前用户输入,用于后续同步聊天记录
+ currentUserInput = text
+
+ Thread {
+ try {
+ val messages = JSONArray().apply {
+ put(openAIService.createUserMessage(text))
+ }
+
+ // 创建响应构建器
+ val responseBuilder = StringBuilder()
+
+ openAIService.sendMessageStream(
+ messages = messages,
+ callback = object : OpenAIService.StreamCallback {
+ override fun onToken(token: String) {
+ // 累加响应内容
+ responseBuilder.append(token)
+ }
+
+ override fun onComplete() {
+ // 处理完整响应
+ val response = responseBuilder.toString()
+ if (response.isNotEmpty()) {
+ // 播放AI回复
+ Log.d(TAG, "AI 回复: $response")
+
+ speakAIResponse(response)
+
+ // 同步聊天记录到Flutter端
+ sendChatHistoryUpdate("personal_assistant", text, response)
+ }
+ }
+
+ override fun onError(e: Exception) {
+ FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e)
+ callback?.onError("AI处理出错")
+ }
+
+ override fun onFunctionCall(call: JSONObject) {
+ FileLogger.d(TAG, "processWithOpenAI 收到函数调用请求: ${call.getString("name")}")
+
+
+ }
+ }
+ )
+
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "AI处理出错: ${e.message}", e)
+ callback?.onError("AI处理出错")
+ }
+ }.start()
+ }
+
+ /**
+ * 播放TTS
+ */
+ fun playTts(text: String, callback: TtsCallback? = null) {
+ isTtsSpeaking = true
+ notifyStateChanged()
+
+ azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback {
+ override fun onSuccess(message: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ callback?.onComplete()
+ }
+
+ override fun onError(error: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ callback?.onError(error)
+ }
+ })
+ }
+
+ /**
+ * 停止TTS播放
+ */
+ fun stopTts() {
+ if (isTtsSpeaking) {
+ azureTtsHelper?.stopSpeaking()
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+ }
+
+ /**
+ * 播放AI回复
+ */
+ private fun speakAIResponse(text: String) {
+ isTtsSpeaking = true
+ notifyStateChanged()
+
+ azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback {
+ override fun onSuccess(message: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+
+ override fun onError(error: String) {
+ isTtsSpeaking = false
+ notifyStateChanged()
+ }
+ })
+ }
+
+ /**
+ * 释放资源
+ */
+ fun dispose() {
+ // 停止语音识别
+ stopRecognition()
+
+ // 停止TTS播放
+ stopTts()
+
+ // 注销广播接收器
+ try {
+ context.unregisterReceiver(exitInteractionReceiver)
+ } catch (e: Exception) {
+ FileLogger.e(TAG, "注销广播接收器失败: ${e.message}", e)
+ }
+
+ // 释放Azure资源
+ azureAsrHelper?.let {
+ FileLogger.d(TAG, "关闭Azure ASR服务")
+ it.dispose()
+ }
+
+ azureTtsHelper?.let {
+ FileLogger.d(TAG, "关闭Azure TTS服务")
+ it.dispose()
+ }
+
+ FileLogger.d(TAG, "语音交互处理器资源已释放")
+ }
+
+ /**
+ * 检查录音权限
+ */
+ private fun checkRecordAudioPermission(): Boolean {
+ val permission = android.Manifest.permission.RECORD_AUDIO
+ val result = context.checkCallingOrSelfPermission(permission)
+ return result == android.content.pm.PackageManager.PERMISSION_GRANTED
+ }
+
+ /**
+ * 通知状态变化
+ */
+ private fun notifyStateChanged() {
+ callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected)
+ }
+
+ /**
+ * 发送聊天历史更新
+ */
+ private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) {
+ val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply {
+ putExtra("agentId", agentId)
+ putExtra("userMessage", userMessage)
+ putExtra("assistantMessage", assistantMessage)
+ putExtra("timestamp", System.currentTimeMillis())
+ setPackage(context.packageName)
+ }
+
+ // 发送广播
+ context.sendBroadcast(intent)
+ }
+
+ /**
+ * 交互回调接口
+ */
+ interface InteractionCallback {
+ fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean)
+ fun onError(message: String)
+ fun onPromptRequest(message: String)
+ }
+
+ /**
+ * TTS回调接口
+ */
+ interface TtsCallback {
+ fun onComplete()
+ fun onError(error: String)
+ }
+}
\ No newline at end of file
diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt
similarity index 67%
rename from android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt
rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt
index f469c136e..dbe64c79e 100644
--- a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt
+++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt
@@ -22,16 +22,21 @@ import android.os.Handler
import android.os.Looper
import java.util.concurrent.atomic.AtomicBoolean
import org.json.JSONArray
+import org.json.JSONObject
import android.media.MediaPlayer
import android.media.AudioAttributes
import android.net.Uri
import com.yunqiinnovation.deepsound.core.utils.FileLogger
+import com.yunqiinnovation.azure_speech.AzureAsrHelper
+import com.yunqiinnovation.volcano_speech.VolcanoTtsHelper
+import com.yunqiinnovation.open_ai_service.OpenAIService
+
/**
* 后台语音交互 Service:
* 1) 前台服务,确保不会被系统轻易杀死
* 2) MediaSession 捕获蓝牙耳机按键
- * 3) 处理录音/语音识别
+ * 3) 负责唤醒控制和服务生命周期管理
*/
class VoiceInteractionService : Service() {
@@ -41,7 +46,7 @@ class VoiceInteractionService : Service() {
private const val CHANNEL_ID = "voice_interaction_channel"
// 语音识别超时时间(毫秒)
- private const val RECOGNITION_TIMEOUT = 8000L
+ private const val RECOGNITION_TIMEOUT = 10000L
// 用于跟踪服务是否正在运行
private val isRunning = AtomicBoolean(false)
@@ -53,14 +58,12 @@ class VoiceInteractionService : Service() {
const val ACTION_RECOGNITION_STARTED = "com.yunqiinnovation.deepsound.ACTION_RECOGNITION_STARTED"
const val ACTION_PAUSE_VOICE_INTERACTION = "com.yunqiinnovation.deepsound.ACTION_PAUSE_VOICE_INTERACTION"
const val ACTION_CHAT_HISTORY_UPDATED = "com.yunqiinnovation.deepsound.ACTION_CHAT_HISTORY_UPDATED"
+ const val ACTION_ENTER_TRANSLATION_MODE = "com.yunqiinnovation.deepsound.ACTION_ENTER_TRANSLATION_MODE"
}
// 服务状态
private var isActive = false // 服务是否活跃
- private var isRecognitionActive = false // 语音识别是否活跃
private var isTimeoutPaused = false // 是否因超时暂停
- private var hasSpeechDetected = false // 是否检测到语音
- private var isTtsSpeaking = false // 是否正在播放TTS
// 按键处理
private var lastKeyEventTime = 0L
@@ -69,16 +72,12 @@ class VoiceInteractionService : Service() {
// 活动时间
private var lastActivityTime = 0L
- // 当前用户输入
- private var currentUserInput = ""
-
-
// 服务组件
private lateinit var mediaSession: MediaSessionCompat
private lateinit var audioManager: AudioManager
- private lateinit var azureAsrHelper: AzureAsrHelper
- private lateinit var azureTtsHelper: AzureTtsHelper
- private lateinit var volcanoAIService: VolcanoAIService
+
+ // 语音交互处理器
+ private lateinit var voiceInteractionHandler: VoiceInteractionHandler
// 定时器
private val handler = Handler(Looper.getMainLooper())
@@ -88,15 +87,6 @@ class VoiceInteractionService : Service() {
handler.postDelayed(this, 1000) // 每秒执行一次
}
}
-
- // 系统提示词
- private val systemPrompt = """
- 你是一个智能语音助手,能够简洁明了地回答用户的问题。
- 请保持回答简短、准确,避免过长的解释。
- 如果用户的问题不清楚,请礼貌地请求澄清。
- 不要使用复杂的术语,除非用户明确要求。
- 用户用语音和你交互.
- """.trimIndent()
// 添加媒体播放器
private var audioPlayer: MiniMediaPlayer? = null
@@ -115,7 +105,9 @@ class VoiceInteractionService : Service() {
audioManager = getSystemService(Context.AUDIO_SERVICE) as AudioManager
FileLogger.d(TAG, "AudioManager初始化完成")
- initServices()
+ // 初始化语音交互处理器
+ initVoiceInteractionHandler()
+
initMediaSession()
registerMediaButtonReceiver()
@@ -125,7 +117,6 @@ class VoiceInteractionService : Service() {
// 设置为媒体播放状态
setPlaybackState(PlaybackStateCompat.STATE_PAUSED)
- // FileLogger.d(TAG, "设置播放状态为STATE_PAUSED")
// 启动监控和前台服务
startMonitoring()
@@ -139,28 +130,26 @@ class VoiceInteractionService : Service() {
*/
private fun resetState() {
isActive = false
- isRecognitionActive = false
isTimeoutPaused = false
- hasSpeechDetected = false
- isTtsSpeaking = false
}
/**
- * 初始化所有服务
+ * 初始化语音交互处理器
*/
- private fun initServices() {
- // 创建新的Azure服务实例
- FileLogger.d(TAG, "创建新的Azure服务实例")
- azureAsrHelper = AzureAsrHelper(this)
- azureTtsHelper = AzureTtsHelper(this)
+ private fun initVoiceInteractionHandler() {
+ FileLogger.d(TAG, "初始化语音交互处理器")
// 尝试从静态变量获取配置
var subscriptionKey = MainActivity.azureSpeechKey
var serviceRegion = MainActivity.azureSpeechRegion
- var volcanoKey = MainActivity.volcanoAiApiKey
+ var openaiKey = MainActivity.openaiApiKey
+ var openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" // OpenAI API基本URL
+ var openaiModel = MainActivity.openaiModel ?: "" // OpenAI模型
+ var volcanoSpeechAppId = MainActivity.volcanoSpeechAppId ?: ""
+ var volcanoSpeechAppToken = MainActivity.volcanoSpeechAppToken ?: ""
// 如果静态变量中没有配置,尝试从加密存储中加载
- if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || volcanoKey.isEmpty()) {
+ if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || openaiKey.isEmpty()) {
FileLogger.d(TAG, "静态变量中的配置信息不完整,尝试从加密存储加载")
// 从加密存储加载密钥
@@ -170,7 +159,11 @@ class VoiceInteractionService : Service() {
// 更新本地变量
subscriptionKey = MainActivity.azureSpeechKey
serviceRegion = MainActivity.azureSpeechRegion
- volcanoKey = MainActivity.volcanoAiApiKey
+ openaiKey = MainActivity.openaiApiKey
+ openaiBaseUrl = MainActivity.openaiBaseUrl ?: ""
+ openaiModel = MainActivity.openaiModel ?: ""
+ volcanoSpeechAppId = MainActivity.volcanoSpeechAppId ?: ""
+ volcanoSpeechAppToken = MainActivity.volcanoSpeechAppToken ?: ""
FileLogger.d(TAG, "已从加密存储加载配置信息")
} else {
@@ -178,28 +171,34 @@ class VoiceInteractionService : Service() {
}
}
- // 初始化语音服务
- if (subscriptionKey.isNotEmpty() && serviceRegion.isNotEmpty()) {
- // 初始化ASR
- azureAsrHelper.initialize(subscriptionKey, serviceRegion, arrayOf("zh-CN"))
+ // 初始化语音交互处理器
+ voiceInteractionHandler = VoiceInteractionHandler(applicationContext,
+ subscriptionKey, serviceRegion,
+ openaiKey, openaiBaseUrl, openaiModel,
+ volcanoSpeechAppId, volcanoSpeechAppToken)
+
+ // 初始化回调
+ voiceInteractionHandler.setCallback(object : VoiceInteractionHandler.InteractionCallback {
+ override fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) {
+ // 更新活动时间
+ updateLastActivityTime()
+ }
- // 初始化TTS
- azureTtsHelper.initialize(subscriptionKey, serviceRegion, "zh-CN")
+ override fun onError(message: String) {
+ playNotification(message)
+ }
- FileLogger.d(TAG, "Azure语音服务已初始化")
- } else {
- FileLogger.e(TAG, "Azure配置信息不完整,无法初始化Azure服务")
- }
-
- // 初始化火山AI服务
- volcanoAIService = VolcanoAIService()
+ override fun onPromptRequest(message: String) {
+ playPrompt(message)
+ }
+ })
- // 初始化火山AI服务
- if (volcanoKey.isNotEmpty()) {
- volcanoAIService.initialize(volcanoKey)
- FileLogger.d(TAG, "火山AI服务已初始化")
+ // 初始化处理器
+ val initialized = voiceInteractionHandler.initialize()
+ if (initialized) {
+ FileLogger.d(TAG, "语音交互处理器初始化成功")
} else {
- FileLogger.e(TAG, "火山AI配置信息不完整,无法初始化火山AI服务")
+ FileLogger.e(TAG, "语音交互处理器初始化失败")
}
}
@@ -302,24 +301,23 @@ class VoiceInteractionService : Service() {
if (!isActive) {
isActive = true
}
-
+ // FileLogger.d(TAG, "服务状态: isActive=${isActive}, isRecognitionActive=${voiceInteractionHandler.isRecognitionActive}, isTimeoutPaused=${isTimeoutPaused}")
// 检查语音识别状态
- if (isRecognitionActive) {
+ if (voiceInteractionHandler.isRecognitionActive) {
val currentTime = System.currentTimeMillis()
val elapsedTime = currentTime - lastActivityTime
-
- // 如果超过5秒没有检测到语音,且不在TTS播放中,暂停语音识别
- if (!hasSpeechDetected && !isTtsSpeaking && elapsedTime >= RECOGNITION_TIMEOUT) {
+ // FileLogger.d(TAG, "hasSpeechDetected=${voiceInteractionHandler.hasSpeechDetected}, isTtsSpeaking=${voiceInteractionHandler.isTtsSpeaking}, elapsedTime=${elapsedTime}")
+ // 如果超过指定时间没有检测到语音,且不在TTS播放中,暂停语音识别
+ if (!voiceInteractionHandler.hasSpeechDetected &&
+ !voiceInteractionHandler.isTtsSpeaking &&
+ elapsedTime >= RECOGNITION_TIMEOUT) {
+
FileLogger.d(TAG, "超过${RECOGNITION_TIMEOUT/1000}秒未检测到语音,停止识别")
isTimeoutPaused = true
playNotification("没有听到您说话,已暂停对话。双击耳机按钮可重新开始。")
- // 停止语音识别并确保资源完全释放
- stopVoiceRecognition()
-
- // 清理识别状态
- isRecognitionActive = false
- hasSpeechDetected = false
+ // 停止语音识别
+ voiceInteractionHandler.stopRecognition()
}
}
}
@@ -340,14 +338,12 @@ class VoiceInteractionService : Service() {
lastKeyEventTime = currentTime
// 停止当前TTS播放
- stopCurrentTTS()
+ voiceInteractionHandler.stopTts()
setPlaybackState(PlaybackStateCompat.STATE_PLAYING)
// 播放提示音
playPrompt("我在!")
- // audioPlayer?.play(R.raw.listening)
-
// 设置为播放状态
setPlaybackState(PlaybackStateCompat.STATE_PAUSED)
@@ -355,221 +351,48 @@ class VoiceInteractionService : Service() {
// 重置超时暂停标志
isTimeoutPaused = false
- // 启动或重置语音识别
- if (!isRecognitionActive) {
+ // 启动语音识别
+ if (!voiceInteractionHandler.isRecognitionActive) {
FileLogger.d(TAG, "语音识别未激活,开始启动")
- // 如果之前是因为超时暂停,重新初始化语音识别组件
- // if (isTimeoutPaused) {
- // Log.d(TAG, "之前因超时暂停,重新初始化Azure服务")
- // restartAsr()
- // }
+ // 通知 Flutter 语音识别已启动
+ notifyRecognitionStarted()
- startVoiceRecognition()
+ // 启动语音识别
+ voiceInteractionHandler.startRecognition()
} else {
FileLogger.d(TAG, "语音识别已激活,更新活动时间")
updateLastActivityTime()
- hasSpeechDetected = false
- }
- }
-
- /**
- * 开始语音识别
- */
- private fun startVoiceRecognition() {
- if (isRecognitionActive) return
-
- // 检查录音权限
- if (!checkRecordAudioPermission()) {
- playNotification("需要录音权限,请在设置中授予权限")
- return
- }
- // 通知 Flutter 语音识别已启动
- notifyRecognitionStarted()
-
- isActive = true
- isRecognitionActive = true
- hasSpeechDetected = false
- updateLastActivityTime()
-
- try {
- azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
- override fun onRecognizing(recognizing: String, detectedLanguage: String) {
- if (recognizing.isNotEmpty()) {
- hasSpeechDetected = true
- stopCurrentTTS()
- updateLastActivityTime()
- }
- }
-
- override fun onResult(result: String, detectedLanguage: String) {
- if (result.isNotEmpty()) {
- processWithVolcanoAI(result)
- }
-
- // 重置状态,继续识别
- hasSpeechDetected = false
- updateLastActivityTime()
- }
-
- override fun onSessionStarted() {
- updateLastActivityTime()
-
- }
-
- override fun onSessionStopped() {
- isRecognitionActive = false
-
-
- }
-
- override fun onCanceled(reason: String, errorDetails: String) {
- isRecognitionActive = false
-
-
- }
-
- override fun onError(error: String) {
- isRecognitionActive = false
- playNotification("语音识别出错")
-
- }
- })
- } catch (e: Exception) {
- isRecognitionActive = false
- FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e)
- playNotification("启动语音识别失败")
-
-
- }
- }
-
- /**
- * 停止语音识别
- */
- private fun stopVoiceRecognition() {
- if (!isRecognitionActive) return
-
- FileLogger.d(TAG, "停止语音识别")
-
- try {
- azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
- override fun onResult(result: String, detectedLanguage: String) {}
- override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
- override fun onSessionStarted() {}
- override fun onSessionStopped() {
- isRecognitionActive = false
- FileLogger.d(TAG, "语音识别会话已停止")
- }
- override fun onCanceled(reason: String, errorDetails: String) {
- isRecognitionActive = false
- FileLogger.d(TAG, "语音识别已取消: $reason")
- }
- override fun onError(error: String) {
- isRecognitionActive = false
- FileLogger.e(TAG, "停止语音识别时出错: $error")
- }
- })
- } catch (e: Exception) {
- FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e)
- // 确保状态一致性
- isRecognitionActive = false
}
-
- setPlaybackState(PlaybackStateCompat.STATE_PAUSED)
- isRecognitionActive = false
- hasSpeechDetected = false
- // 不重置 isTimeoutPaused,保留暂停原因
- }
-
- /**
- * 使用VolcanoAI处理语音识别结果
- */
- private fun processWithVolcanoAI(text: String) {
- // 保存当前用户输入,用于后续同步聊天记录
- currentUserInput = text
-
- Thread {
- try {
- val messages = JSONArray().apply {
- put(volcanoAIService.createUserMessage(text))
- }
-
- val response = volcanoAIService.sendMessage(messages, systemPrompt)
-
- // 播放AI回复
- speakAIResponse(response)
-
- // 同步聊天记录到Flutter端
- notifyChatHistoryUpdated("personal_assistant", text, response)
- } catch (e: Exception) {
- FileLogger.e(TAG, "AI处理出错: ${e.message}")
- playNotification("AI处理出错")
- }
- }.start()
- }
-
- /**
- * 播放AI回复
- */
- private fun speakAIResponse(text: String) {
- isTtsSpeaking = true
- azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback {
- override fun onSuccess(message: String) {
- isTtsSpeaking = false
- updateLastActivityTime()
- }
-
- override fun onError(error: String) {
- isTtsSpeaking = false
- }
- })
}
/**
* 播放提示音
*/
private fun playPrompt(message: String) {
- isTtsSpeaking = true
- azureTtsHelper.speakText(message, object : AzureTtsHelper.TTSCallback {
- override fun onSuccess(message: String) { isTtsSpeaking = false }
- override fun onError(error: String) { isTtsSpeaking = false }
- })
+ voiceInteractionHandler.playTts(message)
}
/**
* 播放通知提示音
*/
private fun playNotification(message: String) {
- isTtsSpeaking = true
// 更新播放状态为播放中,增加接收蓝牙按键事件的几率
setPlaybackState(PlaybackStateCompat.STATE_PLAYING)
// 确保媒体会话处于活跃状态
mediaSession.isActive = true
- azureTtsHelper.speakText(message, object : AzureTtsHelper.TTSCallback {
- override fun onSuccess(message: String) {
- isTtsSpeaking = false
+ voiceInteractionHandler.playTts(message, object : VoiceInteractionHandler.TtsCallback {
+ override fun onComplete() {
setPlaybackState(PlaybackStateCompat.STATE_PAUSED)
}
override fun onError(error: String) {
- isTtsSpeaking = false
setPlaybackState(PlaybackStateCompat.STATE_PAUSED)
}
})
}
- /**
- * 停止当前TTS播放
- */
- private fun stopCurrentTTS() {
- if (isTtsSpeaking) {
- azureTtsHelper.stopSpeaking()
- isTtsSpeaking = false
- }
- }
-
/**
* 更新最后活动时间
*/
@@ -603,15 +426,6 @@ class VoiceInteractionService : Service() {
}
}
- /**
- * 检查录音权限
- */
- private fun checkRecordAudioPermission(): Boolean {
- val permission = android.Manifest.permission.RECORD_AUDIO
- val result = applicationContext.checkCallingOrSelfPermission(permission)
- return result == android.content.pm.PackageManager.PERMISSION_GRANTED
- }
-
/**
* 启动前台服务
*/
@@ -746,38 +560,32 @@ class VoiceInteractionService : Service() {
override fun onBind(intent: Intent?): IBinder? = null
override fun onDestroy() {
- FileLogger.d(TAG, "onDestroy - 语音交互服务正在销毁")
-
- // 释放音频播放器
- audioPlayer?.release()
- audioPlayer = null
+ super.onDestroy()
+ FileLogger.d(TAG, "onDestroy - 语音交互服务即将销毁")
- // 停止监控
+ // 停止服务监控
stopMonitoring()
// 停止语音识别
- if (isRecognitionActive) {
- stopVoiceRecognition()
- }
+ voiceInteractionHandler.stopRecognition()
+
+ // 停止媒体会话
+ mediaSession.release()
+ FileLogger.d(TAG, "媒体会话已释放")
// 停止TTS
- stopCurrentTTS()
+ voiceInteractionHandler.stopTts()
- // 释放 MediaSession
- mediaSession.release()
+ // 释放语音交互处理器资源
+ voiceInteractionHandler.dispose()
- // 释放Azure服务实例
- FileLogger.d(TAG, "释放Azure服务实例")
- azureAsrHelper.dispose()
- azureTtsHelper.dispose()
+ // 关闭音频播放器
+ audioPlayer?.release()
- // 更新服务状态
+ // 设置服务状态
isRunning.set(false)
- // 关闭日志系统
- FileLogger.shutdown()
-
- super.onDestroy()
+ FileLogger.d(TAG, "语音交互服务已销毁")
}
/**
@@ -788,11 +596,11 @@ class VoiceInteractionService : Service() {
FileLogger.d(TAG, "暂停后台语音交互(来自Flutter的请求)")
// 停止当前TTS播放
- stopCurrentTTS()
+ voiceInteractionHandler.stopTts()
// 停止语音识别
- if (isRecognitionActive) {
- stopVoiceRecognition()
+ if (voiceInteractionHandler.isRecognitionActive) {
+ voiceInteractionHandler.stopRecognition()
}
// 设置为暂停状态,但保持服务活跃
@@ -804,10 +612,6 @@ class VoiceInteractionService : Service() {
* 通知 Flutter 端聊天记录已更新
*/
private fun notifyChatHistoryUpdated(agentId: String, userMessage: String, assistantMessage: String) {
- FileLogger.d(TAG, "通知Flutter聊天记录已更新: agentId=$agentId")
- FileLogger.d(TAG, "用户消息: ${userMessage.take(50)}...")
- FileLogger.d(TAG, "助手回复: ${assistantMessage.take(50)}...")
-
// 创建广播 Intent
val intent = Intent(ACTION_CHAT_HISTORY_UPDATED).apply {
putExtra("agentId", agentId)
@@ -818,7 +622,6 @@ class VoiceInteractionService : Service() {
// 发送广播
sendBroadcast(intent)
- FileLogger.d(TAG, "已发送聊天记录广播")
}
/**
@@ -831,6 +634,7 @@ class VoiceInteractionService : Service() {
val intent = Intent(ACTION_RECOGNITION_STARTED).apply {
// 可以添加额外数据
putExtra("timestamp", System.currentTimeMillis())
+ putExtra("packageName", applicationContext.packageName)
}
// 发送广播
@@ -909,5 +713,4 @@ class VoiceInteractionService : Service() {
*/
fun isPlaying() = mediaPlayer?.isPlaying == true
}
-
}
\ No newline at end of file
diff --git a/android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt
similarity index 100%
rename from android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt
rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt
diff --git a/android/build.gradle.kts b/android/build.gradle.kts
index 35365df1a..5421f73bd 100644
--- a/android/build.gradle.kts
+++ b/android/build.gradle.kts
@@ -2,6 +2,11 @@ allprojects {
repositories {
google()
mavenCentral()
+ maven { url = uri("https://storage.googleapis.com/download.flutter.io") }
+ // 添加火山引擎Maven仓库
+ maven {
+ url = uri("https://artifact.bytedance.com/repository/Volcengine/")
+ }
}
}
@@ -20,3 +25,14 @@ subprojects {
tasks.register("clean") {
delete(rootProject.layout.buildDirectory)
}
+
+buildscript {
+ repositories {
+ google()
+ mavenCentral()
+ }
+ dependencies {
+ classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:2.1.10")
+ // 其他依赖...
+ }
+}
diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts
index 6296e34ed..e332da5f9 100644
--- a/android/settings.gradle.kts
+++ b/android/settings.gradle.kts
@@ -25,8 +25,15 @@ pluginManagement {
plugins {
id("dev.flutter.flutter-plugin-loader")
id("com.android.application") version "8.7.0" apply false
- id("org.jetbrains.kotlin.android") version "1.8.22" apply false
+ id("org.jetbrains.kotlin.android") version "2.1.10" apply false
}
include(":app")
+include(":azure_speech")
+include(":open_ai_service")
+include(":volcano_speech")
+// 设置azure_speech项目的路径
+project(":azure_speech").projectDir = file("../local_plugins/azure_speech/android")
+project(":open_ai_service").projectDir = file("../local_plugins/open_ai_service/android")
+project(":volcano_speech").projectDir = file("../local_plugins/volcano_speech/android")
diff --git a/azure/LICENSE b/azure/LICENSE
deleted file mode 100644
index d3df29b8a..000000000
--- a/azure/LICENSE
+++ /dev/null
@@ -1,21 +0,0 @@
-MIT License
-
-Copyright (c) 2024 Your Company
-
-Permission is hereby granted, free of charge, to any person obtaining a copy
-of this software and associated documentation files (the "Software"), to deal
-in the Software without restriction, including without limitation the rights
-to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
-copies of the Software, and to permit persons to whom the Software is
-furnished to do so, subject to the following conditions:
-
-The above copyright notice and this permission notice shall be included in all
-copies or substantial portions of the Software.
-
-THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
-IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
-FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
-AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
-LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
-OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
-SOFTWARE.
\ No newline at end of file
diff --git a/azure/README.md b/azure/README.md
deleted file mode 100644
index a19025680..000000000
--- a/azure/README.md
+++ /dev/null
@@ -1,66 +0,0 @@
-# Azure Speech Recognition
-
-A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities.
-
-## Features
-
-- Speech-to-text (Azure Speech Recognition)
-- Text-to-speech (Azure Speech Synthesis)
-- Support for multiple languages
-- Language detection
-- Continuous recognition
-- Streaming synthesis
-
-## Getting Started
-
-### Prerequisites
-
-- Azure Speech service subscription key
-- Azure Speech service region
-
-### Installation
-
-Add this to your package's `pubspec.yaml` file:
-
-```yaml
-dependencies:
- azure_speech_recognition:
- path: ./azure
-```
-
-### Usage
-
-```dart
-import 'package:azure_speech_recognition/azure_speech_recognition.dart';
-
-// Initialize the service
-await AzureSpeechRecognition.initialize(
- subscriptionKey: 'your_subscription_key',
- region: 'your_region',
- supportedLanguages: ['zh-CN', 'en-US'],
-);
-
-// Start continuous recognition
-await AzureSpeechRecognition.startContinuousRecognition();
-
-// Listen for recognition events
-AzureSpeechRecognition.onRecognitionEvent.listen((event) {
- if (event['type'] == 'result') {
- print('Recognized: ${event['text']}');
- print('Detected language: ${event['detectedLanguage']}');
- }
-});
-
-// Stop recognition when done
-await AzureSpeechRecognition.stopContinuousRecognition();
-
-// Speak text
-await AzureSpeechRecognition.speakText('Hello, world!');
-
-// Clean up
-await AzureSpeechRecognition.dispose();
-```
-
-## License
-
-This project is licensed under the MIT License - see the LICENSE file for details.
\ No newline at end of file
diff --git a/azure/ios/Classes/AzureAsrHelper.swift b/azure/ios/Classes/AzureAsrHelper.swift
deleted file mode 100644
index 58b142f60..000000000
--- a/azure/ios/Classes/AzureAsrHelper.swift
+++ /dev/null
@@ -1,757 +0,0 @@
-import Foundation
-import MicrosoftCognitiveServicesSpeech
-import AVFoundation
-import AudioToolbox
-
-/// Azure ASR工具类,负责实现语音识别服务接口
-@available(iOS 13.0, *)
-class AzureAsrHelper: NSObject {
- // MARK: - 属性
-
- /// 事件处理回调
- private var eventHandler: (String, [String: Any]) -> Void
-
- /// 语音配置信息
- private var speechSubscriptionKey: String = ""
- private var serviceRegion: String = ""
-
- /// 语音识别相关
- private var speechConfig: SPXSpeechConfiguration?
- private var recognizer: SPXSpeechRecognizer?
- private var audioConfig: SPXAudioConfiguration?
- private var pushStream: SPXPushAudioInputStream?
-
- /// 音频处理相关
- private var audioProcessor: CustomAudioProcessor?
- private var isProcessingAudio = false
- private var audioProcessingTimer: Timer?
-
- /// 状态标志
- private var isInitialized = false
- private var _isContinuousRecognitionActive = false
-
- /// 当前语言和支持的语言
- private var currentLanguage = "zh-CN"
- private var supportedLanguages: [String] = ["zh-CN", "en-US"]
- private var isAutoDetectLanguage = false
-
- // MARK: - 初始化
-
- init(eventHandler: @escaping (String, [String: Any]) -> Void) {
- self.eventHandler = eventHandler
- super.init()
- }
-
- deinit {
- dispose()
- }
-
- // MARK: - ASR Service 接口实现
-
- /// 初始化语音识别服务
- /// - Parameters:
- /// - speechSubscriptionKey: Azure 语音服务订阅密钥
- /// - serviceRegion: Azure 服务区域 (如 eastasia)
- /// - supportedLanguages: 支持的语言代码数组 (可选)
- /// - Returns: 初始化是否成功
- func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool {
- print("[AzureAsrHelper] 初始化 Azure 语音服务")
-
- // 检查配置是否为空
- if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
- print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
- eventHandler("error", ["message": "Azure 配置信息不完整"])
- return false
- }
-
- // 释放之前的资源
- dispose()
-
- // 记录配置信息
- self.speechSubscriptionKey = speechSubscriptionKey
- self.serviceRegion = serviceRegion
-
- // 设置语言
- if let languages = supportedLanguages, !languages.isEmpty {
- self.supportedLanguages = languages
- }
-
- // 根据支持的语言数量决定是否启用自动语言检测
- isAutoDetectLanguage = self.supportedLanguages.count >= 2
-
- // 如果只有一种语言,设置为当前语言
- if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty {
- currentLanguage = self.supportedLanguages[0]
- }
-
- // 创建识别器和设置回调
- if !createRecognizerAndSetupCallbacks() {
- return false
- }
-
- print("[AzureAsrHelper] Azure 语音服务初始化成功")
- isInitialized = true
- return true
- }
-
- /// 创建识别器并设置回调
- private func createRecognizerAndSetupCallbacks() -> Bool {
- // 释放之前的 recognizer
- recognizer = nil
- audioConfig = nil
-
- do {
- // 创建语音配置
- speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
-
- // 设置音频输入参数
- try setupAudioSession()
-
- // 创建自定义推送流,替代默认的麦克风输入
- pushStream = try SPXPushAudioInputStream()
- audioConfig = try SPXAudioConfiguration(streamInput: pushStream!)
-
- // 初始化自定义音频处理器
- audioProcessor = CustomAudioProcessor()
-
- // 设置语言配置
- if isAutoDetectLanguage {
- // 设置自动语言检测
- speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode)
-
- // 创建自动语言检测配置
- let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages)
-
- // 创建识别器
- recognizer = try SPXSpeechRecognizer(
- speechConfiguration: speechConfig!,
- autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig,
- audioConfiguration: audioConfig!
- )
- } else {
- // 设置指定的识别语言
- speechConfig?.speechRecognitionLanguage = currentLanguage
-
- // 创建识别器
- recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
- }
-
- // 设置所有回调
- setupAllCallbacks()
-
- return true
- } catch {
- print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)")
- eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"])
- return false
- }
- }
-
- /// 设置音频会话
- private func setupAudioSession() throws {
- let audioSession = AVAudioSession.sharedInstance()
-
- // 使用playAndRecord类别允许同时录音和播放
- try audioSession.setCategory(.playAndRecord,
- mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除
- options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers])
-
- // 设置首选的输入和输出
- let currentRoute = audioSession.currentRoute
-
- // 获取当前是否连接了耳机或外部麦克风
- let hasHeadphones = currentRoute.outputs.contains {
- $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
- }
-
- // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除
- if !hasHeadphones {
- try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除
-
- // 启用回音消除和噪声抑制
- try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性
- } else {
- // 耳机模式,可以使用不同的设置
- try audioSession.setMode(.voiceChat)
- try audioSession.setInputGain(1.0)
- }
-
- // 设置合适的采样率
- try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率
- try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟
-
- // 激活音频会话
- try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
-
- print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除")
- }
-
- /// 设置所有回调
- private func setupAllCallbacks() {
- guard let recognizer = recognizer else { return }
-
- // 最终识别结果
- recognizer.addRecognizedEventHandler { [weak self] _, event in
- guard let self = self else { return }
-
- if event.result.reason == SPXResultReason.recognizedSpeech {
- let detectedLanguage = self.getDetectedLanguage(from: event.result)
- print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
- self.eventHandler("result", [
- "text": event.result.text ?? "",
- "detectedLanguage": detectedLanguage
- ])
- }
- }
-
- // 识别中事件
- recognizer.addRecognizingEventHandler { [weak self] _, event in
- guard let self = self else { return }
-
- if event.result.reason == SPXResultReason.recognizingSpeech {
- let detectedLanguage = self.getDetectedLanguage(from: event.result)
- // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
- self.eventHandler("recognizing", [
- "text": event.result.text ?? "",
- "detectedLanguage": detectedLanguage
- ])
- }
- }
-
- // 会话事件
- recognizer.addSessionStartedEventHandler { [weak self] _, _ in
- guard let self = self else { return }
-
- print("[AzureAsrHelper] 识别会话已开始")
- self._isContinuousRecognitionActive = true
- self.eventHandler("sessionStarted", [:])
- }
-
- recognizer.addSessionStoppedEventHandler { [weak self] _, _ in
- guard let self = self else { return }
-
- print("[AzureAsrHelper] 识别会话已结束")
- self._isContinuousRecognitionActive = false
- self.eventHandler("sessionStopped", [:])
- }
-
- // 取消事件
- recognizer.addCanceledEventHandler { [weak self] _, event in
- guard let self = self else { return }
-
- let reason = event.reason.rawValue
- let errorDetails = event.errorDetails ?? "未知错误"
-
- print("[AzureAsrHelper] 识别取消: \(errorDetails)")
-
- self.eventHandler("canceled", [
- "reason": reason,
- "errorDetails": errorDetails
- ])
-
- self._isContinuousRecognitionActive = false
- }
- }
-
- /// 执行一次性语音识别
- /// - Returns: 是否成功启动识别
- func recognizeOnce() -> Bool {
- if !isInitialized {
- print("[AzureAsrHelper] 错误: 语音服务未初始化")
- eventHandler("error", ["message": "语音服务未初始化"])
- return false
- }
-
- // 如果正在连续识别,先停止
- if _isContinuousRecognitionActive {
- stopContinuousRecognition()
- }
-
- // 确保识别器已创建
- if recognizer == nil && !createRecognizerAndSetupCallbacks() {
- return false
- }
-
- do {
- // 启动音频处理
- startAudioProcessing()
-
- // 通知会话开始
- eventHandler("sessionStarted", [:])
-
- // 执行识别
- try recognizer?.recognizeOnceAsync { [weak self] result in
- guard let self = self else { return }
-
- // 停止音频处理
- self.stopAudioProcessing()
-
- if result.reason == SPXResultReason.recognizedSpeech {
- let detectedLanguage = self.getDetectedLanguage(from: result)
- self.eventHandler("result", [
- "text": result.text ?? "",
- "detectedLanguage": detectedLanguage
- ])
- } else if result.reason == SPXResultReason.noMatch {
- print("[AzureAsrHelper] 无匹配结果")
- self.eventHandler("noMatch", [:])
- } else if result.reason == SPXResultReason.canceled {
- do {
- let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result)
- let errorDetails = details.errorDetails ?? "未知错误"
- self.eventHandler("error", ["message": "识别取消: \(errorDetails)"])
- } catch {
- print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)")
- self.eventHandler("error", ["message": "识别取消,无法获取详细原因"])
- }
- }
- }
-
- return true
- } catch {
- print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
- eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"])
- stopAudioProcessing()
- return false
- }
- }
-
- /// 开始连续语音识别
- /// - Returns: 是否成功启动识别
- func startContinuousRecognition() -> Bool {
- if !isInitialized {
- print("[AzureAsrHelper] 错误: 语音服务未初始化")
- eventHandler("error", ["message": "语音服务未初始化"])
- return false
- }
-
- // 如果已经在进行连续识别,先停止
- if _isContinuousRecognitionActive {
- stopContinuousRecognition()
- }
-
- // 确保识别器已创建
- if recognizer == nil && !createRecognizerAndSetupCallbacks() {
- return false
- }
-
- // 重新确保音频设置正确
- do {
- try setupAudioSession()
- } catch {
- print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)")
- }
-
- do {
- // 启动音频处理
- startAudioProcessing()
-
- // 启动连续识别
- try recognizer?.startContinuousRecognition()
- _isContinuousRecognitionActive = true
-
- print("[AzureAsrHelper] 连续识别开始")
- return true
- } catch {
- print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
- eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"])
- _isContinuousRecognitionActive = false
- stopAudioProcessing()
- return false
- }
- }
-
- /// 停止连续语音识别
- /// - Returns: 是否成功停止识别
- func stopContinuousRecognition() -> Bool {
- // 停止音频处理
- stopAudioProcessing()
-
- if !_isContinuousRecognitionActive || recognizer == nil {
- return true
- }
-
- do {
- try recognizer?.stopContinuousRecognition()
- _isContinuousRecognitionActive = false
- print("[AzureAsrHelper] 连续识别已停止")
- return true
- } catch {
- print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)")
- eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"])
- _isContinuousRecognitionActive = false
- return false
- }
- }
-
- /// 检查连续识别是否活跃
- /// - Returns: 连续识别是否处于活跃状态
- func isContinuousRecognitionActive() -> Bool {
- return _isContinuousRecognitionActive
- }
-
- /// 释放资源
- func dispose() {
- print("[AzureAsrHelper] 释放资源")
-
- // 停止音频处理
- stopAudioProcessing()
-
- // 停止连续识别
- if _isContinuousRecognitionActive {
- stopContinuousRecognition()
- }
-
- // 释放音频会话
- do {
- try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation)
- } catch {
- print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)")
- }
-
- // 释放资源
- recognizer = nil
- speechConfig = nil
- audioConfig = nil
- pushStream = nil
- audioProcessor = nil
-
- // 重置状态
- _isContinuousRecognitionActive = false
- isInitialized = false
- }
-
- /// 从结果中获取检测到的语言
- private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
- if isAutoDetectLanguage {
- do {
- let langResult = try SPXAutoDetectSourceLanguageResult(result)
- return langResult.language ?? currentLanguage
- } catch {
- print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)")
- return currentLanguage
- }
- } else {
- return currentLanguage
- }
- }
-
- // MARK: - 音频处理
-
- /// 开始音频处理
- private func startAudioProcessing() {
- guard !isProcessingAudio, let audioProcessor = audioProcessor else { return }
-
- isProcessingAudio = true
-
- // 启动音频处理器
- if !audioProcessor.startRecord() {
- print("[AzureAsrHelper] 错误: 启动音频处理器失败")
- eventHandler("error", ["message": "启动音频处理器失败"])
- return
- }
-
- // 启动音频处理定时器
- audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in
- guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else {
- return
- }
-
- // 读取处理后的音频数据
- var bytes = [UInt8](repeating: 0, count: 2560)
- let bytesRead = processor.read(bytes: &bytes)
-
- if bytesRead > 0 {
- // 推送数据到Azure语音服务
- let data = Data(bytes: bytes, count: bytesRead)
- stream.write(data)
-
- // 通知音频数据可用
- self.eventHandler("audioData", ["data": bytes])
- }
- }
-
- print("[AzureAsrHelper] 音频处理已启动")
- }
-
- /// 停止音频处理
- private func stopAudioProcessing() {
- // 停止定时器
- audioProcessingTimer?.invalidate()
- audioProcessingTimer = nil
-
- // 停止音频处理器
- audioProcessor?.stopRecord()
-
- isProcessingAudio = false
- print("[AzureAsrHelper] 音频处理已停止")
- }
-}
-
-// MARK: - 自定义音频处理器
-
-@available(iOS 13.0, *)
-class CustomAudioProcessor: NSObject {
- // 音频单元
- private var ioUnit: AudioUnit?
-
- // 音频格式
- private var audioFormat: AudioStreamBasicDescription
-
- // 音频缓冲
- private var audioBufferList: AudioBufferList
- private var audioList: [Float] = []
- private let audioListQueue = DispatchQueue(label: "audioListQueue")
-
- // 回音消除状态
- private var isEchoCancellationEnabled = true
-
- override init() {
- // 设置音频格式 - 16kHz, 16位, 单声道
- audioFormat = AudioStreamBasicDescription(
- mSampleRate: 16000.0,
- mFormatID: kAudioFormatLinearPCM,
- mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
- mBytesPerPacket: 2,
- mFramesPerPacket: 1,
- mBytesPerFrame: 2,
- mChannelsPerFrame: 1,
- mBitsPerChannel: 16,
- mReserved: 0
- )
-
- // 初始化音频缓冲
- audioBufferList = AudioBufferList(
- mNumberBuffers: 1,
- mBuffers: AudioBuffer(
- mNumberChannels: 1,
- mDataByteSize: 4096,
- mData: malloc(4096)
- )
- )
-
- super.init()
- }
-
- deinit {
- stopRecord()
- free(audioBufferList.mBuffers.mData)
- }
-
- /// 启动音频处理
- /// - Returns: 是否成功启动
- func startRecord() -> Bool {
- print("[CustomAudioProcessor] 配置音频单元")
-
- // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除
- var ioUnitDescription = AudioComponentDescription(
- componentType: kAudioUnitType_Output,
- componentSubType: kAudioUnitSubType_VoiceProcessingIO,
- componentManufacturer: kAudioUnitManufacturer_Apple,
- componentFlags: 0,
- componentFlagsMask: 0
- )
-
- // 查找音频组件
- guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
- print("[CustomAudioProcessor] 错误: 未找到音频组件")
- return false
- }
-
- // 创建音频单元实例
- if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") {
- ioUnit = nil
- return false
- }
-
- // 启用输入端口
- var enableInput: UInt32 = 1
- let kInputBus: AudioUnitElement = 1
- let kOutputBus: AudioUnitElement = 0
- if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
- kAudioUnitScope_Input, kInputBus, &enableInput,
- UInt32(MemoryLayout.size)), "启用输入端口") {
- return false
- }
-
- // 禁用输出端口 (我们只需要输入)
- var enableOutput: UInt32 = 0
- if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
- kAudioUnitScope_Output, kOutputBus,
- &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") {
- return false
- }
-
- // 设置缓冲区分配标志
- var flag: UInt32 = 0
- if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
- kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") {
- return false
- }
-
- // 设置音频格式
- let size = UInt32(MemoryLayout.size)
- if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
- kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") {
- return false
- }
-
- if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
- kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") {
- return false
- }
-
- // 启用回音消除
- if isEchoCancellationEnabled {
- var echoCancellation: UInt32 = 1
- AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing,
- kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size))
- }
-
- // 设置输入回调 - 当有新音频数据时调用
- var inputCallback = AURenderCallbackStruct(
- inputProc: CustomAudioProcessor.onAudioDataAvailable,
- inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
- )
-
- if checkError(AudioUnitSetProperty(ioUnit!,
- kAudioOutputUnitProperty_SetInputCallback,
- kAudioUnitScope_Global, kInputBus,
- &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") {
- return false
- }
-
- // 初始化音频单元
- var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
- while hasError {
- Thread.sleep(forTimeInterval: 0.1)
- hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
- }
-
- // 启动音频单元
- hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元")
-
- print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")")
- return !hasError
- }
-
- /// 停止音频处理
- func stopRecord() {
- print("[CustomAudioProcessor] 停止音频处理器")
-
- if let ioUnit = ioUnit {
- // 停止音频单元
- _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元")
-
- // 关闭音频单元
- _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元")
- _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元")
-
- self.ioUnit = nil
- }
-
- // 清空音频数据缓冲
- audioListQueue.sync {
- audioList.removeAll()
- }
- }
-
- /// 音频数据回调 - 当有新的音频数据可用时调用
- private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
- // 获取实例
- let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue()
-
- // 计算预期数据大小
- let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame
-
- // 确保缓冲区足够大
- if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
- processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
- processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
- }
-
- // 渲染音频数据
- let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp,
- inBusNumber, inNumberFrames, &processor.audioBufferList),
- "渲染音频数据")
-
- // 将Int16数据转换为浮点数据进行处理
- var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
- let buffer = processor.audioBufferList.mBuffers
- let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
-
- for j in 0...size)) {
- // 归一化到[-1.0, 1.0]范围
- audioDataFloat[j] = Float(bufferData[j]) / 32768.0
- }
-
- // 应用附加处理 (如有需要)
- // processor.applyAdditionalProcessing(&audioDataFloat)
-
- // 保存处理后的数据
- if status == noErr {
- processor.audioListQueue.async {
- processor.audioList.append(contentsOf: audioDataFloat)
- }
- }
-
- return status
- }
-
- /// 读取处理后的音频数据
- /// - Parameter bytes: 输出字节数组
- /// - Returns: 读取的字节数
- func read(bytes: inout [UInt8]) -> Int {
- return audioListQueue.sync {
- // 如果没有数据,返回0
- if audioList.isEmpty {
- return 0
- }
-
- // 确保有足够的数据 (至少1280个样本)
- if audioList.count < 1280 {
- return 0
- }
-
- // 读取一帧数据 (1280个样本)
- let frameLength = 1280
- let buffer = Array(audioList.prefix(frameLength))
- audioList.removeFirst(frameLength)
-
- // 将浮点数据转回Int16格式
- var int16Data = buffer.map { Int16($0 * 32767) }
-
- // 转换为字节数组
- let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
- bytes = [UInt8](data)
-
- // 每个样本2字节 (16位PCM)
- return frameLength * 2
- }
- }
-
- /// 检查错误并打印日志
- /// - Parameters:
- /// - status: 操作状态
- /// - operation: 操作描述
- /// - Returns: 是否发生错误
- private func checkError(_ status: OSStatus, _ operation: String) -> Bool {
- if status != noErr {
- print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
- return true
- }
- return false
- }
-
- /// 检查OSStatus并返回状态
- /// - Parameters:
- /// - status: 操作状态
- /// - operation: 操作描述
- /// - Returns: 原始状态
- private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
- if status != noErr {
- print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
- }
- return status
- }
-}
\ No newline at end of file
diff --git a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift
deleted file mode 100644
index 141b77ef8..000000000
--- a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift
+++ /dev/null
@@ -1,18 +0,0 @@
-import Flutter
-import UIKit
-
-public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
- public static func register(with registrar: FlutterPluginRegistrar) {
- if #available(iOS 13.0, *) {
- SwiftAzureSpeechRecognitionPlugin.register(with: registrar)
- } else {
- // 如果低于iOS 13.0,返回不支持的错误
- let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger())
- channel.setMethodCallHandler { (call, result) in
- result(FlutterError(code: "UNSUPPORTED",
- message: "需要iOS 13.0及以上系统",
- details: nil))
- }
- }
- }
-}
\ No newline at end of file
diff --git a/azure/ios/Classes/AzureTtsHelper.swift b/azure/ios/Classes/AzureTtsHelper.swift
deleted file mode 100644
index 589df617c..000000000
--- a/azure/ios/Classes/AzureTtsHelper.swift
+++ /dev/null
@@ -1,427 +0,0 @@
-import Foundation
-import MicrosoftCognitiveServicesSpeech
-import AVFoundation
-
-/// Azure TTS工具类,负责实现TTS服务接口
-@available(iOS 13.0, *)
-class AzureTtsHelper: NSObject {
- // MARK: - 属性
-
- /// 事件处理回调
- private var eventHandler: (String, [String: Any]) -> Void
-
- /// 语音配置信息
- private var speechSubscriptionKey: String = ""
- private var serviceRegion: String = ""
-
- /// 语音合成配置
- private var speechConfig: SPXSpeechConfiguration?
-
- /// 语音合成器
- private var synthesizer: SPXSpeechSynthesizer?
-
- /// 是否初始化成功
- private var isInitialized = false
-
- /// 当前是否正在播放
- private var _isSpeaking = false
-
- /// 音频会话配置
- private var isAudioSessionConfigured = false
-
- // MARK: - 语音设置
-
- /// 当前语音
- private var currentVoice = "zh-CN-XiaoxiaoNeural"
-
- /// 支持的语音映射
- private var voiceMap: [String: String] = [
- "zh-CN": "zh-CN-XiaoxiaoNeural",
- "en-US": "en-US-JennyNeural",
- "ja-JP": "ja-JP-NanamiNeural",
- "ko-KR": "ko-KR-SunHiNeural",
- "zh-TW": "zh-TW-HsiaoChenNeural",
- "zh-HK": "zh-HK-HiuMaanNeural"
- ]
-
- /// 当前语音合成参数
- private var currentSpeechRate = "0%"
- private var currentPitch = "0%"
- private var currentVolume = "100%"
-
- // MARK: - 初始化
-
- init(eventHandler: @escaping (String, [String: Any]) -> Void) {
- self.eventHandler = eventHandler
- super.init()
- }
-
- deinit {
- dispose()
- }
-
- // MARK: - TTS 接口实现
-
- /// 初始化语音合成服务
- /// - Parameters:
- /// - speechSubscriptionKey: Azure 语音服务订阅密钥
- /// - serviceRegion: Azure 服务区域 (如 eastasia)
- /// - language: 语言代码 (默认 zh-CN)
- /// - Returns: 初始化是否成功
- func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool {
- print("[AzureTtsHelper] 初始化语音合成服务")
-
- // 检查配置是否为空
- if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
- print("[AzureTtsHelper] 错误: Azure 配置信息不完整")
- eventHandler("error", ["error": "Azure 配置信息不完整"])
- return false
- }
-
- // 释放之前的资源
- dispose()
-
- // 记录配置信息
- self.speechSubscriptionKey = speechSubscriptionKey
- self.serviceRegion = serviceRegion
-
- // 配置音频会话
- if !configureAudioSession() {
- print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化")
- }
-
- do {
- // 创建语音配置
- speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
-
- // 设置默认语音
- let defaultVoice = getDefaultVoiceForLanguage(language)
- currentVoice = defaultVoice
- speechConfig?.speechSynthesisVoiceName = defaultVoice
-
- // 创建语音合成器
- synthesizer = try SPXSpeechSynthesizer(speechConfig!)
-
- // 设置事件处理器
- setupSynthesizerEvents()
-
- isInitialized = true
- print("[AzureTtsHelper] TTS 引擎初始化成功")
-
- return true
- } catch {
- print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)")
- eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"])
- return false
- }
- }
-
- /// 配置音频会话
- private func configureAudioSession() -> Bool {
- let audioSession = AVAudioSession.sharedInstance()
- do {
- // 使用playback类别,但支持混合和空中播放
- try audioSession.setCategory(.playback,
- mode: .spokenAudio,
- options: [.mixWithOthers, .allowAirPlay, .duckOthers])
-
- // 根据设备类型选择最佳配置
- let currentRoute = audioSession.currentRoute
- let hasHeadphones = currentRoute.outputs.contains {
- $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
- }
-
- // 优化音频路由
- if hasHeadphones {
- // 耳机模式,使用默认设置
- try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟
- } else {
- // 扬声器模式
- try audioSession.setPreferredIOBufferDuration(0.005)
- }
-
- // 避免完全激活音频会话,因为ASR可能已经激活
- // 这里使用setActive(false)是为了不与ASR冲突
- if !audioSession.isOtherAudioPlaying {
- try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
- }
-
- isAudioSessionConfigured = true
- print("[AzureTtsHelper] 音频会话配置成功")
- return true
- } catch {
- print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)")
- isAudioSessionConfigured = false
- return false
- }
- }
-
- /// 设置语音
- /// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural")
- /// - Returns: 设置是否成功
- func setVoice(voiceName: String) -> Bool {
- if !isInitialized {
- print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
- eventHandler("error", ["error": "TTS 引擎尚未初始化"])
- return false
- }
-
- if voiceName.isEmpty {
- print("[AzureTtsHelper] 错误: 声音名称为空")
- eventHandler("error", ["error": "声音名称不能为空"])
- return false
- }
-
- if voiceName == currentVoice {
- print("[AzureTtsHelper] 已设置语音: \(voiceName)")
- return true
- }
-
- print("[AzureTtsHelper] 设置声音: \(voiceName)")
- currentVoice = voiceName
-
- // 更新语音配置
- if let speechConfig = speechConfig {
- speechConfig.speechSynthesisVoiceName = voiceName
- return true
- }
-
- return false
- }
-
- /// 设置语音合成参数
- /// - Parameters:
- /// - rate: 语速,范围 -100 到 100,默认为 0
- /// - pitch: 音调,范围 -100 到 100,默认为 0
- /// - volume: 音量,范围 0 到 100,默认为 100
- /// - Returns: 是否设置成功
- func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
- if !isInitialized {
- print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
- eventHandler("error", ["error": "TTS 引擎尚未初始化"])
- return false
- }
-
- // 转换参数格式
- currentSpeechRate = formatRateParam(rate)
- currentPitch = formatPitchParam(pitch)
- currentVolume = formatVolumeParam(volume)
-
- print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)")
- return true
- }
-
- /// 合成文本为语音并播放
- /// - Parameter text: 要合成的文本
- /// - Returns: 操作是否成功启动
- func speakText(text: String) -> Bool {
- if !isInitialized {
- print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
- eventHandler("error", ["error": "TTS 引擎尚未初始化"])
- return false
- }
-
- if text.isEmpty {
- print("[AzureTtsHelper] 警告: 要播放的文本为空")
- return true
- }
-
- // 确保音频会话已配置
- if !isAudioSessionConfigured {
- _ = configureAudioSession()
- }
-
- print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...")
-
- // 生成SSML
- let ssml = generateSsml(text: text)
-
- // 直接进行SSML合成
- return speakSsmlInternal(text: ssml)
- }
-
- /// 内部SSML合成和播放
- private func speakSsmlInternal(text: String) -> Bool {
- guard let synthesizer = synthesizer else {
- print("[AzureTtsHelper] 错误: 合成器未初始化")
- eventHandler("error", ["error": "合成器未初始化"])
- return false
- }
-
- _isSpeaking = true
- eventHandler("started", [:])
-
- Task {
- do {
- // 使用异步方法进行合成并直接播放
- _ = try await synthesizer.startSpeakingSsml(text)
-
- } catch {
- print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)")
- DispatchQueue.main.async {
- self._isSpeaking = false
- self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"])
- }
- }
- }
-
- return true
- }
-
- /// 停止当前语音合成
- /// - Returns: 操作是否成功
- func stopSpeaking() -> Bool {
- if !isInitialized || !_isSpeaking {
- return true
- }
-
- // 停止合成
- do {
- try synthesizer?.stopSpeaking()
- _isSpeaking = false
- eventHandler("canceled", [:])
- print("[AzureTtsHelper] 已停止语音合成")
- return true
- } catch {
- print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)")
- eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"])
- return false
- }
- }
-
- /// 检查是否正在播放
- /// - Returns: 当前是否正在播放语音
- func isSpeaking() -> Bool {
- return _isSpeaking
- }
-
- /// 释放资源
- func dispose() {
- try? stopSpeaking()
-
- // 释放合成器和配置
- synthesizer = nil
- speechConfig = nil
-
- isInitialized = false
- _isSpeaking = false
- isAudioSessionConfigured = false
- print("[AzureTtsHelper] TTS 引擎已释放")
- }
-
- // MARK: - 私有辅助方法
-
- /// 设置合成器事件处理
- private func setupSynthesizerEvents() {
- guard let synthesizer = synthesizer else { return }
-
- // 添加书签到达事件处理
- synthesizer.addBookmarkReachedEventHandler { _, e in
- print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"")
- }
-
- // 合成完成事件
- synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in
- guard let self = self else { return }
- print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)")
- DispatchQueue.main.async {
- self._isSpeaking = false
- self.eventHandler("completed", [:])
- }
- }
-
- // 合成取消事件
- synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in
- guard let self = self else { return }
-
- let result = e.result
- do {
- let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result)
- print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)")
-
- if cancellationDetails.reason == SPXCancellationReason.error {
- print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)")
- print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")")
- }
-
- DispatchQueue.main.async {
- self._isSpeaking = false
- self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"])
- }
- } catch {
- print("[AzureTtsHelper] 获取取消详情时出错: \(error)")
-
- DispatchQueue.main.async {
- self._isSpeaking = false
- self.eventHandler("error", ["error": "语音合成被取消"])
- }
- }
- }
-
- // 合成开始事件
- synthesizer.addSynthesisStartedEventHandler { _, _ in
- // print("[AzureTtsHelper] 语音合成开始")
- }
-
- // 合成中事件
- synthesizer.addSynthesizingEventHandler { _, _ in
- // print("[AzureTtsHelper] 语音合成中")
- }
- }
-
- /// 生成 SSML 文本
- private func generateSsml(text: String) -> String {
- return """
-
-
-
- \(text)
-
-
-
- """
- }
-
- /// 格式化语速参数
- private func formatRateParam(_ rate: Int) -> String {
- let clampedRate = rate.clamp(min: -100, max: 100)
- if clampedRate == 0 {
- return "0%"
- } else if clampedRate < 0 {
- return "\(Int(Double(clampedRate) * 0.9))%"
- } else {
- return "+\(clampedRate)%"
- }
- }
-
- /// 格式化音调参数
- private func formatPitchParam(_ pitch: Int) -> String {
- let clampedPitch = pitch.clamp(min: -100, max: 100)
- if clampedPitch == 0 {
- return "0%"
- } else {
- return "\(Int(Double(clampedPitch) * 0.5))%"
- }
- }
-
- /// 格式化音量参数
- private func formatVolumeParam(_ volume: Int) -> String {
- let clampedVolume = volume.clamp(min: 0, max: 100)
- return "\(clampedVolume)%"
- }
-
- /// 获取指定语言的默认语音
- private func getDefaultVoiceForLanguage(_ language: String) -> String {
- return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural"
- }
-}
-
-// MARK: - 扩展
-
-extension Int {
- func clamp(min: Int, max: Int) -> Int {
- if self < min { return min }
- if self > max { return max }
- return self
- }
-}
\ No newline at end of file
diff --git a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift
deleted file mode 100644
index ea393e215..000000000
--- a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift
+++ /dev/null
@@ -1,259 +0,0 @@
-import Flutter
-import UIKit
-import MicrosoftCognitiveServicesSpeech
-import AVFoundation
-
-@available(iOS 13.0, *)
-public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
- private var azureChannel: FlutterMethodChannel
- private var ttsChannel: FlutterMethodChannel
- private var asrHelper: AzureAsrHelper
- private var ttsHelper: AzureTtsHelper
- private static var eventStreamHandler: AzureEventStreamHandler?
-
- // 创建方法到通道的映射
- private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]()
- private static var asrMethodHandlers = [String: FlutterMethodCallHandler]()
-
- public static func register(with registrar: FlutterPluginRegistrar) {
- // ASR通道
- let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger())
-
- // TTS通道
- let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger())
-
- // 设置ASR事件通道
- let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger())
- eventStreamHandler = AzureEventStreamHandler()
- eventChannel.setStreamHandler(eventStreamHandler)
-
- let instance = SwiftAzureSpeechRecognitionPlugin(
- azureChannel: channel,
- ttsChannel: ttsChannel,
- eventStreamHandler: eventStreamHandler!
- )
-
- // 直接设置各自通道的处理器
- channel.setMethodCallHandler(instance.handleAsrMethodCalls)
- ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls)
- }
-
-
-
- // 新增直接处理方法调用的函数
- private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
- handleTtsMethod(call, result)
- }
-
- private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
- handleAsrMethod(call, result)
- }
-
-
- init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) {
- self.azureChannel = azureChannel
- self.ttsChannel = ttsChannel
-
- // 创建辅助类实例,使用自定义事件回调处理器
- let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in
- DispatchQueue.main.async {
- if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink {
- var eventData = arguments
- eventData["type"] = eventName
- eventSink(eventData)
- }
- }
- }
-
- asrHelper = AzureAsrHelper(eventHandler: eventHandler)
- ttsHelper = AzureTtsHelper(eventHandler: eventHandler)
-
- super.init()
- }
-
- private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
-
- let args = call.arguments as? Dictionary
-
- switch call.method {
- case "initialize":
- // 仅在初始化时读取必要参数
- guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else {
- let errorMsg = "语音订阅密钥不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil))
- return
- }
-
- guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else {
- let errorMsg = "服务区域不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil))
- return
- }
-
- let supportedLanguages = args?["supportedLanguages"] as? [String] ?? []
-
- let success = asrHelper.initialize(
- speechSubscriptionKey: speechSubscriptionKey,
- serviceRegion: serviceRegion,
- supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages
- )
- result(success)
-
- case "startContinuousRecognition":
- // 只有使用参数时才验证
- let success = asrHelper.startContinuousRecognition()
- result(success)
-
- case "stopContinuousRecognition":
- // 不需要额外参数
- let success = asrHelper.stopContinuousRecognition()
- result(success)
-
- case "recognizeOnce":
- // 只有使用参数时才验证
- let success = asrHelper.recognizeOnce()
- result(success)
-
- case "isContinuousRecognitionActive":
- // 不需要额外参数
- result(asrHelper.isContinuousRecognitionActive())
-
- case "dispose":
- // 不需要额外参数
- print("[AzurePlugin] 释放ASR资源")
- asrHelper.dispose()
- result(true)
-
- default:
- print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)")
- result(FlutterMethodNotImplemented)
- }
- }
-
- private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
-
- let args = call.arguments as? Dictionary
-
- switch call.method {
- case "initialize":
- // 仅在初始化时验证参数
- guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else {
- let errorMsg = "语音订阅密钥不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil))
- return
- }
-
- guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else {
- let errorMsg = "服务区域不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil))
- return
- }
-
- let language = args?["language"] as? String ?? "zh-CN"
-
- print("[AzurePlugin] 初始化TTS,语言: \(language)")
-
- let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language)
- result(success)
-
- case "setVoice":
- // 仅获取voice参数
- guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else {
- let errorMsg = "声音名称不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil))
- return
- }
-
- print("[AzurePlugin] 设置声音: \(voiceName)")
-
- let success = ttsHelper.setVoice(voiceName: voiceName)
- result(success)
-
- case "speakText":
- // 仅获取text参数
- let text = args?["text"] as? String ?? ""
-
- if text.isEmpty {
- print("[AzurePlugin] 警告: 要播放的文本为空")
- result("OK")
- return
- }
-
- print("[AzurePlugin] 播放文本: \(text.prefix(50))...")
-
- let success = ttsHelper.speakText(text: text)
- result(success ? "OK" : "ERROR")
-
- case "speakSsml":
- // 仅获取ssml参数
- guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else {
- let errorMsg = "SSML内容不能为空"
- print("[AzurePlugin] 错误: \(errorMsg)")
- result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil))
- return
- }
-
- print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...")
-
- // 由于我们移除了speakSsml方法,这里改用speakText方法
- // Azure SDK内部会自动检测是普通文本还是SSML
- let success = ttsHelper.speakText(text: ssml)
- result(success)
-
- case "stopSpeaking":
- // 不需要参数
- print("[AzurePlugin] 停止播放")
- let success = ttsHelper.stopSpeaking()
- result(success)
-
- case "isSpeaking":
- // 不需要参数
- result(ttsHelper.isSpeaking())
-
- case "setSpeechParams":
- // 仅获取语音参数
- let rate = args?["rate"] as? Int ?? 0
- let pitch = args?["pitch"] as? Int ?? 0
- let volume = args?["volume"] as? Int ?? 100
-
- print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)")
- let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume)
- result(success)
-
- case "dispose":
- // 释放TTS资源
- print("[AzurePlugin] 释放TTS资源")
- ttsHelper.dispose()
- result(true)
-
- default:
- print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)")
- result(FlutterMethodNotImplemented)
- }
- }
-}
-
-// 用于处理事件流的辅助类
-@available(iOS 13.0, *)
-class AzureEventStreamHandler: NSObject, FlutterStreamHandler {
- var eventSink: FlutterEventSink?
-
- func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? {
- self.eventSink = events
- // 通知Flutter端事件通道已准备好
- DispatchQueue.main.async {
- events(["type": "channelReady"])
- }
- return nil
- }
-
- func onCancel(withArguments arguments: Any?) -> FlutterError? {
- self.eventSink = nil
- return nil
- }
-}
\ No newline at end of file
diff --git a/azure/ios/azure_speech_recognition.podspec b/azure/ios/azure_speech_recognition.podspec
deleted file mode 100644
index 88797fb0c..000000000
--- a/azure/ios/azure_speech_recognition.podspec
+++ /dev/null
@@ -1,24 +0,0 @@
-#
-# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html.
-# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing.
-#
-Pod::Spec.new do |s|
- s.name = 'azure_speech_recognition'
- s.version = '0.1.0'
- s.summary = 'Azure Speech Recognition plugin for Flutter'
- s.description = <<-DESC
-A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities.
- DESC
- s.homepage = 'https://github.com/yourusername/azure_speech_recognition'
- s.license = { :type => 'MIT', :file => '../LICENSE' }
- s.author = { 'Your Company' => 'your-email@example.com' }
- s.source = { :path => '.' }
- s.source_files = 'Classes/**/*'
- s.dependency 'Flutter'
- s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0'
- s.platform = :ios, '12.0'
-
- # Flutter.framework does not contain a i386 slice.
- s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' }
- s.swift_version = '5.0'
-end
\ No newline at end of file
diff --git a/azure/lib/azure_speech_recognition.dart b/azure/lib/azure_speech_recognition.dart
deleted file mode 100644
index 02b3f92d8..000000000
--- a/azure/lib/azure_speech_recognition.dart
+++ /dev/null
@@ -1,6 +0,0 @@
-// This is a placeholder file that exports nothing.
-// The actual implementation is in the app's services folder.
-// This file exists just to satisfy the Flutter plugin structure requirements.
-
-// Empty library to satisfy plugin structure
-library azure_speech_recognition;
\ No newline at end of file
diff --git a/azure/pubspec.yaml b/azure/pubspec.yaml
deleted file mode 100644
index a36c77bdc..000000000
--- a/azure/pubspec.yaml
+++ /dev/null
@@ -1,23 +0,0 @@
-name: azure_speech_recognition
-description: Azure Speech Recognition and Text-to-Speech services Flutter plugin
-version: 0.1.0
-homepage: https://github.com/yourusername/azure_speech_recognition
-
-environment:
- sdk: '>=2.12.0 <3.0.0'
- flutter: ">=2.0.0"
-
-dependencies:
- flutter:
- sdk: flutter
-
-dev_dependencies:
- flutter_test:
- sdk: flutter
- flutter_lints: ^1.0.0
-
-flutter:
- plugin:
- platforms:
- ios:
- pluginClass: AzureSpeechRecognitionPlugin
\ No newline at end of file
diff --git a/lib/core/bindings/initial_binding.dart b/lib/core/bindings/initial_binding.dart
index 3f42d9560..748589f63 100644
--- a/lib/core/bindings/initial_binding.dart
+++ b/lib/core/bindings/initial_binding.dart
@@ -8,6 +8,7 @@ import '../../data/services/classic_bluetooth_service.dart';
import '../../data/services/bluetooth_media_button_service.dart';
import '../../core/utils/logger.dart';
import '../../data/services/voice_interaction_service.dart';
+import '../../data/services/open_ai_service_adapter.dart';
/// 初始绑定,用于管理全局依赖
class InitialBinding extends Bindings {
@@ -29,6 +30,13 @@ class InitialBinding extends Bindings {
Get.lazyPut(() => VolcanoTranslationService(),
fenix: true);
+ // OpenAI服务适配器
+ Get.lazyPut(() {
+ final adapter = OpenAIServiceAdapter();
+ adapter.initialize();
+ return adapter;
+ }, fenix: true);
+
// 火山AI服务
Get.lazyPut(() => VolcanoAIService(), fenix: true);
@@ -41,6 +49,7 @@ class InitialBinding extends Bindings {
() => BluetoothMediaButtonService(),
fenix: true);
+ // 语音交互服务
Get.lazyPut(() => VoiceInteractionService(),
fenix: true);
diff --git a/lib/data/models/events/voice_interaction_event.dart b/lib/data/models/events/voice_interaction_event.dart
index 4bfa636df..81860de23 100644
--- a/lib/data/models/events/voice_interaction_event.dart
+++ b/lib/data/models/events/voice_interaction_event.dart
@@ -26,4 +26,17 @@ class RecognitionStartedEvent extends VoiceInteractionEvent {
RecognitionStartedEvent({
required int timestamp,
}) : super(timestamp: timestamp);
+}
+
+/// 通用语音交互事件
+/// 用于处理其他类型的事件
+class GenericVoiceInteractionEvent extends VoiceInteractionEvent {
+ final String type;
+ final Map? data;
+
+ GenericVoiceInteractionEvent({
+ required this.type,
+ this.data,
+ required int timestamp,
+ }) : super(timestamp: timestamp);
}
\ No newline at end of file
diff --git a/lib/data/services/ai_service.dart b/lib/data/services/ai_service.dart
index 949dfca67..a2232b9b5 100644
--- a/lib/data/services/ai_service.dart
+++ b/lib/data/services/ai_service.dart
@@ -1,5 +1,7 @@
/// AI回复服务接口
abstract class AiService {
+
+
/// 非流式输出方法
Future sendMessage({
required List