From fd5478dbf608b5d304743cc556fe4d70f44b922a Mon Sep 17 00:00:00 2001 From: wolfplus Date: Sat, 5 Apr 2025 13:29:57 +0100 Subject: [PATCH] init --- android/app/build.gradle.kts | 10 + .../com/example/deep_voice/AzureAsrHelper.kt | 654 --------------- .../example/deep_voice/VolcanoAIService.kt | 344 -------- .../deepsound}/ClassicBluetoothHelper.kt | 0 .../deepsound}/MainActivity.kt | 255 +----- .../deepsound/OpenAIService.kt | 606 ++++++++++++++ .../deepsound/VoiceFunctionHandler.kt | 180 +++++ .../deepsound}/VoiceInteractionService.kt | 210 +++-- .../deepsound}/core/utils/FileLogger.kt | 0 android/settings.gradle.kts | 4 + azure/LICENSE | 21 - azure/README.md | 66 -- azure/ios/Classes/AzureAsrHelper.swift | 757 ------------------ .../AzureSpeechRecognitionPlugin.swift | 18 - azure/ios/Classes/AzureTtsHelper.swift | 427 ---------- .../SwiftAzureSpeechRecognitionPlugin.swift | 259 ------ azure/ios/azure_speech_recognition.podspec | 24 - azure/lib/azure_speech_recognition.dart | 6 - azure/pubspec.yaml | 23 - .../speech_impl/azure_asr_service.dart | 4 +- .../speech_impl/azure_tts_service.dart | 2 +- .../services/voice_interaction_service.dart | 12 +- local_plugins/azure_speech/README.md | 136 ++++ .../azure_speech/android/build.gradle.kts | 65 ++ .../azure_speech/android/settings.gradle.kts | 1 + .../android/src/main/AndroidManifest.xml | 6 + .../azure_speech/AzureAsrHelper.kt | 593 ++++++++++++++ .../azure_speech/AzureSpeechPlugin.kt | 297 +++++++ .../azure_speech}/AzureTtsHelper.kt | 12 +- .../azure_speech/utils/FileLogger.kt | 45 ++ .../ios/Classes/AzureAsrHelper.swift | 461 +++++++++++ .../ios/Classes/AzureSpeechPlugin.swift | 182 +++++ .../ios/Classes/AzureTtsHelper.swift | 330 ++++++++ local_plugins/azure_speech/pubspec.yaml | 32 + pubspec.yaml | 2 + 35 files changed, 3148 insertions(+), 2896 deletions(-) delete mode 100644 android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt delete mode 100644 android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt rename android/app/src/main/kotlin/com/{example/deep_voice => yunqiinnovation/deepsound}/ClassicBluetoothHelper.kt (100%) rename android/app/src/main/kotlin/com/{example/deep_voice => yunqiinnovation/deepsound}/MainActivity.kt (67%) create mode 100644 android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt create mode 100644 android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt rename android/app/src/main/kotlin/com/{example/deep_voice => yunqiinnovation/deepsound}/VoiceInteractionService.kt (78%) rename android/app/src/main/kotlin/com/{example/deep_voice => yunqiinnovation/deepsound}/core/utils/FileLogger.kt (100%) delete mode 100644 azure/LICENSE delete mode 100644 azure/README.md delete mode 100644 azure/ios/Classes/AzureAsrHelper.swift delete mode 100644 azure/ios/Classes/AzureSpeechRecognitionPlugin.swift delete mode 100644 azure/ios/Classes/AzureTtsHelper.swift delete mode 100644 azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift delete mode 100644 azure/ios/azure_speech_recognition.podspec delete mode 100644 azure/lib/azure_speech_recognition.dart delete mode 100644 azure/pubspec.yaml create mode 100644 local_plugins/azure_speech/README.md create mode 100644 local_plugins/azure_speech/android/build.gradle.kts create mode 100644 local_plugins/azure_speech/android/settings.gradle.kts create mode 100644 local_plugins/azure_speech/android/src/main/AndroidManifest.xml create mode 100644 local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt create mode 100644 local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt rename {android/app/src/main/kotlin/com/example/deep_voice => local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech}/AzureTtsHelper.kt (98%) create mode 100644 local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt create mode 100644 local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift create mode 100644 local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift create mode 100644 local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift create mode 100644 local_plugins/azure_speech/pubspec.yaml diff --git a/android/app/build.gradle.kts b/android/app/build.gradle.kts index 542cf24d8..78b816b75 100644 --- a/android/app/build.gradle.kts +++ b/android/app/build.gradle.kts @@ -34,6 +34,11 @@ android { jvmTarget = JavaVersion.VERSION_11.toString() } + // 添加lint选项 + lintOptions { + isCheckReleaseBuilds = false + } + defaultConfig { // TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html). applicationId = "com.yunqiinnovation.deepsound" @@ -82,6 +87,11 @@ android { dependencies { + implementation("io.modelcontextprotocol:kotlin-sdk:0.4.0") + + + // 添加本地插件模块依赖 + implementation(project(":azure_speech")) // 添加OkHttp依赖 implementation("com.squareup.okhttp3:okhttp:4.9.3") diff --git a/android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt b/android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt deleted file mode 100644 index 171adf80e..000000000 --- a/android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt +++ /dev/null @@ -1,654 +0,0 @@ -package com.yunqiinnovation.deepsound - -import android.content.Context -import android.media.AudioAttributes -import android.media.AudioFormat -import android.media.AudioRecord -import android.media.MediaRecorder -import android.media.audiofx.AcousticEchoCanceler -import android.media.audiofx.NoiseSuppressor -import android.media.audiofx.AutomaticGainControl -import android.os.Process -import com.yunqiinnovation.deepsound.core.utils.FileLogger -import com.microsoft.cognitiveservices.speech.* -import com.microsoft.cognitiveservices.speech.audio.* -import com.microsoft.cognitiveservices.speech.util.EventHandler -import java.util.concurrent.ExecutionException -import java.util.concurrent.atomic.AtomicBoolean -// 移除WebRTC相关导入 - -class AzureAsrHelper(private val context: Context) { - private var recognizer: SpeechRecognizer? = null - private var speechConfig: SpeechConfig? = null - private val TAG = "AzureAsrHelper" - private var isContinuousRecognitionActive = false - private var currentLanguage = "zh-CN" - private var subscriptionKey = "" - private var serviceRegion = "" - private var isAutoDetectLanguage = false - private var supportedLanguages = arrayOf("zh-CN", "en-US") - - // 是否使用回音消除 - 内部控制常量 - private val useEchoCancellation = true - - // 自定义音频处理相关 - private var customAudioProcessor: CustomAudioProcessor? = null - private var pushStream: PushAudioInputStream? = null - private var audioConfig: AudioConfig? = null - - // 初始化SDK并创建recognizer - fun initialize(subscriptionKey: String, serviceRegion: String, - supportedLanguages: Array = arrayOf("zh-CN", "en-US")): Boolean { - try { - FileLogger.d(TAG, "初始化 Azure 语音服务") - - // 检查配置是否为空 - if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) { - FileLogger.e(TAG, "Azure 配置信息不完整") - return false - } - - // 释放之前的资源 - dispose() - - this.subscriptionKey = subscriptionKey - this.serviceRegion = serviceRegion - - // 设置语言 - if (supportedLanguages.isNotEmpty()) { - this.supportedLanguages = supportedLanguages - } - - // 根据支持的语言数量决定是否启用自动语言检测 - this.isAutoDetectLanguage = supportedLanguages.size >= 2 - - // 如果只有一种语言,设置为当前语言 - if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { - this.currentLanguage = supportedLanguages[0] - } - - // 创建语音配置 - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) - - // 设置语言配置 - if (isAutoDetectLanguage) { - // 设置自动语言检测 - speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") - } else { - // 设置指定的识别语言 - speechConfig?.speechRecognitionLanguage = currentLanguage - } - - // 创建识别器 - try { - if (useEchoCancellation) { - // 如果使用回音消除,创建自定义音频输入流 - setupCustomAudioProcessing() - - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) - } else { - recognizer = SpeechRecognizer(speechConfig, audioConfig) - } - } else { - // 使用默认麦克风输入 - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) - } else { - recognizer = SpeechRecognizer(speechConfig) - } - } - - FileLogger.d(TAG, "Azure 语音服务初始化成功") - return true - } catch (e: Exception) { - FileLogger.e(TAG, "创建识别器失败: ${e.message}") - stopCustomAudioProcessing() - return false - } - } catch (e: Exception) { - FileLogger.e(TAG, "初始化失败: ${e.message}") - return false - } - } - - // 重置 recognizer - private fun resetRecognizer(): Boolean { - try { - // 释放之前的 recognizer - recognizer?.close() - recognizer = null - - // 停止当前的音频处理 - stopCustomAudioProcessing() - - // 使用现有配置重新创建 recognizer - if (speechConfig != null) { - if (useEchoCancellation) { - // 如果使用回音消除,创建自定义音频输入流 - setupCustomAudioProcessing() - - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) - } else { - recognizer = SpeechRecognizer(speechConfig, audioConfig) - } - } else { - // 使用默认麦克风输入 - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) - } else { - recognizer = SpeechRecognizer(speechConfig) - } - } - return true - } else { - FileLogger.e(TAG, "语音配置未初始化") - return false - } - } catch (e: Exception) { - FileLogger.e(TAG, "重置识别器失败: ${e.message}") - return false - } - } - - // 开始一次性语音识别 - fun recognizeOnce(callback: RecognizeCallback) { - if (speechConfig == null) { - callback.onError("语音服务未初始化") - return - } - - // 重置 recognizer - if (!resetRecognizer()) { - callback.onError("重置识别器失败") - return - } - - try { - // 启动音频处理 - startCustomAudioProcessing() - - // 执行识别 - val result = recognizer?.recognizeOnceAsync()?.get() - - // 停止音频处理 - stopCustomAudioProcessing() - - if (result != null && result.reason == ResultReason.RecognizedSpeech) { - val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language - callback.onResult(result.text, detectedLanguage ?: "") - } else { - callback.onError("未能识别语音") - } - } catch (e: Exception) { - // 停止音频处理 - stopCustomAudioProcessing() - callback.onError("识别异常: ${e.message}") - } - } - - // 开始连续语音识别 - fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { - if (speechConfig == null) { - callback.onError("语音服务未初始化") - return false - } - - // 如果已经在进行连续识别,先停止 - if (isContinuousRecognitionActive) { - stopContinuousRecognition(callback) - } - - // 重置 recognizer - if (!resetRecognizer()) { - callback.onError("重置识别器失败") - return false - } - - try { - // 启动音频处理 - startCustomAudioProcessing() - - // 设置识别事件处理 - // 最终识别结果 - recognizer?.recognized?.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizedSpeech) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" - } else { - currentLanguage - } - // FileLogger.d(TAG, "最终识别结果: ${event.result.text}") - callback.onResult(event.result.text, detectedLanguage) - } - } - ) - - // 识别中事件 - recognizer?.recognizing?.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizingSpeech) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" - } else { - currentLanguage - } - // FileLogger.d(TAG, "识别中结果: ${event.result.text}") - callback.onRecognizing(event.result.text, detectedLanguage) - } - } - ) - - // 会话事件 - recognizer?.sessionStarted?.addEventListener( - EventHandler { _, _ -> - isContinuousRecognitionActive = true - callback.onSessionStarted() - } - ) - - recognizer?.sessionStopped?.addEventListener( - EventHandler { _, _ -> - isContinuousRecognitionActive = false - callback.onSessionStopped() - } - ) - - // 取消事件 - recognizer?.canceled?.addEventListener( - EventHandler { _, event -> - val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else "" - callback.onCanceled(event.reason.toString(), errorDetails) - isContinuousRecognitionActive = false - } - ) - - // 开始连续识别 - recognizer?.startContinuousRecognitionAsync()?.get() - isContinuousRecognitionActive = true - - return true - } catch (e: Exception) { - callback.onError("开始连续识别失败: ${e.message}") - isContinuousRecognitionActive = false - stopCustomAudioProcessing() - return false - } - } - - // 停止连续语音识别 - fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { - if (!isContinuousRecognitionActive || recognizer == null) { - return true - } - - try { - recognizer?.stopContinuousRecognitionAsync() - isContinuousRecognitionActive = false - callback.onSessionStopped() - - // 停止音频处理 - stopCustomAudioProcessing() - - return true - } catch (e: Exception) { - callback.onError("停止连续识别失败: ${e.message}") - isContinuousRecognitionActive = false - stopCustomAudioProcessing() - return false - } - } - - // 检查连续识别是否活跃 - fun isContinuousRecognitionActive(): Boolean { - return isContinuousRecognitionActive - } - - // 设置自定义音频处理 - private fun setupCustomAudioProcessing() { - if (!useEchoCancellation) { - return - } - - try { - // 1. 创建PushAudioInputStream - pushStream = PushAudioInputStream.create() - - // 2. 创建AudioConfig - audioConfig = AudioConfig.fromStreamInput(pushStream) - - // 3. 创建自定义音频处理器 - customAudioProcessor = CustomAudioProcessor(pushStream) - - FileLogger.d(TAG, "自定义音频处理设置完成") - } catch (e: Exception) { - FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") - releaseCustomAudioProcessing() - - // 降级处理:如果自定义处理设置失败,尝试使用默认麦克风 - try { - FileLogger.d(TAG, "尝试降级到默认麦克风输入") - audioConfig = AudioConfig.fromDefaultMicrophoneInput() - } catch (e2: Exception) { - FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}") - audioConfig = null - } - } - } - - // 启动自定义音频处理 - private fun startCustomAudioProcessing() { - if (!useEchoCancellation || customAudioProcessor == null) { - return - } - - try { - customAudioProcessor?.startRecording() - FileLogger.d(TAG, "自定义音频处理已启动") - } catch (e: Exception) { - FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}") - } - } - - // 停止自定义音频处理 - private fun stopCustomAudioProcessing() { - if (!useEchoCancellation || customAudioProcessor == null) { - return - } - - try { - customAudioProcessor?.stopRecording() - FileLogger.d(TAG, "自定义音频处理已停止") - } catch (e: Exception) { - FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}") - } - } - - // 释放自定义音频处理资源 - private fun releaseCustomAudioProcessing() { - stopCustomAudioProcessing() - - try { - customAudioProcessor = null - pushStream?.close() - pushStream = null - audioConfig?.close() - audioConfig = null - - FileLogger.d(TAG, "自定义音频处理资源已释放") - } catch (e: Exception) { - FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}") - } - } - - // 释放所有资源 - fun dispose() { - try { - // 停止和释放音频处理 - releaseCustomAudioProcessing() - - recognizer?.close() - recognizer = null - - speechConfig?.close() - speechConfig = null - - isContinuousRecognitionActive = false - - FileLogger.d(TAG, "语音识别资源已释放") - } catch (e: Exception) { - FileLogger.e(TAG, "释放资源出错: ${e.message}") - } - } - - // 自定义音频处理器 - 使用Android原生回音消除 - private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) { - private val SAMPLE_RATE = 16000 - private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO - private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT - private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定 - - private var audioRecord: AudioRecord? = null - private var echoCanceler: AcousticEchoCanceler? = null - private val isRecording = AtomicBoolean(false) - private var recordingThread: Thread? = null - - // 启动录音并处理音频数据 - fun startRecording() { - if (isRecording.get() || pushStream == null) { - return - } - - try { - // 使用Builder模式构建AudioFormat - val audioFormat = AudioFormat.Builder() - .setSampleRate(SAMPLE_RATE) - .setEncoding(AUDIO_FORMAT) - .setChannelMask(CHANNEL_CONFIG) - .build() - - // 使用Builder模式创建AudioRecord实例 - audioRecord = AudioRecord.Builder() - .setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION) - .setAudioFormat(audioFormat) - .setBufferSizeInBytes(BUFFER_SIZE) - .build() - - // 检查AudioRecord初始化状态 - if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { - FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}") - // 尝试使用DEFAULT音频源重试一次 - audioRecord?.release() - audioRecord = AudioRecord.Builder() - .setAudioSource(MediaRecorder.AudioSource.DEFAULT) - .setAudioFormat(audioFormat) - .setBufferSizeInBytes(BUFFER_SIZE) - .build() - - if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { - FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃") - releaseAudioResources() - return - } else { - FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord") - } - } - - // 启用音频效果(回音消除、噪声抑制等) - enableAudioEffects() - - // 启动录音 - audioRecord?.startRecording() - isRecording.set(true) - - // 创建录音线程 - recordingThread = Thread({ - val buffer = ByteArray(BUFFER_SIZE) - - while (isRecording.get()) { - try { - val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0 - - if (readSize > 0) { - try { - // 将处理后的音频数据推送到流 - if (readSize == buffer.size) { - // 如果读取的大小等于buffer的大小,直接写入整个buffer - pushStream.write(buffer) - } else { - // 如果只读取了部分数据,创建新的数组只包含有效数据 - val validData = buffer.copyOfRange(0, readSize) - pushStream.write(validData) - } - } catch (e: Exception) { - FileLogger.e(TAG, "写入音频数据失败: ${e.message}") - break - } - } else if (readSize == 0) { - // 读取为0,可能是临时的,等待一下继续尝试 - Thread.sleep(10) - } else { - // 负值表示错误 - FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize") - break - } - } catch (e: Exception) { - FileLogger.e(TAG, "录音线程异常: ${e.message}") - break - } - } - }, "AudioRecordingThread") - - // 设置线程优先级并启动 - recordingThread?.priority = Thread.MAX_PRIORITY - recordingThread?.start() - - FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else "")) - } catch (e: Exception) { - FileLogger.e(TAG, "启动音频录制失败: ${e.message}") - releaseAudioResources() - } - } - - // 启用音频效果(回音消除、噪声抑制等) - private fun enableAudioEffects() { - try { - val audioSessionId = audioRecord?.audioSessionId ?: -1 - - if (audioSessionId != -1) { - // 启用回音消除 - if (AcousticEchoCanceler.isAvailable()) { - try { - echoCanceler = AcousticEchoCanceler.create(audioSessionId) - if (echoCanceler != null) { - echoCanceler?.enabled = true - FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId") - } else { - FileLogger.w(TAG, "回音消除器创建返回null") - } - } catch (e: Exception) { - FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}") - } - } else { - FileLogger.d(TAG, "设备不支持回音消除") - } - - // 以下功能暂时不启用,可根据需要取消注释 - /* - // 启用噪声抑制 - if (NoiseSuppressor.isAvailable()) { - try { - val ns = NoiseSuppressor.create(audioSessionId) - ns?.enabled = true - FileLogger.d(TAG, "噪声抑制已启用") - } catch (e: Exception) { - FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}") - } - } - - // 启用自动增益控制 - if (AutomaticGainControl.isAvailable()) { - try { - val agc = AutomaticGainControl.create(audioSessionId) - agc?.enabled = true - FileLogger.d(TAG, "自动增益控制已启用") - } catch (e: Exception) { - FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}") - } - } - */ - } else { - FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果") - } - } catch (e: Exception) { - FileLogger.e(TAG, "启用音频效果时出错: ${e.message}") - } - } - - // 停止录音 - fun stopRecording() { - if (!isRecording.get()) { - return - } - - isRecording.set(false) - - try { - // 等待录音线程结束 - recordingThread?.join(1000) - - // 释放资源 - releaseAudioResources() - - FileLogger.d(TAG, "音频录制已停止") - } catch (e: Exception) { - FileLogger.e(TAG, "停止音频录制失败: ${e.message}") - } - } - - // 释放音频资源 - private fun releaseAudioResources() { - try { - // 停止录音 - try { - if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) { - audioRecord?.stop() - } - } catch (e: Exception) { - // 忽略可能的IllegalStateException - FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}") - } - - // 释放回音消除器 - try { - if (echoCanceler != null) { - echoCanceler?.enabled = false - echoCanceler?.release() - echoCanceler = null - } - } catch (e: Exception) { - FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}") - } finally { - echoCanceler = null - } - - // 释放音频记录器 - try { - audioRecord?.release() - } catch (e: Exception) { - FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}") - } finally { - audioRecord = null - } - - // 重置线程 - recordingThread = null - - } catch (e: Exception) { - FileLogger.e(TAG, "释放音频资源失败: ${e.message}") - } - } - } - - // 一次性识别回调接口 - interface RecognizeCallback { - fun onResult(result: String, detectedLanguage: String = "") - fun onError(error: String) - } - - // 连续识别回调接口 - interface ContinuousRecognizeCallback { - fun onResult(result: String, detectedLanguage: String = "") - fun onRecognizing(recognizing: String, detectedLanguage: String = "") - fun onSessionStarted() - fun onSessionStopped() - fun onCanceled(reason: String, errorDetails: String) - fun onError(error: String) - } -} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt b/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt deleted file mode 100644 index 2fbcd8da8..000000000 --- a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt +++ /dev/null @@ -1,344 +0,0 @@ -package com.yunqiinnovation.deepsound - -import android.util.Log -import okhttp3.* -import okhttp3.MediaType.Companion.toMediaTypeOrNull -import okhttp3.RequestBody.Companion.toRequestBody -import org.json.JSONArray -import org.json.JSONObject -import java.io.IOException -import java.util.concurrent.CountDownLatch -import java.util.concurrent.TimeUnit - -/** - * 火山AI服务的原生实现 - * - * 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式 - */ -class VolcanoAIService() { - private val TAG = "VolcanoAIService" - private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3" - private val chatEndpoint = "/chat/completions" - private val client = OkHttpClient.Builder() - .connectTimeout(30, TimeUnit.SECONDS) - .readTimeout(30, TimeUnit.SECONDS) - .writeTimeout(30, TimeUnit.SECONDS) - .build() - - private var apiKey: String = "" - private var isInitialized = false - - /** - * 初始化火山AI服务 - * - * @param apiKey 火山AI API密钥 - * @return 初始化是否成功 - */ - fun initialize(apiKey: String): Boolean { - this.apiKey = apiKey - isInitialized = apiKey.isNotEmpty() - - if (!isInitialized) { - Log.e(TAG, "初始化失败:API key 不能为空") - } else { - Log.d(TAG, "火山AI服务初始化成功") - } - - return isInitialized - } - - /** - * 生成个性化问候语 - * - * @param agentName 代理名称 - * @param systemPrompt 系统提示词 - * @param callback 回调函数,返回生成的问候语 - */ - fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) { - val messages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - put(JSONObject().apply { - put("role", "user") - put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。") - }) - } - - sendMessageStream(messages, systemPrompt, object : StreamCallback { - val stringBuilder = StringBuilder() - - override fun onToken(token: String) { - stringBuilder.append(token) - } - - override fun onComplete() { - callback(stringBuilder.toString(), null) - } - - override fun onError(e: Exception) { - callback(null, e) - } - }) - } - - /** - * 发送消息(非流式输出) - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @return 返回AI的回复 - * @throws VolcanoAIException 如果API调用失败 - */ - @Throws(VolcanoAIException::class) - fun sendMessage(messages: JSONArray, systemPrompt: String): String { - // 检查是否已初始化 - if (!isInitialized || apiKey.isEmpty()) { - throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法") - } - - val fullMessages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - for (i in 0 until messages.length()) { - put(messages.getJSONObject(i)) - } - } - - val requestBody = JSONObject().apply { - put("model", "doubao-1-5-lite-32k-250115") - put("messages", fullMessages) - put("temperature", 0.7) - put("max_tokens", 2000) - put("stream", false) - } - - val mediaType = "application/json".toMediaTypeOrNull() - val request = Request.Builder() - .url("$baseUrl$chatEndpoint") - .addHeader("Content-Type", "application/json") - .addHeader("Authorization", "Bearer $apiKey") - .post(requestBody.toString().toRequestBody(mediaType)) - .build() - - try { - client.newCall(request).execute().use { response -> - if (!response.isSuccessful) { - val errorBody = response.body?.string() ?: "" - val errorMessage = try { - JSONObject(errorBody).getJSONObject("error").getString("message") - } catch (e: Exception) { - "Unknown error occurred" - } - throw VolcanoAIException(errorMessage) - } - - val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response") - val jsonResponse = JSONObject(responseBody) - - if (jsonResponse.has("choices") && - jsonResponse.getJSONArray("choices").length() > 0 && - jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) { - return jsonResponse.getJSONArray("choices") - .getJSONObject(0) - .getJSONObject("message") - .getString("content") - } - - throw VolcanoAIException("Invalid response format") - } - } catch (e: Exception) { - if (e is VolcanoAIException) throw e - throw VolcanoAIException("Failed to communicate with AI service: ${e.message}") - } - } - - /** - * 发送消息(流式输出) - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @param callback 回调函数,用于接收流式输出的结果 - */ - fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { - // 检查是否已初始化 - if (!isInitialized || apiKey.isEmpty()) { - callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")) - return - } - - val fullMessages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - for (i in 0 until messages.length()) { - put(messages.getJSONObject(i)) - } - } - - val requestBody = JSONObject().apply { - put("model", "doubao-1-5-lite-32k-250115") - put("messages", fullMessages) - put("temperature", 0.7) - put("max_tokens", 2000) - put("stream", true) - } - - val mediaType = "application/json".toMediaTypeOrNull() - val request = Request.Builder() - .url("$baseUrl$chatEndpoint") - .addHeader("Content-Type", "application/json") - .addHeader("Authorization", "Bearer $apiKey") - .addHeader("Accept", "text/event-stream") - .post(requestBody.toString().toRequestBody(mediaType)) - .build() - - client.newCall(request).enqueue(object : Callback { - override fun onFailure(call: Call, e: IOException) { - callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}")) - } - - override fun onResponse(call: Call, response: Response) { - if (!response.isSuccessful) { - val errorBody = response.body?.string() ?: "" - val errorMessage = try { - JSONObject(errorBody).getJSONObject("error").getString("message") - } catch (e: Exception) { - "Unknown error occurred" - } - callback.onError(VolcanoAIException(errorMessage)) - return - } - - val responseBody = response.body ?: return - val source = responseBody.source() - val bufferedSource = source.buffer - - try { - while (!bufferedSource.exhausted()) { - val line = bufferedSource.readUtf8Line() ?: continue - - if (line.isEmpty()) continue - if (line.startsWith("data: ")) { - val data = line.substring(6) - if (data == "[DONE]") { - callback.onComplete() - break - } - - try { - val jsonData = JSONObject(data) - if (jsonData.has("choices") && - jsonData.getJSONArray("choices").length() > 0 && - jsonData.getJSONArray("choices").getJSONObject(0).has("delta") && - jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) { - val content = jsonData.getJSONArray("choices") - .getJSONObject(0) - .getJSONObject("delta") - .getString("content") - callback.onToken(content) - } - } catch (e: Exception) { - // 忽略无效的JSON数据 - continue - } - } - } - } catch (e: Exception) { - callback.onError(VolcanoAIException("Error processing stream: ${e.message}")) - } finally { - response.close() - } - } - }) - } - - /** - * 同步方式发送消息(流式输出) - * - * 注意:此方法会阻塞当前线程,请在后台线程中调用 - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @return 返回完整的AI回复 - * @throws VolcanoAIException 如果API调用失败 - */ - @Throws(VolcanoAIException::class) - fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String { - val result = StringBuilder() - val latch = CountDownLatch(1) - var exception: Exception? = null - - sendMessageStream(messages, systemPrompt, object : StreamCallback { - override fun onToken(token: String) { - result.append(token) - } - - override fun onComplete() { - latch.countDown() - } - - override fun onError(e: Exception) { - exception = e - latch.countDown() - } - }) - - // 等待流式输出完成或出错 - latch.await(60, TimeUnit.SECONDS) - - if (exception != null) { - throw exception as VolcanoAIException - } - - return result.toString() - } - - /** - * 创建用户消息 - */ - fun createUserMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "user") - put("content", content) - } - } - - /** - * 创建系统消息 - */ - fun createSystemMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "system") - put("content", content) - } - } - - /** - * 创建助手消息 - */ - fun createAssistantMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "assistant") - put("content", content) - } - } - - /** - * 流式输出回调接口 - */ - interface StreamCallback { - fun onToken(token: String) - fun onComplete() - fun onError(e: Exception) - } -} - -/** - * 火山AI异常 - */ -class VolcanoAIException(message: String) : Exception(message) \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt similarity index 100% rename from android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt diff --git a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt similarity index 67% rename from android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt index 49fd8fdf6..df36efb44 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt @@ -23,18 +23,12 @@ import com.yunqiinnovation.deepsound.core.utils.FileLogger class MainActivity: FlutterActivity() { - private val AZURE_ASR_CHANNEL = "com.deep_voice.azure_asr" - private val AZURE_ASR_EVENT_CHANNEL = "com.deep_voice.azure_asr_events" - private val AZURE_TTS_CHANNEL = "com.deep_voice.azure_tts" private val VOICE_INTERACTION_CHANNEL = "com.deep_voice.voice_interaction" private val VOICE_INTERACTION_EVENT_CHANNEL = "com.deep_voice.voice_interaction_events" private val CLASSIC_BLUETOOTH_CHANNEL = "com.deep_voice.classic_bluetooth" private val CLASSIC_BLUETOOTH_EVENT_CHANNEL = "com.deep_voice.classic_bluetooth_events" private val TAG = "MainActivity" - private lateinit var azureAsrHelper: AzureAsrHelper - private lateinit var azureTtsHelper: AzureTtsHelper private lateinit var classicBluetoothHelper: ClassicBluetoothHelper - private var azureAsrEventSink: EventChannel.EventSink? = null private var voiceInteractionEventSink: EventChannel.EventSink? = null private var bluetoothEventSink: EventChannel.EventSink? = null @@ -58,10 +52,6 @@ class MainActivity: FlutterActivity() { val assistantMessage = intent.getStringExtra("assistantMessage") ?: "" val timestamp = intent.getLongExtra("timestamp", System.currentTimeMillis()) - Log.d(TAG, "收到聊天记录更新广播: agentId=$agentId, timestamp=$timestamp") - Log.d(TAG, "用户消息: ${userMessage.take(50)}...") - Log.d(TAG, "助手回复: ${assistantMessage.take(50)}...") - sendChatHistoryEvent(agentId, userMessage, assistantMessage, timestamp) } } @@ -143,13 +133,17 @@ class MainActivity: FlutterActivity() { var azureSpeechKey: String = "" var azureSpeechRegion: String = "" var volcanoAiApiKey: String = "" + var openaiApiKey: String = "" + var openaiBaseUrl: String? = null // 安全存储相关常量 private const val SECURE_PREFS_FILENAME = "deep_voice_secure_prefs" private const val KEY_AZURE_SPEECH_KEY = "azure_speech_key" private const val KEY_AZURE_SPEECH_REGION = "azure_speech_region" private const val KEY_VOLCANO_AI_API_KEY = "volcano_ai_api_key" - + private const val KEY_OPENAI_API_KEY = "openai_api_key" + private const val KEY_OPENAI_BASE_URL = "openai_base_url" + private const val KEY_MCP_SERVER_ENDPOINT = "mcp_server_endpoint" // 会话管理 private const val KEY_SESSION_ID = "session_id" private var currentSessionId = "" @@ -183,6 +177,8 @@ class MainActivity: FlutterActivity() { .putString(KEY_AZURE_SPEECH_KEY, azureSpeechKey) .putString(KEY_AZURE_SPEECH_REGION, azureSpeechRegion) .putString(KEY_VOLCANO_AI_API_KEY, volcanoAiApiKey) + .putString(KEY_OPENAI_API_KEY, openaiApiKey) + .putString(KEY_OPENAI_BASE_URL, openaiBaseUrl) .putString(KEY_SESSION_ID, currentSessionId) .apply() @@ -227,11 +223,13 @@ class MainActivity: FlutterActivity() { azureSpeechKey = sharedPreferences.getString(KEY_AZURE_SPEECH_KEY, "") ?: "" azureSpeechRegion = sharedPreferences.getString(KEY_AZURE_SPEECH_REGION, "") ?: "" volcanoAiApiKey = sharedPreferences.getString(KEY_VOLCANO_AI_API_KEY, "") ?: "" - + openaiApiKey = sharedPreferences.getString(KEY_OPENAI_API_KEY, "") ?: "" + openaiBaseUrl = sharedPreferences.getString(KEY_OPENAI_BASE_URL, null) FileLogger.d("MainActivity", "已从加密存储加载密钥") - // 检查是否成功获取所有密钥 - return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && volcanoAiApiKey.isNotEmpty() + // 检查是否成功获取所有必要密钥 + return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && + (openaiApiKey.isNotEmpty() || volcanoAiApiKey.isNotEmpty()) } catch (e: Exception) { FileLogger.e("MainActivity", "从加密存储加载密钥失败: ${e.message}") e.printStackTrace() @@ -247,8 +245,6 @@ class MainActivity: FlutterActivity() { FileLogger.init(applicationContext) // 初始化 Azure 语音服务 - azureAsrHelper = AzureAsrHelper(applicationContext) - azureTtsHelper = AzureTtsHelper(applicationContext) classicBluetoothHelper = ClassicBluetoothHelper(applicationContext) // 注册广播接收器 @@ -300,8 +296,7 @@ class MainActivity: FlutterActivity() { setupMethodChannels(flutterEngine) // 初始化 Azure 语音服务 - azureAsrHelper = AzureAsrHelper(this) - azureTtsHelper = AzureTtsHelper(this) + classicBluetoothHelper = ClassicBluetoothHelper(this) Log.d(TAG, "Flutter 引擎配置完成") } @@ -327,20 +322,6 @@ class MainActivity: FlutterActivity() { } ) - // Azure ASR 事件通道 - EventChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_EVENT_CHANNEL).setStreamHandler( - object : EventChannel.StreamHandler { - override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { - Log.d(TAG, "ASR事件通道开始监听") - azureAsrEventSink = events - } - - override fun onCancel(arguments: Any?) { - azureAsrEventSink = null - } - } - ) - // 设置蓝牙事件通道 EventChannel(flutterEngine.dartExecutor.binaryMessenger, CLASSIC_BLUETOOTH_EVENT_CHANNEL).setStreamHandler( object : EventChannel.StreamHandler { @@ -367,195 +348,6 @@ class MainActivity: FlutterActivity() { private fun setupMethodChannels(flutterEngine: FlutterEngine) { Log.d(TAG, "开始设置方法通道") - // 设置 Azure ASR 方法通道 - MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_CHANNEL).setMethodCallHandler { call, result -> - when (call.method) { - "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") - val region = call.argument("region") - val supportedLanguages = call.argument>("supportedLanguages")?.toTypedArray() ?: arrayOf("zh-CN", "en-US") - - if (subscriptionKey == null || region == null) { - result.error("INVALID_ARGUMENTS", "订阅密钥和区域不能为空", null) - return@setMethodCallHandler - } - - try { - val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages) - result.success(success) - } catch (e: Exception) { - result.error("INITIALIZATION_ERROR", e.message, null) - } - } - "recognizeOnce" -> { - - azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) { - result.success(mapOf( - "text" to text, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onError(error: String) { - result.error("RECOGNITION_ERROR", error, null) - } - }) - } - "startContinuousRecognition" -> { - // 确保事件通道已准备好 - if (azureAsrEventSink == null) { - result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) - return@setMethodCallHandler - } - - val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) { - sendAsrEvent(mapOf( - "type" to "result", - "text" to text, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onRecognizing(recognizing: String, detectedLanguage: String) { - sendAsrEvent(mapOf( - "type" to "recognizing", - "text" to recognizing, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onSessionStarted() { - sendAsrEvent(mapOf("type" to "sessionStarted")) - } - - override fun onSessionStopped() { - sendAsrEvent(mapOf("type" to "sessionStopped")) - } - - override fun onCanceled(reason: String, errorDetails: String) { - sendAsrEvent(mapOf( - "type" to "canceled", - "reason" to reason, - "errorDetails" to errorDetails - )) - } - - override fun onError(error: String) { - sendAsrEvent(mapOf("type" to "error", "message" to error)) - } - }) - result.success(success) - } - "stopContinuousRecognition" -> { - try { - if (!azureAsrHelper.isContinuousRecognitionActive()) { - result.success(true) - return@setMethodCallHandler - } - - val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) {} - override fun onRecognizing(recognizing: String, detectedLanguage: String) {} - override fun onSessionStarted() {} - override fun onSessionStopped() {} - override fun onCanceled(reason: String, errorDetails: String) {} - override fun onError(error: String) { - result.error("STOP_ERROR", error, null) - } - }) - result.success(success) - } catch (e: Exception) { - result.error("STOP_ERROR", e.message, null) - } - } - "isContinuousRecognitionActive" -> { - result.success(azureAsrHelper.isContinuousRecognitionActive()) - } - "dispose" -> { - azureAsrHelper.dispose() - result.success(true) - } - else -> { - result.notImplemented() - } - } - } - - // 设置 Azure TTS 方法通道 - MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_TTS_CHANNEL).setMethodCallHandler { call, result -> - when (call.method) { - "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") ?: "" - val region = call.argument("region") ?: "" - val language = call.argument("language") ?: "zh-CN" - - val success = azureTtsHelper.initialize(subscriptionKey, region, language) - result.success(success) - } - "setVoice" -> { - val voiceName = call.argument("voiceName") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) - result.success(azureTtsHelper.setVoice(voiceName)) - } - "setSpeechParams" -> { - val rate = call.argument("rate") ?: 0 - val pitch = call.argument("pitch") ?: 0 - val volume = call.argument("volume") ?: 100 - result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) - } - "setAudioOutputType" -> { - val outputTypeStr = call.argument("outputType") ?: "speaker" - val outputType = when (outputTypeStr.lowercase()) { - "speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER - "earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE - "auto" -> AzureTtsHelper.AudioOutputType.AUTO - else -> AzureTtsHelper.AudioOutputType.SPEAKER - } - result.success(azureTtsHelper.setAudioOutputType(outputType)) - } - "speakText" -> { - val text = call.argument("text") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "文本不能为空", null) - - azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - runOnUiThread { result.success(message) } - } - - override fun onError(error: String) { - runOnUiThread { result.error("SPEAK_ERROR", error, null) } - } - }) - } - "speakSsml" -> { - val ssml = call.argument("ssml") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "SSML不能为空", null) - - azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - runOnUiThread { result.success(message) } - } - - override fun onError(error: String) { - runOnUiThread { result.error("SPEAK_ERROR", error, null) } - } - }) - } - "stopSpeaking" -> { - result.success(azureTtsHelper.stopSpeaking()) - } - "isSpeaking" -> { - result.success(azureTtsHelper.isSpeaking()) - } - "dispose" -> { - azureTtsHelper.dispose() - result.success(true) - } - else -> { - result.notImplemented() - } - } - } - // 设置语音交互方法通道 MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOICE_INTERACTION_CHANNEL).setMethodCallHandler { call, result -> when (call.method) { @@ -657,15 +449,6 @@ class MainActivity: FlutterActivity() { } } - // ASR 事件发送方法 - private fun sendAsrEvent(event: Map) { - if (azureAsrEventSink == null) return - - runOnUiThread { - azureAsrEventSink?.success(event) - } - } - /** * 启动语音交互服务(带配置参数) */ @@ -675,12 +458,16 @@ class MainActivity: FlutterActivity() { // 获取配置参数 val key = call.argument("azure_speech_key") ?: "" val region = call.argument("azure_speech_region") ?: "" - val aiKey = call.argument("volcano_ai_api_key") ?: "" + val openaiKey = call.argument("openai_api_key") ?: "" + val baseUrl = call.argument("openai_base_url") - // 设置 Azure Speech 配置 + // 设置 Azure Speech 和 AI 配置 azureSpeechKey = key azureSpeechRegion = region - volcanoAiApiKey = aiKey + openaiApiKey = openaiKey + if (baseUrl != null) { + openaiBaseUrl = baseUrl + } // 保存密钥到安全存储 saveKeysToSecureStorage(applicationContext) @@ -799,8 +586,6 @@ class MainActivity: FlutterActivity() { } // 释放资源 - azureAsrHelper.dispose() - azureTtsHelper.dispose() classicBluetoothHelper.dispose() super.onDestroy() diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt new file mode 100644 index 000000000..39907e6d5 --- /dev/null +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt @@ -0,0 +1,606 @@ +package com.yunqiinnovation.deepsound + +import android.util.Log +import okhttp3.* +import okhttp3.MediaType.Companion.toMediaTypeOrNull +import okhttp3.RequestBody.Companion.toRequestBody +import org.json.JSONArray +import org.json.JSONObject +import java.io.IOException +import java.util.concurrent.TimeUnit +import com.yunqiinnovation.deepsound.core.utils.FileLogger + +/** + * OpenAI服务的原生实现 + */ +class OpenAIService() { + private val TAG = "OpenAIService" + private var baseUrl = "https://api.openai.com/v1/chat/completions" + private val client = OkHttpClient.Builder() + .connectTimeout(30, TimeUnit.SECONDS) + .readTimeout(30, TimeUnit.SECONDS) + .writeTimeout(30, TimeUnit.SECONDS) + .build() + + private var apiKey: String = "" + private var isInitialized = false + private var model: String = "doubao-1-5-lite-32k-250115" // 默认模型 + + // 用于存储注册的函数 + private val registeredFunctions = mutableListOf() + + + /** + * 初始化OpenAI服务 + */ + fun initialize(apiKey: String, baseUrl: String = ""): Boolean { + this.apiKey = apiKey + if (baseUrl.isNotEmpty()) { + this.baseUrl = baseUrl + } + isInitialized = apiKey.isNotEmpty() + return isInitialized + } + + /** + * 注册函数 + */ + fun registerFunction(name: String, description: String, parameters: JSONObject): Boolean { + try { + val function = JSONObject().apply { + put("name", name) + put("description", description) + put("parameters", parameters) + } + + // 检查是否已存在相同名称的函数 + val existingIndex = registeredFunctions.indexOfFirst { + it.getString("name") == name + } + + if (existingIndex >= 0) { + // 如果已存在,则替换 + registeredFunctions[existingIndex] = function + } else { + // 如果不存在,则添加 + registeredFunctions.add(function) + } + + return true + } catch (e: Exception) { + return false + } + } + + /** + * 发送消息(非流式输出) + */ + @Throws(OpenAIException::class) + fun sendMessage(messages: JSONArray, systemPrompt: String): String { + if (!isInitialized || apiKey.isEmpty()) { + throw OpenAIException("OpenAI服务未初始化") + } + + val fullMessages = JSONArray().apply { + put(JSONObject().apply { + put("role", "system") + put("content", systemPrompt) + }) + for (i in 0 until messages.length()) { + put(messages.getJSONObject(i)) + } + } + + val requestBody = JSONObject().apply { + put("model", model) + put("messages", fullMessages) + put("temperature", 0.7) + put("max_tokens", 2000) + put("stream", false) + + // 如果有注册的函数,则添加到请求中 + if (registeredFunctions.isNotEmpty()) { + val tools = JSONArray() + for (function in registeredFunctions) { + val tool = JSONObject().apply { + put("type", "function") + put("function", function) + } + tools.put(tool) + } + put("tools", tools) + } + } + + val mediaType = "application/json".toMediaTypeOrNull() + val request = Request.Builder() + .url(baseUrl) + .addHeader("Content-Type", "application/json") + .addHeader("Authorization", "Bearer $apiKey") + .post(requestBody.toString().toRequestBody(mediaType)) + .build() + + try { + client.newCall(request).execute().use { response -> + if (!response.isSuccessful) { + throw OpenAIException("API调用失败: ${response.code}") + } + + val responseBody = response.body?.string() ?: throw OpenAIException("Empty response") + val jsonResponse = JSONObject(responseBody) + + // 检查是否有函数调用 + if (jsonResponse.has("choices") && + jsonResponse.getJSONArray("choices").length() > 0) { + + val choice = jsonResponse.getJSONArray("choices").getJSONObject(0) + + // 检查是否是函数调用 + if (choice.has("message")) { + val message = choice.getJSONObject("message") + + // 检查是否有工具调用 + if (message.has("tool_calls")) { + val toolCalls = message.getJSONArray("tool_calls") + if (toolCalls.length() > 0) { + val toolCall = toolCalls.getJSONObject(0) + if (toolCall.has("function")) { + val function = toolCall.getJSONObject("function") + val functionCall = JSONObject().apply { + put("name", function.getString("name")) + put("arguments", function.getString("arguments")) + put("id", toolCall.getString("id")) + } + return functionCall.toString() + } + } + } + + // 如果没有工具调用,返回消息内容 + if (message.has("content")) { + return message.getString("content") + } + } + } + + throw OpenAIException("Invalid response format") + } + } catch (e: Exception) { + if (e is OpenAIException) throw e + throw OpenAIException("Failed to communicate with AI service: ${e.message}") + } + } + + /** + * 发送消息(流式输出) + */ + fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { + if (!isInitialized || apiKey.isEmpty()) { + callback.onError(OpenAIException("OpenAI服务未初始化")) + return + } + + val fullMessages = JSONArray().apply { + put(JSONObject().apply { + put("role", "system") + put("content", systemPrompt) + }) + for (i in 0 until messages.length()) { + put(messages.getJSONObject(i)) + } + } + + val requestBody = JSONObject().apply { + put("model", model) + put("messages", fullMessages) + put("temperature", 0.7) + put("max_tokens", 2000) + put("stream", true) + + // 如果有注册的函数,则添加到请求中 + if (registeredFunctions.isNotEmpty()) { + val tools = JSONArray() + for (function in registeredFunctions) { + val tool = JSONObject().apply { + put("type", "function") + put("function", function) + } + tools.put(tool) + } + put("tools", tools) + } + } + + val mediaType = "application/json".toMediaTypeOrNull() + val request = Request.Builder() + .url(baseUrl) + .addHeader("Content-Type", "application/json") + .addHeader("Authorization", "Bearer $apiKey") + .addHeader("Accept", "text/event-stream") + .post(requestBody.toString().toRequestBody(mediaType)) + .build() + + client.newCall(request).enqueue(object : Callback { + override fun onFailure(call: Call, e: IOException) { + callback.onError(OpenAIException(e.message ?: "请求失败")) + } + + override fun onResponse(call: Call, response: Response) { + if (!response.isSuccessful) { + callback.onError(OpenAIException("API调用失败: ${response.code}")) + return + } + + val responseBody = response.body ?: return + val source = responseBody.source() + + try { + // 预取数据到缓冲区 + source.request(Long.MAX_VALUE) + val bufferedSource = source.buffer + + // 用于存储函数调用的各个部分 + val finalToolCalls = mutableMapOf() + + while (!bufferedSource.exhausted()) { + val line = bufferedSource.readUtf8Line() ?: continue + + if (line.isEmpty()) continue + if (line.startsWith("data: ")) { + val data = line.substring(6) + if (data == "[DONE]") { + callback.onComplete() + break + } + + try { + val jsonData = JSONObject(data) + if (jsonData.has("choices") && + jsonData.getJSONArray("choices").length() > 0) { + + val choice = jsonData.getJSONArray("choices").getJSONObject(0) + + // 检查是否有delta + if (choice.has("delta")) { + val delta = choice.getJSONObject("delta") + + // 检查是否有工具调用 + if (delta.has("tool_calls")) { + val toolCalls = delta.getJSONArray("tool_calls") + for (i in 0 until toolCalls.length()) { + val toolCall = toolCalls.getJSONObject(i) + val index = toolCall.optInt("index", i) + + // 如果是新的工具调用,初始化 + if (!finalToolCalls.containsKey(index)) { + finalToolCalls[index] = ToolCallInfo() + } + + // 获取ID + if (toolCall.has("id")) { + finalToolCalls[index]?.id = toolCall.getString("id") + } + + // 处理函数信息 + if (toolCall.has("function")) { + val function = toolCall.getJSONObject("function") + + if (function.has("name")) { + finalToolCalls[index]?.name = function.getString("name") + } + + if (function.has("arguments")) { + finalToolCalls[index]?.arguments += function.getString("arguments") + } + } + } + continue + } + + // 如果有内容,发送给回调 + if (delta.has("content") && !delta.isNull("content")) { + val content = delta.getString("content") + callback.onToken(content) + } + } + } + } catch (e: Exception) { + // 忽略无效的JSON + continue + } + } + } + + // 处理完整的函数调用 + for ((_, toolCallInfo) in finalToolCalls) { + if (toolCallInfo.name.isNotEmpty()) { + try { + // 创建函数调用对象 + val functionCall = JSONObject().apply { + put("id", toolCallInfo.id) + put("name", toolCallInfo.name) + put("arguments", toolCallInfo.arguments.trim()) + } + + callback.onFunctionCall(functionCall) + } catch (e: Exception) { + // 出错时使用空参数 + val functionCall = JSONObject().apply { + put("id", toolCallInfo.id) + put("name", toolCallInfo.name) + put("arguments", "{}") + } + callback.onFunctionCall(functionCall) + } + } + } + } catch (e: Exception) { + callback.onError(OpenAIException("处理流式响应出错: ${e.message}")) + } finally { + response.close() + } + } + }) + } + + /** + * 发送函数调用结果 + */ + fun sendFunctionCallResult( + messages: JSONArray, + systemPrompt: String, + functionCall: JSONObject, + functionResult: String, + callback: StreamCallback + ) { + if (!isInitialized || apiKey.isEmpty()) { + callback.onError(OpenAIException("OpenAI服务未初始化")) + return + } + + // 构建完整的消息历史 + val fullMessages = JSONArray().apply { + // 添加系统提示 + put(JSONObject().apply { + put("role", "system") + put("content", systemPrompt) + }) + + // 添加历史消息 + for (i in 0 until messages.length()) { + put(messages.getJSONObject(i)) + } + + // 添加函数调用信息 + put(JSONObject().apply { + put("role", "assistant") + put("content", null) + put("tool_calls", JSONArray().apply { + put(JSONObject().apply { + put("id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) + put("type", "function") + put("function", JSONObject().apply { + put("name", functionCall.getString("name")) + put("arguments", functionCall.getString("arguments")) + }) + }) + }) + }) + + // 添加函数返回结果 + put(JSONObject().apply { + put("role", "tool") + put("tool_call_id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) + put("content", functionResult) + }) + } + + // 构建请求 + val requestBody = JSONObject().apply { + put("model", model) + put("messages", fullMessages) + put("temperature", 0.7) + put("max_tokens", 2000) + put("stream", true) + + // 如果有注册的函数,则添加到请求中 + if (registeredFunctions.isNotEmpty()) { + val tools = JSONArray() + for (function in registeredFunctions) { + val tool = JSONObject().apply { + put("type", "function") + put("function", function) + } + tools.put(tool) + } + put("tools", tools) + } + } + + val mediaType = "application/json".toMediaTypeOrNull() + val request = Request.Builder() + .url(baseUrl) + .addHeader("Content-Type", "application/json") + .addHeader("Authorization", "Bearer $apiKey") + .addHeader("Accept", "text/event-stream") + .post(requestBody.toString().toRequestBody(mediaType)) + .build() + + // 发送请求 + client.newCall(request).enqueue(object : Callback { + override fun onFailure(call: Call, e: IOException) { + callback.onError(OpenAIException(e.message ?: "请求失败")) + } + + override fun onResponse(call: Call, response: Response) { + if (!response.isSuccessful) { + callback.onError(OpenAIException("API调用失败: ${response.code}")) + return + } + + val responseBody = response.body ?: return + val source = responseBody.source() + + try { + // 预取数据到缓冲区 + source.request(Long.MAX_VALUE) + val bufferedSource = source.buffer + + // 用于存储函数调用的各个部分 + val finalToolCalls = mutableMapOf() + + while (!bufferedSource.exhausted()) { + val line = bufferedSource.readUtf8Line() ?: continue + + if (line.isEmpty()) continue + if (line.startsWith("data: ")) { + val data = line.substring(6) + if (data == "[DONE]") { + callback.onComplete() + break + } + + try { + val jsonData = JSONObject(data) + if (jsonData.has("choices") && + jsonData.getJSONArray("choices").length() > 0) { + + val choice = jsonData.getJSONArray("choices").getJSONObject(0) + + // 检查是否有delta + if (choice.has("delta")) { + val delta = choice.getJSONObject("delta") + + // 检查是否有工具调用 + if (delta.has("tool_calls")) { + val toolCalls = delta.getJSONArray("tool_calls") + for (i in 0 until toolCalls.length()) { + val toolCall = toolCalls.getJSONObject(i) + val index = toolCall.optInt("index", i) + + // 如果是新的工具调用,初始化 + if (!finalToolCalls.containsKey(index)) { + finalToolCalls[index] = ToolCallInfo() + } + + // 获取ID + if (toolCall.has("id")) { + finalToolCalls[index]?.id = toolCall.getString("id") + } + + // 处理函数信息 + if (toolCall.has("function")) { + val function = toolCall.getJSONObject("function") + + if (function.has("name")) { + finalToolCalls[index]?.name = function.getString("name") + } + + if (function.has("arguments")) { + finalToolCalls[index]?.arguments += function.getString("arguments") + } + } + } + continue + } + + // 如果有内容,发送给回调 + if (delta.has("content") && !delta.isNull("content")) { + val content = delta.getString("content") + callback.onToken(content) + } + } + } + } catch (e: Exception) { + // 忽略无效的JSON + continue + } + } + } + + // 处理完整的函数调用 + for ((_, toolCallInfo) in finalToolCalls) { + if (toolCallInfo.name.isNotEmpty()) { + try { + // 创建函数调用对象 + val functionCall = JSONObject().apply { + put("id", toolCallInfo.id) + put("name", toolCallInfo.name) + put("arguments", toolCallInfo.arguments.trim()) + } + + callback.onFunctionCall(functionCall) + } catch (e: Exception) { + // 出错时使用空参数 + val functionCall = JSONObject().apply { + put("id", toolCallInfo.id) + put("name", toolCallInfo.name) + put("arguments", "{}") + } + callback.onFunctionCall(functionCall) + } + } + } + } catch (e: Exception) { + callback.onError(OpenAIException("处理流式响应出错: ${e.message}")) + } finally { + response.close() + } + } + }) + } + + /** + * 创建用户消息 + */ + fun createUserMessage(content: String): JSONObject { + return JSONObject().apply { + put("role", "user") + put("content", content) + } + } + + /** + * 创建系统消息 + */ + fun createSystemMessage(content: String): JSONObject { + return JSONObject().apply { + put("role", "system") + put("content", content) + } + } + + /** + * 创建助手消息 + */ + fun createAssistantMessage(content: String): JSONObject { + return JSONObject().apply { + put("role", "assistant") + put("content", content) + } + } + + /** + * 流式输出回调接口 + */ + interface StreamCallback { + fun onToken(token: String) + fun onComplete() + fun onError(e: Exception) + fun onFunctionCall(functionCall: JSONObject) {} + } + + /** + * 用于存储工具调用信息的辅助类 + */ + private class ToolCallInfo { + var id: String = "" + var name: String = "" + var arguments: String = "" + } +} + +/** + * OpenAI异常 + */ +class OpenAIException(message: String) : Exception(message) \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt new file mode 100644 index 000000000..a89441506 --- /dev/null +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt @@ -0,0 +1,180 @@ +package com.yunqiinnovation.deepsound + +import org.json.JSONArray +import org.json.JSONObject +import com.yunqiinnovation.deepsound.OpenAIService +import com.yunqiinnovation.deepsound.core.utils.FileLogger + +/** + * 语音功能处理器 - 处理AI函数调用 + */ +class VoiceFunctionHandler( + private val openAIService: OpenAIService, + private val systemPrompt: String +) { + companion object { + private const val TAG = "VoiceFunctionHandler" + } + + /** + * 初始化并注册所有可用的函数 + */ + fun initialize() { + try { + // 注册退出交互函数 + registerExitInteractionFunction() + + // 在这里可以注册更多函数 + + } catch (e: Exception) { + FileLogger.e(TAG, "初始化函数处理器失败: ${e.message}", e) + } + } + + /** + * 注册退出交互函数 + */ + private fun registerExitInteractionFunction() { + try { + openAIService.registerFunction( + "exit_interaction", + "退出当前语音交互", + JSONObject(""" + { + "type": "object", + "properties": {}, + "required": [] + } + """) + ) + FileLogger.d(TAG, "退出交互功能已注册") + } catch (e: Exception) { + FileLogger.e(TAG, "注册退出交互函数失败: ${e.message}", e) + } + } + + /** + * 处理函数调用 + * + * @param functionCall 函数调用信息 + * @param messages 消息历史 + * @param callback 回调,处理退出等操作 + * @return 是否已处理函数调用 + */ + fun handleFunctionCall( + functionCall: JSONObject, + messages: JSONArray, + callback: FunctionCallCallback + ): Boolean { + val functionName = functionCall.getString("name") + FileLogger.d(TAG, "处理函数调用: $functionName") + + return when (functionName) { + "exit_interaction" -> { + handleExitInteraction(functionCall, messages, callback) + true + } + else -> { + // 未知函数,返回默认结果 + handleUnknownFunction(functionCall, messages, callback) + false + } + } + } + + /** + * 处理退出交互函数 + */ + private fun handleExitInteraction( + functionCall: JSONObject, + messages: JSONArray, + callback: FunctionCallCallback + ) { + FileLogger.d(TAG, "处理退出交互函数") + + val responseBuilder = StringBuilder() + + openAIService.sendFunctionCallResult( + messages = messages, + systemPrompt = systemPrompt, + functionCall = functionCall, + functionResult = "{\"result\": \"已退出语音交互\"}", + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + responseBuilder.append(token) + } + + override fun onComplete() { + FileLogger.d(TAG, "handleExitInteraction onComplete: ${responseBuilder.toString()}") + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + callback.onExitWithMessage(response) + } else { + callback.onExitWithMessage("已退出语音交互") + } + } + + override fun onError(e: Exception) { + FileLogger.e(TAG, "处理退出交互函数调用出错: ${e.message}") + callback.onError("退出交互时出错") + } + + override fun onFunctionCall(nestedCall: JSONObject) { + FileLogger.e(TAG, "意外收到嵌套函数调用: ${nestedCall.getString("name")}") + } + } + ) + } + + /** + * 处理未知函数调用 + */ + private fun handleUnknownFunction( + functionCall: JSONObject, + messages: JSONArray, + callback: FunctionCallCallback + ) { + FileLogger.d(TAG, "处理未知函数: ${functionCall.getString("name")}") + + try { + openAIService.sendFunctionCallResult( + messages = messages, + systemPrompt = systemPrompt, + functionCall = functionCall, + functionResult = "{\"result\": \"处理函数调用中\"}", + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + callback.onTokenReceived(token) + } + + override fun onComplete() { + callback.onComplete() + } + + override fun onError(e: Exception) { + FileLogger.e(TAG, "处理函数调用失败: ${e.message}") + callback.onError("处理函数调用失败: ${e.message}") + } + + override fun onFunctionCall(nestedCall: JSONObject) { + callback.onFunctionCall(nestedCall) + } + } + ) + } catch (e: Exception) { + FileLogger.e(TAG, "处理函数调用失败: ${e.message}") + callback.onError("处理函数调用失败: ${e.message}") + } + } + + /** + * 函数调用回调接口 + */ + interface FunctionCallCallback { + fun onTokenReceived(token: String) + fun onComplete() + fun onError(message: String) + fun onFunctionCall(functionCall: JSONObject) + fun onExitWithMessage(farewell: String) + } +} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt similarity index 78% rename from android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt index f469c136e..ad20ab6b7 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt @@ -22,10 +22,15 @@ import android.os.Handler import android.os.Looper import java.util.concurrent.atomic.AtomicBoolean import org.json.JSONArray +import org.json.JSONObject import android.media.MediaPlayer import android.media.AudioAttributes import android.net.Uri import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.AzureAsrHelper +import com.yunqiinnovation.azure_speech.AzureTtsHelper +import com.yunqiinnovation.deepsound.OpenAIService + /** * 后台语音交互 Service: @@ -78,7 +83,8 @@ class VoiceInteractionService : Service() { private lateinit var audioManager: AudioManager private lateinit var azureAsrHelper: AzureAsrHelper private lateinit var azureTtsHelper: AzureTtsHelper - private lateinit var volcanoAIService: VolcanoAIService + private lateinit var openAIService: OpenAIService + private lateinit var functionHandler: VoiceFunctionHandler // 定时器 private val handler = Handler(Looper.getMainLooper()) @@ -95,7 +101,10 @@ class VoiceInteractionService : Service() { 请保持回答简短、准确,避免过长的解释。 如果用户的问题不清楚,请礼貌地请求澄清。 不要使用复杂的术语,除非用户明确要求。 - 用户用语音和你交互. + 用户用语音和你交互。 + + 当用户说"退出"、"再见"、"结束对话"等类似意图时,你应该使用exit_interaction函数来结束对话, + 并在结束前说一句友好的告别语,例如"再见,有需要随时找我"。 """.trimIndent() // 添加媒体播放器 @@ -157,10 +166,11 @@ class VoiceInteractionService : Service() { // 尝试从静态变量获取配置 var subscriptionKey = MainActivity.azureSpeechKey var serviceRegion = MainActivity.azureSpeechRegion - var volcanoKey = MainActivity.volcanoAiApiKey + var openaiKey = MainActivity.openaiApiKey + var openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" // OpenAI API基本URL // 如果静态变量中没有配置,尝试从加密存储中加载 - if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || volcanoKey.isEmpty()) { + if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || openaiKey.isEmpty()) { FileLogger.d(TAG, "静态变量中的配置信息不完整,尝试从加密存储加载") // 从加密存储加载密钥 @@ -170,7 +180,8 @@ class VoiceInteractionService : Service() { // 更新本地变量 subscriptionKey = MainActivity.azureSpeechKey serviceRegion = MainActivity.azureSpeechRegion - volcanoKey = MainActivity.volcanoAiApiKey + openaiKey = MainActivity.openaiApiKey + openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" FileLogger.d(TAG, "已从加密存储加载配置信息") } else { @@ -191,15 +202,29 @@ class VoiceInteractionService : Service() { FileLogger.e(TAG, "Azure配置信息不完整,无法初始化Azure服务") } - // 初始化火山AI服务 - volcanoAIService = VolcanoAIService() + // 初始化OpenAI服务 + openAIService = OpenAIService() - // 初始化火山AI服务 - if (volcanoKey.isNotEmpty()) { - volcanoAIService.initialize(volcanoKey) - FileLogger.d(TAG, "火山AI服务已初始化") + // 初始化OpenAI服务 + if (openaiKey.isNotEmpty()) { + val initialized = if (openaiBaseUrl.isNotEmpty()) { + openAIService.initialize(openaiKey, openaiBaseUrl) + } else { + openAIService.initialize(openaiKey) + } + + if (initialized) { + FileLogger.d(TAG, "OpenAI服务已初始化") + + // 初始化函数处理器 + functionHandler = VoiceFunctionHandler(openAIService, systemPrompt) + functionHandler.initialize() + + } else { + FileLogger.e(TAG, "OpenAI服务初始化失败") + } } else { - FileLogger.e(TAG, "火山AI配置信息不完整,无法初始化火山AI服务") + FileLogger.e(TAG, "OpenAI配置信息不完整,无法初始化OpenAI服务") } } @@ -404,7 +429,7 @@ class VoiceInteractionService : Service() { override fun onResult(result: String, detectedLanguage: String) { if (result.isNotEmpty()) { - processWithVolcanoAI(result) + processWithOpenAI(result) } // 重置状态,继续识别 @@ -483,27 +508,114 @@ class VoiceInteractionService : Service() { } /** - * 使用VolcanoAI处理语音识别结果 + * 使用OpenAI处理语音识别结果 */ - private fun processWithVolcanoAI(text: String) { + private fun processWithOpenAI(text: String) { // 保存当前用户输入,用于后续同步聊天记录 currentUserInput = text Thread { try { val messages = JSONArray().apply { - put(volcanoAIService.createUserMessage(text)) + put(openAIService.createUserMessage(text)) } + + // 创建响应构建器 + val responseBuilder = StringBuilder() - val response = volcanoAIService.sendMessage(messages, systemPrompt) - - // 播放AI回复 - speakAIResponse(response) + openAIService.sendMessageStream( + messages = messages, + systemPrompt = systemPrompt, + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + // 累加响应内容 + responseBuilder.append(token) + } + + override fun onComplete() { + // 处理完整响应 + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + // 播放AI回复 + Log.d(TAG, "AI 回复: $response") + + speakAIResponse(response) + + // 同步聊天记录到Flutter端 + notifyChatHistoryUpdated("personal_assistant", text, response) + } + } + + override fun onError(e: Exception) { + FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e) + playNotification("AI处理出错") + } + + override fun onFunctionCall(call: JSONObject) { + FileLogger.d(TAG, "收到函数调用请求: ${call.getString("name")}") + + // 使用函数处理器处理函数调用 + val handled = functionHandler.handleFunctionCall( + functionCall = call, + messages = messages, + callback = object : VoiceFunctionHandler.FunctionCallCallback { + override fun onTokenReceived(token: String) { + responseBuilder.append(token) + } + + override fun onComplete() { + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + // 播放AI回复 + Log.d(TAG, "AI Function Call 回复: $response") + speakAIResponse(response) + + // 同步聊天记录到Flutter端 + notifyChatHistoryUpdated("personal_assistant", text, response) + } + updateLastActivityTime() + } + + override fun onError(message: String) { + FileLogger.e(TAG, "函数处理出错: $message") + playNotification(message) + } + + override fun onFunctionCall(nestedCall: JSONObject) { + FileLogger.d(TAG, "收到嵌套函数调用: ${nestedCall.getString("name")}") + // 递归处理嵌套函数调用 + functionHandler.handleFunctionCall( + functionCall = nestedCall, + messages = messages, + callback = this + ) + } + + override fun onExitWithMessage(farewell: String) { + // 播放退出消息 + speakAIResponse(farewell) + + // 同步聊天记录 + notifyChatHistoryUpdated("personal_assistant", text, farewell) + + // 停止语音识别 + stopVoiceRecognition() + } + } + ) + + if (!handled) { + // 如果函数没有被处理,作为普通文本处理 + FileLogger.d(TAG, "函数未处理,作为普通文本处理") + speakAIResponse("我无法处理这个请求") + notifyChatHistoryUpdated("personal_assistant", text, "我无法处理这个请求") + } + } + } + ) - // 同步聊天记录到Flutter端 - notifyChatHistoryUpdated("personal_assistant", text, response) } catch (e: Exception) { - FileLogger.e(TAG, "AI处理出错: ${e.message}") + FileLogger.e(TAG, "AI处理出错: ${e.message}", e) playNotification("AI处理出错") } }.start() @@ -746,38 +858,45 @@ class VoiceInteractionService : Service() { override fun onBind(intent: Intent?): IBinder? = null override fun onDestroy() { - FileLogger.d(TAG, "onDestroy - 语音交互服务正在销毁") - - // 释放音频播放器 - audioPlayer?.release() - audioPlayer = null + super.onDestroy() + FileLogger.d(TAG, "onDestroy - 语音交互服务即将销毁") - // 停止监控 + // 停止服务监控 stopMonitoring() // 停止语音识别 - if (isRecognitionActive) { - stopVoiceRecognition() - } + stopVoiceRecognition() + + // 停止媒体会话 + mediaSession.release() + FileLogger.d(TAG, "媒体会话已释放") // 停止TTS stopCurrentTTS() - // 释放 MediaSession - mediaSession.release() + // 关闭Azure语音服务 + if (::azureAsrHelper.isInitialized) { + FileLogger.d(TAG, "关闭Azure语音服务") + azureAsrHelper.dispose() + } - // 释放Azure服务实例 - FileLogger.d(TAG, "释放Azure服务实例") - azureAsrHelper.dispose() - azureTtsHelper.dispose() + if (::azureTtsHelper.isInitialized) { + FileLogger.d(TAG, "关闭Azure TTS服务") + azureTtsHelper.dispose() + } - // 更新服务状态 - isRunning.set(false) + // 关闭OpenAI服务 + if (::openAIService.isInitialized) { + FileLogger.d(TAG, "关闭OpenAI服务") + } - // 关闭日志系统 - FileLogger.shutdown() + // 关闭音频播放器 + audioPlayer?.release() - super.onDestroy() + // 设置服务状态 + isRunning.set(false) + + FileLogger.d(TAG, "语音交互服务已销毁") } /** @@ -804,9 +923,7 @@ class VoiceInteractionService : Service() { * 通知 Flutter 端聊天记录已更新 */ private fun notifyChatHistoryUpdated(agentId: String, userMessage: String, assistantMessage: String) { - FileLogger.d(TAG, "通知Flutter聊天记录已更新: agentId=$agentId") - FileLogger.d(TAG, "用户消息: ${userMessage.take(50)}...") - FileLogger.d(TAG, "助手回复: ${assistantMessage.take(50)}...") + // 创建广播 Intent val intent = Intent(ACTION_CHAT_HISTORY_UPDATED).apply { @@ -818,7 +935,6 @@ class VoiceInteractionService : Service() { // 发送广播 sendBroadcast(intent) - FileLogger.d(TAG, "已发送聊天记录广播") } /** diff --git a/android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt similarity index 100% rename from android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts index 6296e34ed..6a3483c9c 100644 --- a/android/settings.gradle.kts +++ b/android/settings.gradle.kts @@ -29,4 +29,8 @@ plugins { } include(":app") +include(":azure_speech") + +// 设置azure_speech项目的路径 +project(":azure_speech").projectDir = file("../local_plugins/azure_speech/android") diff --git a/azure/LICENSE b/azure/LICENSE deleted file mode 100644 index d3df29b8a..000000000 --- a/azure/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2024 Your Company - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. \ No newline at end of file diff --git a/azure/README.md b/azure/README.md deleted file mode 100644 index a19025680..000000000 --- a/azure/README.md +++ /dev/null @@ -1,66 +0,0 @@ -# Azure Speech Recognition - -A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. - -## Features - -- Speech-to-text (Azure Speech Recognition) -- Text-to-speech (Azure Speech Synthesis) -- Support for multiple languages -- Language detection -- Continuous recognition -- Streaming synthesis - -## Getting Started - -### Prerequisites - -- Azure Speech service subscription key -- Azure Speech service region - -### Installation - -Add this to your package's `pubspec.yaml` file: - -```yaml -dependencies: - azure_speech_recognition: - path: ./azure -``` - -### Usage - -```dart -import 'package:azure_speech_recognition/azure_speech_recognition.dart'; - -// Initialize the service -await AzureSpeechRecognition.initialize( - subscriptionKey: 'your_subscription_key', - region: 'your_region', - supportedLanguages: ['zh-CN', 'en-US'], -); - -// Start continuous recognition -await AzureSpeechRecognition.startContinuousRecognition(); - -// Listen for recognition events -AzureSpeechRecognition.onRecognitionEvent.listen((event) { - if (event['type'] == 'result') { - print('Recognized: ${event['text']}'); - print('Detected language: ${event['detectedLanguage']}'); - } -}); - -// Stop recognition when done -await AzureSpeechRecognition.stopContinuousRecognition(); - -// Speak text -await AzureSpeechRecognition.speakText('Hello, world!'); - -// Clean up -await AzureSpeechRecognition.dispose(); -``` - -## License - -This project is licensed under the MIT License - see the LICENSE file for details. \ No newline at end of file diff --git a/azure/ios/Classes/AzureAsrHelper.swift b/azure/ios/Classes/AzureAsrHelper.swift deleted file mode 100644 index 58b142f60..000000000 --- a/azure/ios/Classes/AzureAsrHelper.swift +++ /dev/null @@ -1,757 +0,0 @@ -import Foundation -import MicrosoftCognitiveServicesSpeech -import AVFoundation -import AudioToolbox - -/// Azure ASR工具类,负责实现语音识别服务接口 -@available(iOS 13.0, *) -class AzureAsrHelper: NSObject { - // MARK: - 属性 - - /// 事件处理回调 - private var eventHandler: (String, [String: Any]) -> Void - - /// 语音配置信息 - private var speechSubscriptionKey: String = "" - private var serviceRegion: String = "" - - /// 语音识别相关 - private var speechConfig: SPXSpeechConfiguration? - private var recognizer: SPXSpeechRecognizer? - private var audioConfig: SPXAudioConfiguration? - private var pushStream: SPXPushAudioInputStream? - - /// 音频处理相关 - private var audioProcessor: CustomAudioProcessor? - private var isProcessingAudio = false - private var audioProcessingTimer: Timer? - - /// 状态标志 - private var isInitialized = false - private var _isContinuousRecognitionActive = false - - /// 当前语言和支持的语言 - private var currentLanguage = "zh-CN" - private var supportedLanguages: [String] = ["zh-CN", "en-US"] - private var isAutoDetectLanguage = false - - // MARK: - 初始化 - - init(eventHandler: @escaping (String, [String: Any]) -> Void) { - self.eventHandler = eventHandler - super.init() - } - - deinit { - dispose() - } - - // MARK: - ASR Service 接口实现 - - /// 初始化语音识别服务 - /// - Parameters: - /// - speechSubscriptionKey: Azure 语音服务订阅密钥 - /// - serviceRegion: Azure 服务区域 (如 eastasia) - /// - supportedLanguages: 支持的语言代码数组 (可选) - /// - Returns: 初始化是否成功 - func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool { - print("[AzureAsrHelper] 初始化 Azure 语音服务") - - // 检查配置是否为空 - if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { - print("[AzureAsrHelper] 错误: Azure 配置信息不完整") - eventHandler("error", ["message": "Azure 配置信息不完整"]) - return false - } - - // 释放之前的资源 - dispose() - - // 记录配置信息 - self.speechSubscriptionKey = speechSubscriptionKey - self.serviceRegion = serviceRegion - - // 设置语言 - if let languages = supportedLanguages, !languages.isEmpty { - self.supportedLanguages = languages - } - - // 根据支持的语言数量决定是否启用自动语言检测 - isAutoDetectLanguage = self.supportedLanguages.count >= 2 - - // 如果只有一种语言,设置为当前语言 - if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty { - currentLanguage = self.supportedLanguages[0] - } - - // 创建识别器和设置回调 - if !createRecognizerAndSetupCallbacks() { - return false - } - - print("[AzureAsrHelper] Azure 语音服务初始化成功") - isInitialized = true - return true - } - - /// 创建识别器并设置回调 - private func createRecognizerAndSetupCallbacks() -> Bool { - // 释放之前的 recognizer - recognizer = nil - audioConfig = nil - - do { - // 创建语音配置 - speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - - // 设置音频输入参数 - try setupAudioSession() - - // 创建自定义推送流,替代默认的麦克风输入 - pushStream = try SPXPushAudioInputStream() - audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) - - // 初始化自定义音频处理器 - audioProcessor = CustomAudioProcessor() - - // 设置语言配置 - if isAutoDetectLanguage { - // 设置自动语言检测 - speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode) - - // 创建自动语言检测配置 - let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) - - // 创建识别器 - recognizer = try SPXSpeechRecognizer( - speechConfiguration: speechConfig!, - autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig, - audioConfiguration: audioConfig! - ) - } else { - // 设置指定的识别语言 - speechConfig?.speechRecognitionLanguage = currentLanguage - - // 创建识别器 - recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) - } - - // 设置所有回调 - setupAllCallbacks() - - return true - } catch { - print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") - eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) - return false - } - } - - /// 设置音频会话 - private func setupAudioSession() throws { - let audioSession = AVAudioSession.sharedInstance() - - // 使用playAndRecord类别允许同时录音和播放 - try audioSession.setCategory(.playAndRecord, - mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 - options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) - - // 设置首选的输入和输出 - let currentRoute = audioSession.currentRoute - - // 获取当前是否连接了耳机或外部麦克风 - let hasHeadphones = currentRoute.outputs.contains { - $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP - } - - // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 - if !hasHeadphones { - try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 - - // 启用回音消除和噪声抑制 - try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 - } else { - // 耳机模式,可以使用不同的设置 - try audioSession.setMode(.voiceChat) - try audioSession.setInputGain(1.0) - } - - // 设置合适的采样率 - try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 - try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 - - // 激活音频会话 - try audioSession.setActive(true, options: .notifyOthersOnDeactivation) - - print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") - } - - /// 设置所有回调 - private func setupAllCallbacks() { - guard let recognizer = recognizer else { return } - - // 最终识别结果 - recognizer.addRecognizedEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("result", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 识别中事件 - recognizer.addRecognizingEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizingSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("recognizing", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 会话事件 - recognizer.addSessionStartedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - print("[AzureAsrHelper] 识别会话已开始") - self._isContinuousRecognitionActive = true - self.eventHandler("sessionStarted", [:]) - } - - recognizer.addSessionStoppedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - print("[AzureAsrHelper] 识别会话已结束") - self._isContinuousRecognitionActive = false - self.eventHandler("sessionStopped", [:]) - } - - // 取消事件 - recognizer.addCanceledEventHandler { [weak self] _, event in - guard let self = self else { return } - - let reason = event.reason.rawValue - let errorDetails = event.errorDetails ?? "未知错误" - - print("[AzureAsrHelper] 识别取消: \(errorDetails)") - - self.eventHandler("canceled", [ - "reason": reason, - "errorDetails": errorDetails - ]) - - self._isContinuousRecognitionActive = false - } - } - - /// 执行一次性语音识别 - /// - Returns: 是否成功启动识别 - func recognizeOnce() -> Bool { - if !isInitialized { - print("[AzureAsrHelper] 错误: 语音服务未初始化") - eventHandler("error", ["message": "语音服务未初始化"]) - return false - } - - // 如果正在连续识别,先停止 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 确保识别器已创建 - if recognizer == nil && !createRecognizerAndSetupCallbacks() { - return false - } - - do { - // 启动音频处理 - startAudioProcessing() - - // 通知会话开始 - eventHandler("sessionStarted", [:]) - - // 执行识别 - try recognizer?.recognizeOnceAsync { [weak self] result in - guard let self = self else { return } - - // 停止音频处理 - self.stopAudioProcessing() - - if result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: result) - self.eventHandler("result", [ - "text": result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } else if result.reason == SPXResultReason.noMatch { - print("[AzureAsrHelper] 无匹配结果") - self.eventHandler("noMatch", [:]) - } else if result.reason == SPXResultReason.canceled { - do { - let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) - let errorDetails = details.errorDetails ?? "未知错误" - self.eventHandler("error", ["message": "识别取消: \(errorDetails)"]) - } catch { - print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)") - self.eventHandler("error", ["message": "识别取消,无法获取详细原因"]) - } - } - } - - return true - } catch { - print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") - eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) - stopAudioProcessing() - return false - } - } - - /// 开始连续语音识别 - /// - Returns: 是否成功启动识别 - func startContinuousRecognition() -> Bool { - if !isInitialized { - print("[AzureAsrHelper] 错误: 语音服务未初始化") - eventHandler("error", ["message": "语音服务未初始化"]) - return false - } - - // 如果已经在进行连续识别,先停止 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 确保识别器已创建 - if recognizer == nil && !createRecognizerAndSetupCallbacks() { - return false - } - - // 重新确保音频设置正确 - do { - try setupAudioSession() - } catch { - print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") - } - - do { - // 启动音频处理 - startAudioProcessing() - - // 启动连续识别 - try recognizer?.startContinuousRecognition() - _isContinuousRecognitionActive = true - - print("[AzureAsrHelper] 连续识别开始") - return true - } catch { - print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") - eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) - _isContinuousRecognitionActive = false - stopAudioProcessing() - return false - } - } - - /// 停止连续语音识别 - /// - Returns: 是否成功停止识别 - func stopContinuousRecognition() -> Bool { - // 停止音频处理 - stopAudioProcessing() - - if !_isContinuousRecognitionActive || recognizer == nil { - return true - } - - do { - try recognizer?.stopContinuousRecognition() - _isContinuousRecognitionActive = false - print("[AzureAsrHelper] 连续识别已停止") - return true - } catch { - print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") - eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"]) - _isContinuousRecognitionActive = false - return false - } - } - - /// 检查连续识别是否活跃 - /// - Returns: 连续识别是否处于活跃状态 - func isContinuousRecognitionActive() -> Bool { - return _isContinuousRecognitionActive - } - - /// 释放资源 - func dispose() { - print("[AzureAsrHelper] 释放资源") - - // 停止音频处理 - stopAudioProcessing() - - // 停止连续识别 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 释放音频会话 - do { - try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) - } catch { - print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") - } - - // 释放资源 - recognizer = nil - speechConfig = nil - audioConfig = nil - pushStream = nil - audioProcessor = nil - - // 重置状态 - _isContinuousRecognitionActive = false - isInitialized = false - } - - /// 从结果中获取检测到的语言 - private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { - if isAutoDetectLanguage { - do { - let langResult = try SPXAutoDetectSourceLanguageResult(result) - return langResult.language ?? currentLanguage - } catch { - print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") - return currentLanguage - } - } else { - return currentLanguage - } - } - - // MARK: - 音频处理 - - /// 开始音频处理 - private func startAudioProcessing() { - guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } - - isProcessingAudio = true - - // 启动音频处理器 - if !audioProcessor.startRecord() { - print("[AzureAsrHelper] 错误: 启动音频处理器失败") - eventHandler("error", ["message": "启动音频处理器失败"]) - return - } - - // 启动音频处理定时器 - audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in - guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { - return - } - - // 读取处理后的音频数据 - var bytes = [UInt8](repeating: 0, count: 2560) - let bytesRead = processor.read(bytes: &bytes) - - if bytesRead > 0 { - // 推送数据到Azure语音服务 - let data = Data(bytes: bytes, count: bytesRead) - stream.write(data) - - // 通知音频数据可用 - self.eventHandler("audioData", ["data": bytes]) - } - } - - print("[AzureAsrHelper] 音频处理已启动") - } - - /// 停止音频处理 - private func stopAudioProcessing() { - // 停止定时器 - audioProcessingTimer?.invalidate() - audioProcessingTimer = nil - - // 停止音频处理器 - audioProcessor?.stopRecord() - - isProcessingAudio = false - print("[AzureAsrHelper] 音频处理已停止") - } -} - -// MARK: - 自定义音频处理器 - -@available(iOS 13.0, *) -class CustomAudioProcessor: NSObject { - // 音频单元 - private var ioUnit: AudioUnit? - - // 音频格式 - private var audioFormat: AudioStreamBasicDescription - - // 音频缓冲 - private var audioBufferList: AudioBufferList - private var audioList: [Float] = [] - private let audioListQueue = DispatchQueue(label: "audioListQueue") - - // 回音消除状态 - private var isEchoCancellationEnabled = true - - override init() { - // 设置音频格式 - 16kHz, 16位, 单声道 - audioFormat = AudioStreamBasicDescription( - mSampleRate: 16000.0, - mFormatID: kAudioFormatLinearPCM, - mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, - mBytesPerPacket: 2, - mFramesPerPacket: 1, - mBytesPerFrame: 2, - mChannelsPerFrame: 1, - mBitsPerChannel: 16, - mReserved: 0 - ) - - // 初始化音频缓冲 - audioBufferList = AudioBufferList( - mNumberBuffers: 1, - mBuffers: AudioBuffer( - mNumberChannels: 1, - mDataByteSize: 4096, - mData: malloc(4096) - ) - ) - - super.init() - } - - deinit { - stopRecord() - free(audioBufferList.mBuffers.mData) - } - - /// 启动音频处理 - /// - Returns: 是否成功启动 - func startRecord() -> Bool { - print("[CustomAudioProcessor] 配置音频单元") - - // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 - var ioUnitDescription = AudioComponentDescription( - componentType: kAudioUnitType_Output, - componentSubType: kAudioUnitSubType_VoiceProcessingIO, - componentManufacturer: kAudioUnitManufacturer_Apple, - componentFlags: 0, - componentFlagsMask: 0 - ) - - // 查找音频组件 - guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { - print("[CustomAudioProcessor] 错误: 未找到音频组件") - return false - } - - // 创建音频单元实例 - if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { - ioUnit = nil - return false - } - - // 启用输入端口 - var enableInput: UInt32 = 1 - let kInputBus: AudioUnitElement = 1 - let kOutputBus: AudioUnitElement = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Input, kInputBus, &enableInput, - UInt32(MemoryLayout.size)), "启用输入端口") { - return false - } - - // 禁用输出端口 (我们只需要输入) - var enableOutput: UInt32 = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Output, kOutputBus, - &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") { - return false - } - - // 设置缓冲区分配标志 - var flag: UInt32 = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, - kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") { - return false - } - - // 设置音频格式 - let size = UInt32(MemoryLayout.size) - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { - return false - } - - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { - return false - } - - // 启用回音消除 - if isEchoCancellationEnabled { - var echoCancellation: UInt32 = 1 - AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, - kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size)) - } - - // 设置输入回调 - 当有新音频数据时调用 - var inputCallback = AURenderCallbackStruct( - inputProc: CustomAudioProcessor.onAudioDataAvailable, - inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) - ) - - if checkError(AudioUnitSetProperty(ioUnit!, - kAudioOutputUnitProperty_SetInputCallback, - kAudioUnitScope_Global, kInputBus, - &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { - return false - } - - // 初始化音频单元 - var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") - while hasError { - Thread.sleep(forTimeInterval: 0.1) - hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") - } - - // 启动音频单元 - hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") - - print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") - return !hasError - } - - /// 停止音频处理 - func stopRecord() { - print("[CustomAudioProcessor] 停止音频处理器") - - if let ioUnit = ioUnit { - // 停止音频单元 - _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") - - // 关闭音频单元 - _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") - _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") - - self.ioUnit = nil - } - - // 清空音频数据缓冲 - audioListQueue.sync { - audioList.removeAll() - } - } - - /// 音频数据回调 - 当有新的音频数据可用时调用 - private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in - // 获取实例 - let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() - - // 计算预期数据大小 - let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame - - // 确保缓冲区足够大 - if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { - processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) - processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize - } - - // 渲染音频数据 - let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, - inBusNumber, inNumberFrames, &processor.audioBufferList), - "渲染音频数据") - - // 将Int16数据转换为浮点数据进行处理 - var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) - let buffer = processor.audioBufferList.mBuffers - let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) - - for j in 0...size)) { - // 归一化到[-1.0, 1.0]范围 - audioDataFloat[j] = Float(bufferData[j]) / 32768.0 - } - - // 应用附加处理 (如有需要) - // processor.applyAdditionalProcessing(&audioDataFloat) - - // 保存处理后的数据 - if status == noErr { - processor.audioListQueue.async { - processor.audioList.append(contentsOf: audioDataFloat) - } - } - - return status - } - - /// 读取处理后的音频数据 - /// - Parameter bytes: 输出字节数组 - /// - Returns: 读取的字节数 - func read(bytes: inout [UInt8]) -> Int { - return audioListQueue.sync { - // 如果没有数据,返回0 - if audioList.isEmpty { - return 0 - } - - // 确保有足够的数据 (至少1280个样本) - if audioList.count < 1280 { - return 0 - } - - // 读取一帧数据 (1280个样本) - let frameLength = 1280 - let buffer = Array(audioList.prefix(frameLength)) - audioList.removeFirst(frameLength) - - // 将浮点数据转回Int16格式 - var int16Data = buffer.map { Int16($0 * 32767) } - - // 转换为字节数组 - let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) - bytes = [UInt8](data) - - // 每个样本2字节 (16位PCM) - return frameLength * 2 - } - } - - /// 检查错误并打印日志 - /// - Parameters: - /// - status: 操作状态 - /// - operation: 操作描述 - /// - Returns: 是否发生错误 - private func checkError(_ status: OSStatus, _ operation: String) -> Bool { - if status != noErr { - print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") - return true - } - return false - } - - /// 检查OSStatus并返回状态 - /// - Parameters: - /// - status: 操作状态 - /// - operation: 操作描述 - /// - Returns: 原始状态 - private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { - if status != noErr { - print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") - } - return status - } -} \ No newline at end of file diff --git a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift deleted file mode 100644 index 141b77ef8..000000000 --- a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift +++ /dev/null @@ -1,18 +0,0 @@ -import Flutter -import UIKit - -public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { - public static func register(with registrar: FlutterPluginRegistrar) { - if #available(iOS 13.0, *) { - SwiftAzureSpeechRecognitionPlugin.register(with: registrar) - } else { - // 如果低于iOS 13.0,返回不支持的错误 - let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) - channel.setMethodCallHandler { (call, result) in - result(FlutterError(code: "UNSUPPORTED", - message: "需要iOS 13.0及以上系统", - details: nil)) - } - } - } -} \ No newline at end of file diff --git a/azure/ios/Classes/AzureTtsHelper.swift b/azure/ios/Classes/AzureTtsHelper.swift deleted file mode 100644 index 589df617c..000000000 --- a/azure/ios/Classes/AzureTtsHelper.swift +++ /dev/null @@ -1,427 +0,0 @@ -import Foundation -import MicrosoftCognitiveServicesSpeech -import AVFoundation - -/// Azure TTS工具类,负责实现TTS服务接口 -@available(iOS 13.0, *) -class AzureTtsHelper: NSObject { - // MARK: - 属性 - - /// 事件处理回调 - private var eventHandler: (String, [String: Any]) -> Void - - /// 语音配置信息 - private var speechSubscriptionKey: String = "" - private var serviceRegion: String = "" - - /// 语音合成配置 - private var speechConfig: SPXSpeechConfiguration? - - /// 语音合成器 - private var synthesizer: SPXSpeechSynthesizer? - - /// 是否初始化成功 - private var isInitialized = false - - /// 当前是否正在播放 - private var _isSpeaking = false - - /// 音频会话配置 - private var isAudioSessionConfigured = false - - // MARK: - 语音设置 - - /// 当前语音 - private var currentVoice = "zh-CN-XiaoxiaoNeural" - - /// 支持的语音映射 - private var voiceMap: [String: String] = [ - "zh-CN": "zh-CN-XiaoxiaoNeural", - "en-US": "en-US-JennyNeural", - "ja-JP": "ja-JP-NanamiNeural", - "ko-KR": "ko-KR-SunHiNeural", - "zh-TW": "zh-TW-HsiaoChenNeural", - "zh-HK": "zh-HK-HiuMaanNeural" - ] - - /// 当前语音合成参数 - private var currentSpeechRate = "0%" - private var currentPitch = "0%" - private var currentVolume = "100%" - - // MARK: - 初始化 - - init(eventHandler: @escaping (String, [String: Any]) -> Void) { - self.eventHandler = eventHandler - super.init() - } - - deinit { - dispose() - } - - // MARK: - TTS 接口实现 - - /// 初始化语音合成服务 - /// - Parameters: - /// - speechSubscriptionKey: Azure 语音服务订阅密钥 - /// - serviceRegion: Azure 服务区域 (如 eastasia) - /// - language: 语言代码 (默认 zh-CN) - /// - Returns: 初始化是否成功 - func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { - print("[AzureTtsHelper] 初始化语音合成服务") - - // 检查配置是否为空 - if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { - print("[AzureTtsHelper] 错误: Azure 配置信息不完整") - eventHandler("error", ["error": "Azure 配置信息不完整"]) - return false - } - - // 释放之前的资源 - dispose() - - // 记录配置信息 - self.speechSubscriptionKey = speechSubscriptionKey - self.serviceRegion = serviceRegion - - // 配置音频会话 - if !configureAudioSession() { - print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化") - } - - do { - // 创建语音配置 - speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - - // 设置默认语音 - let defaultVoice = getDefaultVoiceForLanguage(language) - currentVoice = defaultVoice - speechConfig?.speechSynthesisVoiceName = defaultVoice - - // 创建语音合成器 - synthesizer = try SPXSpeechSynthesizer(speechConfig!) - - // 设置事件处理器 - setupSynthesizerEvents() - - isInitialized = true - print("[AzureTtsHelper] TTS 引擎初始化成功") - - return true - } catch { - print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)") - eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"]) - return false - } - } - - /// 配置音频会话 - private func configureAudioSession() -> Bool { - let audioSession = AVAudioSession.sharedInstance() - do { - // 使用playback类别,但支持混合和空中播放 - try audioSession.setCategory(.playback, - mode: .spokenAudio, - options: [.mixWithOthers, .allowAirPlay, .duckOthers]) - - // 根据设备类型选择最佳配置 - let currentRoute = audioSession.currentRoute - let hasHeadphones = currentRoute.outputs.contains { - $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP - } - - // 优化音频路由 - if hasHeadphones { - // 耳机模式,使用默认设置 - try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟 - } else { - // 扬声器模式 - try audioSession.setPreferredIOBufferDuration(0.005) - } - - // 避免完全激活音频会话,因为ASR可能已经激活 - // 这里使用setActive(false)是为了不与ASR冲突 - if !audioSession.isOtherAudioPlaying { - try audioSession.setActive(true, options: .notifyOthersOnDeactivation) - } - - isAudioSessionConfigured = true - print("[AzureTtsHelper] 音频会话配置成功") - return true - } catch { - print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)") - isAudioSessionConfigured = false - return false - } - } - - /// 设置语音 - /// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") - /// - Returns: 设置是否成功 - func setVoice(voiceName: String) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - if voiceName.isEmpty { - print("[AzureTtsHelper] 错误: 声音名称为空") - eventHandler("error", ["error": "声音名称不能为空"]) - return false - } - - if voiceName == currentVoice { - print("[AzureTtsHelper] 已设置语音: \(voiceName)") - return true - } - - print("[AzureTtsHelper] 设置声音: \(voiceName)") - currentVoice = voiceName - - // 更新语音配置 - if let speechConfig = speechConfig { - speechConfig.speechSynthesisVoiceName = voiceName - return true - } - - return false - } - - /// 设置语音合成参数 - /// - Parameters: - /// - rate: 语速,范围 -100 到 100,默认为 0 - /// - pitch: 音调,范围 -100 到 100,默认为 0 - /// - volume: 音量,范围 0 到 100,默认为 100 - /// - Returns: 是否设置成功 - func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - // 转换参数格式 - currentSpeechRate = formatRateParam(rate) - currentPitch = formatPitchParam(pitch) - currentVolume = formatVolumeParam(volume) - - print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)") - return true - } - - /// 合成文本为语音并播放 - /// - Parameter text: 要合成的文本 - /// - Returns: 操作是否成功启动 - func speakText(text: String) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - if text.isEmpty { - print("[AzureTtsHelper] 警告: 要播放的文本为空") - return true - } - - // 确保音频会话已配置 - if !isAudioSessionConfigured { - _ = configureAudioSession() - } - - print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") - - // 生成SSML - let ssml = generateSsml(text: text) - - // 直接进行SSML合成 - return speakSsmlInternal(text: ssml) - } - - /// 内部SSML合成和播放 - private func speakSsmlInternal(text: String) -> Bool { - guard let synthesizer = synthesizer else { - print("[AzureTtsHelper] 错误: 合成器未初始化") - eventHandler("error", ["error": "合成器未初始化"]) - return false - } - - _isSpeaking = true - eventHandler("started", [:]) - - Task { - do { - // 使用异步方法进行合成并直接播放 - _ = try await synthesizer.startSpeakingSsml(text) - - } catch { - print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)") - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"]) - } - } - } - - return true - } - - /// 停止当前语音合成 - /// - Returns: 操作是否成功 - func stopSpeaking() -> Bool { - if !isInitialized || !_isSpeaking { - return true - } - - // 停止合成 - do { - try synthesizer?.stopSpeaking() - _isSpeaking = false - eventHandler("canceled", [:]) - print("[AzureTtsHelper] 已停止语音合成") - return true - } catch { - print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)") - eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"]) - return false - } - } - - /// 检查是否正在播放 - /// - Returns: 当前是否正在播放语音 - func isSpeaking() -> Bool { - return _isSpeaking - } - - /// 释放资源 - func dispose() { - try? stopSpeaking() - - // 释放合成器和配置 - synthesizer = nil - speechConfig = nil - - isInitialized = false - _isSpeaking = false - isAudioSessionConfigured = false - print("[AzureTtsHelper] TTS 引擎已释放") - } - - // MARK: - 私有辅助方法 - - /// 设置合成器事件处理 - private func setupSynthesizerEvents() { - guard let synthesizer = synthesizer else { return } - - // 添加书签到达事件处理 - synthesizer.addBookmarkReachedEventHandler { _, e in - print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"") - } - - // 合成完成事件 - synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in - guard let self = self else { return } - print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)") - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("completed", [:]) - } - } - - // 合成取消事件 - synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in - guard let self = self else { return } - - let result = e.result - do { - let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result) - print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)") - - if cancellationDetails.reason == SPXCancellationReason.error { - print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)") - print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")") - } - - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"]) - } - } catch { - print("[AzureTtsHelper] 获取取消详情时出错: \(error)") - - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成被取消"]) - } - } - } - - // 合成开始事件 - synthesizer.addSynthesisStartedEventHandler { _, _ in - // print("[AzureTtsHelper] 语音合成开始") - } - - // 合成中事件 - synthesizer.addSynthesizingEventHandler { _, _ in - // print("[AzureTtsHelper] 语音合成中") - } - } - - /// 生成 SSML 文本 - private func generateSsml(text: String) -> String { - return """ - - - - \(text) - - - - """ - } - - /// 格式化语速参数 - private func formatRateParam(_ rate: Int) -> String { - let clampedRate = rate.clamp(min: -100, max: 100) - if clampedRate == 0 { - return "0%" - } else if clampedRate < 0 { - return "\(Int(Double(clampedRate) * 0.9))%" - } else { - return "+\(clampedRate)%" - } - } - - /// 格式化音调参数 - private func formatPitchParam(_ pitch: Int) -> String { - let clampedPitch = pitch.clamp(min: -100, max: 100) - if clampedPitch == 0 { - return "0%" - } else { - return "\(Int(Double(clampedPitch) * 0.5))%" - } - } - - /// 格式化音量参数 - private func formatVolumeParam(_ volume: Int) -> String { - let clampedVolume = volume.clamp(min: 0, max: 100) - return "\(clampedVolume)%" - } - - /// 获取指定语言的默认语音 - private func getDefaultVoiceForLanguage(_ language: String) -> String { - return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural" - } -} - -// MARK: - 扩展 - -extension Int { - func clamp(min: Int, max: Int) -> Int { - if self < min { return min } - if self > max { return max } - return self - } -} \ No newline at end of file diff --git a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift deleted file mode 100644 index ea393e215..000000000 --- a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift +++ /dev/null @@ -1,259 +0,0 @@ -import Flutter -import UIKit -import MicrosoftCognitiveServicesSpeech -import AVFoundation - -@available(iOS 13.0, *) -public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { - private var azureChannel: FlutterMethodChannel - private var ttsChannel: FlutterMethodChannel - private var asrHelper: AzureAsrHelper - private var ttsHelper: AzureTtsHelper - private static var eventStreamHandler: AzureEventStreamHandler? - - // 创建方法到通道的映射 - private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]() - private static var asrMethodHandlers = [String: FlutterMethodCallHandler]() - - public static func register(with registrar: FlutterPluginRegistrar) { - // ASR通道 - let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) - - // TTS通道 - let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger()) - - // 设置ASR事件通道 - let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger()) - eventStreamHandler = AzureEventStreamHandler() - eventChannel.setStreamHandler(eventStreamHandler) - - let instance = SwiftAzureSpeechRecognitionPlugin( - azureChannel: channel, - ttsChannel: ttsChannel, - eventStreamHandler: eventStreamHandler! - ) - - // 直接设置各自通道的处理器 - channel.setMethodCallHandler(instance.handleAsrMethodCalls) - ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls) - } - - - - // 新增直接处理方法调用的函数 - private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - handleTtsMethod(call, result) - } - - private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - handleAsrMethod(call, result) - } - - - init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) { - self.azureChannel = azureChannel - self.ttsChannel = ttsChannel - - // 创建辅助类实例,使用自定义事件回调处理器 - let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in - DispatchQueue.main.async { - if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink { - var eventData = arguments - eventData["type"] = eventName - eventSink(eventData) - } - } - } - - asrHelper = AzureAsrHelper(eventHandler: eventHandler) - ttsHelper = AzureTtsHelper(eventHandler: eventHandler) - - super.init() - } - - private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - - let args = call.arguments as? Dictionary - - switch call.method { - case "initialize": - // 仅在初始化时读取必要参数 - guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { - let errorMsg = "语音订阅密钥不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) - return - } - - guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { - let errorMsg = "服务区域不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) - return - } - - let supportedLanguages = args?["supportedLanguages"] as? [String] ?? [] - - let success = asrHelper.initialize( - speechSubscriptionKey: speechSubscriptionKey, - serviceRegion: serviceRegion, - supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages - ) - result(success) - - case "startContinuousRecognition": - // 只有使用参数时才验证 - let success = asrHelper.startContinuousRecognition() - result(success) - - case "stopContinuousRecognition": - // 不需要额外参数 - let success = asrHelper.stopContinuousRecognition() - result(success) - - case "recognizeOnce": - // 只有使用参数时才验证 - let success = asrHelper.recognizeOnce() - result(success) - - case "isContinuousRecognitionActive": - // 不需要额外参数 - result(asrHelper.isContinuousRecognitionActive()) - - case "dispose": - // 不需要额外参数 - print("[AzurePlugin] 释放ASR资源") - asrHelper.dispose() - result(true) - - default: - print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)") - result(FlutterMethodNotImplemented) - } - } - - private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - - let args = call.arguments as? Dictionary - - switch call.method { - case "initialize": - // 仅在初始化时验证参数 - guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { - let errorMsg = "语音订阅密钥不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) - return - } - - guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { - let errorMsg = "服务区域不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) - return - } - - let language = args?["language"] as? String ?? "zh-CN" - - print("[AzurePlugin] 初始化TTS,语言: \(language)") - - let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language) - result(success) - - case "setVoice": - // 仅获取voice参数 - guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else { - let errorMsg = "声音名称不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil)) - return - } - - print("[AzurePlugin] 设置声音: \(voiceName)") - - let success = ttsHelper.setVoice(voiceName: voiceName) - result(success) - - case "speakText": - // 仅获取text参数 - let text = args?["text"] as? String ?? "" - - if text.isEmpty { - print("[AzurePlugin] 警告: 要播放的文本为空") - result("OK") - return - } - - print("[AzurePlugin] 播放文本: \(text.prefix(50))...") - - let success = ttsHelper.speakText(text: text) - result(success ? "OK" : "ERROR") - - case "speakSsml": - // 仅获取ssml参数 - guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else { - let errorMsg = "SSML内容不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil)) - return - } - - print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...") - - // 由于我们移除了speakSsml方法,这里改用speakText方法 - // Azure SDK内部会自动检测是普通文本还是SSML - let success = ttsHelper.speakText(text: ssml) - result(success) - - case "stopSpeaking": - // 不需要参数 - print("[AzurePlugin] 停止播放") - let success = ttsHelper.stopSpeaking() - result(success) - - case "isSpeaking": - // 不需要参数 - result(ttsHelper.isSpeaking()) - - case "setSpeechParams": - // 仅获取语音参数 - let rate = args?["rate"] as? Int ?? 0 - let pitch = args?["pitch"] as? Int ?? 0 - let volume = args?["volume"] as? Int ?? 100 - - print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)") - let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume) - result(success) - - case "dispose": - // 释放TTS资源 - print("[AzurePlugin] 释放TTS资源") - ttsHelper.dispose() - result(true) - - default: - print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)") - result(FlutterMethodNotImplemented) - } - } -} - -// 用于处理事件流的辅助类 -@available(iOS 13.0, *) -class AzureEventStreamHandler: NSObject, FlutterStreamHandler { - var eventSink: FlutterEventSink? - - func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { - self.eventSink = events - // 通知Flutter端事件通道已准备好 - DispatchQueue.main.async { - events(["type": "channelReady"]) - } - return nil - } - - func onCancel(withArguments arguments: Any?) -> FlutterError? { - self.eventSink = nil - return nil - } -} \ No newline at end of file diff --git a/azure/ios/azure_speech_recognition.podspec b/azure/ios/azure_speech_recognition.podspec deleted file mode 100644 index 88797fb0c..000000000 --- a/azure/ios/azure_speech_recognition.podspec +++ /dev/null @@ -1,24 +0,0 @@ -# -# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html. -# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing. -# -Pod::Spec.new do |s| - s.name = 'azure_speech_recognition' - s.version = '0.1.0' - s.summary = 'Azure Speech Recognition plugin for Flutter' - s.description = <<-DESC -A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. - DESC - s.homepage = 'https://github.com/yourusername/azure_speech_recognition' - s.license = { :type => 'MIT', :file => '../LICENSE' } - s.author = { 'Your Company' => 'your-email@example.com' } - s.source = { :path => '.' } - s.source_files = 'Classes/**/*' - s.dependency 'Flutter' - s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0' - s.platform = :ios, '12.0' - - # Flutter.framework does not contain a i386 slice. - s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' } - s.swift_version = '5.0' -end \ No newline at end of file diff --git a/azure/lib/azure_speech_recognition.dart b/azure/lib/azure_speech_recognition.dart deleted file mode 100644 index 02b3f92d8..000000000 --- a/azure/lib/azure_speech_recognition.dart +++ /dev/null @@ -1,6 +0,0 @@ -// This is a placeholder file that exports nothing. -// The actual implementation is in the app's services folder. -// This file exists just to satisfy the Flutter plugin structure requirements. - -// Empty library to satisfy plugin structure -library azure_speech_recognition; \ No newline at end of file diff --git a/azure/pubspec.yaml b/azure/pubspec.yaml deleted file mode 100644 index a36c77bdc..000000000 --- a/azure/pubspec.yaml +++ /dev/null @@ -1,23 +0,0 @@ -name: azure_speech_recognition -description: Azure Speech Recognition and Text-to-Speech services Flutter plugin -version: 0.1.0 -homepage: https://github.com/yourusername/azure_speech_recognition - -environment: - sdk: '>=2.12.0 <3.0.0' - flutter: ">=2.0.0" - -dependencies: - flutter: - sdk: flutter - -dev_dependencies: - flutter_test: - sdk: flutter - flutter_lints: ^1.0.0 - -flutter: - plugin: - platforms: - ios: - pluginClass: AzureSpeechRecognitionPlugin \ No newline at end of file diff --git a/lib/data/services/speech_impl/azure_asr_service.dart b/lib/data/services/speech_impl/azure_asr_service.dart index ffb67ed46..e09defb66 100644 --- a/lib/data/services/speech_impl/azure_asr_service.dart +++ b/lib/data/services/speech_impl/azure_asr_service.dart @@ -10,8 +10,8 @@ import '../asr_service.dart'; /// 该服务提供了通过平台通道与 Android 上的 Microsoft Speech SDK 交互的接口 class AzureAsrService extends GetxService implements AsrService { static final AzureAsrService to = Get.put(AzureAsrService()); - static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_asr'); - static const EventChannel _eventChannel = EventChannel('com.deep_voice.azure_asr_events'); + static const MethodChannel _channel = MethodChannel('azure_speech/asr'); + static const EventChannel _eventChannel = EventChannel('azure_speech/asr_events'); bool _isInitialized = false; late final String _subscriptionKey; diff --git a/lib/data/services/speech_impl/azure_tts_service.dart b/lib/data/services/speech_impl/azure_tts_service.dart index ae5ab9369..1cdb62507 100644 --- a/lib/data/services/speech_impl/azure_tts_service.dart +++ b/lib/data/services/speech_impl/azure_tts_service.dart @@ -12,7 +12,7 @@ import '../tts_service.dart'; /// 提供文本转语音功能。 class AzureTtsService extends GetxService implements TtsService { static final AzureTtsService to = Get.put(AzureTtsService()); - static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_tts'); + static const MethodChannel _channel = MethodChannel('azure_speech/tts'); bool _isInitialized = false; late final String _subscriptionKey; diff --git a/lib/data/services/voice_interaction_service.dart b/lib/data/services/voice_interaction_service.dart index a1d26d509..1bb2406f5 100644 --- a/lib/data/services/voice_interaction_service.dart +++ b/lib/data/services/voice_interaction_service.dart @@ -38,7 +38,8 @@ class VoiceInteractionService extends GetxService { // 配置信息 late String _azureSpeechKey; late String _azureSpeechRegion; - late String _volcanoAiKey; + late String _openaiApiKey; + late String _openaiBaseUrl; // 聊天历史服务 late final ChatHistoryService _chatHistoryService; @@ -57,15 +58,13 @@ class VoiceInteractionService extends GetxService { void _loadConfig() { _azureSpeechKey = dotenv.env['AZURE_SPEECH_KEY'] ?? ''; _azureSpeechRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? ''; - _volcanoAiKey = dotenv.env['VOLCANO_AI_API_KEY'] ?? ''; + _openaiApiKey = dotenv.env['OPENAI_API_KEY'] ?? ''; + _openaiBaseUrl = dotenv.env['OPENAI_BASE_URL'] ?? ''; if (_azureSpeechKey.isEmpty || _azureSpeechRegion.isEmpty) { Logger.warning('未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION'); } - if (_volcanoAiKey.isEmpty) { - Logger.warning('未找到火山 AI API 密钥。请在 .env 文件中设置 VOLCANO_AI_API_KEY'); - } } /// 处理来自原生层的事件 @@ -195,7 +194,8 @@ class VoiceInteractionService extends GetxService { final result = await _channel.invokeMethod('startService', { 'azure_speech_key': _azureSpeechKey, 'azure_speech_region': _azureSpeechRegion, - 'volcano_ai_api_key': _volcanoAiKey, + 'openai_api_key': _openaiApiKey, + 'openai_base_url': _openaiBaseUrl, }) ?? false; if (result) { diff --git a/local_plugins/azure_speech/README.md b/local_plugins/azure_speech/README.md new file mode 100644 index 000000000..067b344eb --- /dev/null +++ b/local_plugins/azure_speech/README.md @@ -0,0 +1,136 @@ +# Azure Speech 插件 + +本插件为Flutter提供了Azure语音服务的集成,包括: + +- 语音合成(TTS) +- 语音识别(ASR) + +## 功能 + +### 语音合成(TTS) + +- 支持多种语音(如中文、英文等) +- 语音参数调整(语速、音调、音量) +- 音频输出设备选择(扬声器、听筒、自动) +- SSML支持 + +### 语音识别(ASR) + +- 一次性语音识别 +- 连续语音识别 +- 自动语言检测 +- 音频处理优化(回音消除、噪声抑制等) + +## 平台支持 + +- Android +- iOS + +## 如何使用 + +### 初始化 + +```dart +import 'package:azure_speech/azure_speech.dart'; + +// 初始化TTS +await AzureSpeech.initializeTts( + 'your_subscription_key', + 'your_service_region', + language: 'zh-CN', +); + +// 初始化ASR +await AzureSpeech.initializeAsr( + 'your_subscription_key', + 'your_service_region', + ['zh-CN', 'en-US'], +); +``` + +### 语音合成 + +```dart +// 设置语音 +await AzureSpeech.setTtsVoice('zh-CN-XiaoxiaoNeural'); + +// 设置语音参数 +await AzureSpeech.setTtsSpeechParams( + rate: 0, // 语速 -100~100 + pitch: 0, // 音调 -100~100 + volume: 100, // 音量 0~100 +); + +// 设置音频输出设备 +await AzureSpeech.setTtsAudioOutputType('AUTO'); // 'SPEAKER', 'EARPIECE', 'AUTO' + +// 播放文本 +await AzureSpeech.speakText('你好,世界!'); + +// 停止播放 +await AzureSpeech.stopSpeaking(); + +// 检查是否正在播放 +bool isSpeaking = await AzureSpeech.isSpeaking(); +``` + +### 语音识别 + +```dart +// 一次性识别 +final result = await AzureSpeech.recognizeOnce(); +if (result['success']) { + print('识别文本: ${result['text']}'); + print('识别语言: ${result['language']}'); +} else { + print('识别失败: ${result['error']}'); +} + +// 连续识别 +// 监听识别结果 +AzureSpeech.asrResultStream.listen((event) { + switch (event['eventType']) { + case 'recognizing': + // 实时识别中的结果 + print('识别中: ${event['text']}'); + break; + case 'finalResult': + // 最终识别结果 + print('最终结果: ${event['text']}'); + break; + case 'error': + // 错误 + print('错误: ${event['error']}'); + break; + } +}); + +// 开始连续识别 +await AzureSpeech.startContinuousRecognition(); + +// 停止连续识别 +await AzureSpeech.stopContinuousRecognition(); + +// 检查连续识别是否活跃 +bool isActive = await AzureSpeech.isContinuousRecognitionActive(); +``` + +### 释放资源 + +```dart +// 释放所有资源 +await AzureSpeech.dispose(); +``` + +## 依赖项 + +本插件依赖于: + +- Microsoft Cognitive Services Speech SDK +- Flutter + +## 注意事项 + +- 使用前需要在Azure门户中创建语音服务资源,并获取订阅密钥和区域 +- Android需要相关权限:RECORD_AUDIO, INTERNET等 +- iOS需要在Info.plist中添加麦克风使用权限描述 \ No newline at end of file diff --git a/local_plugins/azure_speech/android/build.gradle.kts b/local_plugins/azure_speech/android/build.gradle.kts new file mode 100644 index 000000000..665d495d1 --- /dev/null +++ b/local_plugins/azure_speech/android/build.gradle.kts @@ -0,0 +1,65 @@ +import com.android.build.gradle.LibraryExtension + +buildscript { + repositories { + google() + mavenCentral() + } + dependencies { + classpath("com.android.tools.build:gradle:7.3.0") + classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") + } +} + +allprojects { + repositories { + google() + mavenCentral() + } +} + +plugins { + id("com.android.library") + kotlin("android") +} + +// 配置android扩展 +configure { + namespace = "com.yunqiinnovation.azure_speech" + compileSdkVersion(33) + + defaultConfig { + minSdk = 21 + } + + compileOptions { + sourceCompatibility = JavaVersion.VERSION_1_8 + targetCompatibility = JavaVersion.VERSION_1_8 + } + + sourceSets { + getByName("main") { + manifest.srcFile("src/main/AndroidManifest.xml") + java.srcDirs("src/main/kotlin") + } + } + + // 添加lint选项 + lintOptions { + isCheckReleaseBuilds = false + } +} + +// 显式设置Kotlin JVM目标版本 +tasks.withType { + kotlinOptions { + jvmTarget = "1.8" + } +} + +dependencies { + // 直接通过本地依赖方式添加Flutter + implementation(fileTree(mapOf("dir" to "libs", "include" to listOf("*.jar")))) + // 添加Microsoft语音SDK + implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.30.0") +} \ No newline at end of file diff --git a/local_plugins/azure_speech/android/settings.gradle.kts b/local_plugins/azure_speech/android/settings.gradle.kts new file mode 100644 index 000000000..14ba24107 --- /dev/null +++ b/local_plugins/azure_speech/android/settings.gradle.kts @@ -0,0 +1 @@ +rootProject.name = "azure_speech" diff --git a/local_plugins/azure_speech/android/src/main/AndroidManifest.xml b/local_plugins/azure_speech/android/src/main/AndroidManifest.xml new file mode 100644 index 000000000..df2a770c6 --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/AndroidManifest.xml @@ -0,0 +1,6 @@ + + + + + \ No newline at end of file diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt new file mode 100644 index 000000000..ac6697cdb --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -0,0 +1,593 @@ +package com.yunqiinnovation.azure_speech + +import android.content.Context +import android.media.AudioAttributes +import android.media.AudioFormat +import android.media.AudioRecord +import android.media.MediaRecorder +import android.media.audiofx.AcousticEchoCanceler +import android.media.audiofx.NoiseSuppressor +import android.media.audiofx.AutomaticGainControl +import android.os.Process +import com.yunqiinnovation.azure_speech.utils.FileLogger +import com.microsoft.cognitiveservices.speech.* +import com.microsoft.cognitiveservices.speech.audio.* +import com.microsoft.cognitiveservices.speech.util.EventHandler +import java.util.concurrent.ExecutionException +import java.util.concurrent.atomic.AtomicBoolean + +class AzureAsrHelper(private val context: Context) { + private var recognizer: SpeechRecognizer? = null + private var speechConfig: SpeechConfig? = null + private val TAG = "AzureAsrHelper" + private var isContinuousRecognitionActive = false + private var currentLanguage = "zh-CN" + private var subscriptionKey = "" + private var region = "" + private var isAutoDetectLanguage = false + private var supportedLanguages = arrayOf("zh-CN", "en-US") + + // 是否使用回音消除 - 内部控制常量 + private val useEchoCancellation = false + + // 自定义音频处理相关 + private var customAudioProcessor: CustomAudioProcessor? = null + private var pushStream: PushAudioInputStream? = null + private var audioConfig: AudioConfig? = null + + // 初始化SDK并创建recognizer + fun initialize(subscriptionKey: String, region: String, + supportedLanguages: Array = arrayOf("zh-CN", "en-US")): Boolean { + try { + FileLogger.d(TAG, "初始化 Azure 语音服务") + + // 检查配置是否为空 + if (subscriptionKey.isEmpty() || region.isEmpty()) { + FileLogger.e(TAG, "Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + this.subscriptionKey = subscriptionKey + this.region = region + + // 设置语言 + if (supportedLanguages.isNotEmpty()) { + this.supportedLanguages = supportedLanguages + } + + // 根据支持的语言数量决定是否启用自动语言检测 + this.isAutoDetectLanguage = supportedLanguages.size >= 2 + + // 如果只有一种语言,设置为当前语言 + if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { + this.currentLanguage = supportedLanguages[0] + } + + // 创建语音配置 + speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) + + // 设置语言配置 + if (isAutoDetectLanguage) { + // 设置自动语言检测 + speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") + } else { + // 设置指定的识别语言 + speechConfig?.speechRecognitionLanguage = currentLanguage + } + + // 创建识别器 + try { + if (useEchoCancellation) { + // 如果使用回音消除,创建自定义音频输入流 + setupCustomAudioProcessing() + + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) + } else { + recognizer = SpeechRecognizer(speechConfig, audioConfig) + } + } else { + // 使用默认麦克风输入 + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) + } else { + recognizer = SpeechRecognizer(speechConfig) + } + } + + FileLogger.d(TAG, "Azure 语音服务初始化成功") + return true + } catch (e: Exception) { + FileLogger.e(TAG, "创建识别器失败: ${e.message}") + stopCustomAudioProcessing() + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "初始化失败: ${e.message}") + return false + } + } + + // 重置 recognizer + private fun resetRecognizer(): Boolean { + try { + // 释放之前的 recognizer + recognizer?.close() + recognizer = null + + // 停止当前的音频处理 + stopCustomAudioProcessing() + + // 使用现有配置重新创建 recognizer + if (speechConfig != null) { + if (useEchoCancellation) { + // 如果使用回音消除,创建自定义音频输入流 + setupCustomAudioProcessing() + + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) + } else { + recognizer = SpeechRecognizer(speechConfig, audioConfig) + } + } else { + // 使用默认麦克风输入 + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) + } else { + recognizer = SpeechRecognizer(speechConfig) + } + } + return true + } else { + FileLogger.e(TAG, "语音配置未初始化") + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "重置识别器失败: ${e.message}") + return false + } + } + + // 开始一次性语音识别 + fun recognizeOnce(callback: RecognizeCallback) { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return + } + + // 重置 recognizer + if (!resetRecognizer()) { + callback.onError("重置识别器失败") + return + } + + try { + // 启动音频处理 + startCustomAudioProcessing() + + // 执行识别 + val result = recognizer?.recognizeOnceAsync()?.get() + + // 停止音频处理 + stopCustomAudioProcessing() + + if (result != null && result.reason == ResultReason.RecognizedSpeech) { + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language + callback.onResult(result.text, detectedLanguage ?: "") + } else { + callback.onError("未能识别语音") + } + } catch (e: Exception) { + // 停止音频处理 + stopCustomAudioProcessing() + callback.onError("识别异常: ${e.message}") + } + } + + // 开始连续语音识别 + fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return false + } + + if (isContinuousRecognitionActive) { + FileLogger.w(TAG, "已在进行连续识别,忽略请求") + return true + } + + // 重置 recognizer + if (!resetRecognizer()) { + callback.onError("重置识别器失败") + return false + } + + try { + // 启动音频处理 + startCustomAudioProcessing() + + // 识别中事件 + recognizer?.recognizing?.addEventListener( + EventHandler { _, event -> + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language + FileLogger.d(TAG, "识别中: ${event.result.text}, 语言: $detectedLanguage") + callback.onRecognizing(event.result.text, detectedLanguage ?: "") + } + ) + + // 识别完成事件 + recognizer?.recognized?.addEventListener( + EventHandler { _, event -> + if (event.result.reason == ResultReason.RecognizedSpeech) { + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language + FileLogger.d(TAG, "识别完成: ${event.result.text}, 语言: $detectedLanguage") + callback.onResult(event.result.text, detectedLanguage ?: "") + } + } + ) + + // 会话开始事件 + recognizer?.sessionStarted?.addEventListener( + EventHandler { _, _ -> + FileLogger.d(TAG, "识别会话已开始") + callback.onSessionStarted() + callback.onSuccess("开始识别") // 兼容旧接口 + } + ) + + // 会话结束事件 + recognizer?.sessionStopped?.addEventListener( + EventHandler { _, _ -> + FileLogger.d(TAG, "识别会话已结束") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onSessionStopped() + } + ) + + // 取消事件 + recognizer?.canceled?.addEventListener( + EventHandler { _, event -> + val errorDetails = try { + event.errorDetails ?: "未知错误" + } catch (e: Exception) { + "未知错误" + } + val reason = event.reason.toString() + FileLogger.e(TAG, "识别取消: $errorDetails") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onCanceled(reason, errorDetails) + callback.onError("识别取消: $errorDetails") // 兼容旧接口 + } + ) + + // 开始连续识别 + recognizer?.startContinuousRecognitionAsync() + isContinuousRecognitionActive = true + FileLogger.d(TAG, "连续识别已启动") + + return true + } catch (e: Exception) { + // 停止音频处理 + stopCustomAudioProcessing() + FileLogger.e(TAG, "开始连续识别失败: ${e.message}") + e.printStackTrace() + callback.onError("开始连续识别失败: ${e.message}") + return false + } + } + + // 停止连续语音识别 + fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return false + } + + if (!isContinuousRecognitionActive) { + FileLogger.d(TAG, "未进行连续识别,忽略停止请求") + return true + } + + try { + FileLogger.d(TAG, "停止连续语音识别") + + if (recognizer == null) { + if (isContinuousRecognitionActive) { + FileLogger.w(TAG, "识别器为空,但状态显示活跃") + } + isContinuousRecognitionActive = false + return true + } + + // 停止连续识别 + val future = recognizer?.stopContinuousRecognitionAsync() + future?.get() + + // 停止音频处理 + stopCustomAudioProcessing() + + isContinuousRecognitionActive = false + FileLogger.d(TAG, "连续识别已停止") + callback.onSuccess("连续识别已停止") + + return true + } catch (e: Exception) { + // 强制重置状态 + isContinuousRecognitionActive = false + FileLogger.e(TAG, "停止连续识别失败: ${e.message}") + e.printStackTrace() + callback.onError("停止连续识别失败: ${e.message}") + + // 停止音频处理 + stopCustomAudioProcessing() + + // 尝试强制关闭识别器 + try { + recognizer?.close() + recognizer = null + } catch (ex: Exception) { + FileLogger.e(TAG, "关闭识别器失败: ${ex.message}") + } + + return false + } + } + + // 释放资源 + fun dispose() { + try { + // 如果正在进行连续识别,先停止 + if (isContinuousRecognitionActive) { + recognizer?.stopContinuousRecognitionAsync()?.get() + isContinuousRecognitionActive = false + } + + // 停止音频处理 + stopCustomAudioProcessing() + + // 释放recognizer + recognizer?.close() + recognizer = null + + // 释放speechConfig + speechConfig?.close() + speechConfig = null + + FileLogger.d(TAG, "资源已释放") + } catch (e: Exception) { + FileLogger.e(TAG, "释放资源失败: ${e.message}") + + // 确保状态被重置 + isContinuousRecognitionActive = false + customAudioProcessor = null + pushStream = null + audioConfig = null + recognizer = null + speechConfig = null + } + } + + // 检查连续识别是否处于活跃状态 + fun isContinuousRecognitionActive(): Boolean { + return isContinuousRecognitionActive + } + + // 设置自定义音频处理 + private fun setupCustomAudioProcessing() { + try { + // 创建音频推送流 + pushStream = PushAudioInputStream.create() + + // 创建音频配置 + audioConfig = AudioConfig.fromStreamInput(pushStream) + + // 创建自定义音频处理器 + customAudioProcessor = CustomAudioProcessor(pushStream!!) + } catch (e: Exception) { + FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") + e.printStackTrace() + } + } + + // 启动自定义音频处理 + private fun startCustomAudioProcessing() { + if (useEchoCancellation && customAudioProcessor != null) { + try { + customAudioProcessor?.startProcessing() + } catch (e: Exception) { + FileLogger.e(TAG, "启动音频处理器失败") + e.printStackTrace() + } + } + } + + // 停止自定义音频处理 + private fun stopCustomAudioProcessing() { + if (customAudioProcessor != null) { + try { + customAudioProcessor?.stopProcessing() + customAudioProcessor = null + } catch (e: Exception) { + FileLogger.e(TAG, "停止音频处理器失败: ${e.message}") + e.printStackTrace() + } + } + } + + // 修改认证取消事件处理代码 + private fun setupCancelledEventHandler(callback: ContinuousRecognizeCallback) { + recognizer?.canceled?.addEventListener( + EventHandler { _, event -> + val errorDetails = try { + event.errorDetails ?: "未知错误" + } catch (e: Exception) { + "未知错误" + } + val reason = event.reason.toString() + FileLogger.e(TAG, "识别取消: $errorDetails") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onCanceled(reason, errorDetails) + callback.onError("识别取消: $errorDetails") // 兼容旧接口 + } + ) + } + + // 自定义音频处理器 + private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream) { + private val isProcessing = AtomicBoolean(false) + private var audioRecord: AudioRecord? = null + private var echoCanceler: AcousticEchoCanceler? = null + private var noiseSuppressor: NoiseSuppressor? = null + private var automaticGainControl: AutomaticGainControl? = null + + // 音频配置 + private val sampleRate = 16000 // 16kHz,适合语音识别 + private val channelConfig = AudioFormat.CHANNEL_IN_MONO + private val audioFormat = AudioFormat.ENCODING_PCM_16BIT + + // 计算最小 buffer 大小 + private val bufferSize = AudioRecord.getMinBufferSize( + sampleRate, channelConfig, audioFormat + ) + + // 启动音频处理 + fun startProcessing() { + if (isProcessing.get()) return + + // 创建录音对象 + try { + audioRecord = AudioRecord( + MediaRecorder.AudioSource.VOICE_RECOGNITION, + sampleRate, + channelConfig, + audioFormat, + bufferSize * 2 // 使用更大的缓冲区以确保不会丢失数据 + ) + + // 创建音频处理效果 + if (AcousticEchoCanceler.isAvailable()) { + echoCanceler = AcousticEchoCanceler.create(audioRecord!!.audioSessionId) + echoCanceler?.enabled = true + } + + if (NoiseSuppressor.isAvailable()) { + noiseSuppressor = NoiseSuppressor.create(audioRecord!!.audioSessionId) + noiseSuppressor?.enabled = true + } + + if (AutomaticGainControl.isAvailable()) { + automaticGainControl = AutomaticGainControl.create(audioRecord!!.audioSessionId) + automaticGainControl?.enabled = true + } + + // 开始录音 + audioRecord?.startRecording() + + // 处理线程 + Thread { + android.os.Process.setThreadPriority(Process.THREAD_PRIORITY_AUDIO) + processAudio() + }.start() + + isProcessing.set(true) + FileLogger.d(TAG, "音频处理已启动") + } catch (e: Exception) { + FileLogger.e(TAG, "创建音频处理器失败: ${e.message}") + releaseAudioResources() + throw e + } + } + + // 停止音频处理 + fun stopProcessing() { + if (!isProcessing.get()) return + + isProcessing.set(false) + releaseAudioResources() + FileLogger.d(TAG, "音频处理已停止") + } + + // 释放音频资源 + private fun releaseAudioResources() { + try { + audioRecord?.stop() + + echoCanceler?.release() + echoCanceler = null + + noiseSuppressor?.release() + noiseSuppressor = null + + automaticGainControl?.release() + automaticGainControl = null + + audioRecord?.release() + audioRecord = null + } catch (e: Exception) { + FileLogger.e(TAG, "释放音频资源失败: ${e.message}") + } + } + + // 音频处理线程 + private fun processAudio() { + // 设置线程优先级 + try { + Process.setThreadPriority(Process.THREAD_PRIORITY_URGENT_AUDIO) + } catch (e: Exception) { + FileLogger.e(TAG, "设置线程优先级失败") + } + + val buffer = ByteArray(bufferSize) + + while (isProcessing.get()) { + try { + val readSize = audioRecord?.read(buffer, 0, buffer.size) ?: -1 + + if (readSize > 0) { + // 修复: 只传入buffer,不传readSize + // 创建新的byte数组,只包含读取到的数据 + val audioData = buffer.copyOfRange(0, readSize) + pushStream.write(audioData) + } + + // 适当休眠,避免占用过多 CPU + Thread.sleep(5) + } catch (e: Exception) { + if (isProcessing.get()) { + FileLogger.e(TAG, "处理音频数据异常: ${e.message}") + } + break + } + } + } + } + + // 一次性识别回调接口 + interface RecognizeCallback { + fun onResult(text: String, detectedLanguage: String) + fun onError(error: String) + } + + // 连续识别回调接口 + interface ContinuousRecognizeCallback { + fun onResult(text: String, detectedLanguage: String) + fun onRecognizing(recognizing: String, detectedLanguage: String) + fun onSessionStarted() + fun onSessionStopped() + fun onCanceled(reason: String, errorDetails: String) + fun onError(error: String) + + // 兼容旧版本的接口 + fun onSuccess(message: String) {} + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt new file mode 100644 index 000000000..090a72efc --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -0,0 +1,297 @@ +package com.yunqiinnovation.azure_speech + +import android.content.Context +import android.app.Activity +import android.os.Handler +import android.os.Looper +import androidx.annotation.NonNull +import com.yunqiinnovation.azure_speech.utils.FileLogger + +import io.flutter.embedding.engine.plugins.FlutterPlugin +import io.flutter.plugin.common.MethodCall +import io.flutter.plugin.common.MethodChannel +import io.flutter.plugin.common.MethodChannel.MethodCallHandler +import io.flutter.plugin.common.MethodChannel.Result +import io.flutter.plugin.common.EventChannel + +/** AzureSpeechPlugin */ +class AzureSpeechPlugin: FlutterPlugin { + private val TAG = "AzureSpeechPlugin" + private lateinit var context: Context + private val mainHandler = Handler(Looper.getMainLooper()) + + // ASR相关 + private lateinit var asrChannel : MethodChannel + private lateinit var asrEventChannel: EventChannel + private var asrEventSink: EventChannel.EventSink? = null + private lateinit var azureAsrHelper: AzureAsrHelper + + // TTS相关 + private lateinit var ttsChannel : MethodChannel + private lateinit var azureTtsHelper: AzureTtsHelper + + // ASR 事件发送方法 + private fun sendAsrEvent(event: Map) { + FileLogger.d(TAG, "发送ASR事件: $event") + if (asrEventSink == null) { + FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") + return + } + + mainHandler.post { + try { + asrEventSink?.success(event) + FileLogger.d(TAG, "ASR事件发送成功") + } catch (e: Exception) { + FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") + } + } + } + + override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { + context = flutterPluginBinding.applicationContext + + // 初始化ASR通道 + asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr") + asrChannel.setMethodCallHandler(AsrMethodHandler()) + + // 初始化TTS通道 + ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/tts") + ttsChannel.setMethodCallHandler(TtsMethodHandler()) + + // 初始化ASR事件通道 + asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events") + asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { + override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { + asrEventSink = events + } + + override fun onCancel(arguments: Any?) { + asrEventSink = null + } + }) + + // 初始化Azure语音服务 + azureTtsHelper = AzureTtsHelper(context) + azureAsrHelper = AzureAsrHelper(context) + } + + // ASR方法处理器 + inner class AsrMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + val subscriptionKey = call.argument("subscriptionKey") ?: "" + val region = call.argument("region") ?: "" + val supportedLanguages = call.argument>("supportedLanguages") ?: listOf("zh-CN") + + try { + val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages.toTypedArray()) + result.success(success) + } catch (e: Exception) { + result.error("INITIALIZATION_ERROR", e.message, null) + } + } + "recognizeOnce" -> { + azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) { + mainHandler.post { + result.success(mapOf( + "text" to text, + "detectedLanguage" to detectedLanguage + )) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("RECOGNITION_ERROR", error, null) + } + } + }) + } + "startContinuousRecognition" -> { + // 确保事件通道已准备好 + if (asrEventSink == null) { + result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) + return + } + + val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) { + sendAsrEvent(mapOf( + "type" to "result", + "text" to text, + "detectedLanguage" to detectedLanguage + )) + } + + override fun onRecognizing(recognizing: String, detectedLanguage: String) { + sendAsrEvent(mapOf( + "type" to "recognizing", + "text" to recognizing, + "detectedLanguage" to detectedLanguage + )) + } + + override fun onSessionStarted() { + sendAsrEvent(mapOf("type" to "sessionStarted")) + } + + override fun onSessionStopped() { + sendAsrEvent(mapOf("type" to "sessionStopped")) + } + + override fun onCanceled(reason: String, errorDetails: String) { + sendAsrEvent(mapOf( + "type" to "canceled", + "reason" to reason, + "errorDetails" to errorDetails + )) + } + + override fun onError(error: String) { + sendAsrEvent(mapOf("type" to "error", "message" to error)) + } + + override fun onSuccess(message: String) { + sendAsrEvent(mapOf("type" to "success", "message" to message)) + } + }) + result.success(success) + } + "stopContinuousRecognition" -> { + try { + if (!azureAsrHelper.isContinuousRecognitionActive()) { + result.success(true) + return + } + + val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) {} + override fun onRecognizing(recognizing: String, detectedLanguage: String) {} + override fun onSessionStarted() {} + override fun onSessionStopped() {} + override fun onCanceled(reason: String, errorDetails: String) {} + override fun onError(error: String) { + mainHandler.post { + result.error("STOP_ERROR", error, null) + } + } + override fun onSuccess(message: String) {} + }) + result.success(success) + } catch (e: Exception) { + result.error("STOP_ERROR", e.message, null) + } + } + "isContinuousRecognitionActive" -> { + result.success(azureAsrHelper.isContinuousRecognitionActive()) + } + "dispose" -> { + azureAsrHelper.dispose() + result.success(true) + } + else -> { + result.notImplemented() + } + } + } + } + + // TTS方法处理器 + inner class TtsMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + val subscriptionKey = call.argument("subscriptionKey") ?: "" + val region = call.argument("region") ?: "" + val language = call.argument("language") ?: "zh-CN" + + val success = azureTtsHelper.initialize(subscriptionKey, region, language) + result.success(success) + } + "setVoice" -> { + val voiceName = call.argument("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) + result.success(azureTtsHelper.setVoice(voiceName)) + } + "setSpeechParams" -> { + val rate = call.argument("rate") ?: 0 + val pitch = call.argument("pitch") ?: 0 + val volume = call.argument("volume") ?: 100 + result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) + } + "setAudioOutputType" -> { + val outputTypeStr = call.argument("outputType") ?: "speaker" + val outputType = when (outputTypeStr.lowercase()) { + "speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER + "earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE + "auto" -> AzureTtsHelper.AudioOutputType.AUTO + else -> AzureTtsHelper.AudioOutputType.SPEAKER + } + result.success(azureTtsHelper.setAudioOutputType(outputType)) + } + "speakText" -> { + val text = call.argument("text") ?: return result.error("INVALID_ARGUMENTS", "文本不能为空", null) + + azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + mainHandler.post { + result.success(true) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("SPEAK_ERROR", error, null) + } + } + }) + } + "speakSsml" -> { + val ssml = call.argument("ssml") ?: return result.error("INVALID_ARGUMENTS", "SSML不能为空", null) + + azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + mainHandler.post { + result.success(true) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("SPEAK_ERROR", error, null) + } + } + }) + } + "stopSpeaking" -> { + result.success(azureTtsHelper.stopSpeaking()) + } + "isSpeaking" -> { + result.success(azureTtsHelper.isSpeaking()) + } + "dispose" -> { + azureTtsHelper.dispose() + result.success(true) + } + else -> { + result.notImplemented() + } + } + } + } + + override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { + asrChannel.setMethodCallHandler(null) + ttsChannel.setMethodCallHandler(null) + asrEventChannel.setStreamHandler(null) + + try { + azureTtsHelper.dispose() + azureAsrHelper.dispose() + } catch (e: Exception) { + FileLogger.e(TAG, "Dispose resources error: ${e.message}") + } + } +} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt similarity index 98% rename from android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt rename to local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 31971c2a4..0ae180fa9 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -1,11 +1,11 @@ -package com.yunqiinnovation.deepsound +package com.yunqiinnovation.azure_speech import android.content.Context import android.media.AudioAttributes import android.media.AudioDeviceInfo import android.media.AudioManager import android.os.Build -import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.utils.FileLogger import com.microsoft.cognitiveservices.speech.* import com.microsoft.cognitiveservices.speech.audio.* import java.util.concurrent.Future @@ -50,16 +50,16 @@ class AzureTtsHelper(private val context: Context) { * 初始化 TTS 引擎 * * @param subscriptionKey Azure 语音服务订阅密钥 - * @param serviceRegion Azure 语音服务区域 + * @param region Azure 语音服务区域 * @param language 可选,默认语言,默认为 "zh-CN" */ - fun initialize(subscriptionKey: String, serviceRegion: String, language: String = "zh-CN"): Boolean { + fun initialize(subscriptionKey: String, region: String, language: String = "zh-CN"): Boolean { try { audioManager = context.getSystemService(Context.AUDIO_SERVICE) as AudioManager // 创建语音配置 - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) + speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) // 设置语音合成输出格式为高质量音频 speechConfig?.setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff24Khz16BitMonoPcm) @@ -96,7 +96,7 @@ class AzureTtsHelper(private val context: Context) { return true } catch (e: Exception) { - FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${serviceRegion}") + FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${region}") e.printStackTrace() return false } diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt new file mode 100644 index 000000000..d210ec48f --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt @@ -0,0 +1,45 @@ +package com.yunqiinnovation.azure_speech.utils + +import android.util.Log + +/** + * 简单的文件日志工具类 + */ +object FileLogger { + private const val TAG_PREFIX = "AzureSpeech_" + + /** + * 记录调试信息 + */ + fun d(tag: String, message: String) { + Log.d("$TAG_PREFIX$tag", message) + } + + /** + * 记录信息 + */ + fun i(tag: String, message: String) { + Log.i("$TAG_PREFIX$tag", message) + } + + /** + * 记录警告信息 + */ + fun w(tag: String, message: String) { + Log.w("$TAG_PREFIX$tag", message) + } + + /** + * 记录错误信息 + */ + fun e(tag: String, message: String) { + Log.e("$TAG_PREFIX$tag", message) + } + + /** + * 记录异常 + */ + fun e(tag: String, message: String, throwable: Throwable) { + Log.e("$TAG_PREFIX$tag", message, throwable) + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift new file mode 100644 index 000000000..d3dd8d0b8 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift @@ -0,0 +1,461 @@ +import Foundation +import AVFoundation +import MicrosoftCognitiveServicesSpeech + +/// Azure 语音识别辅助类 +class AzureAsrHelper: NSObject { + private var recognizer: SPXSpeechRecognizer? + private var speechConfig: SPXSpeechConfig? + private var audioConfig: SPXAudioConfig? + private var initialized = false + private var isContinuousRecognitionActive = false + private var currentLanguage = "zh-CN" + private var subscriptionKey = "" + private var serviceRegion = "" + private var isAutoDetectLanguage = false + private var supportedLanguages = ["zh-CN", "en-US"] + + // 音频会话管理 + private let audioSession = AVAudioSession.sharedInstance() + + // 事件回调 + private var eventHandler: (([String: Any]) -> Void)? + + /// 设置事件处理器 + /// + /// - Parameter handler: 事件处理回调 + func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) { + self.eventHandler = handler + } + + /// 初始化语音识别服务 + /// + /// - Parameters: + /// - speechSubscriptionKey: Azure 语音服务订阅密钥 + /// - serviceRegion: Azure 语音服务区域 + /// - supportedLanguages: 支持的语言列表,默认为 ["zh-CN", "en-US"] + /// - Returns: 是否初始化成功 + func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String] = ["zh-CN", "en-US"]) -> Bool { + print("[AzureAsrHelper] 初始化 Azure 语音服务") + + // 检查配置是否为空 + if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + // 保存配置 + self.subscriptionKey = speechSubscriptionKey + self.serviceRegion = serviceRegion + + // 设置语言 + if supportedLanguages.isEmpty { + print("[AzureAsrHelper] 警告: 传入的支持语言列表为空,将使用默认语言") + } else { + self.supportedLanguages = supportedLanguages + } + + // 根据支持的语言数量决定是否启用自动语言检测 + self.isAutoDetectLanguage = supportedLanguages.count >= 2 + + // 如果只有一种语言,设置为当前语言 + if !isAutoDetectLanguage && !supportedLanguages.isEmpty { + self.currentLanguage = supportedLanguages[0] + } + + // 创建语音配置 + do { + speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) + + // 设置语言配置 + if isAutoDetectLanguage { + // 设置自动语言检测 + try speechConfig?.setPropertyTo("Continuous", byId: SPXPropertyId.SpeechServiceConnection_LanguageIdMode) + } else { + // 设置指定的识别语言 + speechConfig?.speechRecognitionLanguage = currentLanguage + } + + // 创建音频配置 - 使用默认麦克风 + audioConfig = SPXAudioConfig.default() + + // 创建识别器 + if isAutoDetectLanguage { + let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) + } else { + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) + } + + // 配置音频会话 + try configureAudioSession() + + initialized = true + print("[AzureAsrHelper] Azure 语音服务初始化成功") + return true + } catch { + print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") + return false + } + } + + /// 配置音频会话 + private func configureAudioSession() throws { + print("[AzureAsrHelper] 开始配置音频会话...") + + do { + // 设置音频会话类别和模式 + try audioSession.setCategory(.record, mode: .measurement, options: [.duckOthers, .allowBluetooth]) + try audioSession.setActive(true, options: .notifyOthersOnDeactivation) + } catch { + print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败") + throw error + } + } + + /// 一次性语音识别 + /// + /// - Parameter completion: 完成回调,返回是否成功、识别文本、识别语言和可能的错误信息 + func recognizeOnce(completion: @escaping (Bool, String?, String?, String?) -> Void) { + if !initialized { + completion(false, nil, nil, "语音服务未初始化") + return + } + + // 重置 recognizer + if !resetRecognizer() { + completion(false, nil, nil, "重置识别器失败") + return + } + + do { + // 激活音频会话 + try audioSession.setActive(true) + + // 添加识别事件处理 + recognizer?.addRecognizedEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizedSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + completion(true, event.result.text, detectedLanguage, nil) + } + } + + recognizer?.addRecognizingEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizingSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + } + } + + // 添加会话事件处理 + recognizer?.addSessionStartedEventHandler { _, _ in + print("[AzureAsrHelper] 识别会话已开始") + } + + recognizer?.addSessionStoppedEventHandler { _, _ in + print("[AzureAsrHelper] 识别会话已结束") + } + + // 添加取消事件处理 + recognizer?.addCanceledEventHandler { _, event in + if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { + let errorDetails = cancellationDetails.errorDetails ?? "未知错误" + print("[AzureAsrHelper] 识别取消: \(errorDetails)") + completion(false, nil, nil, "识别取消: \(errorDetails)") + } + } + + // 执行识别 + let result = try recognizer?.recognizeOnceAsync().get() + + if result?.reason != .recognizedSpeech { + completion(false, nil, nil, "未能识别语音") + } + } catch { + completion(false, nil, nil, "识别异常: \(error.localizedDescription)") + } + } + + /// 重置识别器 + /// + /// - Returns: 是否重置成功 + private func resetRecognizer() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + // 检查配置是否为空 + if subscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + do { + // 释放之前的 recognizer + recognizer = nil + + // 创建音频配置 - 使用默认麦克风 + audioConfig = SPXAudioConfig.default() + + // 重新创建识别器 + if isAutoDetectLanguage { + let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) + } else { + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) + } + + return true + } catch { + print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") + return false + } + } + + /// 获取检测到的语言 + /// + /// - Parameter result: 识别结果 + /// - Returns: 检测到的语言代码 + private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { + if isAutoDetectLanguage { + do { + if let autoDetectResult = try SPXAutoDetectSourceLanguageResult(fromRecognitionResult: result) { + return autoDetectResult.language + } + return "" + } catch { + print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") + return "" + } + } else { + return currentLanguage + } + } + + /// 开始连续语音识别 + /// + /// - Returns: 是否成功启动连续识别 + func startContinuousRecognition() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + // 检查配置是否为空 + if subscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + // 如果已经在进行连续识别,直接返回 + if isContinuousRecognitionActive { + print("[AzureAsrHelper] 已经在进行连续识别中,忽略请求") + return true + } + + // 重置 recognizer + if !resetRecognizer() { + print("[AzureAsrHelper] 尝试重新创建识别器...") + return false + } + + do { + // 激活音频会话 + try audioSession.setActive(true) + + // 添加识别事件处理 + recognizer?.addRecognizedEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizedSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + let eventData: [String: Any] = [ + "eventType": "finalResult", + "text": event.result.text ?? "", + "language": detectedLanguage + ] + self.eventHandler?(eventData) + } + } + + // 识别中事件 + recognizer?.addRecognizingEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizingSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + let eventData: [String: Any] = [ + "eventType": "recognizing", + "text": event.result.text ?? "", + "language": detectedLanguage + ] + self.eventHandler?(eventData) + } + } + + // 会话事件 + recognizer?.addSessionStartedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + let eventData: [String: Any] = [ + "eventType": "sessionStarted" + ] + self.eventHandler?(eventData) + } + + recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + self.isContinuousRecognitionActive = false + let eventData: [String: Any] = [ + "eventType": "sessionStopped" + ] + self.eventHandler?(eventData) + } + + // 取消事件 + recognizer?.addCanceledEventHandler { [weak self] _, event in + guard let self = self else { return } + + self.isContinuousRecognitionActive = false + var errorMessage = "未知错误" + + if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { + errorMessage = cancellationDetails.errorDetails ?? "未知错误" + } + + let eventData: [String: Any] = [ + "eventType": "error", + "error": "识别取消: \(errorMessage)" + ] + self.eventHandler?(eventData) + } + + // 开始连续识别 + try recognizer?.startContinuousRecognition() + isContinuousRecognitionActive = true + print("[AzureAsrHelper] 连续识别已启动") + + return true + } catch { + print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") + return false + } + } + + /// 停止连续语音识别 + /// + /// - Returns: 是否成功停止连续识别 + func stopContinuousRecognition() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + if !isContinuousRecognitionActive { + print("[AzureAsrHelper] 未进行连续识别,忽略停止请求") + return true + } + + do { + print("[AzureAsrHelper] 停止连续语音识别") + + if recognizer == nil { + if isContinuousRecognitionActive { + print("[AzureAsrHelper] 警告: 识别器为空,但状态显示活跃") + } + isContinuousRecognitionActive = false + + // 通知停止成功 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止" + ] + eventHandler?(eventData) + return true + } + + // 停止连续识别 + try recognizer?.stopContinuousRecognition() + + // 延迟一点时间确保处理完成 + DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { [weak self] in + guard let self = self else { return } + + // 重置状态 + self.isContinuousRecognitionActive = false + + // 恢复音频会话 + do { + try self.audioSession.setActive(false, options: .notifyOthersOnDeactivation) + } catch { + // 忽略错误 + } + + print("[AzureAsrHelper] 连续识别已停止") + + // 通知停止成功 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止" + ] + self.eventHandler?(eventData) + } + + return true + } catch { + // 强制重置状态 + isContinuousRecognitionActive = false + print("[AzureAsrHelper] 警告: 停止连续识别失败: \(error.localizedDescription)") + + // 通知停止失败,但仍然视为处理完成 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止(但有错误)" + ] + eventHandler?(eventData) + + return false + } + } + + /// 释放资源 + func dispose() { + // 如果正在进行连续识别,先停止 + if isContinuousRecognitionActive { + _ = stopContinuousRecognition() + } + + // 恢复音频会话 + do { + try audioSession.setActive(false, options: .notifyOthersOnDeactivation) + } catch { + // 忽略错误 + } + + // 释放资源 + recognizer = nil + speechConfig = nil + audioConfig = nil + + initialized = false + isContinuousRecognitionActive = false + print("[AzureAsrHelper] 资源已释放") + } + + /// 检查连续识别是否处于活跃状态 + /// + /// - Returns: 是否正在进行连续识别 + func isContinuousRecognitionActive() -> Bool { + return isContinuousRecognitionActive + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift new file mode 100644 index 000000000..dcb2d57b8 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift @@ -0,0 +1,182 @@ +import Flutter +import UIKit + +public class AzureSpeechPlugin: NSObject, FlutterPlugin { + private var ttsHelper: AzureTtsHelper? + private var asrHelper: AzureAsrHelper? + private var eventSink: FlutterEventSink? + + public static func register(with registrar: FlutterPluginRegistrar) { + let channel = FlutterMethodChannel(name: "azure_speech", binaryMessenger: registrar.messenger()) + let instance = AzureSpeechPlugin() + registrar.addMethodCallDelegate(instance, channel: channel) + + // 初始化事件通道 + let eventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) + eventChannel.setStreamHandler(AsrStreamHandler(instance: instance)) + } + + override init() { + super.init() + ttsHelper = AzureTtsHelper() + asrHelper = AzureAsrHelper() + + // 设置ASR事件处理 + asrHelper?.setEventHandler { [weak self] event in + self?.handleAsrEvent(event) + } + } + + public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { + switch call.method { + // TTS相关方法 + case "initializeTts": + guard let args = call.arguments as? [String: Any], + let subscriptionKey = args["subscriptionKey"] as? String, + let serviceRegion = args["serviceRegion"] as? String else { + result(false) + return + } + + let language = args["language"] as? String ?? "zh-CN" + let success = ttsHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, language: language) ?? false + result(success) + + case "setTtsVoice": + guard let args = call.arguments as? [String: Any], + let voiceName = args["voiceName"] as? String else { + result(false) + return + } + + let success = ttsHelper?.setVoice(voiceName: voiceName) ?? false + result(success) + + case "setTtsSpeechParams": + guard let args = call.arguments as? [String: Any], + let rate = args["rate"] as? Int, + let pitch = args["pitch"] as? Int, + let volume = args["volume"] as? Int else { + result(false) + return + } + + let success = ttsHelper?.setSpeechParams(rate: rate, pitch: pitch, volume: volume) ?? false + result(success) + + case "speakText": + guard let args = call.arguments as? [String: Any], + let text = args["text"] as? String else { + result(false) + return + } + + ttsHelper?.speakText(text: text) { success, _ in + result(success) + } + + case "stopSpeaking": + let success = ttsHelper?.stopSpeaking() ?? false + result(success) + + case "isSpeaking": + let speaking = ttsHelper?.isSpeaking() ?? false + result(speaking) + + case "setTtsAudioOutputType": + guard let args = call.arguments as? [String: Any], + let outputType = args["outputType"] as? String else { + result(false) + return + } + + var type: AzureTtsHelper.AudioOutputType = .auto + switch outputType.uppercased() { + case "SPEAKER": + type = .speaker + case "EARPIECE": + type = .earpiece + default: + type = .auto + } + + let success = ttsHelper?.setAudioOutputType(outputType: type) ?? false + result(success) + + // ASR相关方法 + case "initializeAsr": + guard let args = call.arguments as? [String: Any], + let subscriptionKey = args["subscriptionKey"] as? String, + let serviceRegion = args["serviceRegion"] as? String, + let supportedLanguages = args["supportedLanguages"] as? [String] else { + result(false) + return + } + + let success = asrHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, supportedLanguages: supportedLanguages) ?? false + result(success) + + case "recognizeOnce": + asrHelper?.recognizeOnce { success, text, language, error in + var resultMap: [String: Any] = ["success": success] + if success { + resultMap["text"] = text + resultMap["language"] = language + } else { + resultMap["error"] = error + } + result(resultMap) + } + + case "startContinuousRecognition": + let success = asrHelper?.startContinuousRecognition() ?? false + result(success) + + case "stopContinuousRecognition": + let success = asrHelper?.stopContinuousRecognition() ?? false + result(success) + + case "isContinuousRecognitionActive": + let isActive = asrHelper?.isContinuousRecognitionActive() ?? false + result(isActive) + + case "dispose": + ttsHelper?.dispose() + asrHelper?.dispose() + result(nil) + + default: + result(FlutterMethodNotImplemented) + } + } + + // 设置事件接收器 + func setEventSink(_ sink: FlutterEventSink?) { + self.eventSink = sink + } + + // 处理ASR事件 + private func handleAsrEvent(_ event: [String: Any]) { + self.eventSink?(event) + } +} + +// ASR事件流处理器 +class AsrStreamHandler: NSObject, FlutterStreamHandler { + private weak var plugin: AzureSpeechPlugin? + + init(instance: AzureSpeechPlugin) { + self.plugin = instance + super.init() + } + + func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { + plugin?.setEventSink(events) + return nil + } + + func onCancel(withArguments arguments: Any?) -> FlutterError? { + plugin?.setEventSink(nil) + return nil + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift new file mode 100644 index 000000000..4242a9037 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift @@ -0,0 +1,330 @@ +import Foundation +import AVFoundation +import MicrosoftCognitiveServicesSpeech + +/// Azure 语音合成辅助类 +class AzureTtsHelper: NSObject { + private var synthesizer: SPXSpeechSynthesizer? + private var speechConfig: SPXSpeechConfig? + private var audioConfig: SPXAudioConfig? + private var initialized = false + private var speaking = false + + // 音频输出类型 + enum AudioOutputType { + case speaker // 扬声器 + case earpiece // 听筒 + case auto // 自动选择 + } + + // 当前设置 + private var currentVoiceName = "zh-CN-XiaoxiaoNeural" + private var currentSpeechRate = 0 + private var currentPitch = 0 + private var currentVolume = 100 + private var currentAudioOutputType: AudioOutputType = .auto + + // 音频会话管理 + private let audioSession = AVAudioSession.sharedInstance() + + /// 初始化 TTS 引擎 + /// + /// - Parameters: + /// - speechSubscriptionKey: Azure 语音服务订阅密钥 + /// - serviceRegion: Azure 语音服务区域 + /// - language: 可选,默认语言,默认为 "zh-CN" + /// - Returns: 是否初始化成功 + func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { + print("[AzureTtsHelper] 初始化 Azure 语音服务") + + // 检查配置是否为空 + if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureTtsHelper] 错误: Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + do { + // 创建语音配置 + speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) + + // 设置语音合成输出格式为高质量音频 + speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm) + + // 设置默认语言 + speechConfig?.setSpeechSynthesisLanguage(language) + + // 设置默认语音 + speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) + + // 创建音频配置 - 使用默认扬声器 + audioConfig = SPXAudioConfig.default() + + // 创建语音合成器 + synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) + + initialized = true + + // 设置默认音频输出类型为自动 + setAudioOutputType(outputType: .auto) + + print("[AzureTtsHelper] TTS 引擎初始化成功") + return true + } catch { + print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)") + return false + } + } + + /// 设置音频输出设备类型 + /// + /// - Parameter outputType: 音频输出设备类型 + /// - Returns: 是否设置成功 + func setAudioOutputType(outputType: AudioOutputType) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + do { + currentAudioOutputType = outputType + + switch outputType { + case .speaker: + // 使用扬声器 + try audioSession.setCategory(.playback, mode: .default) + try audioSession.overrideOutputAudioPort(.speaker) + print("[AzureTtsHelper] 已设置音频输出设备为扬声器") + + case .earpiece: + // 使用听筒 + try audioSession.setCategory(.playback, mode: .voiceChat) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为听筒") + + case .auto: + // 检查是否有耳机连接 + let outputs = audioSession.currentRoute.outputs + let hasHeadphones = outputs.contains { output in + return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP + } + + if hasHeadphones { + // 有耳机,使用耳机 + try audioSession.setCategory(.playback, mode: .default) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为耳机") + } else { + // 无耳机,使用听筒 + try audioSession.setCategory(.playback, mode: .voiceChat) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为听筒") + } + } + + try audioSession.setActive(true) + return true + } catch { + print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)") + return false + } + } + + /// 设置语音 + /// + /// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural" + /// - Returns: 是否设置成功 + func setVoice(voiceName: String) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + if voiceName == currentVoiceName { + print("[AzureTtsHelper] 已设置语音: \(voiceName)") + return true + } + + do { + currentVoiceName = voiceName + speechConfig?.setSpeechSynthesisVoiceName(voiceName) + + // 重新创建合成器 + synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) + + print("[AzureTtsHelper] 已设置语音: \(voiceName)") + return true + } catch { + print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)") + return false + } + } + + /// 设置语音合成参数 + /// + /// - Parameters: + /// - rate: 语速,范围 -100 到 100,默认为 0 + /// - pitch: 音调,范围 -100 到 100,默认为 0 + /// - volume: 音量,范围 0 到 100,默认为 100 + /// - Returns: 是否设置成功 + func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + currentSpeechRate = rate + currentPitch = pitch + currentVolume = volume + + print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)") + return true + } + + /// 合成文本为语音并播放 + /// + /// - Parameters: + /// - text: 要合成的文本 + /// - completion: 完成回调,返回是否成功和可能的错误信息 + func speakText(text: String, completion: @escaping (Bool, String?) -> Void) { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + completion(false, "TTS 引擎尚未初始化") + return + } + + do { + print("[AzureTtsHelper] 开始合成文本: \(text)") + + // 生成 SSML + let ssml = generateSsml(text: text) + + // 使用 SSML 合成语音 + speakSsml(ssml: ssml, completion: completion) + } catch { + print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") + completion(false, "语音合成异常: \(error.localizedDescription)") + } + } + + /// 生成 SSML 文本 + /// + /// - Parameter text: 要转换的文本 + /// - Returns: SSML 格式的文本 + private func generateSsml(text: String) -> String { + // 计算 SSML 参数 + let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%") + let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%" + let volumeParam = "\(min(max(currentVolume, 0), 100))%" + + return """ + + + + \(text) + + + + """ + } + + /// 合成 SSML 为语音并播放 + /// + /// - Parameters: + /// - ssml: SSML 格式的文本 + /// - completion: 完成回调,返回是否成功和可能的错误信息 + private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + completion(false, "TTS 引擎尚未初始化") + return + } + + do { + print("[AzureTtsHelper] 开始合成 SSML") + + // 标记为正在播放 + speaking = true + + // 激活音频会话 + try audioSession.setActive(true) + + // 异步合成语音 + let result = try synthesizer!.speakSsml(ssml) + + switch result.reason { + case .synthesizingAudioCompleted: + print("[AzureTtsHelper] 语音合成完成") + speaking = false + completion(true, "语音合成完成") + case .canceled: + if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) { + print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") + speaking = false + completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") + } else { + print("[AzureTtsHelper] 语音合成取消") + speaking = false + completion(false, "语音合成取消") + } + default: + print("[AzureTtsHelper] 语音合成失败: \(result.reason)") + speaking = false + completion(false, "语音合成失败: \(result.reason)") + } + } catch { + print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") + speaking = false + completion(false, "语音合成异常: \(error.localizedDescription)") + } + } + + /// 停止当前语音合成 + /// + /// - Returns: 是否停止成功 + func stopSpeaking() -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + do { + try synthesizer?.stopSpeaking() + speaking = false + print("[AzureTtsHelper] 已停止语音合成") + return true + } catch { + print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)") + return false + } + } + + /// 释放资源 + func dispose() { + do { + stopSpeaking() + + // 恢复音频会话 + try audioSession.setActive(false, options: .notifyOthersOnDeactivation) + + synthesizer = nil + speechConfig = nil + audioConfig = nil + + initialized = false + speaking = false + print("[AzureTtsHelper] TTS 引擎已释放") + } catch { + print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)") + } + } + + /// 检查当前是否正在播放语音 + /// + /// - Returns: 是否正在播放语音 + func isSpeaking() -> Bool { + return speaking + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/pubspec.yaml b/local_plugins/azure_speech/pubspec.yaml new file mode 100644 index 000000000..5935bbce2 --- /dev/null +++ b/local_plugins/azure_speech/pubspec.yaml @@ -0,0 +1,32 @@ +name: azure_speech +description: Azure语音服务插件,包含TTS和ASR服务 +version: 0.0.1 +homepage: + +environment: + sdk: ">=2.17.0 <3.0.0" + flutter: ">=2.5.0" + +dependencies: + flutter: + sdk: flutter + +dev_dependencies: + flutter_test: + sdk: flutter + flutter_lints: ^2.0.0 + +# For information on the generic Dart part of this file, see the +# following page: https://dart.dev/tools/pub/pubspec + +# The following section is specific to Flutter packages. +flutter: + # This section identifies this Flutter project as a plugin project. + plugin: + platforms: + android: + package: com.yunqiinnovation.azure_speech + pluginClass: AzureSpeechPlugin + ios: + pluginClass: AzureSpeechPlugin + diff --git a/pubspec.yaml b/pubspec.yaml index e4b2f4e6e..3e1dd7c90 100644 --- a/pubspec.yaml +++ b/pubspec.yaml @@ -60,6 +60,8 @@ dependencies: dio: ^5.8.0+1 sqflite: ^2.4.2 path: ^1.9.1 + azure_speech: + path: local_plugins/azure_speech dev_dependencies: flutter_test: