diff --git a/android/app/build.gradle.kts b/android/app/build.gradle.kts index 542cf24d8..8c371408e 100644 --- a/android/app/build.gradle.kts +++ b/android/app/build.gradle.kts @@ -34,6 +34,25 @@ android { jvmTarget = JavaVersion.VERSION_11.toString() } + // 添加lint选项 + lintOptions { + isCheckReleaseBuilds = false + } + + // 添加packaging配置,排除冲突文件 + packaging { + resources { + excludes.add("META-INF/DEPENDENCIES") + excludes.add("META-INF/LICENSE") + excludes.add("META-INF/LICENSE.txt") + excludes.add("META-INF/license.txt") + excludes.add("META-INF/NOTICE") + excludes.add("META-INF/NOTICE.txt") + excludes.add("META-INF/notice.txt") + excludes.add("META-INF/*.kotlin_module") + } + } + defaultConfig { // TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html). applicationId = "com.yunqiinnovation.deepsound" @@ -82,14 +101,19 @@ android { dependencies { + implementation("io.modelcontextprotocol:kotlin-sdk:0.4.0") + + + // 添加本地插件模块依赖 + implementation(project(":azure_speech")) + implementation(project(":open_ai_service")) + implementation(project(":volcano_speech")) // 添加OkHttp依赖 implementation("com.squareup.okhttp3:okhttp:4.9.3") // 添加核心库反糖化 coreLibraryDesugaring("com.android.tools:desugar_jdk_libs:2.0.3") - // Microsoft 语音识别SDK - implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.42.0") // 添加androidx.media依赖 implementation("androidx.media:media:1.6.0") @@ -100,11 +124,7 @@ dependencies { // 添加 AndroidX Security 加密 SharedPreferences 依赖 implementation("androidx.security:security-crypto:1.1.0-alpha06") - // // 添加火山语音合成SDK依赖 - // implementation("com.bytedance.speechengine:speechengine_tts_tob:5.4.8") - - // 添加火山语音识别SDK依赖 - // implementation("com.bytedance.speechengine:speechengine_tob:0.0.5") + } flutter { diff --git a/android/app/src/main/AndroidManifest.xml b/android/app/src/main/AndroidManifest.xml index f120cc1f7..23c380fb0 100644 --- a/android/app/src/main/AndroidManifest.xml +++ b/android/app/src/main/AndroidManifest.xml @@ -27,6 +27,15 @@ + + + + + + + + = arrayOf("zh-CN", "en-US")): Boolean { - try { - FileLogger.d(TAG, "初始化 Azure 语音服务") - - // 检查配置是否为空 - if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) { - FileLogger.e(TAG, "Azure 配置信息不完整") - return false - } - - // 释放之前的资源 - dispose() - - this.subscriptionKey = subscriptionKey - this.serviceRegion = serviceRegion - - // 设置语言 - if (supportedLanguages.isNotEmpty()) { - this.supportedLanguages = supportedLanguages - } - - // 根据支持的语言数量决定是否启用自动语言检测 - this.isAutoDetectLanguage = supportedLanguages.size >= 2 - - // 如果只有一种语言,设置为当前语言 - if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { - this.currentLanguage = supportedLanguages[0] - } - - // 创建语音配置 - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) - - // 设置语言配置 - if (isAutoDetectLanguage) { - // 设置自动语言检测 - speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") - } else { - // 设置指定的识别语言 - speechConfig?.speechRecognitionLanguage = currentLanguage - } - - // 创建识别器 - try { - if (useEchoCancellation) { - // 如果使用回音消除,创建自定义音频输入流 - setupCustomAudioProcessing() - - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) - } else { - recognizer = SpeechRecognizer(speechConfig, audioConfig) - } - } else { - // 使用默认麦克风输入 - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) - } else { - recognizer = SpeechRecognizer(speechConfig) - } - } - - FileLogger.d(TAG, "Azure 语音服务初始化成功") - return true - } catch (e: Exception) { - FileLogger.e(TAG, "创建识别器失败: ${e.message}") - stopCustomAudioProcessing() - return false - } - } catch (e: Exception) { - FileLogger.e(TAG, "初始化失败: ${e.message}") - return false - } - } - - // 重置 recognizer - private fun resetRecognizer(): Boolean { - try { - // 释放之前的 recognizer - recognizer?.close() - recognizer = null - - // 停止当前的音频处理 - stopCustomAudioProcessing() - - // 使用现有配置重新创建 recognizer - if (speechConfig != null) { - if (useEchoCancellation) { - // 如果使用回音消除,创建自定义音频输入流 - setupCustomAudioProcessing() - - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) - } else { - recognizer = SpeechRecognizer(speechConfig, audioConfig) - } - } else { - // 使用默认麦克风输入 - if (isAutoDetectLanguage) { - val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) - recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) - } else { - recognizer = SpeechRecognizer(speechConfig) - } - } - return true - } else { - FileLogger.e(TAG, "语音配置未初始化") - return false - } - } catch (e: Exception) { - FileLogger.e(TAG, "重置识别器失败: ${e.message}") - return false - } - } - - // 开始一次性语音识别 - fun recognizeOnce(callback: RecognizeCallback) { - if (speechConfig == null) { - callback.onError("语音服务未初始化") - return - } - - // 重置 recognizer - if (!resetRecognizer()) { - callback.onError("重置识别器失败") - return - } - - try { - // 启动音频处理 - startCustomAudioProcessing() - - // 执行识别 - val result = recognizer?.recognizeOnceAsync()?.get() - - // 停止音频处理 - stopCustomAudioProcessing() - - if (result != null && result.reason == ResultReason.RecognizedSpeech) { - val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language - callback.onResult(result.text, detectedLanguage ?: "") - } else { - callback.onError("未能识别语音") - } - } catch (e: Exception) { - // 停止音频处理 - stopCustomAudioProcessing() - callback.onError("识别异常: ${e.message}") - } - } - - // 开始连续语音识别 - fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { - if (speechConfig == null) { - callback.onError("语音服务未初始化") - return false - } - - // 如果已经在进行连续识别,先停止 - if (isContinuousRecognitionActive) { - stopContinuousRecognition(callback) - } - - // 重置 recognizer - if (!resetRecognizer()) { - callback.onError("重置识别器失败") - return false - } - - try { - // 启动音频处理 - startCustomAudioProcessing() - - // 设置识别事件处理 - // 最终识别结果 - recognizer?.recognized?.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizedSpeech) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" - } else { - currentLanguage - } - // FileLogger.d(TAG, "最终识别结果: ${event.result.text}") - callback.onResult(event.result.text, detectedLanguage) - } - } - ) - - // 识别中事件 - recognizer?.recognizing?.addEventListener( - EventHandler { _, event -> - if (event.result.reason == ResultReason.RecognizingSpeech) { - val detectedLanguage = if (isAutoDetectLanguage) { - AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" - } else { - currentLanguage - } - // FileLogger.d(TAG, "识别中结果: ${event.result.text}") - callback.onRecognizing(event.result.text, detectedLanguage) - } - } - ) - - // 会话事件 - recognizer?.sessionStarted?.addEventListener( - EventHandler { _, _ -> - isContinuousRecognitionActive = true - callback.onSessionStarted() - } - ) - - recognizer?.sessionStopped?.addEventListener( - EventHandler { _, _ -> - isContinuousRecognitionActive = false - callback.onSessionStopped() - } - ) - - // 取消事件 - recognizer?.canceled?.addEventListener( - EventHandler { _, event -> - val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else "" - callback.onCanceled(event.reason.toString(), errorDetails) - isContinuousRecognitionActive = false - } - ) - - // 开始连续识别 - recognizer?.startContinuousRecognitionAsync()?.get() - isContinuousRecognitionActive = true - - return true - } catch (e: Exception) { - callback.onError("开始连续识别失败: ${e.message}") - isContinuousRecognitionActive = false - stopCustomAudioProcessing() - return false - } - } - - // 停止连续语音识别 - fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { - if (!isContinuousRecognitionActive || recognizer == null) { - return true - } - - try { - recognizer?.stopContinuousRecognitionAsync() - isContinuousRecognitionActive = false - callback.onSessionStopped() - - // 停止音频处理 - stopCustomAudioProcessing() - - return true - } catch (e: Exception) { - callback.onError("停止连续识别失败: ${e.message}") - isContinuousRecognitionActive = false - stopCustomAudioProcessing() - return false - } - } - - // 检查连续识别是否活跃 - fun isContinuousRecognitionActive(): Boolean { - return isContinuousRecognitionActive - } - - // 设置自定义音频处理 - private fun setupCustomAudioProcessing() { - if (!useEchoCancellation) { - return - } - - try { - // 1. 创建PushAudioInputStream - pushStream = PushAudioInputStream.create() - - // 2. 创建AudioConfig - audioConfig = AudioConfig.fromStreamInput(pushStream) - - // 3. 创建自定义音频处理器 - customAudioProcessor = CustomAudioProcessor(pushStream) - - FileLogger.d(TAG, "自定义音频处理设置完成") - } catch (e: Exception) { - FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") - releaseCustomAudioProcessing() - - // 降级处理:如果自定义处理设置失败,尝试使用默认麦克风 - try { - FileLogger.d(TAG, "尝试降级到默认麦克风输入") - audioConfig = AudioConfig.fromDefaultMicrophoneInput() - } catch (e2: Exception) { - FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}") - audioConfig = null - } - } - } - - // 启动自定义音频处理 - private fun startCustomAudioProcessing() { - if (!useEchoCancellation || customAudioProcessor == null) { - return - } - - try { - customAudioProcessor?.startRecording() - FileLogger.d(TAG, "自定义音频处理已启动") - } catch (e: Exception) { - FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}") - } - } - - // 停止自定义音频处理 - private fun stopCustomAudioProcessing() { - if (!useEchoCancellation || customAudioProcessor == null) { - return - } - - try { - customAudioProcessor?.stopRecording() - FileLogger.d(TAG, "自定义音频处理已停止") - } catch (e: Exception) { - FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}") - } - } - - // 释放自定义音频处理资源 - private fun releaseCustomAudioProcessing() { - stopCustomAudioProcessing() - - try { - customAudioProcessor = null - pushStream?.close() - pushStream = null - audioConfig?.close() - audioConfig = null - - FileLogger.d(TAG, "自定义音频处理资源已释放") - } catch (e: Exception) { - FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}") - } - } - - // 释放所有资源 - fun dispose() { - try { - // 停止和释放音频处理 - releaseCustomAudioProcessing() - - recognizer?.close() - recognizer = null - - speechConfig?.close() - speechConfig = null - - isContinuousRecognitionActive = false - - FileLogger.d(TAG, "语音识别资源已释放") - } catch (e: Exception) { - FileLogger.e(TAG, "释放资源出错: ${e.message}") - } - } - - // 自定义音频处理器 - 使用Android原生回音消除 - private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) { - private val SAMPLE_RATE = 16000 - private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO - private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT - private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定 - - private var audioRecord: AudioRecord? = null - private var echoCanceler: AcousticEchoCanceler? = null - private val isRecording = AtomicBoolean(false) - private var recordingThread: Thread? = null - - // 启动录音并处理音频数据 - fun startRecording() { - if (isRecording.get() || pushStream == null) { - return - } - - try { - // 使用Builder模式构建AudioFormat - val audioFormat = AudioFormat.Builder() - .setSampleRate(SAMPLE_RATE) - .setEncoding(AUDIO_FORMAT) - .setChannelMask(CHANNEL_CONFIG) - .build() - - // 使用Builder模式创建AudioRecord实例 - audioRecord = AudioRecord.Builder() - .setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION) - .setAudioFormat(audioFormat) - .setBufferSizeInBytes(BUFFER_SIZE) - .build() - - // 检查AudioRecord初始化状态 - if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { - FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}") - // 尝试使用DEFAULT音频源重试一次 - audioRecord?.release() - audioRecord = AudioRecord.Builder() - .setAudioSource(MediaRecorder.AudioSource.DEFAULT) - .setAudioFormat(audioFormat) - .setBufferSizeInBytes(BUFFER_SIZE) - .build() - - if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { - FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃") - releaseAudioResources() - return - } else { - FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord") - } - } - - // 启用音频效果(回音消除、噪声抑制等) - enableAudioEffects() - - // 启动录音 - audioRecord?.startRecording() - isRecording.set(true) - - // 创建录音线程 - recordingThread = Thread({ - val buffer = ByteArray(BUFFER_SIZE) - - while (isRecording.get()) { - try { - val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0 - - if (readSize > 0) { - try { - // 将处理后的音频数据推送到流 - if (readSize == buffer.size) { - // 如果读取的大小等于buffer的大小,直接写入整个buffer - pushStream.write(buffer) - } else { - // 如果只读取了部分数据,创建新的数组只包含有效数据 - val validData = buffer.copyOfRange(0, readSize) - pushStream.write(validData) - } - } catch (e: Exception) { - FileLogger.e(TAG, "写入音频数据失败: ${e.message}") - break - } - } else if (readSize == 0) { - // 读取为0,可能是临时的,等待一下继续尝试 - Thread.sleep(10) - } else { - // 负值表示错误 - FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize") - break - } - } catch (e: Exception) { - FileLogger.e(TAG, "录音线程异常: ${e.message}") - break - } - } - }, "AudioRecordingThread") - - // 设置线程优先级并启动 - recordingThread?.priority = Thread.MAX_PRIORITY - recordingThread?.start() - - FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else "")) - } catch (e: Exception) { - FileLogger.e(TAG, "启动音频录制失败: ${e.message}") - releaseAudioResources() - } - } - - // 启用音频效果(回音消除、噪声抑制等) - private fun enableAudioEffects() { - try { - val audioSessionId = audioRecord?.audioSessionId ?: -1 - - if (audioSessionId != -1) { - // 启用回音消除 - if (AcousticEchoCanceler.isAvailable()) { - try { - echoCanceler = AcousticEchoCanceler.create(audioSessionId) - if (echoCanceler != null) { - echoCanceler?.enabled = true - FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId") - } else { - FileLogger.w(TAG, "回音消除器创建返回null") - } - } catch (e: Exception) { - FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}") - } - } else { - FileLogger.d(TAG, "设备不支持回音消除") - } - - // 以下功能暂时不启用,可根据需要取消注释 - /* - // 启用噪声抑制 - if (NoiseSuppressor.isAvailable()) { - try { - val ns = NoiseSuppressor.create(audioSessionId) - ns?.enabled = true - FileLogger.d(TAG, "噪声抑制已启用") - } catch (e: Exception) { - FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}") - } - } - - // 启用自动增益控制 - if (AutomaticGainControl.isAvailable()) { - try { - val agc = AutomaticGainControl.create(audioSessionId) - agc?.enabled = true - FileLogger.d(TAG, "自动增益控制已启用") - } catch (e: Exception) { - FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}") - } - } - */ - } else { - FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果") - } - } catch (e: Exception) { - FileLogger.e(TAG, "启用音频效果时出错: ${e.message}") - } - } - - // 停止录音 - fun stopRecording() { - if (!isRecording.get()) { - return - } - - isRecording.set(false) - - try { - // 等待录音线程结束 - recordingThread?.join(1000) - - // 释放资源 - releaseAudioResources() - - FileLogger.d(TAG, "音频录制已停止") - } catch (e: Exception) { - FileLogger.e(TAG, "停止音频录制失败: ${e.message}") - } - } - - // 释放音频资源 - private fun releaseAudioResources() { - try { - // 停止录音 - try { - if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) { - audioRecord?.stop() - } - } catch (e: Exception) { - // 忽略可能的IllegalStateException - FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}") - } - - // 释放回音消除器 - try { - if (echoCanceler != null) { - echoCanceler?.enabled = false - echoCanceler?.release() - echoCanceler = null - } - } catch (e: Exception) { - FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}") - } finally { - echoCanceler = null - } - - // 释放音频记录器 - try { - audioRecord?.release() - } catch (e: Exception) { - FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}") - } finally { - audioRecord = null - } - - // 重置线程 - recordingThread = null - - } catch (e: Exception) { - FileLogger.e(TAG, "释放音频资源失败: ${e.message}") - } - } - } - - // 一次性识别回调接口 - interface RecognizeCallback { - fun onResult(result: String, detectedLanguage: String = "") - fun onError(error: String) - } - - // 连续识别回调接口 - interface ContinuousRecognizeCallback { - fun onResult(result: String, detectedLanguage: String = "") - fun onRecognizing(recognizing: String, detectedLanguage: String = "") - fun onSessionStarted() - fun onSessionStopped() - fun onCanceled(reason: String, errorDetails: String) - fun onError(error: String) - } -} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt b/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt deleted file mode 100644 index 2fbcd8da8..000000000 --- a/android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt +++ /dev/null @@ -1,344 +0,0 @@ -package com.yunqiinnovation.deepsound - -import android.util.Log -import okhttp3.* -import okhttp3.MediaType.Companion.toMediaTypeOrNull -import okhttp3.RequestBody.Companion.toRequestBody -import org.json.JSONArray -import org.json.JSONObject -import java.io.IOException -import java.util.concurrent.CountDownLatch -import java.util.concurrent.TimeUnit - -/** - * 火山AI服务的原生实现 - * - * 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式 - */ -class VolcanoAIService() { - private val TAG = "VolcanoAIService" - private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3" - private val chatEndpoint = "/chat/completions" - private val client = OkHttpClient.Builder() - .connectTimeout(30, TimeUnit.SECONDS) - .readTimeout(30, TimeUnit.SECONDS) - .writeTimeout(30, TimeUnit.SECONDS) - .build() - - private var apiKey: String = "" - private var isInitialized = false - - /** - * 初始化火山AI服务 - * - * @param apiKey 火山AI API密钥 - * @return 初始化是否成功 - */ - fun initialize(apiKey: String): Boolean { - this.apiKey = apiKey - isInitialized = apiKey.isNotEmpty() - - if (!isInitialized) { - Log.e(TAG, "初始化失败:API key 不能为空") - } else { - Log.d(TAG, "火山AI服务初始化成功") - } - - return isInitialized - } - - /** - * 生成个性化问候语 - * - * @param agentName 代理名称 - * @param systemPrompt 系统提示词 - * @param callback 回调函数,返回生成的问候语 - */ - fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) { - val messages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - put(JSONObject().apply { - put("role", "user") - put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。") - }) - } - - sendMessageStream(messages, systemPrompt, object : StreamCallback { - val stringBuilder = StringBuilder() - - override fun onToken(token: String) { - stringBuilder.append(token) - } - - override fun onComplete() { - callback(stringBuilder.toString(), null) - } - - override fun onError(e: Exception) { - callback(null, e) - } - }) - } - - /** - * 发送消息(非流式输出) - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @return 返回AI的回复 - * @throws VolcanoAIException 如果API调用失败 - */ - @Throws(VolcanoAIException::class) - fun sendMessage(messages: JSONArray, systemPrompt: String): String { - // 检查是否已初始化 - if (!isInitialized || apiKey.isEmpty()) { - throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法") - } - - val fullMessages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - for (i in 0 until messages.length()) { - put(messages.getJSONObject(i)) - } - } - - val requestBody = JSONObject().apply { - put("model", "doubao-1-5-lite-32k-250115") - put("messages", fullMessages) - put("temperature", 0.7) - put("max_tokens", 2000) - put("stream", false) - } - - val mediaType = "application/json".toMediaTypeOrNull() - val request = Request.Builder() - .url("$baseUrl$chatEndpoint") - .addHeader("Content-Type", "application/json") - .addHeader("Authorization", "Bearer $apiKey") - .post(requestBody.toString().toRequestBody(mediaType)) - .build() - - try { - client.newCall(request).execute().use { response -> - if (!response.isSuccessful) { - val errorBody = response.body?.string() ?: "" - val errorMessage = try { - JSONObject(errorBody).getJSONObject("error").getString("message") - } catch (e: Exception) { - "Unknown error occurred" - } - throw VolcanoAIException(errorMessage) - } - - val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response") - val jsonResponse = JSONObject(responseBody) - - if (jsonResponse.has("choices") && - jsonResponse.getJSONArray("choices").length() > 0 && - jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) { - return jsonResponse.getJSONArray("choices") - .getJSONObject(0) - .getJSONObject("message") - .getString("content") - } - - throw VolcanoAIException("Invalid response format") - } - } catch (e: Exception) { - if (e is VolcanoAIException) throw e - throw VolcanoAIException("Failed to communicate with AI service: ${e.message}") - } - } - - /** - * 发送消息(流式输出) - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @param callback 回调函数,用于接收流式输出的结果 - */ - fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { - // 检查是否已初始化 - if (!isInitialized || apiKey.isEmpty()) { - callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")) - return - } - - val fullMessages = JSONArray().apply { - put(JSONObject().apply { - put("role", "system") - put("content", systemPrompt) - }) - for (i in 0 until messages.length()) { - put(messages.getJSONObject(i)) - } - } - - val requestBody = JSONObject().apply { - put("model", "doubao-1-5-lite-32k-250115") - put("messages", fullMessages) - put("temperature", 0.7) - put("max_tokens", 2000) - put("stream", true) - } - - val mediaType = "application/json".toMediaTypeOrNull() - val request = Request.Builder() - .url("$baseUrl$chatEndpoint") - .addHeader("Content-Type", "application/json") - .addHeader("Authorization", "Bearer $apiKey") - .addHeader("Accept", "text/event-stream") - .post(requestBody.toString().toRequestBody(mediaType)) - .build() - - client.newCall(request).enqueue(object : Callback { - override fun onFailure(call: Call, e: IOException) { - callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}")) - } - - override fun onResponse(call: Call, response: Response) { - if (!response.isSuccessful) { - val errorBody = response.body?.string() ?: "" - val errorMessage = try { - JSONObject(errorBody).getJSONObject("error").getString("message") - } catch (e: Exception) { - "Unknown error occurred" - } - callback.onError(VolcanoAIException(errorMessage)) - return - } - - val responseBody = response.body ?: return - val source = responseBody.source() - val bufferedSource = source.buffer - - try { - while (!bufferedSource.exhausted()) { - val line = bufferedSource.readUtf8Line() ?: continue - - if (line.isEmpty()) continue - if (line.startsWith("data: ")) { - val data = line.substring(6) - if (data == "[DONE]") { - callback.onComplete() - break - } - - try { - val jsonData = JSONObject(data) - if (jsonData.has("choices") && - jsonData.getJSONArray("choices").length() > 0 && - jsonData.getJSONArray("choices").getJSONObject(0).has("delta") && - jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) { - val content = jsonData.getJSONArray("choices") - .getJSONObject(0) - .getJSONObject("delta") - .getString("content") - callback.onToken(content) - } - } catch (e: Exception) { - // 忽略无效的JSON数据 - continue - } - } - } - } catch (e: Exception) { - callback.onError(VolcanoAIException("Error processing stream: ${e.message}")) - } finally { - response.close() - } - } - }) - } - - /** - * 同步方式发送消息(流式输出) - * - * 注意:此方法会阻塞当前线程,请在后台线程中调用 - * - * @param messages 消息列表 - * @param systemPrompt 系统提示词 - * @return 返回完整的AI回复 - * @throws VolcanoAIException 如果API调用失败 - */ - @Throws(VolcanoAIException::class) - fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String { - val result = StringBuilder() - val latch = CountDownLatch(1) - var exception: Exception? = null - - sendMessageStream(messages, systemPrompt, object : StreamCallback { - override fun onToken(token: String) { - result.append(token) - } - - override fun onComplete() { - latch.countDown() - } - - override fun onError(e: Exception) { - exception = e - latch.countDown() - } - }) - - // 等待流式输出完成或出错 - latch.await(60, TimeUnit.SECONDS) - - if (exception != null) { - throw exception as VolcanoAIException - } - - return result.toString() - } - - /** - * 创建用户消息 - */ - fun createUserMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "user") - put("content", content) - } - } - - /** - * 创建系统消息 - */ - fun createSystemMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "system") - put("content", content) - } - } - - /** - * 创建助手消息 - */ - fun createAssistantMessage(content: String): JSONObject { - return JSONObject().apply { - put("role", "assistant") - put("content", content) - } - } - - /** - * 流式输出回调接口 - */ - interface StreamCallback { - fun onToken(token: String) - fun onComplete() - fun onError(e: Exception) - } -} - -/** - * 火山AI异常 - */ -class VolcanoAIException(message: String) : Exception(message) \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt similarity index 100% rename from android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt diff --git a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt similarity index 67% rename from android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt index 49fd8fdf6..1c4717f11 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt @@ -1,5 +1,6 @@ package com.yunqiinnovation.deepsound +import android.Manifest import android.content.Intent import android.os.Build import android.os.Bundle @@ -20,33 +21,54 @@ import android.bluetooth.BluetoothDevice import androidx.security.crypto.EncryptedSharedPreferences import androidx.security.crypto.MasterKey import com.yunqiinnovation.deepsound.core.utils.FileLogger +import android.content.pm.PackageManager +import androidx.core.app.ActivityCompat +import androidx.core.content.ContextCompat class MainActivity: FlutterActivity() { - private val AZURE_ASR_CHANNEL = "com.deep_voice.azure_asr" - private val AZURE_ASR_EVENT_CHANNEL = "com.deep_voice.azure_asr_events" - private val AZURE_TTS_CHANNEL = "com.deep_voice.azure_tts" private val VOICE_INTERACTION_CHANNEL = "com.deep_voice.voice_interaction" private val VOICE_INTERACTION_EVENT_CHANNEL = "com.deep_voice.voice_interaction_events" private val CLASSIC_BLUETOOTH_CHANNEL = "com.deep_voice.classic_bluetooth" private val CLASSIC_BLUETOOTH_EVENT_CHANNEL = "com.deep_voice.classic_bluetooth_events" private val TAG = "MainActivity" - private lateinit var azureAsrHelper: AzureAsrHelper - private lateinit var azureTtsHelper: AzureTtsHelper private lateinit var classicBluetoothHelper: ClassicBluetoothHelper - private var azureAsrEventSink: EventChannel.EventSink? = null private var voiceInteractionEventSink: EventChannel.EventSink? = null private var bluetoothEventSink: EventChannel.EventSink? = null + // 添加权限请求相关常量 + private val PERMISSION_REQUEST_CODE = 100 + private val REQUIRED_PERMISSIONS = if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.S) { + arrayOf( + Manifest.permission.RECORD_AUDIO, + Manifest.permission.BLUETOOTH_CONNECT, + Manifest.permission.BLUETOOTH_SCAN, + Manifest.permission.ACCESS_FINE_LOCATION, + Manifest.permission.SEND_SMS, + Manifest.permission.READ_CONTACTS, + Manifest.permission.CALL_PHONE, + Manifest.permission.POST_NOTIFICATIONS + ) + } else { + arrayOf( + Manifest.permission.RECORD_AUDIO, + Manifest.permission.BLUETOOTH, + Manifest.permission.BLUETOOTH_ADMIN, + Manifest.permission.ACCESS_FINE_LOCATION, + Manifest.permission.SEND_SMS, + Manifest.permission.READ_CONTACTS, + Manifest.permission.CALL_PHONE + ) + } + // 广播接收器 private val voiceInteractionReceiver = object : BroadcastReceiver() { override fun onReceive(context: Context?, intent: Intent?) { - Log.d(TAG, "收到广播: ${intent?.action}") + FileLogger.d(TAG, "收到广播: ${intent?.action}") when (intent?.action) { VoiceInteractionService.ACTION_RECOGNITION_STARTED -> { val timestamp = intent.getLongExtra("timestamp", 0) - Log.d(TAG, "收到语音识别启动广播: timestamp=$timestamp") sendVoiceInteractionEvent(mapOf( "type" to "recognition_started", "timestamp" to timestamp @@ -58,12 +80,14 @@ class MainActivity: FlutterActivity() { val assistantMessage = intent.getStringExtra("assistantMessage") ?: "" val timestamp = intent.getLongExtra("timestamp", System.currentTimeMillis()) - Log.d(TAG, "收到聊天记录更新广播: agentId=$agentId, timestamp=$timestamp") - Log.d(TAG, "用户消息: ${userMessage.take(50)}...") - Log.d(TAG, "助手回复: ${assistantMessage.take(50)}...") - sendChatHistoryEvent(agentId, userMessage, assistantMessage, timestamp) } + VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE -> { + sendVoiceInteractionEvent(mapOf( + "type" to "enter_translation_mode", + "timestamp" to System.currentTimeMillis() + )) + } } } } @@ -143,13 +167,24 @@ class MainActivity: FlutterActivity() { var azureSpeechKey: String = "" var azureSpeechRegion: String = "" var volcanoAiApiKey: String = "" + var openaiApiKey: String = "" + var openaiBaseUrl: String? = null + var openaiModel: String? = null + var volcanoSpeechAppId: String = "" + var volcanoSpeechAppToken: String = "" // 安全存储相关常量 private const val SECURE_PREFS_FILENAME = "deep_voice_secure_prefs" private const val KEY_AZURE_SPEECH_KEY = "azure_speech_key" private const val KEY_AZURE_SPEECH_REGION = "azure_speech_region" private const val KEY_VOLCANO_AI_API_KEY = "volcano_ai_api_key" - + private const val KEY_OPENAI_API_KEY = "openai_api_key" + private const val KEY_OPENAI_BASE_URL = "openai_base_url" + private const val KEY_OPENAI_MODEL = "openai_model" + private const val KEY_MCP_SERVER_ENDPOINT = "mcp_server_endpoint" + private const val KEY_VOLCANO_SPEECH_APP_ID = "volcano_speech_app_id" + private const val KEY_VOLCANO_SPEECH_APP_TOKEN = "volcano_speech_app_token" + // 会话管理 private const val KEY_SESSION_ID = "session_id" private var currentSessionId = "" @@ -183,7 +218,12 @@ class MainActivity: FlutterActivity() { .putString(KEY_AZURE_SPEECH_KEY, azureSpeechKey) .putString(KEY_AZURE_SPEECH_REGION, azureSpeechRegion) .putString(KEY_VOLCANO_AI_API_KEY, volcanoAiApiKey) + .putString(KEY_OPENAI_API_KEY, openaiApiKey) + .putString(KEY_OPENAI_BASE_URL, openaiBaseUrl) .putString(KEY_SESSION_ID, currentSessionId) + .putString(KEY_OPENAI_MODEL, openaiModel) + .putString(KEY_VOLCANO_SPEECH_APP_ID, volcanoSpeechAppId) + .putString(KEY_VOLCANO_SPEECH_APP_TOKEN, volcanoSpeechAppToken) .apply() FileLogger.d("MainActivity", "密钥已安全保存到加密存储中") @@ -227,11 +267,16 @@ class MainActivity: FlutterActivity() { azureSpeechKey = sharedPreferences.getString(KEY_AZURE_SPEECH_KEY, "") ?: "" azureSpeechRegion = sharedPreferences.getString(KEY_AZURE_SPEECH_REGION, "") ?: "" volcanoAiApiKey = sharedPreferences.getString(KEY_VOLCANO_AI_API_KEY, "") ?: "" - + openaiApiKey = sharedPreferences.getString(KEY_OPENAI_API_KEY, "") ?: "" + openaiBaseUrl = sharedPreferences.getString(KEY_OPENAI_BASE_URL, null) + openaiModel = sharedPreferences.getString(KEY_OPENAI_MODEL, null) + volcanoSpeechAppId = sharedPreferences.getString(KEY_VOLCANO_SPEECH_APP_ID, "") ?: "" + volcanoSpeechAppToken = sharedPreferences.getString(KEY_VOLCANO_SPEECH_APP_TOKEN, "") ?: "" FileLogger.d("MainActivity", "已从加密存储加载密钥") - // 检查是否成功获取所有密钥 - return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && volcanoAiApiKey.isNotEmpty() + // 检查是否成功获取所有必要密钥 + return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && + (openaiApiKey.isNotEmpty() || volcanoAiApiKey.isNotEmpty()) } catch (e: Exception) { FileLogger.e("MainActivity", "从加密存储加载密钥失败: ${e.message}") e.printStackTrace() @@ -246,9 +291,10 @@ class MainActivity: FlutterActivity() { // 初始化 FileLogger FileLogger.init(applicationContext) + // 请求必要权限 + requestRequiredPermissions() + // 初始化 Azure 语音服务 - azureAsrHelper = AzureAsrHelper(applicationContext) - azureTtsHelper = AzureTtsHelper(applicationContext) classicBluetoothHelper = ClassicBluetoothHelper(applicationContext) // 注册广播接收器 @@ -266,7 +312,14 @@ class MainActivity: FlutterActivity() { val voiceInteractionFilter = IntentFilter().apply { addAction(VoiceInteractionService.ACTION_RECOGNITION_STARTED) addAction(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED) + addAction(VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE) } + + // 打印已注册的广播 + FileLogger.d(TAG, "已注册语音交互广播:${VoiceInteractionService.ACTION_RECOGNITION_STARTED}, " + + "${VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED}, " + + "${VoiceInteractionService.ACTION_ENTER_TRANSLATION_MODE}") + if (android.os.Build.VERSION.SDK_INT >= android.os.Build.VERSION_CODES.UPSIDE_DOWN_CAKE) { registerReceiver(voiceInteractionReceiver, voiceInteractionFilter, Context.RECEIVER_NOT_EXPORTED) } else { @@ -300,8 +353,7 @@ class MainActivity: FlutterActivity() { setupMethodChannels(flutterEngine) // 初始化 Azure 语音服务 - azureAsrHelper = AzureAsrHelper(this) - azureTtsHelper = AzureTtsHelper(this) + classicBluetoothHelper = ClassicBluetoothHelper(this) Log.d(TAG, "Flutter 引擎配置完成") } @@ -327,20 +379,6 @@ class MainActivity: FlutterActivity() { } ) - // Azure ASR 事件通道 - EventChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_EVENT_CHANNEL).setStreamHandler( - object : EventChannel.StreamHandler { - override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { - Log.d(TAG, "ASR事件通道开始监听") - azureAsrEventSink = events - } - - override fun onCancel(arguments: Any?) { - azureAsrEventSink = null - } - } - ) - // 设置蓝牙事件通道 EventChannel(flutterEngine.dartExecutor.binaryMessenger, CLASSIC_BLUETOOTH_EVENT_CHANNEL).setStreamHandler( object : EventChannel.StreamHandler { @@ -367,195 +405,6 @@ class MainActivity: FlutterActivity() { private fun setupMethodChannels(flutterEngine: FlutterEngine) { Log.d(TAG, "开始设置方法通道") - // 设置 Azure ASR 方法通道 - MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_CHANNEL).setMethodCallHandler { call, result -> - when (call.method) { - "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") - val region = call.argument("region") - val supportedLanguages = call.argument>("supportedLanguages")?.toTypedArray() ?: arrayOf("zh-CN", "en-US") - - if (subscriptionKey == null || region == null) { - result.error("INVALID_ARGUMENTS", "订阅密钥和区域不能为空", null) - return@setMethodCallHandler - } - - try { - val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages) - result.success(success) - } catch (e: Exception) { - result.error("INITIALIZATION_ERROR", e.message, null) - } - } - "recognizeOnce" -> { - - azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) { - result.success(mapOf( - "text" to text, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onError(error: String) { - result.error("RECOGNITION_ERROR", error, null) - } - }) - } - "startContinuousRecognition" -> { - // 确保事件通道已准备好 - if (azureAsrEventSink == null) { - result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) - return@setMethodCallHandler - } - - val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) { - sendAsrEvent(mapOf( - "type" to "result", - "text" to text, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onRecognizing(recognizing: String, detectedLanguage: String) { - sendAsrEvent(mapOf( - "type" to "recognizing", - "text" to recognizing, - "detectedLanguage" to detectedLanguage - )) - } - - override fun onSessionStarted() { - sendAsrEvent(mapOf("type" to "sessionStarted")) - } - - override fun onSessionStopped() { - sendAsrEvent(mapOf("type" to "sessionStopped")) - } - - override fun onCanceled(reason: String, errorDetails: String) { - sendAsrEvent(mapOf( - "type" to "canceled", - "reason" to reason, - "errorDetails" to errorDetails - )) - } - - override fun onError(error: String) { - sendAsrEvent(mapOf("type" to "error", "message" to error)) - } - }) - result.success(success) - } - "stopContinuousRecognition" -> { - try { - if (!azureAsrHelper.isContinuousRecognitionActive()) { - result.success(true) - return@setMethodCallHandler - } - - val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(text: String, detectedLanguage: String) {} - override fun onRecognizing(recognizing: String, detectedLanguage: String) {} - override fun onSessionStarted() {} - override fun onSessionStopped() {} - override fun onCanceled(reason: String, errorDetails: String) {} - override fun onError(error: String) { - result.error("STOP_ERROR", error, null) - } - }) - result.success(success) - } catch (e: Exception) { - result.error("STOP_ERROR", e.message, null) - } - } - "isContinuousRecognitionActive" -> { - result.success(azureAsrHelper.isContinuousRecognitionActive()) - } - "dispose" -> { - azureAsrHelper.dispose() - result.success(true) - } - else -> { - result.notImplemented() - } - } - } - - // 设置 Azure TTS 方法通道 - MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_TTS_CHANNEL).setMethodCallHandler { call, result -> - when (call.method) { - "initialize" -> { - val subscriptionKey = call.argument("subscriptionKey") ?: "" - val region = call.argument("region") ?: "" - val language = call.argument("language") ?: "zh-CN" - - val success = azureTtsHelper.initialize(subscriptionKey, region, language) - result.success(success) - } - "setVoice" -> { - val voiceName = call.argument("voiceName") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) - result.success(azureTtsHelper.setVoice(voiceName)) - } - "setSpeechParams" -> { - val rate = call.argument("rate") ?: 0 - val pitch = call.argument("pitch") ?: 0 - val volume = call.argument("volume") ?: 100 - result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) - } - "setAudioOutputType" -> { - val outputTypeStr = call.argument("outputType") ?: "speaker" - val outputType = when (outputTypeStr.lowercase()) { - "speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER - "earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE - "auto" -> AzureTtsHelper.AudioOutputType.AUTO - else -> AzureTtsHelper.AudioOutputType.SPEAKER - } - result.success(azureTtsHelper.setAudioOutputType(outputType)) - } - "speakText" -> { - val text = call.argument("text") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "文本不能为空", null) - - azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - runOnUiThread { result.success(message) } - } - - override fun onError(error: String) { - runOnUiThread { result.error("SPEAK_ERROR", error, null) } - } - }) - } - "speakSsml" -> { - val ssml = call.argument("ssml") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "SSML不能为空", null) - - azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - runOnUiThread { result.success(message) } - } - - override fun onError(error: String) { - runOnUiThread { result.error("SPEAK_ERROR", error, null) } - } - }) - } - "stopSpeaking" -> { - result.success(azureTtsHelper.stopSpeaking()) - } - "isSpeaking" -> { - result.success(azureTtsHelper.isSpeaking()) - } - "dispose" -> { - azureTtsHelper.dispose() - result.success(true) - } - else -> { - result.notImplemented() - } - } - } - // 设置语音交互方法通道 MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOICE_INTERACTION_CHANNEL).setMethodCallHandler { call, result -> when (call.method) { @@ -657,15 +506,6 @@ class MainActivity: FlutterActivity() { } } - // ASR 事件发送方法 - private fun sendAsrEvent(event: Map) { - if (azureAsrEventSink == null) return - - runOnUiThread { - azureAsrEventSink?.success(event) - } - } - /** * 启动语音交互服务(带配置参数) */ @@ -673,14 +513,14 @@ class MainActivity: FlutterActivity() { FileLogger.d(TAG, "启动语音交互服务") // 获取配置参数 - val key = call.argument("azure_speech_key") ?: "" - val region = call.argument("azure_speech_region") ?: "" - val aiKey = call.argument("volcano_ai_api_key") ?: "" - - // 设置 Azure Speech 配置 - azureSpeechKey = key - azureSpeechRegion = region - volcanoAiApiKey = aiKey + azureSpeechKey = call.argument("azure_speech_key") ?: "" + azureSpeechRegion = call.argument("azure_speech_region") ?: "" + openaiApiKey = call.argument("openai_api_key") ?: "" + openaiBaseUrl = call.argument("openai_base_url") + openaiModel = call.argument("openai_model") + volcanoSpeechAppId = call.argument("volcano_speech_app_id") ?: "" + volcanoSpeechAppToken = call.argument("volcano_speech_app_token") ?: "" + // 保存密钥到安全存储 saveKeysToSecureStorage(applicationContext) @@ -755,14 +595,19 @@ class MainActivity: FlutterActivity() { private fun sendVoiceInteractionEvent(event: Map) { if (voiceInteractionEventSink == null) { Log.e(TAG, "无法发送语音交互事件:事件通道未准备好") + FileLogger.e(TAG, "无法发送语音交互事件:事件通道未准备好") return } + FileLogger.d(TAG, "准备发送事件到Flutter: ${event["type"]}") + runOnUiThread { try { voiceInteractionEventSink?.success(event) + FileLogger.d(TAG, "成功发送事件到Flutter: ${event["type"]}") } catch (e: Exception) { - Log.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}", e) + Log.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}") + FileLogger.e(TAG, "发送语音交互事件到Flutter失败: ${e.message}") } } } @@ -799,11 +644,58 @@ class MainActivity: FlutterActivity() { } // 释放资源 - azureAsrHelper.dispose() - azureTtsHelper.dispose() classicBluetoothHelper.dispose() super.onDestroy() } + + /** + * 请求必要权限 + */ + private fun requestRequiredPermissions() { + val permissionsToRequest = ArrayList() + + for (permission in REQUIRED_PERMISSIONS) { + if (ContextCompat.checkSelfPermission(this, permission) != PackageManager.PERMISSION_GRANTED) { + permissionsToRequest.add(permission) + } + } + + if (permissionsToRequest.isNotEmpty()) { + ActivityCompat.requestPermissions( + this, + permissionsToRequest.toTypedArray(), + PERMISSION_REQUEST_CODE + ) + } + } + + /** + * 处理权限请求结果 + */ + override fun onRequestPermissionsResult( + requestCode: Int, + permissions: Array, + grantResults: IntArray + ) { + super.onRequestPermissionsResult(requestCode, permissions, grantResults) + + if (requestCode == PERMISSION_REQUEST_CODE) { + val deniedPermissions = ArrayList() + + for (i in permissions.indices) { + if (grantResults[i] != PackageManager.PERMISSION_GRANTED) { + deniedPermissions.add(permissions[i]) + } + } + + if (deniedPermissions.isNotEmpty()) { + // 记录未授权的权限 + FileLogger.w(TAG, "未授权的权限: ${deniedPermissions.joinToString()}") + } else { + FileLogger.d(TAG, "所有必要权限已授权") + } + } + } } diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak new file mode 100644 index 000000000..1a9e2df13 --- /dev/null +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler copy.kt.bak @@ -0,0 +1,456 @@ +package com.yunqiinnovation.deepsound + +import android.content.Context +import org.json.JSONArray +import org.json.JSONObject +import android.util.Log +import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.AzureAsrHelper +import com.yunqiinnovation.volcano_speech.VolcanoTtsHelper +import com.yunqiinnovation.open_ai_service.OpenAIService + + +/** + * 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑 + */ +class VoiceInteractionHandler( + private val context: Context, + private val azureSpeechKey: String, + private val azureSpeechRegion: String, + private val openaiApiKey: String, + private val openaiBaseUrl: String = "", + private val openaiModel: String = "", + private val volcanoSpeechAppId: String, + private val volcanoSpeechAppToken: String +) { + private val TAG = "VoiceInteractionHandler" + + // Azure服务 + private var azureAsrHelper: AzureAsrHelper? = null + private var volcanoTtsHelper: VolcanoTtsHelper? = null + + // OpenAI服务 + private val openAIService = OpenAIService() + + + // 语音功能处理 + private val voiceFunctionHandler = VoiceFunctionHandler(openAIService, context) + + // 当前用户输入 + private var currentUserInput = "" + + // 状态 + private var isInitialized = false + var isRecognitionActive = false + private set + var isTtsSpeaking = false + private set + var hasSpeechDetected = false + private set + + // 回调 + private var callback: InteractionCallback? = null + + /** + * 初始化 + */ + fun initialize(): Boolean { + if (isInitialized) return true + + try { + // 初始化Azure ASR + azureAsrHelper = AzureAsrHelper(context).apply { + initialize(azureSpeechKey, azureSpeechRegion) + } + + // 初始化Volcano TTS (使用大模型TTS) + volcanoTtsHelper = VolcanoTtsHelper(context).apply { + // 这里需要替换为实际的Volcano SDK初始化参数 + // 暂时使用假参数,实际使用时需要替换为真实值 + initialize(volcanoSpeechAppId, volcanoSpeechAppToken, "volc.bigasr.sauc.duration") + + } + + // 初始化OpenAI服务 + openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel) + + // 初始化语音功能处理器 + voiceFunctionHandler.initialize() + + isInitialized = true + return true + } catch (e: Exception) { + FileLogger.e(TAG, "初始化失败: ${e.message}", e) + return false + } + } + + /** + * 设置回调 + */ + fun setCallback(callback: InteractionCallback) { + this.callback = callback + } + + /** + * 开始语音识别 + */ + fun startRecognition() { + if (isRecognitionActive) return + + // 检查录音权限 + if (!checkRecordAudioPermission()) { + callback?.onError("需要录音权限,请在设置中授予权限") + return + } + + isRecognitionActive = true + hasSpeechDetected = false + notifyStateChanged() + + try { + azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onRecognizing(recognizing: String, detectedLanguage: String) { + if (recognizing.isNotEmpty()) { + hasSpeechDetected = true + stopTts() + notifyStateChanged() + } + } + + override fun onResult(result: String, detectedLanguage: String) { + if (result.isNotEmpty()) { + notifyStateChanged() + + processWithOpenAI(result) + + } + + // 重置状态,继续识别 + hasSpeechDetected = false + } + + override fun onSessionStarted() { + notifyStateChanged() + } + + override fun onSessionStopped() { + isRecognitionActive = false + notifyStateChanged() + } + + override fun onCanceled(reason: String, errorDetails: String) { + isRecognitionActive = false + notifyStateChanged() + } + + override fun onError(error: String) { + isRecognitionActive = false + callback?.onError("语音识别出错") + notifyStateChanged() + } + + override fun onSuccess(message: String) { + // 处理成功事件 + } + }) + } catch (e: Exception) { + isRecognitionActive = false + FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e) + callback?.onError("启动语音识别失败") + notifyStateChanged() + } + } + + /** + * 停止语音识别 + */ + fun stopRecognition() { + if (!isRecognitionActive) return + + FileLogger.d(TAG, "停止语音识别") + + try { + azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(result: String, detectedLanguage: String) {} + override fun onRecognizing(recognizing: String, detectedLanguage: String) {} + override fun onSessionStarted() {} + override fun onSessionStopped() { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别会话已停止") + notifyStateChanged() + } + override fun onCanceled(reason: String, errorDetails: String) { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别已取消: $reason") + notifyStateChanged() + } + override fun onError(error: String) { + isRecognitionActive = false + FileLogger.e(TAG, "停止语音识别时出错: $error") + notifyStateChanged() + } + override fun onSuccess(message: String) { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别已停止: $message") + notifyStateChanged() + } + }) + } catch (e: Exception) { + FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e) + // 确保状态一致性 + isRecognitionActive = false + notifyStateChanged() + } + } + + /** + * 使用OpenAI处理语音识别结果 + */ + private fun processWithOpenAI(text: String) { + // 保存当前用户输入,用于后续同步聊天记录 + currentUserInput = text + + Thread { + try { + val messages = JSONArray().apply { + put(openAIService.createUserMessage(text)) + } + + // 创建响应构建器 + val responseBuilder = StringBuilder() + + openAIService.sendMessageStream( + messages = messages, + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + // 累加响应内容 + responseBuilder.append(token) + } + + override fun onComplete() { + // 处理完整响应 + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + // 播放AI回复 + Log.d(TAG, "AI 回复: $response") + + speakAIResponse(response) + + // 同步聊天记录到Flutter端 + sendChatHistoryUpdate("personal_assistant", text, response) + } + } + + override fun onError(e: Exception) { + FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e) + callback?.onError("AI处理出错") + } + + override fun onFunctionCall(call: JSONObject) { + FileLogger.d(TAG, "收到函数调用请求: ${call.getString("name")}") + + // 使用函数处理器处理函数调用 + val handled = voiceFunctionHandler.handleFunctionCall( + functionCall = call, + messages = messages, + callback = object : VoiceFunctionHandler.FunctionCallCallback { + override fun onTokenReceived(token: String) { + responseBuilder.append(token) + } + + override fun onComplete() { + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + // 播放AI回复 + Log.d(TAG, "AI Function Call 回复: $response") + speakAIResponse(response) + + // 同步聊天记录到Flutter端 + sendChatHistoryUpdate("personal_assistant", text, response) + } + notifyStateChanged() + } + + override fun onError(message: String) { + FileLogger.e(TAG, "函数处理出错: $message") + callback?.onError(message) + } + + override fun onFunctionCall(nestedCall: JSONObject) { + FileLogger.d(TAG, "收到嵌套函数调用: ${nestedCall.getString("name")}") + // 不处理嵌套函数调用,直接返回错误提示 + // speakAIResponse("抱歉,暂不支持嵌套函数调用") + } + + override fun onExitWithMessage(farewell: String) { + // 播放退出消息 + speakAIResponse(farewell) + + // 同步聊天记录 + sendChatHistoryUpdate("personal_assistant", text, farewell) + + // 停止语音识别 + stopRecognition() + } + } + ) + + if (!handled) { + // 如果函数没有被处理,作为普通文本处理 + FileLogger.d(TAG, "函数未处理,作为普通文本处理") + speakAIResponse("我无法处理这个请求") + sendChatHistoryUpdate("personal_assistant", text, "我无法处理这个请求") + } + } + } + ) + + } catch (e: Exception) { + FileLogger.e(TAG, "AI处理出错: ${e.message}", e) + callback?.onError("AI处理出错") + } + }.start() + } + + /** + * 播放TTS + */ + fun playTts(text: String, callback: TtsCallback? = null) { + isTtsSpeaking = true + notifyStateChanged() + + volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback { + override fun onStart(reqId: String) { + // TTS开始播放事件 + } + + override fun onProgress(reqId: String, progress: Double) { + // TTS播放进度事件 + } + + override fun onComplete(reqId: String) { + isTtsSpeaking = false + notifyStateChanged() + callback?.onComplete() + } + + override fun onError(reqId: String, errorCode: Int, errorMsg: String) { + isTtsSpeaking = false + notifyStateChanged() + callback?.onError(errorMsg) + } + }) + } + + /** + * 停止TTS播放 + */ + fun stopTts() { + if (isTtsSpeaking) { + volcanoTtsHelper?.stop() + isTtsSpeaking = false + notifyStateChanged() + } + } + + /** + * 播放AI回复 + */ + private fun speakAIResponse(text: String) { + isTtsSpeaking = true + notifyStateChanged() + + volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback { + override fun onStart(reqId: String) { + // TTS开始播放事件 + } + + override fun onProgress(reqId: String, progress: Double) { + // TTS播放进度事件 + } + + override fun onComplete(reqId: String) { + isTtsSpeaking = false + notifyStateChanged() + } + + override fun onError(reqId: String, errorCode: Int, errorMsg: String) { + isTtsSpeaking = false + notifyStateChanged() + } + }) + } + + /** + * 释放资源 + */ + fun dispose() { + // 停止语音识别 + stopRecognition() + + // 停止TTS播放 + stopTts() + + // 释放Azure资源 + azureAsrHelper?.let { + FileLogger.d(TAG, "关闭Azure ASR服务") + it.dispose() + } + + volcanoTtsHelper?.let { + FileLogger.d(TAG, "关闭Volcano TTS服务") + it.release() + } + + FileLogger.d(TAG, "语音交互处理器资源已释放") + } + + /** + * 检查录音权限 + */ + private fun checkRecordAudioPermission(): Boolean { + val permission = android.Manifest.permission.RECORD_AUDIO + val result = context.checkCallingOrSelfPermission(permission) + return result == android.content.pm.PackageManager.PERMISSION_GRANTED + } + + /** + * 通知状态变化 + */ + private fun notifyStateChanged() { + callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected) + } + + /** + * 发送聊天历史更新 + */ + private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) { + val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply { + putExtra("agentId", agentId) + putExtra("userMessage", userMessage) + putExtra("assistantMessage", assistantMessage) + putExtra("timestamp", System.currentTimeMillis()) + } + + // 发送广播 + context.sendBroadcast(intent) + } + + /** + * 交互回调接口 + */ + interface InteractionCallback { + fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) + fun onError(message: String) + fun onPromptRequest(message: String) + } + + /** + * TTS回调接口 + */ + interface TtsCallback { + fun onComplete() + fun onError(error: String) + } +} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt new file mode 100644 index 000000000..b885117c0 --- /dev/null +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionHandler.kt @@ -0,0 +1,416 @@ +package com.yunqiinnovation.deepsound + +import android.content.BroadcastReceiver +import android.content.Context +import android.content.Intent +import android.content.IntentFilter +import org.json.JSONArray +import org.json.JSONObject +import android.util.Log +import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.AzureAsrHelper +import com.yunqiinnovation.azure_speech.AzureTtsHelper +import com.yunqiinnovation.open_ai_service.OpenAIService +import com.yunqiinnovation.open_ai_service.SystemFunctionHandler + + +/** + * 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑 + */ +class VoiceInteractionHandler( + private val context: Context, + private val azureSpeechKey: String, + private val azureSpeechRegion: String, + private val openaiApiKey: String, + private val openaiBaseUrl: String = "", + private val openaiModel: String = "", + private val volcanoSpeechAppId: String, + private val volcanoSpeechAppToken: String +) { + private val TAG = "VoiceInteractionHandler" + + // Azure服务 + private var azureAsrHelper: AzureAsrHelper? = null + private var azureTtsHelper: AzureTtsHelper? = null + + // OpenAI服务 + private val openAIService = OpenAIService(context.applicationContext) + + // 当前用户输入 + private var currentUserInput = "" + + // 状态 + private var isInitialized = false + var isRecognitionActive = false + private set + var isTtsSpeaking = false + private set + var hasSpeechDetected = false + private set + + // 回调 + private var callback: InteractionCallback? = null + + // 广播接收器 + private val exitInteractionReceiver = object : BroadcastReceiver() { + override fun onReceive(context: Context, intent: Intent) { + if (intent.action == SystemFunctionHandler.ACTION_EXIT_INTERACTION) { + Log.d(TAG, "收到退出交互广播") + stopRecognition() + + } + } + } + + /** + * 初始化 + */ + fun initialize(): Boolean { + if (isInitialized) return true + + try { + // 初始化Azure ASR + azureAsrHelper = AzureAsrHelper(context).apply { + initialize(azureSpeechKey, azureSpeechRegion) + } + + // 初始化Azure TTS + azureTtsHelper = AzureTtsHelper(context).apply { + initialize(azureSpeechKey, azureSpeechRegion) + } + + // 初始化OpenAI服务 + openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel) + + // 注册广播接收器 + try { + Log.d(TAG, "注册退出交互广播接收器,包名=${context.packageName}, action=${SystemFunctionHandler.ACTION_EXIT_INTERACTION}") + context.registerReceiver( + exitInteractionReceiver, + IntentFilter(SystemFunctionHandler.ACTION_EXIT_INTERACTION), + Context.RECEIVER_NOT_EXPORTED + ) + Log.d(TAG, "退出交互广播接收器注册成功") + } catch (e: Exception) { + // 广播注册失败不应该影响整个应用初始化 + Log.e(TAG, "注册退出交互广播接收器失败: ${e.message}", e) + } + + isInitialized = true + return true + } catch (e: Exception) { + FileLogger.e(TAG, "初始化失败: ${e.message}", e) + return false + } + } + + /** + * 设置回调 + */ + fun setCallback(callback: InteractionCallback) { + this.callback = callback + } + + /** + * 开始语音识别 + */ + fun startRecognition() { + if (isRecognitionActive) return + + // 检查录音权限 + if (!checkRecordAudioPermission()) { + callback?.onError("需要录音权限,请在设置中授予权限") + return + } + + isRecognitionActive = true + hasSpeechDetected = false + notifyStateChanged() + + try { + azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onRecognizing(recognizing: String, detectedLanguage: String) { + if (recognizing.isNotEmpty()) { + hasSpeechDetected = true + stopTts() + notifyStateChanged() + } + } + + override fun onResult(result: String, detectedLanguage: String) { + if (result.isNotEmpty()) { + notifyStateChanged() + + processWithOpenAI(result) + + } + + // 重置状态,继续识别 + hasSpeechDetected = false + } + + override fun onSessionStarted() { + notifyStateChanged() + } + + override fun onSessionStopped() { + isRecognitionActive = false + notifyStateChanged() + } + + override fun onCanceled(reason: String, errorDetails: String) { + isRecognitionActive = false + notifyStateChanged() + } + + override fun onError(error: String) { + isRecognitionActive = false + callback?.onError("语音识别出错") + notifyStateChanged() + } + + override fun onSuccess(message: String) { + // 处理成功事件 + } + }) + } catch (e: Exception) { + isRecognitionActive = false + FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e) + callback?.onError("启动语音识别失败") + notifyStateChanged() + } + } + + /** + * 停止语音识别 + */ + fun stopRecognition() { + if (!isRecognitionActive) return + + FileLogger.d(TAG, "停止语音识别") + + try { + azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(result: String, detectedLanguage: String) {} + override fun onRecognizing(recognizing: String, detectedLanguage: String) {} + override fun onSessionStarted() {} + override fun onSessionStopped() { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别会话已停止") + notifyStateChanged() + } + override fun onCanceled(reason: String, errorDetails: String) { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别已取消: $reason") + notifyStateChanged() + } + override fun onError(error: String) { + isRecognitionActive = false + FileLogger.e(TAG, "停止语音识别时出错: $error") + notifyStateChanged() + } + override fun onSuccess(message: String) { + isRecognitionActive = false + FileLogger.d(TAG, "语音识别已停止: $message") + notifyStateChanged() + } + }) + } catch (e: Exception) { + FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e) + // 确保状态一致性 + isRecognitionActive = false + notifyStateChanged() + } + } + + /** + * 使用OpenAI处理语音识别结果 + */ + private fun processWithOpenAI(text: String) { + // 保存当前用户输入,用于后续同步聊天记录 + currentUserInput = text + + Thread { + try { + val messages = JSONArray().apply { + put(openAIService.createUserMessage(text)) + } + + // 创建响应构建器 + val responseBuilder = StringBuilder() + + openAIService.sendMessageStream( + messages = messages, + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + // 累加响应内容 + responseBuilder.append(token) + } + + override fun onComplete() { + // 处理完整响应 + val response = responseBuilder.toString() + if (response.isNotEmpty()) { + // 播放AI回复 + Log.d(TAG, "AI 回复: $response") + + speakAIResponse(response) + + // 同步聊天记录到Flutter端 + sendChatHistoryUpdate("personal_assistant", text, response) + } + } + + override fun onError(e: Exception) { + FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e) + callback?.onError("AI处理出错") + } + + override fun onFunctionCall(call: JSONObject) { + FileLogger.d(TAG, "processWithOpenAI 收到函数调用请求: ${call.getString("name")}") + + + } + } + ) + + } catch (e: Exception) { + FileLogger.e(TAG, "AI处理出错: ${e.message}", e) + callback?.onError("AI处理出错") + } + }.start() + } + + /** + * 播放TTS + */ + fun playTts(text: String, callback: TtsCallback? = null) { + isTtsSpeaking = true + notifyStateChanged() + + azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + isTtsSpeaking = false + notifyStateChanged() + callback?.onComplete() + } + + override fun onError(error: String) { + isTtsSpeaking = false + notifyStateChanged() + callback?.onError(error) + } + }) + } + + /** + * 停止TTS播放 + */ + fun stopTts() { + if (isTtsSpeaking) { + azureTtsHelper?.stopSpeaking() + isTtsSpeaking = false + notifyStateChanged() + } + } + + /** + * 播放AI回复 + */ + private fun speakAIResponse(text: String) { + isTtsSpeaking = true + notifyStateChanged() + + azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + isTtsSpeaking = false + notifyStateChanged() + } + + override fun onError(error: String) { + isTtsSpeaking = false + notifyStateChanged() + } + }) + } + + /** + * 释放资源 + */ + fun dispose() { + // 停止语音识别 + stopRecognition() + + // 停止TTS播放 + stopTts() + + // 注销广播接收器 + try { + context.unregisterReceiver(exitInteractionReceiver) + } catch (e: Exception) { + FileLogger.e(TAG, "注销广播接收器失败: ${e.message}", e) + } + + // 释放Azure资源 + azureAsrHelper?.let { + FileLogger.d(TAG, "关闭Azure ASR服务") + it.dispose() + } + + azureTtsHelper?.let { + FileLogger.d(TAG, "关闭Azure TTS服务") + it.dispose() + } + + FileLogger.d(TAG, "语音交互处理器资源已释放") + } + + /** + * 检查录音权限 + */ + private fun checkRecordAudioPermission(): Boolean { + val permission = android.Manifest.permission.RECORD_AUDIO + val result = context.checkCallingOrSelfPermission(permission) + return result == android.content.pm.PackageManager.PERMISSION_GRANTED + } + + /** + * 通知状态变化 + */ + private fun notifyStateChanged() { + callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected) + } + + /** + * 发送聊天历史更新 + */ + private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) { + val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply { + putExtra("agentId", agentId) + putExtra("userMessage", userMessage) + putExtra("assistantMessage", assistantMessage) + putExtra("timestamp", System.currentTimeMillis()) + setPackage(context.packageName) + } + + // 发送广播 + context.sendBroadcast(intent) + } + + /** + * 交互回调接口 + */ + interface InteractionCallback { + fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) + fun onError(message: String) + fun onPromptRequest(message: String) + } + + /** + * TTS回调接口 + */ + interface TtsCallback { + fun onComplete() + fun onError(error: String) + } +} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt similarity index 67% rename from android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt index f469c136e..dbe64c79e 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt +++ b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt @@ -22,16 +22,21 @@ import android.os.Handler import android.os.Looper import java.util.concurrent.atomic.AtomicBoolean import org.json.JSONArray +import org.json.JSONObject import android.media.MediaPlayer import android.media.AudioAttributes import android.net.Uri import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.AzureAsrHelper +import com.yunqiinnovation.volcano_speech.VolcanoTtsHelper +import com.yunqiinnovation.open_ai_service.OpenAIService + /** * 后台语音交互 Service: * 1) 前台服务,确保不会被系统轻易杀死 * 2) MediaSession 捕获蓝牙耳机按键 - * 3) 处理录音/语音识别 + * 3) 负责唤醒控制和服务生命周期管理 */ class VoiceInteractionService : Service() { @@ -41,7 +46,7 @@ class VoiceInteractionService : Service() { private const val CHANNEL_ID = "voice_interaction_channel" // 语音识别超时时间(毫秒) - private const val RECOGNITION_TIMEOUT = 8000L + private const val RECOGNITION_TIMEOUT = 10000L // 用于跟踪服务是否正在运行 private val isRunning = AtomicBoolean(false) @@ -53,14 +58,12 @@ class VoiceInteractionService : Service() { const val ACTION_RECOGNITION_STARTED = "com.yunqiinnovation.deepsound.ACTION_RECOGNITION_STARTED" const val ACTION_PAUSE_VOICE_INTERACTION = "com.yunqiinnovation.deepsound.ACTION_PAUSE_VOICE_INTERACTION" const val ACTION_CHAT_HISTORY_UPDATED = "com.yunqiinnovation.deepsound.ACTION_CHAT_HISTORY_UPDATED" + const val ACTION_ENTER_TRANSLATION_MODE = "com.yunqiinnovation.deepsound.ACTION_ENTER_TRANSLATION_MODE" } // 服务状态 private var isActive = false // 服务是否活跃 - private var isRecognitionActive = false // 语音识别是否活跃 private var isTimeoutPaused = false // 是否因超时暂停 - private var hasSpeechDetected = false // 是否检测到语音 - private var isTtsSpeaking = false // 是否正在播放TTS // 按键处理 private var lastKeyEventTime = 0L @@ -69,16 +72,12 @@ class VoiceInteractionService : Service() { // 活动时间 private var lastActivityTime = 0L - // 当前用户输入 - private var currentUserInput = "" - - // 服务组件 private lateinit var mediaSession: MediaSessionCompat private lateinit var audioManager: AudioManager - private lateinit var azureAsrHelper: AzureAsrHelper - private lateinit var azureTtsHelper: AzureTtsHelper - private lateinit var volcanoAIService: VolcanoAIService + + // 语音交互处理器 + private lateinit var voiceInteractionHandler: VoiceInteractionHandler // 定时器 private val handler = Handler(Looper.getMainLooper()) @@ -88,15 +87,6 @@ class VoiceInteractionService : Service() { handler.postDelayed(this, 1000) // 每秒执行一次 } } - - // 系统提示词 - private val systemPrompt = """ - 你是一个智能语音助手,能够简洁明了地回答用户的问题。 - 请保持回答简短、准确,避免过长的解释。 - 如果用户的问题不清楚,请礼貌地请求澄清。 - 不要使用复杂的术语,除非用户明确要求。 - 用户用语音和你交互. - """.trimIndent() // 添加媒体播放器 private var audioPlayer: MiniMediaPlayer? = null @@ -115,7 +105,9 @@ class VoiceInteractionService : Service() { audioManager = getSystemService(Context.AUDIO_SERVICE) as AudioManager FileLogger.d(TAG, "AudioManager初始化完成") - initServices() + // 初始化语音交互处理器 + initVoiceInteractionHandler() + initMediaSession() registerMediaButtonReceiver() @@ -125,7 +117,6 @@ class VoiceInteractionService : Service() { // 设置为媒体播放状态 setPlaybackState(PlaybackStateCompat.STATE_PAUSED) - // FileLogger.d(TAG, "设置播放状态为STATE_PAUSED") // 启动监控和前台服务 startMonitoring() @@ -139,28 +130,26 @@ class VoiceInteractionService : Service() { */ private fun resetState() { isActive = false - isRecognitionActive = false isTimeoutPaused = false - hasSpeechDetected = false - isTtsSpeaking = false } /** - * 初始化所有服务 + * 初始化语音交互处理器 */ - private fun initServices() { - // 创建新的Azure服务实例 - FileLogger.d(TAG, "创建新的Azure服务实例") - azureAsrHelper = AzureAsrHelper(this) - azureTtsHelper = AzureTtsHelper(this) + private fun initVoiceInteractionHandler() { + FileLogger.d(TAG, "初始化语音交互处理器") // 尝试从静态变量获取配置 var subscriptionKey = MainActivity.azureSpeechKey var serviceRegion = MainActivity.azureSpeechRegion - var volcanoKey = MainActivity.volcanoAiApiKey + var openaiKey = MainActivity.openaiApiKey + var openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" // OpenAI API基本URL + var openaiModel = MainActivity.openaiModel ?: "" // OpenAI模型 + var volcanoSpeechAppId = MainActivity.volcanoSpeechAppId ?: "" + var volcanoSpeechAppToken = MainActivity.volcanoSpeechAppToken ?: "" // 如果静态变量中没有配置,尝试从加密存储中加载 - if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || volcanoKey.isEmpty()) { + if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || openaiKey.isEmpty()) { FileLogger.d(TAG, "静态变量中的配置信息不完整,尝试从加密存储加载") // 从加密存储加载密钥 @@ -170,7 +159,11 @@ class VoiceInteractionService : Service() { // 更新本地变量 subscriptionKey = MainActivity.azureSpeechKey serviceRegion = MainActivity.azureSpeechRegion - volcanoKey = MainActivity.volcanoAiApiKey + openaiKey = MainActivity.openaiApiKey + openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" + openaiModel = MainActivity.openaiModel ?: "" + volcanoSpeechAppId = MainActivity.volcanoSpeechAppId ?: "" + volcanoSpeechAppToken = MainActivity.volcanoSpeechAppToken ?: "" FileLogger.d(TAG, "已从加密存储加载配置信息") } else { @@ -178,28 +171,34 @@ class VoiceInteractionService : Service() { } } - // 初始化语音服务 - if (subscriptionKey.isNotEmpty() && serviceRegion.isNotEmpty()) { - // 初始化ASR - azureAsrHelper.initialize(subscriptionKey, serviceRegion, arrayOf("zh-CN")) + // 初始化语音交互处理器 + voiceInteractionHandler = VoiceInteractionHandler(applicationContext, + subscriptionKey, serviceRegion, + openaiKey, openaiBaseUrl, openaiModel, + volcanoSpeechAppId, volcanoSpeechAppToken) + + // 初始化回调 + voiceInteractionHandler.setCallback(object : VoiceInteractionHandler.InteractionCallback { + override fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) { + // 更新活动时间 + updateLastActivityTime() + } - // 初始化TTS - azureTtsHelper.initialize(subscriptionKey, serviceRegion, "zh-CN") + override fun onError(message: String) { + playNotification(message) + } - FileLogger.d(TAG, "Azure语音服务已初始化") - } else { - FileLogger.e(TAG, "Azure配置信息不完整,无法初始化Azure服务") - } - - // 初始化火山AI服务 - volcanoAIService = VolcanoAIService() + override fun onPromptRequest(message: String) { + playPrompt(message) + } + }) - // 初始化火山AI服务 - if (volcanoKey.isNotEmpty()) { - volcanoAIService.initialize(volcanoKey) - FileLogger.d(TAG, "火山AI服务已初始化") + // 初始化处理器 + val initialized = voiceInteractionHandler.initialize() + if (initialized) { + FileLogger.d(TAG, "语音交互处理器初始化成功") } else { - FileLogger.e(TAG, "火山AI配置信息不完整,无法初始化火山AI服务") + FileLogger.e(TAG, "语音交互处理器初始化失败") } } @@ -302,24 +301,23 @@ class VoiceInteractionService : Service() { if (!isActive) { isActive = true } - + // FileLogger.d(TAG, "服务状态: isActive=${isActive}, isRecognitionActive=${voiceInteractionHandler.isRecognitionActive}, isTimeoutPaused=${isTimeoutPaused}") // 检查语音识别状态 - if (isRecognitionActive) { + if (voiceInteractionHandler.isRecognitionActive) { val currentTime = System.currentTimeMillis() val elapsedTime = currentTime - lastActivityTime - - // 如果超过5秒没有检测到语音,且不在TTS播放中,暂停语音识别 - if (!hasSpeechDetected && !isTtsSpeaking && elapsedTime >= RECOGNITION_TIMEOUT) { + // FileLogger.d(TAG, "hasSpeechDetected=${voiceInteractionHandler.hasSpeechDetected}, isTtsSpeaking=${voiceInteractionHandler.isTtsSpeaking}, elapsedTime=${elapsedTime}") + // 如果超过指定时间没有检测到语音,且不在TTS播放中,暂停语音识别 + if (!voiceInteractionHandler.hasSpeechDetected && + !voiceInteractionHandler.isTtsSpeaking && + elapsedTime >= RECOGNITION_TIMEOUT) { + FileLogger.d(TAG, "超过${RECOGNITION_TIMEOUT/1000}秒未检测到语音,停止识别") isTimeoutPaused = true playNotification("没有听到您说话,已暂停对话。双击耳机按钮可重新开始。") - // 停止语音识别并确保资源完全释放 - stopVoiceRecognition() - - // 清理识别状态 - isRecognitionActive = false - hasSpeechDetected = false + // 停止语音识别 + voiceInteractionHandler.stopRecognition() } } } @@ -340,14 +338,12 @@ class VoiceInteractionService : Service() { lastKeyEventTime = currentTime // 停止当前TTS播放 - stopCurrentTTS() + voiceInteractionHandler.stopTts() setPlaybackState(PlaybackStateCompat.STATE_PLAYING) // 播放提示音 playPrompt("我在!") - // audioPlayer?.play(R.raw.listening) - // 设置为播放状态 setPlaybackState(PlaybackStateCompat.STATE_PAUSED) @@ -355,221 +351,48 @@ class VoiceInteractionService : Service() { // 重置超时暂停标志 isTimeoutPaused = false - // 启动或重置语音识别 - if (!isRecognitionActive) { + // 启动语音识别 + if (!voiceInteractionHandler.isRecognitionActive) { FileLogger.d(TAG, "语音识别未激活,开始启动") - // 如果之前是因为超时暂停,重新初始化语音识别组件 - // if (isTimeoutPaused) { - // Log.d(TAG, "之前因超时暂停,重新初始化Azure服务") - // restartAsr() - // } + // 通知 Flutter 语音识别已启动 + notifyRecognitionStarted() - startVoiceRecognition() + // 启动语音识别 + voiceInteractionHandler.startRecognition() } else { FileLogger.d(TAG, "语音识别已激活,更新活动时间") updateLastActivityTime() - hasSpeechDetected = false - } - } - - /** - * 开始语音识别 - */ - private fun startVoiceRecognition() { - if (isRecognitionActive) return - - // 检查录音权限 - if (!checkRecordAudioPermission()) { - playNotification("需要录音权限,请在设置中授予权限") - return - } - // 通知 Flutter 语音识别已启动 - notifyRecognitionStarted() - - isActive = true - isRecognitionActive = true - hasSpeechDetected = false - updateLastActivityTime() - - try { - azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onRecognizing(recognizing: String, detectedLanguage: String) { - if (recognizing.isNotEmpty()) { - hasSpeechDetected = true - stopCurrentTTS() - updateLastActivityTime() - } - } - - override fun onResult(result: String, detectedLanguage: String) { - if (result.isNotEmpty()) { - processWithVolcanoAI(result) - } - - // 重置状态,继续识别 - hasSpeechDetected = false - updateLastActivityTime() - } - - override fun onSessionStarted() { - updateLastActivityTime() - - } - - override fun onSessionStopped() { - isRecognitionActive = false - - - } - - override fun onCanceled(reason: String, errorDetails: String) { - isRecognitionActive = false - - - } - - override fun onError(error: String) { - isRecognitionActive = false - playNotification("语音识别出错") - - } - }) - } catch (e: Exception) { - isRecognitionActive = false - FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e) - playNotification("启动语音识别失败") - - - } - } - - /** - * 停止语音识别 - */ - private fun stopVoiceRecognition() { - if (!isRecognitionActive) return - - FileLogger.d(TAG, "停止语音识别") - - try { - azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { - override fun onResult(result: String, detectedLanguage: String) {} - override fun onRecognizing(recognizing: String, detectedLanguage: String) {} - override fun onSessionStarted() {} - override fun onSessionStopped() { - isRecognitionActive = false - FileLogger.d(TAG, "语音识别会话已停止") - } - override fun onCanceled(reason: String, errorDetails: String) { - isRecognitionActive = false - FileLogger.d(TAG, "语音识别已取消: $reason") - } - override fun onError(error: String) { - isRecognitionActive = false - FileLogger.e(TAG, "停止语音识别时出错: $error") - } - }) - } catch (e: Exception) { - FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e) - // 确保状态一致性 - isRecognitionActive = false } - - setPlaybackState(PlaybackStateCompat.STATE_PAUSED) - isRecognitionActive = false - hasSpeechDetected = false - // 不重置 isTimeoutPaused,保留暂停原因 - } - - /** - * 使用VolcanoAI处理语音识别结果 - */ - private fun processWithVolcanoAI(text: String) { - // 保存当前用户输入,用于后续同步聊天记录 - currentUserInput = text - - Thread { - try { - val messages = JSONArray().apply { - put(volcanoAIService.createUserMessage(text)) - } - - val response = volcanoAIService.sendMessage(messages, systemPrompt) - - // 播放AI回复 - speakAIResponse(response) - - // 同步聊天记录到Flutter端 - notifyChatHistoryUpdated("personal_assistant", text, response) - } catch (e: Exception) { - FileLogger.e(TAG, "AI处理出错: ${e.message}") - playNotification("AI处理出错") - } - }.start() - } - - /** - * 播放AI回复 - */ - private fun speakAIResponse(text: String) { - isTtsSpeaking = true - azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - isTtsSpeaking = false - updateLastActivityTime() - } - - override fun onError(error: String) { - isTtsSpeaking = false - } - }) } /** * 播放提示音 */ private fun playPrompt(message: String) { - isTtsSpeaking = true - azureTtsHelper.speakText(message, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { isTtsSpeaking = false } - override fun onError(error: String) { isTtsSpeaking = false } - }) + voiceInteractionHandler.playTts(message) } /** * 播放通知提示音 */ private fun playNotification(message: String) { - isTtsSpeaking = true // 更新播放状态为播放中,增加接收蓝牙按键事件的几率 setPlaybackState(PlaybackStateCompat.STATE_PLAYING) // 确保媒体会话处于活跃状态 mediaSession.isActive = true - azureTtsHelper.speakText(message, object : AzureTtsHelper.TTSCallback { - override fun onSuccess(message: String) { - isTtsSpeaking = false + voiceInteractionHandler.playTts(message, object : VoiceInteractionHandler.TtsCallback { + override fun onComplete() { setPlaybackState(PlaybackStateCompat.STATE_PAUSED) } override fun onError(error: String) { - isTtsSpeaking = false setPlaybackState(PlaybackStateCompat.STATE_PAUSED) } }) } - /** - * 停止当前TTS播放 - */ - private fun stopCurrentTTS() { - if (isTtsSpeaking) { - azureTtsHelper.stopSpeaking() - isTtsSpeaking = false - } - } - /** * 更新最后活动时间 */ @@ -603,15 +426,6 @@ class VoiceInteractionService : Service() { } } - /** - * 检查录音权限 - */ - private fun checkRecordAudioPermission(): Boolean { - val permission = android.Manifest.permission.RECORD_AUDIO - val result = applicationContext.checkCallingOrSelfPermission(permission) - return result == android.content.pm.PackageManager.PERMISSION_GRANTED - } - /** * 启动前台服务 */ @@ -746,38 +560,32 @@ class VoiceInteractionService : Service() { override fun onBind(intent: Intent?): IBinder? = null override fun onDestroy() { - FileLogger.d(TAG, "onDestroy - 语音交互服务正在销毁") - - // 释放音频播放器 - audioPlayer?.release() - audioPlayer = null + super.onDestroy() + FileLogger.d(TAG, "onDestroy - 语音交互服务即将销毁") - // 停止监控 + // 停止服务监控 stopMonitoring() // 停止语音识别 - if (isRecognitionActive) { - stopVoiceRecognition() - } + voiceInteractionHandler.stopRecognition() + + // 停止媒体会话 + mediaSession.release() + FileLogger.d(TAG, "媒体会话已释放") // 停止TTS - stopCurrentTTS() + voiceInteractionHandler.stopTts() - // 释放 MediaSession - mediaSession.release() + // 释放语音交互处理器资源 + voiceInteractionHandler.dispose() - // 释放Azure服务实例 - FileLogger.d(TAG, "释放Azure服务实例") - azureAsrHelper.dispose() - azureTtsHelper.dispose() + // 关闭音频播放器 + audioPlayer?.release() - // 更新服务状态 + // 设置服务状态 isRunning.set(false) - // 关闭日志系统 - FileLogger.shutdown() - - super.onDestroy() + FileLogger.d(TAG, "语音交互服务已销毁") } /** @@ -788,11 +596,11 @@ class VoiceInteractionService : Service() { FileLogger.d(TAG, "暂停后台语音交互(来自Flutter的请求)") // 停止当前TTS播放 - stopCurrentTTS() + voiceInteractionHandler.stopTts() // 停止语音识别 - if (isRecognitionActive) { - stopVoiceRecognition() + if (voiceInteractionHandler.isRecognitionActive) { + voiceInteractionHandler.stopRecognition() } // 设置为暂停状态,但保持服务活跃 @@ -804,10 +612,6 @@ class VoiceInteractionService : Service() { * 通知 Flutter 端聊天记录已更新 */ private fun notifyChatHistoryUpdated(agentId: String, userMessage: String, assistantMessage: String) { - FileLogger.d(TAG, "通知Flutter聊天记录已更新: agentId=$agentId") - FileLogger.d(TAG, "用户消息: ${userMessage.take(50)}...") - FileLogger.d(TAG, "助手回复: ${assistantMessage.take(50)}...") - // 创建广播 Intent val intent = Intent(ACTION_CHAT_HISTORY_UPDATED).apply { putExtra("agentId", agentId) @@ -818,7 +622,6 @@ class VoiceInteractionService : Service() { // 发送广播 sendBroadcast(intent) - FileLogger.d(TAG, "已发送聊天记录广播") } /** @@ -831,6 +634,7 @@ class VoiceInteractionService : Service() { val intent = Intent(ACTION_RECOGNITION_STARTED).apply { // 可以添加额外数据 putExtra("timestamp", System.currentTimeMillis()) + putExtra("packageName", applicationContext.packageName) } // 发送广播 @@ -909,5 +713,4 @@ class VoiceInteractionService : Service() { */ fun isPlaying() = mediaPlayer?.isPlaying == true } - } \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt b/android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt similarity index 100% rename from android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt rename to android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt diff --git a/android/build.gradle.kts b/android/build.gradle.kts index 35365df1a..5421f73bd 100644 --- a/android/build.gradle.kts +++ b/android/build.gradle.kts @@ -2,6 +2,11 @@ allprojects { repositories { google() mavenCentral() + maven { url = uri("https://storage.googleapis.com/download.flutter.io") } + // 添加火山引擎Maven仓库 + maven { + url = uri("https://artifact.bytedance.com/repository/Volcengine/") + } } } @@ -20,3 +25,14 @@ subprojects { tasks.register("clean") { delete(rootProject.layout.buildDirectory) } + +buildscript { + repositories { + google() + mavenCentral() + } + dependencies { + classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:2.1.10") + // 其他依赖... + } +} diff --git a/android/settings.gradle.kts b/android/settings.gradle.kts index 6296e34ed..e332da5f9 100644 --- a/android/settings.gradle.kts +++ b/android/settings.gradle.kts @@ -25,8 +25,15 @@ pluginManagement { plugins { id("dev.flutter.flutter-plugin-loader") id("com.android.application") version "8.7.0" apply false - id("org.jetbrains.kotlin.android") version "1.8.22" apply false + id("org.jetbrains.kotlin.android") version "2.1.10" apply false } include(":app") +include(":azure_speech") +include(":open_ai_service") +include(":volcano_speech") +// 设置azure_speech项目的路径 +project(":azure_speech").projectDir = file("../local_plugins/azure_speech/android") +project(":open_ai_service").projectDir = file("../local_plugins/open_ai_service/android") +project(":volcano_speech").projectDir = file("../local_plugins/volcano_speech/android") diff --git a/azure/LICENSE b/azure/LICENSE deleted file mode 100644 index d3df29b8a..000000000 --- a/azure/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2024 Your Company - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. \ No newline at end of file diff --git a/azure/README.md b/azure/README.md deleted file mode 100644 index a19025680..000000000 --- a/azure/README.md +++ /dev/null @@ -1,66 +0,0 @@ -# Azure Speech Recognition - -A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. - -## Features - -- Speech-to-text (Azure Speech Recognition) -- Text-to-speech (Azure Speech Synthesis) -- Support for multiple languages -- Language detection -- Continuous recognition -- Streaming synthesis - -## Getting Started - -### Prerequisites - -- Azure Speech service subscription key -- Azure Speech service region - -### Installation - -Add this to your package's `pubspec.yaml` file: - -```yaml -dependencies: - azure_speech_recognition: - path: ./azure -``` - -### Usage - -```dart -import 'package:azure_speech_recognition/azure_speech_recognition.dart'; - -// Initialize the service -await AzureSpeechRecognition.initialize( - subscriptionKey: 'your_subscription_key', - region: 'your_region', - supportedLanguages: ['zh-CN', 'en-US'], -); - -// Start continuous recognition -await AzureSpeechRecognition.startContinuousRecognition(); - -// Listen for recognition events -AzureSpeechRecognition.onRecognitionEvent.listen((event) { - if (event['type'] == 'result') { - print('Recognized: ${event['text']}'); - print('Detected language: ${event['detectedLanguage']}'); - } -}); - -// Stop recognition when done -await AzureSpeechRecognition.stopContinuousRecognition(); - -// Speak text -await AzureSpeechRecognition.speakText('Hello, world!'); - -// Clean up -await AzureSpeechRecognition.dispose(); -``` - -## License - -This project is licensed under the MIT License - see the LICENSE file for details. \ No newline at end of file diff --git a/azure/ios/Classes/AzureAsrHelper.swift b/azure/ios/Classes/AzureAsrHelper.swift deleted file mode 100644 index 58b142f60..000000000 --- a/azure/ios/Classes/AzureAsrHelper.swift +++ /dev/null @@ -1,757 +0,0 @@ -import Foundation -import MicrosoftCognitiveServicesSpeech -import AVFoundation -import AudioToolbox - -/// Azure ASR工具类,负责实现语音识别服务接口 -@available(iOS 13.0, *) -class AzureAsrHelper: NSObject { - // MARK: - 属性 - - /// 事件处理回调 - private var eventHandler: (String, [String: Any]) -> Void - - /// 语音配置信息 - private var speechSubscriptionKey: String = "" - private var serviceRegion: String = "" - - /// 语音识别相关 - private var speechConfig: SPXSpeechConfiguration? - private var recognizer: SPXSpeechRecognizer? - private var audioConfig: SPXAudioConfiguration? - private var pushStream: SPXPushAudioInputStream? - - /// 音频处理相关 - private var audioProcessor: CustomAudioProcessor? - private var isProcessingAudio = false - private var audioProcessingTimer: Timer? - - /// 状态标志 - private var isInitialized = false - private var _isContinuousRecognitionActive = false - - /// 当前语言和支持的语言 - private var currentLanguage = "zh-CN" - private var supportedLanguages: [String] = ["zh-CN", "en-US"] - private var isAutoDetectLanguage = false - - // MARK: - 初始化 - - init(eventHandler: @escaping (String, [String: Any]) -> Void) { - self.eventHandler = eventHandler - super.init() - } - - deinit { - dispose() - } - - // MARK: - ASR Service 接口实现 - - /// 初始化语音识别服务 - /// - Parameters: - /// - speechSubscriptionKey: Azure 语音服务订阅密钥 - /// - serviceRegion: Azure 服务区域 (如 eastasia) - /// - supportedLanguages: 支持的语言代码数组 (可选) - /// - Returns: 初始化是否成功 - func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool { - print("[AzureAsrHelper] 初始化 Azure 语音服务") - - // 检查配置是否为空 - if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { - print("[AzureAsrHelper] 错误: Azure 配置信息不完整") - eventHandler("error", ["message": "Azure 配置信息不完整"]) - return false - } - - // 释放之前的资源 - dispose() - - // 记录配置信息 - self.speechSubscriptionKey = speechSubscriptionKey - self.serviceRegion = serviceRegion - - // 设置语言 - if let languages = supportedLanguages, !languages.isEmpty { - self.supportedLanguages = languages - } - - // 根据支持的语言数量决定是否启用自动语言检测 - isAutoDetectLanguage = self.supportedLanguages.count >= 2 - - // 如果只有一种语言,设置为当前语言 - if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty { - currentLanguage = self.supportedLanguages[0] - } - - // 创建识别器和设置回调 - if !createRecognizerAndSetupCallbacks() { - return false - } - - print("[AzureAsrHelper] Azure 语音服务初始化成功") - isInitialized = true - return true - } - - /// 创建识别器并设置回调 - private func createRecognizerAndSetupCallbacks() -> Bool { - // 释放之前的 recognizer - recognizer = nil - audioConfig = nil - - do { - // 创建语音配置 - speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - - // 设置音频输入参数 - try setupAudioSession() - - // 创建自定义推送流,替代默认的麦克风输入 - pushStream = try SPXPushAudioInputStream() - audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) - - // 初始化自定义音频处理器 - audioProcessor = CustomAudioProcessor() - - // 设置语言配置 - if isAutoDetectLanguage { - // 设置自动语言检测 - speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode) - - // 创建自动语言检测配置 - let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) - - // 创建识别器 - recognizer = try SPXSpeechRecognizer( - speechConfiguration: speechConfig!, - autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig, - audioConfiguration: audioConfig! - ) - } else { - // 设置指定的识别语言 - speechConfig?.speechRecognitionLanguage = currentLanguage - - // 创建识别器 - recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) - } - - // 设置所有回调 - setupAllCallbacks() - - return true - } catch { - print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") - eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) - return false - } - } - - /// 设置音频会话 - private func setupAudioSession() throws { - let audioSession = AVAudioSession.sharedInstance() - - // 使用playAndRecord类别允许同时录音和播放 - try audioSession.setCategory(.playAndRecord, - mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 - options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) - - // 设置首选的输入和输出 - let currentRoute = audioSession.currentRoute - - // 获取当前是否连接了耳机或外部麦克风 - let hasHeadphones = currentRoute.outputs.contains { - $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP - } - - // 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 - if !hasHeadphones { - try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 - - // 启用回音消除和噪声抑制 - try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 - } else { - // 耳机模式,可以使用不同的设置 - try audioSession.setMode(.voiceChat) - try audioSession.setInputGain(1.0) - } - - // 设置合适的采样率 - try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 - try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 - - // 激活音频会话 - try audioSession.setActive(true, options: .notifyOthersOnDeactivation) - - print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") - } - - /// 设置所有回调 - private func setupAllCallbacks() { - guard let recognizer = recognizer else { return } - - // 最终识别结果 - recognizer.addRecognizedEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("result", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 识别中事件 - recognizer.addRecognizingEventHandler { [weak self] _, event in - guard let self = self else { return } - - if event.result.reason == SPXResultReason.recognizingSpeech { - let detectedLanguage = self.getDetectedLanguage(from: event.result) - // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") - self.eventHandler("recognizing", [ - "text": event.result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } - } - - // 会话事件 - recognizer.addSessionStartedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - print("[AzureAsrHelper] 识别会话已开始") - self._isContinuousRecognitionActive = true - self.eventHandler("sessionStarted", [:]) - } - - recognizer.addSessionStoppedEventHandler { [weak self] _, _ in - guard let self = self else { return } - - print("[AzureAsrHelper] 识别会话已结束") - self._isContinuousRecognitionActive = false - self.eventHandler("sessionStopped", [:]) - } - - // 取消事件 - recognizer.addCanceledEventHandler { [weak self] _, event in - guard let self = self else { return } - - let reason = event.reason.rawValue - let errorDetails = event.errorDetails ?? "未知错误" - - print("[AzureAsrHelper] 识别取消: \(errorDetails)") - - self.eventHandler("canceled", [ - "reason": reason, - "errorDetails": errorDetails - ]) - - self._isContinuousRecognitionActive = false - } - } - - /// 执行一次性语音识别 - /// - Returns: 是否成功启动识别 - func recognizeOnce() -> Bool { - if !isInitialized { - print("[AzureAsrHelper] 错误: 语音服务未初始化") - eventHandler("error", ["message": "语音服务未初始化"]) - return false - } - - // 如果正在连续识别,先停止 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 确保识别器已创建 - if recognizer == nil && !createRecognizerAndSetupCallbacks() { - return false - } - - do { - // 启动音频处理 - startAudioProcessing() - - // 通知会话开始 - eventHandler("sessionStarted", [:]) - - // 执行识别 - try recognizer?.recognizeOnceAsync { [weak self] result in - guard let self = self else { return } - - // 停止音频处理 - self.stopAudioProcessing() - - if result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: result) - self.eventHandler("result", [ - "text": result.text ?? "", - "detectedLanguage": detectedLanguage - ]) - } else if result.reason == SPXResultReason.noMatch { - print("[AzureAsrHelper] 无匹配结果") - self.eventHandler("noMatch", [:]) - } else if result.reason == SPXResultReason.canceled { - do { - let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) - let errorDetails = details.errorDetails ?? "未知错误" - self.eventHandler("error", ["message": "识别取消: \(errorDetails)"]) - } catch { - print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)") - self.eventHandler("error", ["message": "识别取消,无法获取详细原因"]) - } - } - } - - return true - } catch { - print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") - eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) - stopAudioProcessing() - return false - } - } - - /// 开始连续语音识别 - /// - Returns: 是否成功启动识别 - func startContinuousRecognition() -> Bool { - if !isInitialized { - print("[AzureAsrHelper] 错误: 语音服务未初始化") - eventHandler("error", ["message": "语音服务未初始化"]) - return false - } - - // 如果已经在进行连续识别,先停止 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 确保识别器已创建 - if recognizer == nil && !createRecognizerAndSetupCallbacks() { - return false - } - - // 重新确保音频设置正确 - do { - try setupAudioSession() - } catch { - print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") - } - - do { - // 启动音频处理 - startAudioProcessing() - - // 启动连续识别 - try recognizer?.startContinuousRecognition() - _isContinuousRecognitionActive = true - - print("[AzureAsrHelper] 连续识别开始") - return true - } catch { - print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") - eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) - _isContinuousRecognitionActive = false - stopAudioProcessing() - return false - } - } - - /// 停止连续语音识别 - /// - Returns: 是否成功停止识别 - func stopContinuousRecognition() -> Bool { - // 停止音频处理 - stopAudioProcessing() - - if !_isContinuousRecognitionActive || recognizer == nil { - return true - } - - do { - try recognizer?.stopContinuousRecognition() - _isContinuousRecognitionActive = false - print("[AzureAsrHelper] 连续识别已停止") - return true - } catch { - print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") - eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"]) - _isContinuousRecognitionActive = false - return false - } - } - - /// 检查连续识别是否活跃 - /// - Returns: 连续识别是否处于活跃状态 - func isContinuousRecognitionActive() -> Bool { - return _isContinuousRecognitionActive - } - - /// 释放资源 - func dispose() { - print("[AzureAsrHelper] 释放资源") - - // 停止音频处理 - stopAudioProcessing() - - // 停止连续识别 - if _isContinuousRecognitionActive { - stopContinuousRecognition() - } - - // 释放音频会话 - do { - try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) - } catch { - print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") - } - - // 释放资源 - recognizer = nil - speechConfig = nil - audioConfig = nil - pushStream = nil - audioProcessor = nil - - // 重置状态 - _isContinuousRecognitionActive = false - isInitialized = false - } - - /// 从结果中获取检测到的语言 - private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { - if isAutoDetectLanguage { - do { - let langResult = try SPXAutoDetectSourceLanguageResult(result) - return langResult.language ?? currentLanguage - } catch { - print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") - return currentLanguage - } - } else { - return currentLanguage - } - } - - // MARK: - 音频处理 - - /// 开始音频处理 - private func startAudioProcessing() { - guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } - - isProcessingAudio = true - - // 启动音频处理器 - if !audioProcessor.startRecord() { - print("[AzureAsrHelper] 错误: 启动音频处理器失败") - eventHandler("error", ["message": "启动音频处理器失败"]) - return - } - - // 启动音频处理定时器 - audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in - guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { - return - } - - // 读取处理后的音频数据 - var bytes = [UInt8](repeating: 0, count: 2560) - let bytesRead = processor.read(bytes: &bytes) - - if bytesRead > 0 { - // 推送数据到Azure语音服务 - let data = Data(bytes: bytes, count: bytesRead) - stream.write(data) - - // 通知音频数据可用 - self.eventHandler("audioData", ["data": bytes]) - } - } - - print("[AzureAsrHelper] 音频处理已启动") - } - - /// 停止音频处理 - private func stopAudioProcessing() { - // 停止定时器 - audioProcessingTimer?.invalidate() - audioProcessingTimer = nil - - // 停止音频处理器 - audioProcessor?.stopRecord() - - isProcessingAudio = false - print("[AzureAsrHelper] 音频处理已停止") - } -} - -// MARK: - 自定义音频处理器 - -@available(iOS 13.0, *) -class CustomAudioProcessor: NSObject { - // 音频单元 - private var ioUnit: AudioUnit? - - // 音频格式 - private var audioFormat: AudioStreamBasicDescription - - // 音频缓冲 - private var audioBufferList: AudioBufferList - private var audioList: [Float] = [] - private let audioListQueue = DispatchQueue(label: "audioListQueue") - - // 回音消除状态 - private var isEchoCancellationEnabled = true - - override init() { - // 设置音频格式 - 16kHz, 16位, 单声道 - audioFormat = AudioStreamBasicDescription( - mSampleRate: 16000.0, - mFormatID: kAudioFormatLinearPCM, - mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, - mBytesPerPacket: 2, - mFramesPerPacket: 1, - mBytesPerFrame: 2, - mChannelsPerFrame: 1, - mBitsPerChannel: 16, - mReserved: 0 - ) - - // 初始化音频缓冲 - audioBufferList = AudioBufferList( - mNumberBuffers: 1, - mBuffers: AudioBuffer( - mNumberChannels: 1, - mDataByteSize: 4096, - mData: malloc(4096) - ) - ) - - super.init() - } - - deinit { - stopRecord() - free(audioBufferList.mBuffers.mData) - } - - /// 启动音频处理 - /// - Returns: 是否成功启动 - func startRecord() -> Bool { - print("[CustomAudioProcessor] 配置音频单元") - - // 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 - var ioUnitDescription = AudioComponentDescription( - componentType: kAudioUnitType_Output, - componentSubType: kAudioUnitSubType_VoiceProcessingIO, - componentManufacturer: kAudioUnitManufacturer_Apple, - componentFlags: 0, - componentFlagsMask: 0 - ) - - // 查找音频组件 - guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { - print("[CustomAudioProcessor] 错误: 未找到音频组件") - return false - } - - // 创建音频单元实例 - if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { - ioUnit = nil - return false - } - - // 启用输入端口 - var enableInput: UInt32 = 1 - let kInputBus: AudioUnitElement = 1 - let kOutputBus: AudioUnitElement = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Input, kInputBus, &enableInput, - UInt32(MemoryLayout.size)), "启用输入端口") { - return false - } - - // 禁用输出端口 (我们只需要输入) - var enableOutput: UInt32 = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, - kAudioUnitScope_Output, kOutputBus, - &enableOutput, UInt32(MemoryLayout.size)), "禁用输出端口") { - return false - } - - // 设置缓冲区分配标志 - var flag: UInt32 = 0 - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, - kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout.size)), "设置缓冲区分配标志") { - return false - } - - // 设置音频格式 - let size = UInt32(MemoryLayout.size) - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { - return false - } - - if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, - kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { - return false - } - - // 启用回音消除 - if isEchoCancellationEnabled { - var echoCancellation: UInt32 = 1 - AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, - kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout.size)) - } - - // 设置输入回调 - 当有新音频数据时调用 - var inputCallback = AURenderCallbackStruct( - inputProc: CustomAudioProcessor.onAudioDataAvailable, - inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) - ) - - if checkError(AudioUnitSetProperty(ioUnit!, - kAudioOutputUnitProperty_SetInputCallback, - kAudioUnitScope_Global, kInputBus, - &inputCallback, UInt32(MemoryLayout.size)), "设置输入回调") { - return false - } - - // 初始化音频单元 - var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") - while hasError { - Thread.sleep(forTimeInterval: 0.1) - hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") - } - - // 启动音频单元 - hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") - - print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") - return !hasError - } - - /// 停止音频处理 - func stopRecord() { - print("[CustomAudioProcessor] 停止音频处理器") - - if let ioUnit = ioUnit { - // 停止音频单元 - _ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") - - // 关闭音频单元 - _ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") - _ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") - - self.ioUnit = nil - } - - // 清空音频数据缓冲 - audioListQueue.sync { - audioList.removeAll() - } - } - - /// 音频数据回调 - 当有新的音频数据可用时调用 - private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in - // 获取实例 - let processor = Unmanaged.fromOpaque(inRefCon).takeUnretainedValue() - - // 计算预期数据大小 - let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame - - // 确保缓冲区足够大 - if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { - processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) - processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize - } - - // 渲染音频数据 - let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, - inBusNumber, inNumberFrames, &processor.audioBufferList), - "渲染音频数据") - - // 将Int16数据转换为浮点数据进行处理 - var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) - let buffer = processor.audioBufferList.mBuffers - let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) - - for j in 0...size)) { - // 归一化到[-1.0, 1.0]范围 - audioDataFloat[j] = Float(bufferData[j]) / 32768.0 - } - - // 应用附加处理 (如有需要) - // processor.applyAdditionalProcessing(&audioDataFloat) - - // 保存处理后的数据 - if status == noErr { - processor.audioListQueue.async { - processor.audioList.append(contentsOf: audioDataFloat) - } - } - - return status - } - - /// 读取处理后的音频数据 - /// - Parameter bytes: 输出字节数组 - /// - Returns: 读取的字节数 - func read(bytes: inout [UInt8]) -> Int { - return audioListQueue.sync { - // 如果没有数据,返回0 - if audioList.isEmpty { - return 0 - } - - // 确保有足够的数据 (至少1280个样本) - if audioList.count < 1280 { - return 0 - } - - // 读取一帧数据 (1280个样本) - let frameLength = 1280 - let buffer = Array(audioList.prefix(frameLength)) - audioList.removeFirst(frameLength) - - // 将浮点数据转回Int16格式 - var int16Data = buffer.map { Int16($0 * 32767) } - - // 转换为字节数组 - let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) - bytes = [UInt8](data) - - // 每个样本2字节 (16位PCM) - return frameLength * 2 - } - } - - /// 检查错误并打印日志 - /// - Parameters: - /// - status: 操作状态 - /// - operation: 操作描述 - /// - Returns: 是否发生错误 - private func checkError(_ status: OSStatus, _ operation: String) -> Bool { - if status != noErr { - print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") - return true - } - return false - } - - /// 检查OSStatus并返回状态 - /// - Parameters: - /// - status: 操作状态 - /// - operation: 操作描述 - /// - Returns: 原始状态 - private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { - if status != noErr { - print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") - } - return status - } -} \ No newline at end of file diff --git a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift deleted file mode 100644 index 141b77ef8..000000000 --- a/azure/ios/Classes/AzureSpeechRecognitionPlugin.swift +++ /dev/null @@ -1,18 +0,0 @@ -import Flutter -import UIKit - -public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { - public static func register(with registrar: FlutterPluginRegistrar) { - if #available(iOS 13.0, *) { - SwiftAzureSpeechRecognitionPlugin.register(with: registrar) - } else { - // 如果低于iOS 13.0,返回不支持的错误 - let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) - channel.setMethodCallHandler { (call, result) in - result(FlutterError(code: "UNSUPPORTED", - message: "需要iOS 13.0及以上系统", - details: nil)) - } - } - } -} \ No newline at end of file diff --git a/azure/ios/Classes/AzureTtsHelper.swift b/azure/ios/Classes/AzureTtsHelper.swift deleted file mode 100644 index 589df617c..000000000 --- a/azure/ios/Classes/AzureTtsHelper.swift +++ /dev/null @@ -1,427 +0,0 @@ -import Foundation -import MicrosoftCognitiveServicesSpeech -import AVFoundation - -/// Azure TTS工具类,负责实现TTS服务接口 -@available(iOS 13.0, *) -class AzureTtsHelper: NSObject { - // MARK: - 属性 - - /// 事件处理回调 - private var eventHandler: (String, [String: Any]) -> Void - - /// 语音配置信息 - private var speechSubscriptionKey: String = "" - private var serviceRegion: String = "" - - /// 语音合成配置 - private var speechConfig: SPXSpeechConfiguration? - - /// 语音合成器 - private var synthesizer: SPXSpeechSynthesizer? - - /// 是否初始化成功 - private var isInitialized = false - - /// 当前是否正在播放 - private var _isSpeaking = false - - /// 音频会话配置 - private var isAudioSessionConfigured = false - - // MARK: - 语音设置 - - /// 当前语音 - private var currentVoice = "zh-CN-XiaoxiaoNeural" - - /// 支持的语音映射 - private var voiceMap: [String: String] = [ - "zh-CN": "zh-CN-XiaoxiaoNeural", - "en-US": "en-US-JennyNeural", - "ja-JP": "ja-JP-NanamiNeural", - "ko-KR": "ko-KR-SunHiNeural", - "zh-TW": "zh-TW-HsiaoChenNeural", - "zh-HK": "zh-HK-HiuMaanNeural" - ] - - /// 当前语音合成参数 - private var currentSpeechRate = "0%" - private var currentPitch = "0%" - private var currentVolume = "100%" - - // MARK: - 初始化 - - init(eventHandler: @escaping (String, [String: Any]) -> Void) { - self.eventHandler = eventHandler - super.init() - } - - deinit { - dispose() - } - - // MARK: - TTS 接口实现 - - /// 初始化语音合成服务 - /// - Parameters: - /// - speechSubscriptionKey: Azure 语音服务订阅密钥 - /// - serviceRegion: Azure 服务区域 (如 eastasia) - /// - language: 语言代码 (默认 zh-CN) - /// - Returns: 初始化是否成功 - func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { - print("[AzureTtsHelper] 初始化语音合成服务") - - // 检查配置是否为空 - if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { - print("[AzureTtsHelper] 错误: Azure 配置信息不完整") - eventHandler("error", ["error": "Azure 配置信息不完整"]) - return false - } - - // 释放之前的资源 - dispose() - - // 记录配置信息 - self.speechSubscriptionKey = speechSubscriptionKey - self.serviceRegion = serviceRegion - - // 配置音频会话 - if !configureAudioSession() { - print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化") - } - - do { - // 创建语音配置 - speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) - - // 设置默认语音 - let defaultVoice = getDefaultVoiceForLanguage(language) - currentVoice = defaultVoice - speechConfig?.speechSynthesisVoiceName = defaultVoice - - // 创建语音合成器 - synthesizer = try SPXSpeechSynthesizer(speechConfig!) - - // 设置事件处理器 - setupSynthesizerEvents() - - isInitialized = true - print("[AzureTtsHelper] TTS 引擎初始化成功") - - return true - } catch { - print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)") - eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"]) - return false - } - } - - /// 配置音频会话 - private func configureAudioSession() -> Bool { - let audioSession = AVAudioSession.sharedInstance() - do { - // 使用playback类别,但支持混合和空中播放 - try audioSession.setCategory(.playback, - mode: .spokenAudio, - options: [.mixWithOthers, .allowAirPlay, .duckOthers]) - - // 根据设备类型选择最佳配置 - let currentRoute = audioSession.currentRoute - let hasHeadphones = currentRoute.outputs.contains { - $0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP - } - - // 优化音频路由 - if hasHeadphones { - // 耳机模式,使用默认设置 - try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟 - } else { - // 扬声器模式 - try audioSession.setPreferredIOBufferDuration(0.005) - } - - // 避免完全激活音频会话,因为ASR可能已经激活 - // 这里使用setActive(false)是为了不与ASR冲突 - if !audioSession.isOtherAudioPlaying { - try audioSession.setActive(true, options: .notifyOthersOnDeactivation) - } - - isAudioSessionConfigured = true - print("[AzureTtsHelper] 音频会话配置成功") - return true - } catch { - print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)") - isAudioSessionConfigured = false - return false - } - } - - /// 设置语音 - /// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") - /// - Returns: 设置是否成功 - func setVoice(voiceName: String) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - if voiceName.isEmpty { - print("[AzureTtsHelper] 错误: 声音名称为空") - eventHandler("error", ["error": "声音名称不能为空"]) - return false - } - - if voiceName == currentVoice { - print("[AzureTtsHelper] 已设置语音: \(voiceName)") - return true - } - - print("[AzureTtsHelper] 设置声音: \(voiceName)") - currentVoice = voiceName - - // 更新语音配置 - if let speechConfig = speechConfig { - speechConfig.speechSynthesisVoiceName = voiceName - return true - } - - return false - } - - /// 设置语音合成参数 - /// - Parameters: - /// - rate: 语速,范围 -100 到 100,默认为 0 - /// - pitch: 音调,范围 -100 到 100,默认为 0 - /// - volume: 音量,范围 0 到 100,默认为 100 - /// - Returns: 是否设置成功 - func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - // 转换参数格式 - currentSpeechRate = formatRateParam(rate) - currentPitch = formatPitchParam(pitch) - currentVolume = formatVolumeParam(volume) - - print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)") - return true - } - - /// 合成文本为语音并播放 - /// - Parameter text: 要合成的文本 - /// - Returns: 操作是否成功启动 - func speakText(text: String) -> Bool { - if !isInitialized { - print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") - eventHandler("error", ["error": "TTS 引擎尚未初始化"]) - return false - } - - if text.isEmpty { - print("[AzureTtsHelper] 警告: 要播放的文本为空") - return true - } - - // 确保音频会话已配置 - if !isAudioSessionConfigured { - _ = configureAudioSession() - } - - print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") - - // 生成SSML - let ssml = generateSsml(text: text) - - // 直接进行SSML合成 - return speakSsmlInternal(text: ssml) - } - - /// 内部SSML合成和播放 - private func speakSsmlInternal(text: String) -> Bool { - guard let synthesizer = synthesizer else { - print("[AzureTtsHelper] 错误: 合成器未初始化") - eventHandler("error", ["error": "合成器未初始化"]) - return false - } - - _isSpeaking = true - eventHandler("started", [:]) - - Task { - do { - // 使用异步方法进行合成并直接播放 - _ = try await synthesizer.startSpeakingSsml(text) - - } catch { - print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)") - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"]) - } - } - } - - return true - } - - /// 停止当前语音合成 - /// - Returns: 操作是否成功 - func stopSpeaking() -> Bool { - if !isInitialized || !_isSpeaking { - return true - } - - // 停止合成 - do { - try synthesizer?.stopSpeaking() - _isSpeaking = false - eventHandler("canceled", [:]) - print("[AzureTtsHelper] 已停止语音合成") - return true - } catch { - print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)") - eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"]) - return false - } - } - - /// 检查是否正在播放 - /// - Returns: 当前是否正在播放语音 - func isSpeaking() -> Bool { - return _isSpeaking - } - - /// 释放资源 - func dispose() { - try? stopSpeaking() - - // 释放合成器和配置 - synthesizer = nil - speechConfig = nil - - isInitialized = false - _isSpeaking = false - isAudioSessionConfigured = false - print("[AzureTtsHelper] TTS 引擎已释放") - } - - // MARK: - 私有辅助方法 - - /// 设置合成器事件处理 - private func setupSynthesizerEvents() { - guard let synthesizer = synthesizer else { return } - - // 添加书签到达事件处理 - synthesizer.addBookmarkReachedEventHandler { _, e in - print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"") - } - - // 合成完成事件 - synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in - guard let self = self else { return } - print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)") - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("completed", [:]) - } - } - - // 合成取消事件 - synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in - guard let self = self else { return } - - let result = e.result - do { - let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result) - print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)") - - if cancellationDetails.reason == SPXCancellationReason.error { - print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)") - print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")") - } - - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"]) - } - } catch { - print("[AzureTtsHelper] 获取取消详情时出错: \(error)") - - DispatchQueue.main.async { - self._isSpeaking = false - self.eventHandler("error", ["error": "语音合成被取消"]) - } - } - } - - // 合成开始事件 - synthesizer.addSynthesisStartedEventHandler { _, _ in - // print("[AzureTtsHelper] 语音合成开始") - } - - // 合成中事件 - synthesizer.addSynthesizingEventHandler { _, _ in - // print("[AzureTtsHelper] 语音合成中") - } - } - - /// 生成 SSML 文本 - private func generateSsml(text: String) -> String { - return """ - - - - \(text) - - - - """ - } - - /// 格式化语速参数 - private func formatRateParam(_ rate: Int) -> String { - let clampedRate = rate.clamp(min: -100, max: 100) - if clampedRate == 0 { - return "0%" - } else if clampedRate < 0 { - return "\(Int(Double(clampedRate) * 0.9))%" - } else { - return "+\(clampedRate)%" - } - } - - /// 格式化音调参数 - private func formatPitchParam(_ pitch: Int) -> String { - let clampedPitch = pitch.clamp(min: -100, max: 100) - if clampedPitch == 0 { - return "0%" - } else { - return "\(Int(Double(clampedPitch) * 0.5))%" - } - } - - /// 格式化音量参数 - private func formatVolumeParam(_ volume: Int) -> String { - let clampedVolume = volume.clamp(min: 0, max: 100) - return "\(clampedVolume)%" - } - - /// 获取指定语言的默认语音 - private func getDefaultVoiceForLanguage(_ language: String) -> String { - return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural" - } -} - -// MARK: - 扩展 - -extension Int { - func clamp(min: Int, max: Int) -> Int { - if self < min { return min } - if self > max { return max } - return self - } -} \ No newline at end of file diff --git a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift b/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift deleted file mode 100644 index ea393e215..000000000 --- a/azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift +++ /dev/null @@ -1,259 +0,0 @@ -import Flutter -import UIKit -import MicrosoftCognitiveServicesSpeech -import AVFoundation - -@available(iOS 13.0, *) -public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { - private var azureChannel: FlutterMethodChannel - private var ttsChannel: FlutterMethodChannel - private var asrHelper: AzureAsrHelper - private var ttsHelper: AzureTtsHelper - private static var eventStreamHandler: AzureEventStreamHandler? - - // 创建方法到通道的映射 - private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]() - private static var asrMethodHandlers = [String: FlutterMethodCallHandler]() - - public static func register(with registrar: FlutterPluginRegistrar) { - // ASR通道 - let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) - - // TTS通道 - let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger()) - - // 设置ASR事件通道 - let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger()) - eventStreamHandler = AzureEventStreamHandler() - eventChannel.setStreamHandler(eventStreamHandler) - - let instance = SwiftAzureSpeechRecognitionPlugin( - azureChannel: channel, - ttsChannel: ttsChannel, - eventStreamHandler: eventStreamHandler! - ) - - // 直接设置各自通道的处理器 - channel.setMethodCallHandler(instance.handleAsrMethodCalls) - ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls) - } - - - - // 新增直接处理方法调用的函数 - private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - handleTtsMethod(call, result) - } - - private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - handleAsrMethod(call, result) - } - - - init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) { - self.azureChannel = azureChannel - self.ttsChannel = ttsChannel - - // 创建辅助类实例,使用自定义事件回调处理器 - let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in - DispatchQueue.main.async { - if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink { - var eventData = arguments - eventData["type"] = eventName - eventSink(eventData) - } - } - } - - asrHelper = AzureAsrHelper(eventHandler: eventHandler) - ttsHelper = AzureTtsHelper(eventHandler: eventHandler) - - super.init() - } - - private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - - let args = call.arguments as? Dictionary - - switch call.method { - case "initialize": - // 仅在初始化时读取必要参数 - guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { - let errorMsg = "语音订阅密钥不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) - return - } - - guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { - let errorMsg = "服务区域不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) - return - } - - let supportedLanguages = args?["supportedLanguages"] as? [String] ?? [] - - let success = asrHelper.initialize( - speechSubscriptionKey: speechSubscriptionKey, - serviceRegion: serviceRegion, - supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages - ) - result(success) - - case "startContinuousRecognition": - // 只有使用参数时才验证 - let success = asrHelper.startContinuousRecognition() - result(success) - - case "stopContinuousRecognition": - // 不需要额外参数 - let success = asrHelper.stopContinuousRecognition() - result(success) - - case "recognizeOnce": - // 只有使用参数时才验证 - let success = asrHelper.recognizeOnce() - result(success) - - case "isContinuousRecognitionActive": - // 不需要额外参数 - result(asrHelper.isContinuousRecognitionActive()) - - case "dispose": - // 不需要额外参数 - print("[AzurePlugin] 释放ASR资源") - asrHelper.dispose() - result(true) - - default: - print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)") - result(FlutterMethodNotImplemented) - } - } - - private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - - let args = call.arguments as? Dictionary - - switch call.method { - case "initialize": - // 仅在初始化时验证参数 - guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { - let errorMsg = "语音订阅密钥不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) - return - } - - guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { - let errorMsg = "服务区域不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) - return - } - - let language = args?["language"] as? String ?? "zh-CN" - - print("[AzurePlugin] 初始化TTS,语言: \(language)") - - let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language) - result(success) - - case "setVoice": - // 仅获取voice参数 - guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else { - let errorMsg = "声音名称不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil)) - return - } - - print("[AzurePlugin] 设置声音: \(voiceName)") - - let success = ttsHelper.setVoice(voiceName: voiceName) - result(success) - - case "speakText": - // 仅获取text参数 - let text = args?["text"] as? String ?? "" - - if text.isEmpty { - print("[AzurePlugin] 警告: 要播放的文本为空") - result("OK") - return - } - - print("[AzurePlugin] 播放文本: \(text.prefix(50))...") - - let success = ttsHelper.speakText(text: text) - result(success ? "OK" : "ERROR") - - case "speakSsml": - // 仅获取ssml参数 - guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else { - let errorMsg = "SSML内容不能为空" - print("[AzurePlugin] 错误: \(errorMsg)") - result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil)) - return - } - - print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...") - - // 由于我们移除了speakSsml方法,这里改用speakText方法 - // Azure SDK内部会自动检测是普通文本还是SSML - let success = ttsHelper.speakText(text: ssml) - result(success) - - case "stopSpeaking": - // 不需要参数 - print("[AzurePlugin] 停止播放") - let success = ttsHelper.stopSpeaking() - result(success) - - case "isSpeaking": - // 不需要参数 - result(ttsHelper.isSpeaking()) - - case "setSpeechParams": - // 仅获取语音参数 - let rate = args?["rate"] as? Int ?? 0 - let pitch = args?["pitch"] as? Int ?? 0 - let volume = args?["volume"] as? Int ?? 100 - - print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)") - let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume) - result(success) - - case "dispose": - // 释放TTS资源 - print("[AzurePlugin] 释放TTS资源") - ttsHelper.dispose() - result(true) - - default: - print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)") - result(FlutterMethodNotImplemented) - } - } -} - -// 用于处理事件流的辅助类 -@available(iOS 13.0, *) -class AzureEventStreamHandler: NSObject, FlutterStreamHandler { - var eventSink: FlutterEventSink? - - func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { - self.eventSink = events - // 通知Flutter端事件通道已准备好 - DispatchQueue.main.async { - events(["type": "channelReady"]) - } - return nil - } - - func onCancel(withArguments arguments: Any?) -> FlutterError? { - self.eventSink = nil - return nil - } -} \ No newline at end of file diff --git a/azure/ios/azure_speech_recognition.podspec b/azure/ios/azure_speech_recognition.podspec deleted file mode 100644 index 88797fb0c..000000000 --- a/azure/ios/azure_speech_recognition.podspec +++ /dev/null @@ -1,24 +0,0 @@ -# -# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html. -# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing. -# -Pod::Spec.new do |s| - s.name = 'azure_speech_recognition' - s.version = '0.1.0' - s.summary = 'Azure Speech Recognition plugin for Flutter' - s.description = <<-DESC -A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. - DESC - s.homepage = 'https://github.com/yourusername/azure_speech_recognition' - s.license = { :type => 'MIT', :file => '../LICENSE' } - s.author = { 'Your Company' => 'your-email@example.com' } - s.source = { :path => '.' } - s.source_files = 'Classes/**/*' - s.dependency 'Flutter' - s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0' - s.platform = :ios, '12.0' - - # Flutter.framework does not contain a i386 slice. - s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' } - s.swift_version = '5.0' -end \ No newline at end of file diff --git a/azure/lib/azure_speech_recognition.dart b/azure/lib/azure_speech_recognition.dart deleted file mode 100644 index 02b3f92d8..000000000 --- a/azure/lib/azure_speech_recognition.dart +++ /dev/null @@ -1,6 +0,0 @@ -// This is a placeholder file that exports nothing. -// The actual implementation is in the app's services folder. -// This file exists just to satisfy the Flutter plugin structure requirements. - -// Empty library to satisfy plugin structure -library azure_speech_recognition; \ No newline at end of file diff --git a/azure/pubspec.yaml b/azure/pubspec.yaml deleted file mode 100644 index a36c77bdc..000000000 --- a/azure/pubspec.yaml +++ /dev/null @@ -1,23 +0,0 @@ -name: azure_speech_recognition -description: Azure Speech Recognition and Text-to-Speech services Flutter plugin -version: 0.1.0 -homepage: https://github.com/yourusername/azure_speech_recognition - -environment: - sdk: '>=2.12.0 <3.0.0' - flutter: ">=2.0.0" - -dependencies: - flutter: - sdk: flutter - -dev_dependencies: - flutter_test: - sdk: flutter - flutter_lints: ^1.0.0 - -flutter: - plugin: - platforms: - ios: - pluginClass: AzureSpeechRecognitionPlugin \ No newline at end of file diff --git a/lib/core/bindings/initial_binding.dart b/lib/core/bindings/initial_binding.dart index 3f42d9560..748589f63 100644 --- a/lib/core/bindings/initial_binding.dart +++ b/lib/core/bindings/initial_binding.dart @@ -8,6 +8,7 @@ import '../../data/services/classic_bluetooth_service.dart'; import '../../data/services/bluetooth_media_button_service.dart'; import '../../core/utils/logger.dart'; import '../../data/services/voice_interaction_service.dart'; +import '../../data/services/open_ai_service_adapter.dart'; /// 初始绑定,用于管理全局依赖 class InitialBinding extends Bindings { @@ -29,6 +30,13 @@ class InitialBinding extends Bindings { Get.lazyPut(() => VolcanoTranslationService(), fenix: true); + // OpenAI服务适配器 + Get.lazyPut(() { + final adapter = OpenAIServiceAdapter(); + adapter.initialize(); + return adapter; + }, fenix: true); + // 火山AI服务 Get.lazyPut(() => VolcanoAIService(), fenix: true); @@ -41,6 +49,7 @@ class InitialBinding extends Bindings { () => BluetoothMediaButtonService(), fenix: true); + // 语音交互服务 Get.lazyPut(() => VoiceInteractionService(), fenix: true); diff --git a/lib/data/models/events/voice_interaction_event.dart b/lib/data/models/events/voice_interaction_event.dart index 4bfa636df..81860de23 100644 --- a/lib/data/models/events/voice_interaction_event.dart +++ b/lib/data/models/events/voice_interaction_event.dart @@ -26,4 +26,17 @@ class RecognitionStartedEvent extends VoiceInteractionEvent { RecognitionStartedEvent({ required int timestamp, }) : super(timestamp: timestamp); +} + +/// 通用语音交互事件 +/// 用于处理其他类型的事件 +class GenericVoiceInteractionEvent extends VoiceInteractionEvent { + final String type; + final Map? data; + + GenericVoiceInteractionEvent({ + required this.type, + this.data, + required int timestamp, + }) : super(timestamp: timestamp); } \ No newline at end of file diff --git a/lib/data/services/ai_service.dart b/lib/data/services/ai_service.dart index 949dfca67..a2232b9b5 100644 --- a/lib/data/services/ai_service.dart +++ b/lib/data/services/ai_service.dart @@ -1,5 +1,7 @@ /// AI回复服务接口 abstract class AiService { + + /// 非流式输出方法 Future sendMessage({ required List> messages, diff --git a/lib/data/services/open_ai_service_adapter.dart b/lib/data/services/open_ai_service_adapter.dart new file mode 100644 index 000000000..2c0da53ac --- /dev/null +++ b/lib/data/services/open_ai_service_adapter.dart @@ -0,0 +1,247 @@ +import 'dart:async'; +import 'package:flutter_dotenv/flutter_dotenv.dart'; +import 'package:open_ai_service/open_ai_service.dart'; +import 'package:get/get.dart'; +import 'ai_service.dart'; + +/// OpenAI服务适配器 - 连接AiService接口与OpenAIService插件 +class OpenAIServiceAdapter implements AiService { + final OpenAIService _openAIService = OpenAIService(); + StreamSubscription? _eventSubscription; + final StreamController _tokenStreamController = StreamController.broadcast(); + bool _isProcessingStream = false; + + /// 构造函数 + OpenAIServiceAdapter() { + printInfo(info: '创建OpenAIServiceAdapter实例'); + _setupEventListener(); + } + + /// 设置事件监听器 + void _setupEventListener() { + try { + // 首先确保访问eventStream以初始化底层事件通道 + _openAIService.eventStream; + + // 设置事件处理 + _eventSubscription = _openAIService.eventStream.listen( + (event) { + if (!_isProcessingStream) return; + + try { + switch (event.type) { + case OpenAIEventType.token: + if (event.content is String) { + _tokenStreamController.add(event.content as String); + } else { + printInfo(info: '收到非字符串类型的token: ${event.content}'); + } + break; + case OpenAIEventType.complete: + _isProcessingStream = false; + break; + case OpenAIEventType.error: + if (event.content is String) { + _tokenStreamController.addError(event.content as String); + } else { + _tokenStreamController.addError('未知错误: ${event.content}'); + } + _isProcessingStream = false; + break; + case OpenAIEventType.functionCall: + try { + if (event.content is Map) { + _tokenStreamController.addError('收到函数调用,该流仅支持文本响应'); + } else { + _tokenStreamController.addError('收到未知格式的函数调用'); + printError(info: '函数调用格式错误: ${event.content}'); + } + } catch (e) { + printError(info: '处理函数调用事件出错: $e'); + _tokenStreamController.addError('处理函数调用失败: $e'); + } + _isProcessingStream = false; + break; + } + } catch (e) { + printError(info: '处理事件出错: $e'); + _tokenStreamController.addError('处理事件失败: $e'); + _isProcessingStream = false; + } + }, + onError: (error) { + printError(info: '事件流错误: $error'); + _tokenStreamController.addError('事件流错误: $error'); + _isProcessingStream = false; + }, + onDone: () { + printInfo(info: '事件流已关闭'); + _isProcessingStream = false; + }, + ); + } catch (e) { + printError(info: '设置事件监听器失败: $e'); + } + } + + /// 初始化OpenAI服务 + Future initialize() async { + try { + // 从.env文件中读取配置 + final apiKey = dotenv.env['OPENAI_API_KEY'] ?? ''; + final baseUrl = dotenv.env['OPENAI_BASE_URL'] ?? ''; + final model = dotenv.env['OPENAI_MODEL'] ?? ''; + + printInfo(info: '从.env读取OpenAI配置'); + printInfo(info: '基础URL: $baseUrl'); + printInfo(info: '模型名称: $model'); + + if (apiKey.isEmpty) { + printError(info: '错误: OpenAI API密钥未配置,请在.env文件中设置OPENAI_API_KEY'); + return false; + } + + // 初始化OpenAI服务 + final result = await _openAIService.initialize( + apiKey: apiKey, + baseUrl: baseUrl, + model: model, + ); + + if (result) { + printInfo(info: 'OpenAI服务初始化成功'); + } else { + printError(info: 'OpenAI服务初始化失败'); + } + + return result; + } catch (e) { + printError(info: 'OpenAI服务初始化异常: $e'); + return false; + } + } + + /// 发送消息并获取回复 + @override + Future sendMessage({ + required List> messages, + required String systemPrompt, + }) async { + try { + // 在方法内直接转换 + final convertedMessages = messages.map((m) => + Map.from(m)).toList(); + + // 发送消息并获取回复 + final response = await _openAIService.sendMessage( + messages: convertedMessages, + ); + + return response; + } catch (e) { + printError(info: 'OpenAI发送消息失败: $e'); + throw '发送消息失败: $e'; + } + } + + /// 发送消息并获取流式回复 + @override + Stream sendMessageStream({ + required List> messages, + required String systemPrompt, + }) async* { + try { + // 在方法内直接转换 + final convertedMessages = messages.map((m) => + Map.from(m)).toList(); + + // 创建用于接收token的控制器 + final localController = StreamController(); + + // 标记开始处理流 + _isProcessingStream = true; + + // 添加从广播流到本地流的订阅 + final subscription = _tokenStreamController.stream.listen( + (token) => localController.add(token), + onError: (error) { + printError(info: '令牌流错误: $error'); + localController.addError(error); + localController.close(); + }, + onDone: () { + printInfo(info: '令牌流已完成'); + localController.close(); + } + ); + + // 当本地控制器关闭时,取消订阅 + localController.onCancel = () { + subscription.cancel(); + }; + + // 启动流式消息请求 + bool started = false; + try { + started = await _openAIService.sendMessageStream( + messages: convertedMessages, + ); + } catch (e) { + printError(info: '启动消息流失败: $e'); + localController.addError('启动消息流失败: $e'); + localController.close(); + _isProcessingStream = false; + throw '启动消息流失败: $e'; + } + + if (!started) { + printError(info: '无法启动消息流'); + localController.addError('无法启动消息流'); + localController.close(); + _isProcessingStream = false; + throw '无法启动消息流'; + } + + // 通过yield*将controller的流转发 + yield* localController.stream; + } catch (e) { + printError(info: 'OpenAI流式消息处理失败: $e'); + throw '流式消息处理失败: $e'; + } + } + + /// 注册函数 + Future registerFunction(String name, String description, Map parameters) async { + try { + final result = await _openAIService.registerFunction( + name: name, + description: description, + parameters: parameters, + ); + + if (result) { + printInfo(info: '函数 "$name" 注册成功'); + } else { + printError(info: '函数 "$name" 注册失败'); + } + + return result; + } catch (e) { + printError(info: '注册函数失败: $e'); + return false; + } + } + + /// 释放资源 + void dispose() { + try { + _isProcessingStream = false; + _eventSubscription?.cancel(); + _tokenStreamController.close(); + printInfo(info: 'OpenAIServiceAdapter资源已释放'); + } catch (e) { + printError(info: '释放资源时出错: $e'); + } + } + +} \ No newline at end of file diff --git a/lib/data/services/speech_impl/azure_asr_service.dart b/lib/data/services/speech_impl/azure_asr_service.dart index ffb67ed46..e09defb66 100644 --- a/lib/data/services/speech_impl/azure_asr_service.dart +++ b/lib/data/services/speech_impl/azure_asr_service.dart @@ -10,8 +10,8 @@ import '../asr_service.dart'; /// 该服务提供了通过平台通道与 Android 上的 Microsoft Speech SDK 交互的接口 class AzureAsrService extends GetxService implements AsrService { static final AzureAsrService to = Get.put(AzureAsrService()); - static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_asr'); - static const EventChannel _eventChannel = EventChannel('com.deep_voice.azure_asr_events'); + static const MethodChannel _channel = MethodChannel('azure_speech/asr'); + static const EventChannel _eventChannel = EventChannel('azure_speech/asr_events'); bool _isInitialized = false; late final String _subscriptionKey; diff --git a/lib/data/services/speech_impl/azure_tts_service.dart b/lib/data/services/speech_impl/azure_tts_service.dart index ae5ab9369..1cdb62507 100644 --- a/lib/data/services/speech_impl/azure_tts_service.dart +++ b/lib/data/services/speech_impl/azure_tts_service.dart @@ -12,7 +12,7 @@ import '../tts_service.dart'; /// 提供文本转语音功能。 class AzureTtsService extends GetxService implements TtsService { static final AzureTtsService to = Get.put(AzureTtsService()); - static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_tts'); + static const MethodChannel _channel = MethodChannel('azure_speech/tts'); bool _isInitialized = false; late final String _subscriptionKey; diff --git a/lib/data/services/voice_interaction_service.dart b/lib/data/services/voice_interaction_service.dart index a1d26d509..cce506a58 100644 --- a/lib/data/services/voice_interaction_service.dart +++ b/lib/data/services/voice_interaction_service.dart @@ -5,6 +5,7 @@ import 'package:flutter_dotenv/flutter_dotenv.dart'; import '../models/events/voice_interaction_event.dart'; import '../../core/utils/logger.dart'; import '../../modules/chat/models/message_model.dart'; +import '../../routes/app_routes.dart'; import 'chat_history_service.dart'; /// 语音交互服务接口 @@ -38,8 +39,9 @@ class VoiceInteractionService extends GetxService { // 配置信息 late String _azureSpeechKey; late String _azureSpeechRegion; - late String _volcanoAiKey; - + late String _openaiApiKey; + late String _openaiBaseUrl; + late String _openaiModel; // 聊天历史服务 late final ChatHistoryService _chatHistoryService; @@ -57,15 +59,14 @@ class VoiceInteractionService extends GetxService { void _loadConfig() { _azureSpeechKey = dotenv.env['AZURE_SPEECH_KEY'] ?? ''; _azureSpeechRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? ''; - _volcanoAiKey = dotenv.env['VOLCANO_AI_API_KEY'] ?? ''; + _openaiApiKey = dotenv.env['OPENAI_API_KEY'] ?? ''; + _openaiBaseUrl = dotenv.env['OPENAI_BASE_URL'] ?? ''; + _openaiModel = dotenv.env['OPENAI_MODEL'] ?? ''; if (_azureSpeechKey.isEmpty || _azureSpeechRegion.isEmpty) { Logger.warning('未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION'); } - - if (_volcanoAiKey.isEmpty) { - Logger.warning('未找到火山 AI API 密钥。请在 .env 文件中设置 VOLCANO_AI_API_KEY'); - } + } /// 处理来自原生层的事件 @@ -104,6 +105,29 @@ class VoiceInteractionService extends GetxService { // 保存聊天历史到ChatHistoryService _saveChatHistory(agentId, userMessage, assistantMessage, DateTime.now().millisecondsSinceEpoch); break; + + case 'enter_translation_mode': + // 进入翻译模式事件 + Logger.info('收到进入翻译模式事件,正在导航到翻译界面'); + _navigateToTranslation(); + + final translationModeEvent = GenericVoiceInteractionEvent( + type: 'enter_translation_mode', + timestamp: DateTime.now().millisecondsSinceEpoch, + ); + _eventStreamController.add(translationModeEvent); + break; + } + } + + /// 导航到翻译界面 + void _navigateToTranslation() { + try { + // 使用GetX导航到翻译页面 + Get.toNamed(Routes.translation); + Logger.info('已导航到翻译界面'); + } catch (e) { + Logger.error('导航到翻译界面失败: $e'); } } @@ -195,7 +219,9 @@ class VoiceInteractionService extends GetxService { final result = await _channel.invokeMethod('startService', { 'azure_speech_key': _azureSpeechKey, 'azure_speech_region': _azureSpeechRegion, - 'volcano_ai_api_key': _volcanoAiKey, + 'openai_api_key': _openaiApiKey, + 'openai_base_url': _openaiBaseUrl, + 'openai_model': _openaiModel, }) ?? false; if (result) { diff --git a/lib/modules/chat/controllers/chat_controller.dart b/lib/modules/chat/controllers/chat_controller.dart index f882e342f..8f446444e 100644 --- a/lib/modules/chat/controllers/chat_controller.dart +++ b/lib/modules/chat/controllers/chat_controller.dart @@ -17,6 +17,7 @@ import '../../../data/services/asr_service.dart'; import '../../../data/services/chat_history_service.dart'; import '../../../data/models/events/voice_interaction_event.dart'; import '../../../data/services/voice_interaction_service.dart'; +import '../../../data/services/open_ai_service_adapter.dart'; class ChatController extends GetxController { // 服务 @@ -97,6 +98,7 @@ class ChatController extends GetxController { agent = foundAgent; + // 根据Agent ID选择不同的AI服务 switch (agent.id) { case 'cyber_girlfriend': // 亲子陪伴 @@ -109,7 +111,15 @@ class ChatController extends GetxController { _aiService = Get.find(); break; default: - _aiService = Get.find(); + // 默认使用OpenAIServiceAdapter + try { + _aiService = Get.find(); + Logger.info('使用OpenAIServiceAdapter'); + } catch (e) { + // 如果找不到OpenAIServiceAdapter,则回退到VolcanoAIService + Logger.info('未找到OpenAIServiceAdapter,回退使用VolcanoAIService: $e'); + _aiService = Get.find(); + } } // 使用克隆音色语音合成 diff --git a/local_plugins/azure_speech/README.md b/local_plugins/azure_speech/README.md new file mode 100644 index 000000000..067b344eb --- /dev/null +++ b/local_plugins/azure_speech/README.md @@ -0,0 +1,136 @@ +# Azure Speech 插件 + +本插件为Flutter提供了Azure语音服务的集成,包括: + +- 语音合成(TTS) +- 语音识别(ASR) + +## 功能 + +### 语音合成(TTS) + +- 支持多种语音(如中文、英文等) +- 语音参数调整(语速、音调、音量) +- 音频输出设备选择(扬声器、听筒、自动) +- SSML支持 + +### 语音识别(ASR) + +- 一次性语音识别 +- 连续语音识别 +- 自动语言检测 +- 音频处理优化(回音消除、噪声抑制等) + +## 平台支持 + +- Android +- iOS + +## 如何使用 + +### 初始化 + +```dart +import 'package:azure_speech/azure_speech.dart'; + +// 初始化TTS +await AzureSpeech.initializeTts( + 'your_subscription_key', + 'your_service_region', + language: 'zh-CN', +); + +// 初始化ASR +await AzureSpeech.initializeAsr( + 'your_subscription_key', + 'your_service_region', + ['zh-CN', 'en-US'], +); +``` + +### 语音合成 + +```dart +// 设置语音 +await AzureSpeech.setTtsVoice('zh-CN-XiaoxiaoNeural'); + +// 设置语音参数 +await AzureSpeech.setTtsSpeechParams( + rate: 0, // 语速 -100~100 + pitch: 0, // 音调 -100~100 + volume: 100, // 音量 0~100 +); + +// 设置音频输出设备 +await AzureSpeech.setTtsAudioOutputType('AUTO'); // 'SPEAKER', 'EARPIECE', 'AUTO' + +// 播放文本 +await AzureSpeech.speakText('你好,世界!'); + +// 停止播放 +await AzureSpeech.stopSpeaking(); + +// 检查是否正在播放 +bool isSpeaking = await AzureSpeech.isSpeaking(); +``` + +### 语音识别 + +```dart +// 一次性识别 +final result = await AzureSpeech.recognizeOnce(); +if (result['success']) { + print('识别文本: ${result['text']}'); + print('识别语言: ${result['language']}'); +} else { + print('识别失败: ${result['error']}'); +} + +// 连续识别 +// 监听识别结果 +AzureSpeech.asrResultStream.listen((event) { + switch (event['eventType']) { + case 'recognizing': + // 实时识别中的结果 + print('识别中: ${event['text']}'); + break; + case 'finalResult': + // 最终识别结果 + print('最终结果: ${event['text']}'); + break; + case 'error': + // 错误 + print('错误: ${event['error']}'); + break; + } +}); + +// 开始连续识别 +await AzureSpeech.startContinuousRecognition(); + +// 停止连续识别 +await AzureSpeech.stopContinuousRecognition(); + +// 检查连续识别是否活跃 +bool isActive = await AzureSpeech.isContinuousRecognitionActive(); +``` + +### 释放资源 + +```dart +// 释放所有资源 +await AzureSpeech.dispose(); +``` + +## 依赖项 + +本插件依赖于: + +- Microsoft Cognitive Services Speech SDK +- Flutter + +## 注意事项 + +- 使用前需要在Azure门户中创建语音服务资源,并获取订阅密钥和区域 +- Android需要相关权限:RECORD_AUDIO, INTERNET等 +- iOS需要在Info.plist中添加麦克风使用权限描述 \ No newline at end of file diff --git a/local_plugins/azure_speech/android/build.gradle.kts b/local_plugins/azure_speech/android/build.gradle.kts new file mode 100644 index 000000000..a26b13967 --- /dev/null +++ b/local_plugins/azure_speech/android/build.gradle.kts @@ -0,0 +1,64 @@ +import com.android.build.gradle.LibraryExtension + +buildscript { + repositories { + google() + mavenCentral() + } + dependencies { + classpath("com.android.tools.build:gradle:7.3.0") + classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") + } +} + +allprojects { + repositories { + google() + mavenCentral() + } +} + +plugins { + id("com.android.library") + kotlin("android") +} + +// 配置android扩展 +configure { + namespace = "com.yunqiinnovation.azure_speech" + compileSdkVersion(33) + + defaultConfig { + minSdk = 21 + } + + compileOptions { + sourceCompatibility = JavaVersion.VERSION_11 + targetCompatibility = JavaVersion.VERSION_11 + } + + sourceSets { + getByName("main") { + manifest.srcFile("src/main/AndroidManifest.xml") + java.srcDirs("src/main/kotlin") + } + } + + // 添加lint选项 + lintOptions { + isCheckReleaseBuilds = false + } +} + +// 显式设置Kotlin JVM目标版本 +tasks.withType { + kotlinOptions { + jvmTarget = "11" + } +} + +dependencies { + + // 添加Microsoft语音SDK + implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.30.0") +} \ No newline at end of file diff --git a/local_plugins/azure_speech/android/settings.gradle.kts b/local_plugins/azure_speech/android/settings.gradle.kts new file mode 100644 index 000000000..14ba24107 --- /dev/null +++ b/local_plugins/azure_speech/android/settings.gradle.kts @@ -0,0 +1 @@ +rootProject.name = "azure_speech" diff --git a/local_plugins/azure_speech/android/src/main/AndroidManifest.xml b/local_plugins/azure_speech/android/src/main/AndroidManifest.xml new file mode 100644 index 000000000..df2a770c6 --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/AndroidManifest.xml @@ -0,0 +1,6 @@ + + + + + \ No newline at end of file diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt new file mode 100644 index 000000000..1dd7cf624 --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -0,0 +1,593 @@ +package com.yunqiinnovation.azure_speech + +import android.content.Context +import android.media.AudioAttributes +import android.media.AudioFormat +import android.media.AudioRecord +import android.media.MediaRecorder +import android.media.audiofx.AcousticEchoCanceler +import android.media.audiofx.NoiseSuppressor +import android.media.audiofx.AutomaticGainControl +import android.os.Process +import com.yunqiinnovation.azure_speech.utils.FileLogger +import com.microsoft.cognitiveservices.speech.* +import com.microsoft.cognitiveservices.speech.audio.* +import com.microsoft.cognitiveservices.speech.util.EventHandler +import java.util.concurrent.ExecutionException +import java.util.concurrent.atomic.AtomicBoolean + +class AzureAsrHelper(private val context: Context) { + private var recognizer: SpeechRecognizer? = null + private var speechConfig: SpeechConfig? = null + private val TAG = "AzureAsrHelper" + private var isContinuousRecognitionActive = false + private var currentLanguage = "zh-CN" + private var subscriptionKey = "" + private var region = "" + private var isAutoDetectLanguage = false + private var supportedLanguages = arrayOf("zh-CN") + + // 是否使用回音消除 - 内部控制常量 + private val useEchoCancellation = false + + // 自定义音频处理相关 + private var customAudioProcessor: CustomAudioProcessor? = null + private var pushStream: PushAudioInputStream? = null + private var audioConfig: AudioConfig? = null + + // 初始化SDK并创建recognizer + fun initialize(subscriptionKey: String, region: String, + supportedLanguages: Array = arrayOf("zh-CN")): Boolean { + try { + FileLogger.d(TAG, "初始化 Azure 语音服务") + + // 检查配置是否为空 + if (subscriptionKey.isEmpty() || region.isEmpty()) { + FileLogger.e(TAG, "Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + this.subscriptionKey = subscriptionKey + this.region = region + + // 设置语言 + if (supportedLanguages.isNotEmpty()) { + this.supportedLanguages = supportedLanguages + } + + // 根据支持的语言数量决定是否启用自动语言检测 + this.isAutoDetectLanguage = supportedLanguages.size >= 2 + + // 如果只有一种语言,设置为当前语言 + if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { + this.currentLanguage = supportedLanguages[0] + } + + // 创建语音配置 + speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) + + // 设置语言配置 + if (isAutoDetectLanguage) { + // 设置自动语言检测 + speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") + } else { + // 设置指定的识别语言 + speechConfig?.speechRecognitionLanguage = currentLanguage + } + + // 创建识别器 + try { + if (useEchoCancellation) { + // 如果使用回音消除,创建自定义音频输入流 + setupCustomAudioProcessing() + + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) + } else { + recognizer = SpeechRecognizer(speechConfig, audioConfig) + } + } else { + // 使用默认麦克风输入 + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) + } else { + recognizer = SpeechRecognizer(speechConfig) + } + } + + FileLogger.d(TAG, "Azure 语音服务初始化成功") + return true + } catch (e: Exception) { + FileLogger.e(TAG, "创建识别器失败: ${e.message}") + stopCustomAudioProcessing() + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "初始化失败: ${e.message}") + return false + } + } + + // 重置 recognizer + private fun resetRecognizer(): Boolean { + try { + // 释放之前的 recognizer + recognizer?.close() + recognizer = null + + // 停止当前的音频处理 + stopCustomAudioProcessing() + + // 使用现有配置重新创建 recognizer + if (speechConfig != null) { + if (useEchoCancellation) { + // 如果使用回音消除,创建自定义音频输入流 + setupCustomAudioProcessing() + + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) + } else { + recognizer = SpeechRecognizer(speechConfig, audioConfig) + } + } else { + // 使用默认麦克风输入 + if (isAutoDetectLanguage) { + val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) + recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) + } else { + recognizer = SpeechRecognizer(speechConfig) + } + } + return true + } else { + FileLogger.e(TAG, "语音配置未初始化") + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "重置识别器失败: ${e.message}") + return false + } + } + + // 开始一次性语音识别 + fun recognizeOnce(callback: RecognizeCallback) { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return + } + + // 重置 recognizer + if (!resetRecognizer()) { + callback.onError("重置识别器失败") + return + } + + try { + // 启动音频处理 + startCustomAudioProcessing() + + // 执行识别 + val result = recognizer?.recognizeOnceAsync()?.get() + + // 停止音频处理 + stopCustomAudioProcessing() + + if (result != null && result.reason == ResultReason.RecognizedSpeech) { + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language + callback.onResult(result.text, detectedLanguage ?: "") + } else { + callback.onError("未能识别语音") + } + } catch (e: Exception) { + // 停止音频处理 + stopCustomAudioProcessing() + callback.onError("识别异常: ${e.message}") + } + } + + // 开始连续语音识别 + fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return false + } + + if (isContinuousRecognitionActive) { + FileLogger.w(TAG, "已在进行连续识别,忽略请求") + return true + } + + // 重置 recognizer + if (!resetRecognizer()) { + callback.onError("重置识别器失败") + return false + } + + try { + // 启动音频处理 + startCustomAudioProcessing() + + // 识别中事件 + recognizer?.recognizing?.addEventListener( + EventHandler { _, event -> + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language + FileLogger.d(TAG, "识别中: ${event.result.text}, 语言: $detectedLanguage") + callback.onRecognizing(event.result.text, detectedLanguage ?: "") + } + ) + + // 识别完成事件 + recognizer?.recognized?.addEventListener( + EventHandler { _, event -> + if (event.result.reason == ResultReason.RecognizedSpeech) { + val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language + FileLogger.d(TAG, "识别完成: ${event.result.text}, 语言: $detectedLanguage") + callback.onResult(event.result.text, detectedLanguage ?: "") + } + } + ) + + // 会话开始事件 + recognizer?.sessionStarted?.addEventListener( + EventHandler { _, _ -> + FileLogger.d(TAG, "识别会话已开始") + callback.onSessionStarted() + callback.onSuccess("开始识别") // 兼容旧接口 + } + ) + + // 会话结束事件 + recognizer?.sessionStopped?.addEventListener( + EventHandler { _, _ -> + FileLogger.d(TAG, "识别会话已结束") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onSessionStopped() + } + ) + + // 取消事件 + recognizer?.canceled?.addEventListener( + EventHandler { _, event -> + val errorDetails = try { + event.errorDetails ?: "未知错误" + } catch (e: Exception) { + "未知错误" + } + val reason = event.reason.toString() + FileLogger.e(TAG, "识别取消: $errorDetails") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onCanceled(reason, errorDetails) + callback.onError("识别取消: $errorDetails") // 兼容旧接口 + } + ) + + // 开始连续识别 + recognizer?.startContinuousRecognitionAsync() + isContinuousRecognitionActive = true + FileLogger.d(TAG, "连续识别已启动") + + return true + } catch (e: Exception) { + // 停止音频处理 + stopCustomAudioProcessing() + FileLogger.e(TAG, "开始连续识别失败: ${e.message}") + e.printStackTrace() + callback.onError("开始连续识别失败: ${e.message}") + return false + } + } + + // 停止连续语音识别 + fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { + if (speechConfig == null) { + callback.onError("语音服务未初始化") + return false + } + + if (!isContinuousRecognitionActive) { + FileLogger.d(TAG, "未进行连续识别,忽略停止请求") + return true + } + + try { + FileLogger.d(TAG, "停止连续语音识别") + + if (recognizer == null) { + if (isContinuousRecognitionActive) { + FileLogger.w(TAG, "识别器为空,但状态显示活跃") + } + isContinuousRecognitionActive = false + return true + } + + // 停止连续识别 + val future = recognizer?.stopContinuousRecognitionAsync() + future?.get() + + // 停止音频处理 + stopCustomAudioProcessing() + + isContinuousRecognitionActive = false + FileLogger.d(TAG, "连续识别已停止") + callback.onSuccess("连续识别已停止") + + return true + } catch (e: Exception) { + // 强制重置状态 + isContinuousRecognitionActive = false + FileLogger.e(TAG, "停止连续识别失败: ${e.message}") + e.printStackTrace() + callback.onError("停止连续识别失败: ${e.message}") + + // 停止音频处理 + stopCustomAudioProcessing() + + // 尝试强制关闭识别器 + try { + recognizer?.close() + recognizer = null + } catch (ex: Exception) { + FileLogger.e(TAG, "关闭识别器失败: ${ex.message}") + } + + return false + } + } + + // 释放资源 + fun dispose() { + try { + // 如果正在进行连续识别,先停止 + if (isContinuousRecognitionActive) { + recognizer?.stopContinuousRecognitionAsync()?.get() + isContinuousRecognitionActive = false + } + + // 停止音频处理 + stopCustomAudioProcessing() + + // 释放recognizer + recognizer?.close() + recognizer = null + + // 释放speechConfig + speechConfig?.close() + speechConfig = null + + FileLogger.d(TAG, "资源已释放") + } catch (e: Exception) { + FileLogger.e(TAG, "释放资源失败: ${e.message}") + + // 确保状态被重置 + isContinuousRecognitionActive = false + customAudioProcessor = null + pushStream = null + audioConfig = null + recognizer = null + speechConfig = null + } + } + + // 检查连续识别是否处于活跃状态 + fun isContinuousRecognitionActive(): Boolean { + return isContinuousRecognitionActive + } + + // 设置自定义音频处理 + private fun setupCustomAudioProcessing() { + try { + // 创建音频推送流 + pushStream = PushAudioInputStream.create() + + // 创建音频配置 + audioConfig = AudioConfig.fromStreamInput(pushStream) + + // 创建自定义音频处理器 + customAudioProcessor = CustomAudioProcessor(pushStream!!) + } catch (e: Exception) { + FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") + e.printStackTrace() + } + } + + // 启动自定义音频处理 + private fun startCustomAudioProcessing() { + if (useEchoCancellation && customAudioProcessor != null) { + try { + customAudioProcessor?.startProcessing() + } catch (e: Exception) { + FileLogger.e(TAG, "启动音频处理器失败") + e.printStackTrace() + } + } + } + + // 停止自定义音频处理 + private fun stopCustomAudioProcessing() { + if (customAudioProcessor != null) { + try { + customAudioProcessor?.stopProcessing() + customAudioProcessor = null + } catch (e: Exception) { + FileLogger.e(TAG, "停止音频处理器失败: ${e.message}") + e.printStackTrace() + } + } + } + + // 修改认证取消事件处理代码 + private fun setupCancelledEventHandler(callback: ContinuousRecognizeCallback) { + recognizer?.canceled?.addEventListener( + EventHandler { _, event -> + val errorDetails = try { + event.errorDetails ?: "未知错误" + } catch (e: Exception) { + "未知错误" + } + val reason = event.reason.toString() + FileLogger.e(TAG, "识别取消: $errorDetails") + isContinuousRecognitionActive = false + stopCustomAudioProcessing() + callback.onCanceled(reason, errorDetails) + callback.onError("识别取消: $errorDetails") // 兼容旧接口 + } + ) + } + + // 自定义音频处理器 + private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream) { + private val isProcessing = AtomicBoolean(false) + private var audioRecord: AudioRecord? = null + private var echoCanceler: AcousticEchoCanceler? = null + private var noiseSuppressor: NoiseSuppressor? = null + private var automaticGainControl: AutomaticGainControl? = null + + // 音频配置 + private val sampleRate = 16000 // 16kHz,适合语音识别 + private val channelConfig = AudioFormat.CHANNEL_IN_MONO + private val audioFormat = AudioFormat.ENCODING_PCM_16BIT + + // 计算最小 buffer 大小 + private val bufferSize = AudioRecord.getMinBufferSize( + sampleRate, channelConfig, audioFormat + ) + + // 启动音频处理 + fun startProcessing() { + if (isProcessing.get()) return + + // 创建录音对象 + try { + audioRecord = AudioRecord( + MediaRecorder.AudioSource.VOICE_RECOGNITION, + sampleRate, + channelConfig, + audioFormat, + bufferSize * 2 // 使用更大的缓冲区以确保不会丢失数据 + ) + + // 创建音频处理效果 + if (AcousticEchoCanceler.isAvailable()) { + echoCanceler = AcousticEchoCanceler.create(audioRecord!!.audioSessionId) + echoCanceler?.enabled = true + } + + if (NoiseSuppressor.isAvailable()) { + noiseSuppressor = NoiseSuppressor.create(audioRecord!!.audioSessionId) + noiseSuppressor?.enabled = true + } + + if (AutomaticGainControl.isAvailable()) { + automaticGainControl = AutomaticGainControl.create(audioRecord!!.audioSessionId) + automaticGainControl?.enabled = true + } + + // 开始录音 + audioRecord?.startRecording() + + // 处理线程 + Thread { + android.os.Process.setThreadPriority(Process.THREAD_PRIORITY_AUDIO) + processAudio() + }.start() + + isProcessing.set(true) + FileLogger.d(TAG, "音频处理已启动") + } catch (e: Exception) { + FileLogger.e(TAG, "创建音频处理器失败: ${e.message}") + releaseAudioResources() + throw e + } + } + + // 停止音频处理 + fun stopProcessing() { + if (!isProcessing.get()) return + + isProcessing.set(false) + releaseAudioResources() + FileLogger.d(TAG, "音频处理已停止") + } + + // 释放音频资源 + private fun releaseAudioResources() { + try { + audioRecord?.stop() + + echoCanceler?.release() + echoCanceler = null + + noiseSuppressor?.release() + noiseSuppressor = null + + automaticGainControl?.release() + automaticGainControl = null + + audioRecord?.release() + audioRecord = null + } catch (e: Exception) { + FileLogger.e(TAG, "释放音频资源失败: ${e.message}") + } + } + + // 音频处理线程 + private fun processAudio() { + // 设置线程优先级 + try { + Process.setThreadPriority(Process.THREAD_PRIORITY_URGENT_AUDIO) + } catch (e: Exception) { + FileLogger.e(TAG, "设置线程优先级失败") + } + + val buffer = ByteArray(bufferSize) + + while (isProcessing.get()) { + try { + val readSize = audioRecord?.read(buffer, 0, buffer.size) ?: -1 + + if (readSize > 0) { + // 修复: 只传入buffer,不传readSize + // 创建新的byte数组,只包含读取到的数据 + val audioData = buffer.copyOfRange(0, readSize) + pushStream.write(audioData) + } + + // 适当休眠,避免占用过多 CPU + Thread.sleep(5) + } catch (e: Exception) { + if (isProcessing.get()) { + FileLogger.e(TAG, "处理音频数据异常: ${e.message}") + } + break + } + } + } + } + + // 一次性识别回调接口 + interface RecognizeCallback { + fun onResult(text: String, detectedLanguage: String) + fun onError(error: String) + } + + // 连续识别回调接口 + interface ContinuousRecognizeCallback { + fun onResult(text: String, detectedLanguage: String) + fun onRecognizing(recognizing: String, detectedLanguage: String) + fun onSessionStarted() + fun onSessionStopped() + fun onCanceled(reason: String, errorDetails: String) + fun onError(error: String) + + // 兼容旧版本的接口 + fun onSuccess(message: String) {} + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt new file mode 100644 index 000000000..090a72efc --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt @@ -0,0 +1,297 @@ +package com.yunqiinnovation.azure_speech + +import android.content.Context +import android.app.Activity +import android.os.Handler +import android.os.Looper +import androidx.annotation.NonNull +import com.yunqiinnovation.azure_speech.utils.FileLogger + +import io.flutter.embedding.engine.plugins.FlutterPlugin +import io.flutter.plugin.common.MethodCall +import io.flutter.plugin.common.MethodChannel +import io.flutter.plugin.common.MethodChannel.MethodCallHandler +import io.flutter.plugin.common.MethodChannel.Result +import io.flutter.plugin.common.EventChannel + +/** AzureSpeechPlugin */ +class AzureSpeechPlugin: FlutterPlugin { + private val TAG = "AzureSpeechPlugin" + private lateinit var context: Context + private val mainHandler = Handler(Looper.getMainLooper()) + + // ASR相关 + private lateinit var asrChannel : MethodChannel + private lateinit var asrEventChannel: EventChannel + private var asrEventSink: EventChannel.EventSink? = null + private lateinit var azureAsrHelper: AzureAsrHelper + + // TTS相关 + private lateinit var ttsChannel : MethodChannel + private lateinit var azureTtsHelper: AzureTtsHelper + + // ASR 事件发送方法 + private fun sendAsrEvent(event: Map) { + FileLogger.d(TAG, "发送ASR事件: $event") + if (asrEventSink == null) { + FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") + return + } + + mainHandler.post { + try { + asrEventSink?.success(event) + FileLogger.d(TAG, "ASR事件发送成功") + } catch (e: Exception) { + FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") + } + } + } + + override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { + context = flutterPluginBinding.applicationContext + + // 初始化ASR通道 + asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr") + asrChannel.setMethodCallHandler(AsrMethodHandler()) + + // 初始化TTS通道 + ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/tts") + ttsChannel.setMethodCallHandler(TtsMethodHandler()) + + // 初始化ASR事件通道 + asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events") + asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { + override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { + asrEventSink = events + } + + override fun onCancel(arguments: Any?) { + asrEventSink = null + } + }) + + // 初始化Azure语音服务 + azureTtsHelper = AzureTtsHelper(context) + azureAsrHelper = AzureAsrHelper(context) + } + + // ASR方法处理器 + inner class AsrMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + val subscriptionKey = call.argument("subscriptionKey") ?: "" + val region = call.argument("region") ?: "" + val supportedLanguages = call.argument>("supportedLanguages") ?: listOf("zh-CN") + + try { + val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages.toTypedArray()) + result.success(success) + } catch (e: Exception) { + result.error("INITIALIZATION_ERROR", e.message, null) + } + } + "recognizeOnce" -> { + azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) { + mainHandler.post { + result.success(mapOf( + "text" to text, + "detectedLanguage" to detectedLanguage + )) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("RECOGNITION_ERROR", error, null) + } + } + }) + } + "startContinuousRecognition" -> { + // 确保事件通道已准备好 + if (asrEventSink == null) { + result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) + return + } + + val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) { + sendAsrEvent(mapOf( + "type" to "result", + "text" to text, + "detectedLanguage" to detectedLanguage + )) + } + + override fun onRecognizing(recognizing: String, detectedLanguage: String) { + sendAsrEvent(mapOf( + "type" to "recognizing", + "text" to recognizing, + "detectedLanguage" to detectedLanguage + )) + } + + override fun onSessionStarted() { + sendAsrEvent(mapOf("type" to "sessionStarted")) + } + + override fun onSessionStopped() { + sendAsrEvent(mapOf("type" to "sessionStopped")) + } + + override fun onCanceled(reason: String, errorDetails: String) { + sendAsrEvent(mapOf( + "type" to "canceled", + "reason" to reason, + "errorDetails" to errorDetails + )) + } + + override fun onError(error: String) { + sendAsrEvent(mapOf("type" to "error", "message" to error)) + } + + override fun onSuccess(message: String) { + sendAsrEvent(mapOf("type" to "success", "message" to message)) + } + }) + result.success(success) + } + "stopContinuousRecognition" -> { + try { + if (!azureAsrHelper.isContinuousRecognitionActive()) { + result.success(true) + return + } + + val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { + override fun onResult(text: String, detectedLanguage: String) {} + override fun onRecognizing(recognizing: String, detectedLanguage: String) {} + override fun onSessionStarted() {} + override fun onSessionStopped() {} + override fun onCanceled(reason: String, errorDetails: String) {} + override fun onError(error: String) { + mainHandler.post { + result.error("STOP_ERROR", error, null) + } + } + override fun onSuccess(message: String) {} + }) + result.success(success) + } catch (e: Exception) { + result.error("STOP_ERROR", e.message, null) + } + } + "isContinuousRecognitionActive" -> { + result.success(azureAsrHelper.isContinuousRecognitionActive()) + } + "dispose" -> { + azureAsrHelper.dispose() + result.success(true) + } + else -> { + result.notImplemented() + } + } + } + } + + // TTS方法处理器 + inner class TtsMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + val subscriptionKey = call.argument("subscriptionKey") ?: "" + val region = call.argument("region") ?: "" + val language = call.argument("language") ?: "zh-CN" + + val success = azureTtsHelper.initialize(subscriptionKey, region, language) + result.success(success) + } + "setVoice" -> { + val voiceName = call.argument("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) + result.success(azureTtsHelper.setVoice(voiceName)) + } + "setSpeechParams" -> { + val rate = call.argument("rate") ?: 0 + val pitch = call.argument("pitch") ?: 0 + val volume = call.argument("volume") ?: 100 + result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) + } + "setAudioOutputType" -> { + val outputTypeStr = call.argument("outputType") ?: "speaker" + val outputType = when (outputTypeStr.lowercase()) { + "speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER + "earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE + "auto" -> AzureTtsHelper.AudioOutputType.AUTO + else -> AzureTtsHelper.AudioOutputType.SPEAKER + } + result.success(azureTtsHelper.setAudioOutputType(outputType)) + } + "speakText" -> { + val text = call.argument("text") ?: return result.error("INVALID_ARGUMENTS", "文本不能为空", null) + + azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + mainHandler.post { + result.success(true) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("SPEAK_ERROR", error, null) + } + } + }) + } + "speakSsml" -> { + val ssml = call.argument("ssml") ?: return result.error("INVALID_ARGUMENTS", "SSML不能为空", null) + + azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { + override fun onSuccess(message: String) { + mainHandler.post { + result.success(true) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("SPEAK_ERROR", error, null) + } + } + }) + } + "stopSpeaking" -> { + result.success(azureTtsHelper.stopSpeaking()) + } + "isSpeaking" -> { + result.success(azureTtsHelper.isSpeaking()) + } + "dispose" -> { + azureTtsHelper.dispose() + result.success(true) + } + else -> { + result.notImplemented() + } + } + } + } + + override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { + asrChannel.setMethodCallHandler(null) + ttsChannel.setMethodCallHandler(null) + asrEventChannel.setStreamHandler(null) + + try { + azureTtsHelper.dispose() + azureAsrHelper.dispose() + } catch (e: Exception) { + FileLogger.e(TAG, "Dispose resources error: ${e.message}") + } + } +} \ No newline at end of file diff --git a/android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt similarity index 98% rename from android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt rename to local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt index 31971c2a4..0ae180fa9 100644 --- a/android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt @@ -1,11 +1,11 @@ -package com.yunqiinnovation.deepsound +package com.yunqiinnovation.azure_speech import android.content.Context import android.media.AudioAttributes import android.media.AudioDeviceInfo import android.media.AudioManager import android.os.Build -import com.yunqiinnovation.deepsound.core.utils.FileLogger +import com.yunqiinnovation.azure_speech.utils.FileLogger import com.microsoft.cognitiveservices.speech.* import com.microsoft.cognitiveservices.speech.audio.* import java.util.concurrent.Future @@ -50,16 +50,16 @@ class AzureTtsHelper(private val context: Context) { * 初始化 TTS 引擎 * * @param subscriptionKey Azure 语音服务订阅密钥 - * @param serviceRegion Azure 语音服务区域 + * @param region Azure 语音服务区域 * @param language 可选,默认语言,默认为 "zh-CN" */ - fun initialize(subscriptionKey: String, serviceRegion: String, language: String = "zh-CN"): Boolean { + fun initialize(subscriptionKey: String, region: String, language: String = "zh-CN"): Boolean { try { audioManager = context.getSystemService(Context.AUDIO_SERVICE) as AudioManager // 创建语音配置 - speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) + speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) // 设置语音合成输出格式为高质量音频 speechConfig?.setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff24Khz16BitMonoPcm) @@ -96,7 +96,7 @@ class AzureTtsHelper(private val context: Context) { return true } catch (e: Exception) { - FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${serviceRegion}") + FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${region}") e.printStackTrace() return false } diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt new file mode 100644 index 000000000..d210ec48f --- /dev/null +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt @@ -0,0 +1,45 @@ +package com.yunqiinnovation.azure_speech.utils + +import android.util.Log + +/** + * 简单的文件日志工具类 + */ +object FileLogger { + private const val TAG_PREFIX = "AzureSpeech_" + + /** + * 记录调试信息 + */ + fun d(tag: String, message: String) { + Log.d("$TAG_PREFIX$tag", message) + } + + /** + * 记录信息 + */ + fun i(tag: String, message: String) { + Log.i("$TAG_PREFIX$tag", message) + } + + /** + * 记录警告信息 + */ + fun w(tag: String, message: String) { + Log.w("$TAG_PREFIX$tag", message) + } + + /** + * 记录错误信息 + */ + fun e(tag: String, message: String) { + Log.e("$TAG_PREFIX$tag", message) + } + + /** + * 记录异常 + */ + fun e(tag: String, message: String, throwable: Throwable) { + Log.e("$TAG_PREFIX$tag", message, throwable) + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift new file mode 100644 index 000000000..d3dd8d0b8 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift @@ -0,0 +1,461 @@ +import Foundation +import AVFoundation +import MicrosoftCognitiveServicesSpeech + +/// Azure 语音识别辅助类 +class AzureAsrHelper: NSObject { + private var recognizer: SPXSpeechRecognizer? + private var speechConfig: SPXSpeechConfig? + private var audioConfig: SPXAudioConfig? + private var initialized = false + private var isContinuousRecognitionActive = false + private var currentLanguage = "zh-CN" + private var subscriptionKey = "" + private var serviceRegion = "" + private var isAutoDetectLanguage = false + private var supportedLanguages = ["zh-CN", "en-US"] + + // 音频会话管理 + private let audioSession = AVAudioSession.sharedInstance() + + // 事件回调 + private var eventHandler: (([String: Any]) -> Void)? + + /// 设置事件处理器 + /// + /// - Parameter handler: 事件处理回调 + func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) { + self.eventHandler = handler + } + + /// 初始化语音识别服务 + /// + /// - Parameters: + /// - speechSubscriptionKey: Azure 语音服务订阅密钥 + /// - serviceRegion: Azure 语音服务区域 + /// - supportedLanguages: 支持的语言列表,默认为 ["zh-CN", "en-US"] + /// - Returns: 是否初始化成功 + func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String] = ["zh-CN", "en-US"]) -> Bool { + print("[AzureAsrHelper] 初始化 Azure 语音服务") + + // 检查配置是否为空 + if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + // 保存配置 + self.subscriptionKey = speechSubscriptionKey + self.serviceRegion = serviceRegion + + // 设置语言 + if supportedLanguages.isEmpty { + print("[AzureAsrHelper] 警告: 传入的支持语言列表为空,将使用默认语言") + } else { + self.supportedLanguages = supportedLanguages + } + + // 根据支持的语言数量决定是否启用自动语言检测 + self.isAutoDetectLanguage = supportedLanguages.count >= 2 + + // 如果只有一种语言,设置为当前语言 + if !isAutoDetectLanguage && !supportedLanguages.isEmpty { + self.currentLanguage = supportedLanguages[0] + } + + // 创建语音配置 + do { + speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) + + // 设置语言配置 + if isAutoDetectLanguage { + // 设置自动语言检测 + try speechConfig?.setPropertyTo("Continuous", byId: SPXPropertyId.SpeechServiceConnection_LanguageIdMode) + } else { + // 设置指定的识别语言 + speechConfig?.speechRecognitionLanguage = currentLanguage + } + + // 创建音频配置 - 使用默认麦克风 + audioConfig = SPXAudioConfig.default() + + // 创建识别器 + if isAutoDetectLanguage { + let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) + } else { + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) + } + + // 配置音频会话 + try configureAudioSession() + + initialized = true + print("[AzureAsrHelper] Azure 语音服务初始化成功") + return true + } catch { + print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") + return false + } + } + + /// 配置音频会话 + private func configureAudioSession() throws { + print("[AzureAsrHelper] 开始配置音频会话...") + + do { + // 设置音频会话类别和模式 + try audioSession.setCategory(.record, mode: .measurement, options: [.duckOthers, .allowBluetooth]) + try audioSession.setActive(true, options: .notifyOthersOnDeactivation) + } catch { + print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败") + throw error + } + } + + /// 一次性语音识别 + /// + /// - Parameter completion: 完成回调,返回是否成功、识别文本、识别语言和可能的错误信息 + func recognizeOnce(completion: @escaping (Bool, String?, String?, String?) -> Void) { + if !initialized { + completion(false, nil, nil, "语音服务未初始化") + return + } + + // 重置 recognizer + if !resetRecognizer() { + completion(false, nil, nil, "重置识别器失败") + return + } + + do { + // 激活音频会话 + try audioSession.setActive(true) + + // 添加识别事件处理 + recognizer?.addRecognizedEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizedSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + completion(true, event.result.text, detectedLanguage, nil) + } + } + + recognizer?.addRecognizingEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizingSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + // print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") + } + } + + // 添加会话事件处理 + recognizer?.addSessionStartedEventHandler { _, _ in + print("[AzureAsrHelper] 识别会话已开始") + } + + recognizer?.addSessionStoppedEventHandler { _, _ in + print("[AzureAsrHelper] 识别会话已结束") + } + + // 添加取消事件处理 + recognizer?.addCanceledEventHandler { _, event in + if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { + let errorDetails = cancellationDetails.errorDetails ?? "未知错误" + print("[AzureAsrHelper] 识别取消: \(errorDetails)") + completion(false, nil, nil, "识别取消: \(errorDetails)") + } + } + + // 执行识别 + let result = try recognizer?.recognizeOnceAsync().get() + + if result?.reason != .recognizedSpeech { + completion(false, nil, nil, "未能识别语音") + } + } catch { + completion(false, nil, nil, "识别异常: \(error.localizedDescription)") + } + } + + /// 重置识别器 + /// + /// - Returns: 是否重置成功 + private func resetRecognizer() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + // 检查配置是否为空 + if subscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + do { + // 释放之前的 recognizer + recognizer = nil + + // 创建音频配置 - 使用默认麦克风 + audioConfig = SPXAudioConfig.default() + + // 重新创建识别器 + if isAutoDetectLanguage { + let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) + } else { + recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) + } + + return true + } catch { + print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") + return false + } + } + + /// 获取检测到的语言 + /// + /// - Parameter result: 识别结果 + /// - Returns: 检测到的语言代码 + private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { + if isAutoDetectLanguage { + do { + if let autoDetectResult = try SPXAutoDetectSourceLanguageResult(fromRecognitionResult: result) { + return autoDetectResult.language + } + return "" + } catch { + print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") + return "" + } + } else { + return currentLanguage + } + } + + /// 开始连续语音识别 + /// + /// - Returns: 是否成功启动连续识别 + func startContinuousRecognition() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + // 检查配置是否为空 + if subscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureAsrHelper] 错误: Azure 配置信息不完整") + return false + } + + // 如果已经在进行连续识别,直接返回 + if isContinuousRecognitionActive { + print("[AzureAsrHelper] 已经在进行连续识别中,忽略请求") + return true + } + + // 重置 recognizer + if !resetRecognizer() { + print("[AzureAsrHelper] 尝试重新创建识别器...") + return false + } + + do { + // 激活音频会话 + try audioSession.setActive(true) + + // 添加识别事件处理 + recognizer?.addRecognizedEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizedSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + let eventData: [String: Any] = [ + "eventType": "finalResult", + "text": event.result.text ?? "", + "language": detectedLanguage + ] + self.eventHandler?(eventData) + } + } + + // 识别中事件 + recognizer?.addRecognizingEventHandler { [weak self] _, event in + guard let self = self else { return } + + if event.result.reason == .recognizingSpeech { + let detectedLanguage = self.getDetectedLanguage(from: event.result) + let eventData: [String: Any] = [ + "eventType": "recognizing", + "text": event.result.text ?? "", + "language": detectedLanguage + ] + self.eventHandler?(eventData) + } + } + + // 会话事件 + recognizer?.addSessionStartedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + let eventData: [String: Any] = [ + "eventType": "sessionStarted" + ] + self.eventHandler?(eventData) + } + + recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in + guard let self = self else { return } + + self.isContinuousRecognitionActive = false + let eventData: [String: Any] = [ + "eventType": "sessionStopped" + ] + self.eventHandler?(eventData) + } + + // 取消事件 + recognizer?.addCanceledEventHandler { [weak self] _, event in + guard let self = self else { return } + + self.isContinuousRecognitionActive = false + var errorMessage = "未知错误" + + if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { + errorMessage = cancellationDetails.errorDetails ?? "未知错误" + } + + let eventData: [String: Any] = [ + "eventType": "error", + "error": "识别取消: \(errorMessage)" + ] + self.eventHandler?(eventData) + } + + // 开始连续识别 + try recognizer?.startContinuousRecognition() + isContinuousRecognitionActive = true + print("[AzureAsrHelper] 连续识别已启动") + + return true + } catch { + print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") + return false + } + } + + /// 停止连续语音识别 + /// + /// - Returns: 是否成功停止连续识别 + func stopContinuousRecognition() -> Bool { + if !initialized { + print("[AzureAsrHelper] 错误: 语音服务未初始化") + return false + } + + if !isContinuousRecognitionActive { + print("[AzureAsrHelper] 未进行连续识别,忽略停止请求") + return true + } + + do { + print("[AzureAsrHelper] 停止连续语音识别") + + if recognizer == nil { + if isContinuousRecognitionActive { + print("[AzureAsrHelper] 警告: 识别器为空,但状态显示活跃") + } + isContinuousRecognitionActive = false + + // 通知停止成功 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止" + ] + eventHandler?(eventData) + return true + } + + // 停止连续识别 + try recognizer?.stopContinuousRecognition() + + // 延迟一点时间确保处理完成 + DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { [weak self] in + guard let self = self else { return } + + // 重置状态 + self.isContinuousRecognitionActive = false + + // 恢复音频会话 + do { + try self.audioSession.setActive(false, options: .notifyOthersOnDeactivation) + } catch { + // 忽略错误 + } + + print("[AzureAsrHelper] 连续识别已停止") + + // 通知停止成功 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止" + ] + self.eventHandler?(eventData) + } + + return true + } catch { + // 强制重置状态 + isContinuousRecognitionActive = false + print("[AzureAsrHelper] 警告: 停止连续识别失败: \(error.localizedDescription)") + + // 通知停止失败,但仍然视为处理完成 + let eventData: [String: Any] = [ + "eventType": "success", + "message": "连续识别已停止(但有错误)" + ] + eventHandler?(eventData) + + return false + } + } + + /// 释放资源 + func dispose() { + // 如果正在进行连续识别,先停止 + if isContinuousRecognitionActive { + _ = stopContinuousRecognition() + } + + // 恢复音频会话 + do { + try audioSession.setActive(false, options: .notifyOthersOnDeactivation) + } catch { + // 忽略错误 + } + + // 释放资源 + recognizer = nil + speechConfig = nil + audioConfig = nil + + initialized = false + isContinuousRecognitionActive = false + print("[AzureAsrHelper] 资源已释放") + } + + /// 检查连续识别是否处于活跃状态 + /// + /// - Returns: 是否正在进行连续识别 + func isContinuousRecognitionActive() -> Bool { + return isContinuousRecognitionActive + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift new file mode 100644 index 000000000..dcb2d57b8 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift @@ -0,0 +1,182 @@ +import Flutter +import UIKit + +public class AzureSpeechPlugin: NSObject, FlutterPlugin { + private var ttsHelper: AzureTtsHelper? + private var asrHelper: AzureAsrHelper? + private var eventSink: FlutterEventSink? + + public static func register(with registrar: FlutterPluginRegistrar) { + let channel = FlutterMethodChannel(name: "azure_speech", binaryMessenger: registrar.messenger()) + let instance = AzureSpeechPlugin() + registrar.addMethodCallDelegate(instance, channel: channel) + + // 初始化事件通道 + let eventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) + eventChannel.setStreamHandler(AsrStreamHandler(instance: instance)) + } + + override init() { + super.init() + ttsHelper = AzureTtsHelper() + asrHelper = AzureAsrHelper() + + // 设置ASR事件处理 + asrHelper?.setEventHandler { [weak self] event in + self?.handleAsrEvent(event) + } + } + + public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { + switch call.method { + // TTS相关方法 + case "initializeTts": + guard let args = call.arguments as? [String: Any], + let subscriptionKey = args["subscriptionKey"] as? String, + let serviceRegion = args["serviceRegion"] as? String else { + result(false) + return + } + + let language = args["language"] as? String ?? "zh-CN" + let success = ttsHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, language: language) ?? false + result(success) + + case "setTtsVoice": + guard let args = call.arguments as? [String: Any], + let voiceName = args["voiceName"] as? String else { + result(false) + return + } + + let success = ttsHelper?.setVoice(voiceName: voiceName) ?? false + result(success) + + case "setTtsSpeechParams": + guard let args = call.arguments as? [String: Any], + let rate = args["rate"] as? Int, + let pitch = args["pitch"] as? Int, + let volume = args["volume"] as? Int else { + result(false) + return + } + + let success = ttsHelper?.setSpeechParams(rate: rate, pitch: pitch, volume: volume) ?? false + result(success) + + case "speakText": + guard let args = call.arguments as? [String: Any], + let text = args["text"] as? String else { + result(false) + return + } + + ttsHelper?.speakText(text: text) { success, _ in + result(success) + } + + case "stopSpeaking": + let success = ttsHelper?.stopSpeaking() ?? false + result(success) + + case "isSpeaking": + let speaking = ttsHelper?.isSpeaking() ?? false + result(speaking) + + case "setTtsAudioOutputType": + guard let args = call.arguments as? [String: Any], + let outputType = args["outputType"] as? String else { + result(false) + return + } + + var type: AzureTtsHelper.AudioOutputType = .auto + switch outputType.uppercased() { + case "SPEAKER": + type = .speaker + case "EARPIECE": + type = .earpiece + default: + type = .auto + } + + let success = ttsHelper?.setAudioOutputType(outputType: type) ?? false + result(success) + + // ASR相关方法 + case "initializeAsr": + guard let args = call.arguments as? [String: Any], + let subscriptionKey = args["subscriptionKey"] as? String, + let serviceRegion = args["serviceRegion"] as? String, + let supportedLanguages = args["supportedLanguages"] as? [String] else { + result(false) + return + } + + let success = asrHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, supportedLanguages: supportedLanguages) ?? false + result(success) + + case "recognizeOnce": + asrHelper?.recognizeOnce { success, text, language, error in + var resultMap: [String: Any] = ["success": success] + if success { + resultMap["text"] = text + resultMap["language"] = language + } else { + resultMap["error"] = error + } + result(resultMap) + } + + case "startContinuousRecognition": + let success = asrHelper?.startContinuousRecognition() ?? false + result(success) + + case "stopContinuousRecognition": + let success = asrHelper?.stopContinuousRecognition() ?? false + result(success) + + case "isContinuousRecognitionActive": + let isActive = asrHelper?.isContinuousRecognitionActive() ?? false + result(isActive) + + case "dispose": + ttsHelper?.dispose() + asrHelper?.dispose() + result(nil) + + default: + result(FlutterMethodNotImplemented) + } + } + + // 设置事件接收器 + func setEventSink(_ sink: FlutterEventSink?) { + self.eventSink = sink + } + + // 处理ASR事件 + private func handleAsrEvent(_ event: [String: Any]) { + self.eventSink?(event) + } +} + +// ASR事件流处理器 +class AsrStreamHandler: NSObject, FlutterStreamHandler { + private weak var plugin: AzureSpeechPlugin? + + init(instance: AzureSpeechPlugin) { + self.plugin = instance + super.init() + } + + func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { + plugin?.setEventSink(events) + return nil + } + + func onCancel(withArguments arguments: Any?) -> FlutterError? { + plugin?.setEventSink(nil) + return nil + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift new file mode 100644 index 000000000..4242a9037 --- /dev/null +++ b/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift @@ -0,0 +1,330 @@ +import Foundation +import AVFoundation +import MicrosoftCognitiveServicesSpeech + +/// Azure 语音合成辅助类 +class AzureTtsHelper: NSObject { + private var synthesizer: SPXSpeechSynthesizer? + private var speechConfig: SPXSpeechConfig? + private var audioConfig: SPXAudioConfig? + private var initialized = false + private var speaking = false + + // 音频输出类型 + enum AudioOutputType { + case speaker // 扬声器 + case earpiece // 听筒 + case auto // 自动选择 + } + + // 当前设置 + private var currentVoiceName = "zh-CN-XiaoxiaoNeural" + private var currentSpeechRate = 0 + private var currentPitch = 0 + private var currentVolume = 100 + private var currentAudioOutputType: AudioOutputType = .auto + + // 音频会话管理 + private let audioSession = AVAudioSession.sharedInstance() + + /// 初始化 TTS 引擎 + /// + /// - Parameters: + /// - speechSubscriptionKey: Azure 语音服务订阅密钥 + /// - serviceRegion: Azure 语音服务区域 + /// - language: 可选,默认语言,默认为 "zh-CN" + /// - Returns: 是否初始化成功 + func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { + print("[AzureTtsHelper] 初始化 Azure 语音服务") + + // 检查配置是否为空 + if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { + print("[AzureTtsHelper] 错误: Azure 配置信息不完整") + return false + } + + // 释放之前的资源 + dispose() + + do { + // 创建语音配置 + speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) + + // 设置语音合成输出格式为高质量音频 + speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm) + + // 设置默认语言 + speechConfig?.setSpeechSynthesisLanguage(language) + + // 设置默认语音 + speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) + + // 创建音频配置 - 使用默认扬声器 + audioConfig = SPXAudioConfig.default() + + // 创建语音合成器 + synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) + + initialized = true + + // 设置默认音频输出类型为自动 + setAudioOutputType(outputType: .auto) + + print("[AzureTtsHelper] TTS 引擎初始化成功") + return true + } catch { + print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)") + return false + } + } + + /// 设置音频输出设备类型 + /// + /// - Parameter outputType: 音频输出设备类型 + /// - Returns: 是否设置成功 + func setAudioOutputType(outputType: AudioOutputType) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + do { + currentAudioOutputType = outputType + + switch outputType { + case .speaker: + // 使用扬声器 + try audioSession.setCategory(.playback, mode: .default) + try audioSession.overrideOutputAudioPort(.speaker) + print("[AzureTtsHelper] 已设置音频输出设备为扬声器") + + case .earpiece: + // 使用听筒 + try audioSession.setCategory(.playback, mode: .voiceChat) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为听筒") + + case .auto: + // 检查是否有耳机连接 + let outputs = audioSession.currentRoute.outputs + let hasHeadphones = outputs.contains { output in + return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP + } + + if hasHeadphones { + // 有耳机,使用耳机 + try audioSession.setCategory(.playback, mode: .default) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为耳机") + } else { + // 无耳机,使用听筒 + try audioSession.setCategory(.playback, mode: .voiceChat) + try audioSession.overrideOutputAudioPort(.none) + print("[AzureTtsHelper] 已设置音频输出设备为听筒") + } + } + + try audioSession.setActive(true) + return true + } catch { + print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)") + return false + } + } + + /// 设置语音 + /// + /// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural" + /// - Returns: 是否设置成功 + func setVoice(voiceName: String) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + if voiceName == currentVoiceName { + print("[AzureTtsHelper] 已设置语音: \(voiceName)") + return true + } + + do { + currentVoiceName = voiceName + speechConfig?.setSpeechSynthesisVoiceName(voiceName) + + // 重新创建合成器 + synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) + + print("[AzureTtsHelper] 已设置语音: \(voiceName)") + return true + } catch { + print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)") + return false + } + } + + /// 设置语音合成参数 + /// + /// - Parameters: + /// - rate: 语速,范围 -100 到 100,默认为 0 + /// - pitch: 音调,范围 -100 到 100,默认为 0 + /// - volume: 音量,范围 0 到 100,默认为 100 + /// - Returns: 是否设置成功 + func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + currentSpeechRate = rate + currentPitch = pitch + currentVolume = volume + + print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)") + return true + } + + /// 合成文本为语音并播放 + /// + /// - Parameters: + /// - text: 要合成的文本 + /// - completion: 完成回调,返回是否成功和可能的错误信息 + func speakText(text: String, completion: @escaping (Bool, String?) -> Void) { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + completion(false, "TTS 引擎尚未初始化") + return + } + + do { + print("[AzureTtsHelper] 开始合成文本: \(text)") + + // 生成 SSML + let ssml = generateSsml(text: text) + + // 使用 SSML 合成语音 + speakSsml(ssml: ssml, completion: completion) + } catch { + print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") + completion(false, "语音合成异常: \(error.localizedDescription)") + } + } + + /// 生成 SSML 文本 + /// + /// - Parameter text: 要转换的文本 + /// - Returns: SSML 格式的文本 + private func generateSsml(text: String) -> String { + // 计算 SSML 参数 + let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%") + let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%" + let volumeParam = "\(min(max(currentVolume, 0), 100))%" + + return """ + + + + \(text) + + + + """ + } + + /// 合成 SSML 为语音并播放 + /// + /// - Parameters: + /// - ssml: SSML 格式的文本 + /// - completion: 完成回调,返回是否成功和可能的错误信息 + private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + completion(false, "TTS 引擎尚未初始化") + return + } + + do { + print("[AzureTtsHelper] 开始合成 SSML") + + // 标记为正在播放 + speaking = true + + // 激活音频会话 + try audioSession.setActive(true) + + // 异步合成语音 + let result = try synthesizer!.speakSsml(ssml) + + switch result.reason { + case .synthesizingAudioCompleted: + print("[AzureTtsHelper] 语音合成完成") + speaking = false + completion(true, "语音合成完成") + case .canceled: + if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) { + print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") + speaking = false + completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") + } else { + print("[AzureTtsHelper] 语音合成取消") + speaking = false + completion(false, "语音合成取消") + } + default: + print("[AzureTtsHelper] 语音合成失败: \(result.reason)") + speaking = false + completion(false, "语音合成失败: \(result.reason)") + } + } catch { + print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") + speaking = false + completion(false, "语音合成异常: \(error.localizedDescription)") + } + } + + /// 停止当前语音合成 + /// + /// - Returns: 是否停止成功 + func stopSpeaking() -> Bool { + if !initialized { + print("[AzureTtsHelper] TTS 引擎尚未初始化") + return false + } + + do { + try synthesizer?.stopSpeaking() + speaking = false + print("[AzureTtsHelper] 已停止语音合成") + return true + } catch { + print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)") + return false + } + } + + /// 释放资源 + func dispose() { + do { + stopSpeaking() + + // 恢复音频会话 + try audioSession.setActive(false, options: .notifyOthersOnDeactivation) + + synthesizer = nil + speechConfig = nil + audioConfig = nil + + initialized = false + speaking = false + print("[AzureTtsHelper] TTS 引擎已释放") + } catch { + print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)") + } + } + + /// 检查当前是否正在播放语音 + /// + /// - Returns: 是否正在播放语音 + func isSpeaking() -> Bool { + return speaking + } +} \ No newline at end of file diff --git a/local_plugins/azure_speech/pubspec.yaml b/local_plugins/azure_speech/pubspec.yaml new file mode 100644 index 000000000..5935bbce2 --- /dev/null +++ b/local_plugins/azure_speech/pubspec.yaml @@ -0,0 +1,32 @@ +name: azure_speech +description: Azure语音服务插件,包含TTS和ASR服务 +version: 0.0.1 +homepage: + +environment: + sdk: ">=2.17.0 <3.0.0" + flutter: ">=2.5.0" + +dependencies: + flutter: + sdk: flutter + +dev_dependencies: + flutter_test: + sdk: flutter + flutter_lints: ^2.0.0 + +# For information on the generic Dart part of this file, see the +# following page: https://dart.dev/tools/pub/pubspec + +# The following section is specific to Flutter packages. +flutter: + # This section identifies this Flutter project as a plugin project. + plugin: + platforms: + android: + package: com.yunqiinnovation.azure_speech + pluginClass: AzureSpeechPlugin + ios: + pluginClass: AzureSpeechPlugin + diff --git a/local_plugins/open_ai_service/README.md b/local_plugins/open_ai_service/README.md new file mode 100644 index 000000000..351eee84f --- /dev/null +++ b/local_plugins/open_ai_service/README.md @@ -0,0 +1,253 @@ +# OpenAI Service Plugin + +一个用于Flutter应用的OpenAI服务插件,支持Android和iOS平台。 + +## 功能 + +- 支持文本生成(completions) +- 支持流式输出(streaming) +- 支持函数调用(function calling) +- 支持自定义API基础URL +- 支持自定义模型选择 + +## 安装 + +在你的`pubspec.yaml`文件中添加以下依赖: + +```yaml +dependencies: + open_ai_service: + path: 本地路径/open_ai_service +``` + +## 使用方法 + +### 初始化服务 + +```dart +import 'package:open_ai_service/open_ai_service.dart'; + +final openAIService = OpenAIService(); + +// 初始化服务 +await openAIService.initialize( + apiKey: 'your_openai_api_key', + baseUrl: 'https://api.openai.com/v1/chat/completions', // 可选 + model: 'gpt-4-turbo', // 可选 +); +``` + +### 发送非流式请求 + +```dart +// 创建消息 +final userMessage = await openAIService.createUserMessage('你好,请介绍一下自己'); + +// 发送请求 +final response = await openAIService.sendMessage( + messages: [userMessage], + systemPrompt: '你是一个有用的AI助手', +); + +print('AI回复: $response'); +``` + +### 发送流式请求(回调方式) + +```dart +// 创建消息 +final userMessage = await openAIService.createUserMessage('写一个短故事'); + +// 发送流式请求 +await openAIService.sendMessageStream( + messages: [userMessage], + systemPrompt: '你是一个善于讲故事的AI助手', +); + +// 处理事件 +final subscription = openAIService.processEvents( + onToken: (token) { + // 处理每个返回的token + print(token); + }, + onComplete: () { + // 处理完成事件 + print('生成完成'); + }, + onError: (error) { + // 处理错误 + print('错误: $error'); + }, + onFunctionCall: (functionCall) { + // 处理函数调用 + print('函数调用: ${functionCall['name']}'); + }, +); + +// 在不需要时取消订阅 +subscription.cancel(); +``` + +### 发送流式请求(Stream方式) + +```dart +// 创建消息 +final userMessage = await openAIService.createUserMessage('写一个短故事'); + +// 获取字符串流 +final stream = openAIService.streamMessage( + messages: [userMessage], + systemPrompt: '你是一个善于讲故事的AI助手', +); + +// 使用流 +final StringBuilder responseBuilder = StringBuilder(); + +stream.listen( + (token) { + // 处理每个token + responseBuilder.write(token); + print(token); // 实时输出 + }, + onDone: () { + // 流结束 + print('完整回复: ${responseBuilder.toString()}'); + }, + onError: (error) { + // 错误处理 + print('错误: $error'); + } +); +``` + +### 注册函数 + +```dart +// 注册一个函数 +await openAIService.registerFunction( + name: 'get_weather', + description: '获取指定城市的天气信息', + parameters: { + 'type': 'object', + 'properties': { + 'city': { + 'type': 'string', + 'description': '城市名称', + }, + 'date': { + 'type': 'string', + 'description': '日期,格式为YYYY-MM-DD', + }, + }, + 'required': ['city'], + }, +); +``` + +### 处理函数调用 + +```dart +// 创建消息 +final userMessage = await openAIService.createUserMessage('明天北京的天气如何?'); + +// 发送流式请求 +await openAIService.sendMessageStream( + messages: [userMessage], + systemPrompt: '你是一个有用的AI助手', +); + +// 处理事件 +openAIService.processEvents( + onToken: (token) { + print(token); + }, + onComplete: () { + print('生成完成'); + }, + onError: (error) { + print('错误: $error'); + }, + onFunctionCall: (functionCall) { + // 处理函数调用 + final name = functionCall['name']; + final arguments = functionCall['arguments']; + + print('收到函数调用: $name, 参数: $arguments'); + + // 假设处理了函数调用并获得结果 + final result = '{"temperature": 25, "condition": "sunny"}'; + + // 发送函数调用结果 + openAIService.sendFunctionCallResult( + messages: [userMessage], + systemPrompt: '你是一个有用的AI助手', + functionCall: functionCall, + functionResult: result, + ); + }, +); +``` + +### 使用Stream API处理函数调用 + +```dart +// 创建消息和响应处理器 +final userMessage = await openAIService.createUserMessage('明天北京的天气如何?'); +final responseBuilder = StringBuilder(); + +// 处理事件流以捕获函数调用 +final subscription = openAIService.processEvents( + onFunctionCall: (functionCall) async { + // 取消当前事件监听 + subscription.cancel(); + + // 处理函数调用 + final name = functionCall['name']; + final arguments = functionCall['arguments']; + + print('收到函数调用: $name, 参数: $arguments'); + + // 假设处理了函数调用并获得结果 + final result = '{"temperature": 25, "condition": "sunny"}'; + + // 使用Stream API发送函数调用结果 + final resultStream = openAIService.streamFunctionResult( + messages: [userMessage], + systemPrompt: '你是一个有用的AI助手', + functionCall: functionCall, + functionResult: result, + ); + + // 处理结果流 + resultStream.listen( + (token) { + responseBuilder.write(token); + print(token); // 实时输出 + }, + onDone: () { + print('完整回复: ${responseBuilder.toString()}'); + }, + onError: (error) { + print('错误: $error'); + } + ); + } +); + +// 启动请求 +await openAIService.sendMessageStream( + messages: [userMessage], + systemPrompt: '你是一个有用的AI助手', +); +``` + +## 注意事项 + +1. 确保在使用前已正确初始化服务 +2. 对于流式请求,确保在不需要时取消订阅 +3. 处理函数调用时,确保提供有效的结果格式 +4. 网络请求可能会失败,请确保加入适当的错误处理 + +## 许可证 + +[MIT License](LICENSE) \ No newline at end of file diff --git a/local_plugins/open_ai_service/android/build.gradle.kts b/local_plugins/open_ai_service/android/build.gradle.kts new file mode 100644 index 000000000..4f7b2f643 --- /dev/null +++ b/local_plugins/open_ai_service/android/build.gradle.kts @@ -0,0 +1,37 @@ +plugins { + // Android Library 插件 + id("com.android.library") + // Kotlin Android 插件 + id("org.jetbrains.kotlin.android") +} + +android { + // 命名空间,对应你插件的包名(需与代码内包名保持一致) + namespace = "com.yunqiinnovation.open_ai_service" + + // 目标 SDK 版本 + compileSdk = 33 + + defaultConfig { + // 最低 SDK 版本 + minSdk = 21 + targetSdk = 33 + } + + // Java 语言级别兼容配置 + compileOptions { + sourceCompatibility = JavaVersion.VERSION_11 + targetCompatibility = JavaVersion.VERSION_11 + } + + // Kotlin 语言级别 + kotlinOptions { + jvmTarget = "11" + } +} + +dependencies { + + implementation("com.squareup.okhttp3:okhttp:4.10.0") + +} \ No newline at end of file diff --git a/local_plugins/open_ai_service/android/settings.gradle.kts b/local_plugins/open_ai_service/android/settings.gradle.kts new file mode 100644 index 000000000..1c33f4971 --- /dev/null +++ b/local_plugins/open_ai_service/android/settings.gradle.kts @@ -0,0 +1 @@ +rootProject.name = "open_ai_service" \ No newline at end of file diff --git a/local_plugins/open_ai_service/android/src/main/AndroidManifest.xml b/local_plugins/open_ai_service/android/src/main/AndroidManifest.xml new file mode 100644 index 000000000..e38cdeb21 --- /dev/null +++ b/local_plugins/open_ai_service/android/src/main/AndroidManifest.xml @@ -0,0 +1,11 @@ + + + + + + + + + + diff --git a/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt new file mode 100644 index 000000000..907be93a2 --- /dev/null +++ b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIService.kt @@ -0,0 +1,469 @@ +package com.yunqiinnovation.open_ai_service + +import android.util.Log +import okhttp3.* +import okhttp3.MediaType.Companion.toMediaTypeOrNull +import okhttp3.RequestBody.Companion.toRequestBody +import org.json.JSONArray +import org.json.JSONObject +import java.io.IOException +import java.util.concurrent.TimeUnit + +/** + * OpenAI服务的原生实现 + */ +class OpenAIService() { + private val TAG = "OpenAIService" + private var baseUrl = "" + private val client = OkHttpClient.Builder() + .connectTimeout(30, TimeUnit.SECONDS) + .readTimeout(30, TimeUnit.SECONDS) + .writeTimeout(30, TimeUnit.SECONDS) + .build() + + private var apiKey: String = "" + private var isInitialized = false + private var model: String = "" // 默认模型 + + // 用于存储注册的函数 + private val registeredFunctions = mutableListOf() + + /** + * 构建curl命令用于测试 + */ + private fun buildCurlCommand(request: Request, body: String): String { + val command = StringBuilder("curl -v -X ${request.method}") + + // 添加请求头 + request.headers.forEach { header -> + // 敏感信息处理:不显示真实的API Key + if (header.first == "Authorization") { + command.append(" -H '${header.first}: Bearer $apiKey'") + } else { + command.append(" -H '${header.first}: ${header.second}'") + } + } + + // 添加请求体 + if (request.method == "POST" || request.method == "PUT") { + // 转义JSON中的单引号,确保curl命令正确 + val escapedBody = body.replace("'", "\\'") + command.append(" -d '${escapedBody}'") + } + + // 添加URL + command.append(" '${request.url}'") + + return command.toString() + } + + /** + * 创建用户消息 + */ + fun createUserMessage(content: String): JSONObject { + return JSONObject().apply { + put("role", "user") + put("content", content) + } + } + + /** + * 创建助手消息 + */ + fun createAssistantMessage(content: String): JSONObject { + return JSONObject().apply { + put("role", "assistant") + put("content", content) + } + } + + /** + * 初始化OpenAI服务 + */ + fun initialize(apiKey: String, baseUrl: String, model: String): Boolean { + this.apiKey = apiKey + if (baseUrl.isNotEmpty()) { + this.baseUrl = baseUrl + } + if (model.isNotEmpty()) { + this.model = model + } + isInitialized = apiKey.isNotEmpty() + return isInitialized + } + + /** + * 注册函数 + */ + fun registerFunction(name: String, description: String, parameters: JSONObject): Boolean { + try { + val function = JSONObject().apply { + put("name", name) + put("description", description) + put("parameters", parameters) + } + + // 检查是否已存在相同名称的函数 + val existingIndex = registeredFunctions.indexOfFirst { + it.getString("name") == name + } + + if (existingIndex >= 0) { + // 如果已存在,则替换 + registeredFunctions[existingIndex] = function + } else { + // 如果不存在,则添加 + registeredFunctions.add(function) + } + + return true + } catch (e: Exception) { + return false + } + } + + /** + * 发送消息(非流式输出) + */ + @Throws(OpenAIException::class) + fun sendMessage(messages: JSONArray): String { + if (!isInitialized || apiKey.isEmpty()) { + throw OpenAIException("OpenAI服务未初始化") + } + + val requestBody = JSONObject().apply { + put("model", model) + put("messages", messages) + put("temperature", 0.7) + put("max_tokens", 2000) + put("stream", false) + + // 如果有注册的函数,则添加到请求中 + if (registeredFunctions.isNotEmpty()) { + val tools = JSONArray() + for (function in registeredFunctions) { + val tool = JSONObject().apply { + put("type", "function") + put("function", function) + } + tools.put(tool) + } + put("tools", tools) + } + } + + val mediaType = "application/json".toMediaTypeOrNull() + val request = Request.Builder() + .url(baseUrl) + .addHeader("Content-Type", "application/json") + .addHeader("Authorization", "Bearer $apiKey") + .post(requestBody.toString().toRequestBody(mediaType)) + .build() + + try { + // 输出用于测试的curl命令 + // val curlCommand = buildCurlCommand(request, requestBody.toString()) + // Log.d(TAG, "curl command: \n$curlCommand") + + client.newCall(request).execute().use { response -> + if (!response.isSuccessful) { + throw OpenAIException("API调用失败: ${response.code}") + } + + val responseBody = response.body?.string() ?: throw OpenAIException("Empty response") + val jsonResponse = JSONObject(responseBody) + + // 检查是否有函数调用 + if (jsonResponse.has("choices") && + jsonResponse.getJSONArray("choices").length() > 0) { + + val choice = jsonResponse.getJSONArray("choices").getJSONObject(0) + + // 检查是否是函数调用 + if (choice.has("message")) { + val message = choice.getJSONObject("message") + + // 检查是否有工具调用 + if (message.has("tool_calls")) { + val toolCalls = message.getJSONArray("tool_calls") + if (toolCalls.length() > 0) { + val toolCall = toolCalls.getJSONObject(0) + if (toolCall.has("function")) { + val function = toolCall.getJSONObject("function") + val functionCall = JSONObject().apply { + put("name", function.getString("name")) + put("arguments", function.getString("arguments")) + put("id", toolCall.getString("id")) + } + return functionCall.toString() + } + } + } + + // 如果没有工具调用,返回消息内容 + if (message.has("content")) { + return message.getString("content") + } + } + } + + throw OpenAIException("Invalid response format") + } + } catch (e: Exception) { + if (e is OpenAIException) throw e + throw OpenAIException("Failed to communicate with AI service: ${e.message}") + } + } + + /** + * 发送消息(流式输出) + */ + fun sendMessageStream(messages: JSONArray, callback: StreamCallback) { + if (!isInitialized || apiKey.isEmpty()) { + callback.onError(OpenAIException("OpenAI服务未初始化")) + return + } + + val requestBody = JSONObject().apply { + put("model", model) + put("messages", messages) + put("temperature", 0.7) + put("max_tokens", 2000) + put("stream", true) + + // 如果有注册的函数,则添加到请求中 + if (registeredFunctions.isNotEmpty()) { + val tools = JSONArray() + for (function in registeredFunctions) { + val tool = JSONObject().apply { + put("type", "function") + put("function", function) + } + tools.put(tool) + } + put("tools", tools) + } + } + + val mediaType = "application/json".toMediaTypeOrNull() + val request = Request.Builder() + .url(baseUrl) + .addHeader("Content-Type", "application/json") + .addHeader("Authorization", "Bearer $apiKey") + .addHeader("Accept", "text/event-stream") + .post(requestBody.toString().toRequestBody(mediaType)) + .build() + + // 输出用于测试的curl命令 + // val curlCommand = buildCurlCommand(request, requestBody.toString()) + // Log.d(TAG, "curl command: $curlCommand") + + client.newCall(request).enqueue(object : Callback { + override fun onFailure(call: Call, e: IOException) { + callback.onError(OpenAIException(e.message ?: "请求失败")) + } + + override fun onResponse(call: Call, response: Response) { + + if (!response.isSuccessful) { + callback.onError(OpenAIException("API调用失败: ${response.code}")) + return + } + + val responseBody = response.body ?: return + val source = responseBody.source() + + try { + // 预取数据到缓冲区 + source.request(Long.MAX_VALUE) + val bufferedSource = source.buffer + + // 用于存储函数调用的各个部分 + val finalToolCalls = mutableMapOf() + + while (!bufferedSource.exhausted()) { + val line = bufferedSource.readUtf8Line()?.trim() ?: continue + if (line.isEmpty()) continue + if (line.startsWith("data:")) { + val data = line.substring(5).trim() + + // 处理[DONE]消息 + if (data == "[DONE]" || data == "[\"DONE\"]") { + processToolCalls(finalToolCalls, callback) + callback.onComplete() + break + } + + try { + val jsonData = JSONObject(data) + + // 处理消息内容 + if (jsonData.has("choices")) { + val choices = jsonData.getJSONArray("choices") + if (choices.length() > 0) { + val choice = choices.getJSONObject(0) + + if (choice.has("delta")) { + val delta = choice.getJSONObject("delta") + + // 处理普通文本内容 + if (delta.has("content")) { + val content = delta.getString("content") + callback.onToken(content) + } + + // 处理工具调用(函数调用) + if (delta.has("tool_calls")) { + val toolCalls = delta.getJSONArray("tool_calls") + for (i in 0 until toolCalls.length()) { + val toolCall = toolCalls.getJSONObject(i) + val index = toolCall.getInt("index") + + // 创建或获取现有的工具调用信息 + val toolCallInfo = finalToolCalls.getOrPut(index) { ToolCallInfo() } + + // 更新ID + if (toolCall.has("id")) { + toolCallInfo.id = toolCall.getString("id") + } + + // 更新函数信息 + if (toolCall.has("function")) { + val function = toolCall.getJSONObject("function") + + if (function.has("name")) { + toolCallInfo.name = function.getString("name") + } + + if (function.has("arguments")) { + toolCallInfo.arguments += function.getString("arguments") + } + } + } + } + } + } + } + } catch (e: Exception) { + // 忽略解析错误 + Log.e(TAG, "解析JSON出错: ${e.message}") + } + } + } + } catch (e: Exception) { + callback.onError(OpenAIException("处理响应流时出错: ${e.message}")) + } finally { + responseBody.close() + } + } + }) + } + + /** + * 发送函数调用结果 + */ + fun sendFunctionCallResult( + messages: JSONArray, + functionCall: JSONObject, + functionResult: String, + callback: StreamCallback + ) { + try { + val fullMessages = JSONArray() + + // 添加用户消息 + for (i in 0 until messages.length()) { + fullMessages.put(messages.getJSONObject(i)) + } + + // 添加函数调用消息 + fullMessages.put(JSONObject().apply { + put("role", "assistant") + put("content", "") + + // 添加工具调用 + val toolCalls = JSONArray().apply { + val toolCall = JSONObject().apply { + put("id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) + put("type", "function") + put("function", JSONObject().apply { + put("name", functionCall.getString("name")) + put("arguments", functionCall.getString("arguments")) + }) + } + put(toolCall) + } + put("tool_calls", toolCalls) + }) + + // 添加函数调用结果 + fullMessages.put(JSONObject().apply { + put("role", "tool") + put("content", functionResult) + put("tool_call_id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) + }) + + // 添加一个带有content的assistant消息,确保API请求不会因为缺少content而失败 + fullMessages.put(JSONObject().apply { + put("role", "assistant") + put("content", "") // 空内容,让模型生成新的回复 + }) + + // 发送完整对话 + sendMessageStream(fullMessages, callback) + + } catch (e: Exception) { + callback.onError(OpenAIException("发送函数调用结果失败: ${e.message}")) + } + } + + /** + * 处理工具调用(处理完整的函数调用并回调) + */ + private fun processToolCalls(toolCalls: Map, callback: StreamCallback) { + if (toolCalls.isEmpty()) return + + // 只处理第一个工具调用 + val firstToolCall = toolCalls.entries.firstOrNull()?.value ?: return + + if (firstToolCall.isValid()) { + // 创建函数调用JSON对象 + val functionCall = JSONObject().apply { + put("name", firstToolCall.name) + put("arguments", firstToolCall.arguments) + put("id", firstToolCall.id) + } + + // 回调 + callback.onFunctionCall(functionCall) + } + } + + /** + * 工具调用信息类 + */ + private class ToolCallInfo { + var id: String = "" + var name: String = "" + var arguments: String = "" + + fun isValid(): Boolean { + return id.isNotEmpty() && name.isNotEmpty() + } + } + + /** + * 流式输出回调接口 + */ + interface StreamCallback { + fun onToken(token: String) + fun onComplete() + fun onError(e: Exception) + fun onFunctionCall(functionCall: JSONObject) + } +} + +/** + * OpenAI服务异常 + */ +class OpenAIException(message: String) : Exception(message) \ No newline at end of file diff --git a/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIServicePlugin.kt b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIServicePlugin.kt new file mode 100644 index 000000000..70af36c13 --- /dev/null +++ b/local_plugins/open_ai_service/android/src/main/kotlin/com/yunqiinnovation/open_ai_service/OpenAIServicePlugin.kt @@ -0,0 +1,317 @@ +package com.yunqiinnovation.open_ai_service + +import android.content.Context +import android.util.Log +import androidx.annotation.NonNull +import io.flutter.embedding.engine.plugins.FlutterPlugin +import io.flutter.plugin.common.MethodCall +import io.flutter.plugin.common.MethodChannel +import io.flutter.plugin.common.MethodChannel.MethodCallHandler +import io.flutter.plugin.common.MethodChannel.Result +import io.flutter.plugin.common.EventChannel +import io.flutter.plugin.common.EventChannel.EventSink +import io.flutter.plugin.common.EventChannel.StreamHandler +import org.json.JSONArray +import org.json.JSONObject +import java.util.concurrent.CountDownLatch +import java.util.concurrent.Executors + +/** OpenAIServicePlugin */ +class OpenAIServicePlugin : FlutterPlugin, MethodCallHandler, StreamHandler { + /// 方法通道名称 + private val methodChannelName = "com.yunqiinnovation.open_ai_service/methods" + + /// 事件通道名称 + private val eventChannelName = "com.yunqiinnovation.open_ai_service/events" + + /// 方法通道 + private lateinit var methodChannel: MethodChannel + + /// 事件通道 + private lateinit var eventChannel: EventChannel + + /// 应用上下文 + private lateinit var context: Context + + /// OpenAI服务实例 + private val openAIService = OpenAIService() + + /// 事件接收器(用于流式输出) + private var eventSink: EventSink? = null + + /// 执行器(用于后台线程) + private val executor = Executors.newSingleThreadExecutor() + + override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { + // 保存上下文 + context = flutterPluginBinding.applicationContext + + // 初始化方法通道 + methodChannel = MethodChannel(flutterPluginBinding.binaryMessenger, methodChannelName) + methodChannel.setMethodCallHandler(this) + + // 初始化事件通道 + eventChannel = EventChannel(flutterPluginBinding.binaryMessenger, eventChannelName) + eventChannel.setStreamHandler(this) + } + + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "initialize" -> { + val apiKey = call.argument("apiKey") ?: "" + val baseUrl = call.argument("baseUrl") ?: "" + val model = call.argument("model") ?: "" + + val initialized = openAIService.initialize(apiKey, baseUrl, model) + result.success(initialized) + } + + "registerFunction" -> { + val name = call.argument("name") ?: "" + val description = call.argument("description") ?: "" + val parameters = call.argument>("parameters") + + if (name.isEmpty() || parameters == null) { + result.error("INVALID_ARGUMENT", "函数注册参数无效", null) + return + } + + val parametersJson = JSONObject(parameters) + val registered = openAIService.registerFunction(name, description, parametersJson) + result.success(registered) + } + + "sendMessage" -> { + val messagesRaw = call.argument>>("messages") ?: emptyList() + + // 转换消息格式 + val messages = JSONArray() + for (message in messagesRaw) { + messages.put(JSONObject(message)) + } + + // 在后台线程执行请求 + executor.execute { + try { + val response = openAIService.sendMessage(messages) + // 在主线程返回结果 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.success(response) + } + } catch (e: Exception) { + // 在主线程返回错误 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.error("OPENAI_ERROR", e.message, null) + } + } + } + } + + "sendMessageStream" -> { + val messagesRaw = call.argument>>("messages") ?: emptyList() + + // 检查事件接收器 + if (eventSink == null) { + result.error("NO_EVENT_SINK", "没有可用的事件流接收器", null) + return + } + + // 转换消息格式 + val messages = JSONArray() + for (message in messagesRaw) { + messages.put(JSONObject(message)) + } + + // 在后台线程执行请求 + executor.execute { + try { + openAIService.sendMessageStream( + messages = messages, + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + // 发送token事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "token", "content" to token)) + } + } + + override fun onComplete() { + // 发送完成事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "complete")) + } + } + + override fun onError(e: Exception) { + // 发送错误事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "error", "content" to e.message)) + } + } + + override fun onFunctionCall(functionCall: JSONObject) { + // 发送函数调用事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + val functionCallMap = functionCall.toMap() + eventSink?.success(mapOf("type" to "functionCall", "content" to functionCallMap)) + } + } + } + ) + + // 请求已开始 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.success(true) + } + } catch (e: Exception) { + // 在主线程返回错误 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.error("OPENAI_ERROR", e.message, null) + } + } + } + } + + "sendFunctionCallResult" -> { + val messagesRaw = call.argument>>("messages") ?: emptyList() + val functionCallRaw = call.argument>("functionCall") ?: emptyMap() + val functionResult = call.argument("functionResult") ?: "" + + // 检查事件接收器 + if (eventSink == null) { + result.error("NO_EVENT_SINK", "没有可用的事件流接收器", null) + return + } + + // 转换消息格式 + val messages = JSONArray() + for (message in messagesRaw) { + messages.put(JSONObject(message)) + } + + // 转换函数调用 + val functionCall = JSONObject(functionCallRaw) + + // 在后台线程执行请求 + executor.execute { + try { + openAIService.sendFunctionCallResult( + messages = messages, + functionCall = functionCall, + functionResult = functionResult, + callback = object : OpenAIService.StreamCallback { + override fun onToken(token: String) { + // 发送token事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "token", "content" to token)) + } + } + + override fun onComplete() { + // 发送完成事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "complete")) + } + } + + override fun onError(e: Exception) { + // 发送错误事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + eventSink?.success(mapOf("type" to "error", "content" to e.message)) + } + } + + override fun onFunctionCall(nestedFunctionCall: JSONObject) { + // 发送函数调用事件 + android.os.Handler(android.os.Looper.getMainLooper()).post { + val functionCallMap = nestedFunctionCall.toMap() + eventSink?.success(mapOf("type" to "functionCall", "content" to functionCallMap)) + } + } + } + ) + + // 请求已开始 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.success(true) + } + } catch (e: Exception) { + // 在主线程返回错误 + android.os.Handler(android.os.Looper.getMainLooper()).post { + result.error("OPENAI_ERROR", e.message, null) + } + } + } + } + + "createUserMessage" -> { + val content = call.argument("content") ?: "" + val message = openAIService.createUserMessage(content) + result.success(message.toMap()) + } + + "createAssistantMessage" -> { + val content = call.argument("content") ?: "" + val message = openAIService.createAssistantMessage(content) + result.success(message.toMap()) + } + + else -> { + result.notImplemented() + } + } + } + + override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { + methodChannel.setMethodCallHandler(null) + eventChannel.setStreamHandler(null) + executor.shutdown() + } + + // Stream事件处理 + override fun onListen(arguments: Any?, eventSink: EventSink?) { + this.eventSink = eventSink + } + + override fun onCancel(arguments: Any?) { + this.eventSink = null + } + + // 工具方法:JSONObject转Map + private fun JSONObject.toMap(): Map { + val map = mutableMapOf() + val keys = this.keys() + while (keys.hasNext()) { + val key = keys.next() + var value: Any? = this.opt(key) + + value = when (value) { + JSONObject.NULL -> null + is JSONObject -> value.toMap() + is JSONArray -> value.toList() + else -> value + } + + map[key] = value + } + return map + } + + // 工具方法:JSONArray转List + private fun JSONArray.toList(): List { + val list = mutableListOf() + for (i in 0 until this.length()) { + var value: Any? = this.opt(i) + + value = when (value) { + JSONObject.NULL -> null + is JSONObject -> value.toMap() + is JSONArray -> value.toList() + else -> value + } + + list.add(value) + } + return list + } +} \ No newline at end of file diff --git a/local_plugins/open_ai_service/ios/Classes/OpenAIService.swift b/local_plugins/open_ai_service/ios/Classes/OpenAIService.swift new file mode 100644 index 000000000..671affd73 --- /dev/null +++ b/local_plugins/open_ai_service/ios/Classes/OpenAIService.swift @@ -0,0 +1,473 @@ +import Foundation + +/// OpenAI服务异常 +public class OpenAIError: Error { + let message: String + + init(_ message: String) { + self.message = message + } +} + +/// 工具调用信息 +private class ToolCallInfo { + var id: String = "" + var name: String = "" + var arguments: String = "" + + var isValid: Bool { + return !id.isEmpty && !name.isEmpty + } +} + +/// OpenAI服务iOS原生实现 +public class OpenAIService { + private let TAG = "OpenAIService" + private var baseUrl = "https://api.openai.com/v1/chat/completions" + private var apiKey: String = "" + private var isInitialized = false + private var model: String = "doubao-1-5-lite-32k-250115" // 默认模型 + + // 用于存储注册的函数 + private var registeredFunctions: [[String: Any]] = [] + + // URL会话 + private let session: URLSession + + public init() { + // 创建URL会话配置 + let config = URLSessionConfiguration.default + config.timeoutIntervalForRequest = 30.0 + config.timeoutIntervalForResource = 30.0 + session = URLSession(configuration: config) + } + + /// 创建用户消息 + public func createUserMessage(content: String) -> [String: Any] { + return ["role": "user", "content": content] + } + + /// 创建助手消息 + public func createAssistantMessage(content: String) -> [String: Any] { + return ["role": "assistant", "content": content] + } + + /// 初始化OpenAI服务 + public func initialize(apiKey: String, baseUrl: String = "", model: String = "") -> Bool { + self.apiKey = apiKey + if !baseUrl.isEmpty { + self.baseUrl = baseUrl + } + if !model.isEmpty { + self.model = model + } + isInitialized = !apiKey.isEmpty + return isInitialized + } + + /// 注册函数 + public func registerFunction(name: String, description: String, parameters: [String: Any]) -> Bool { + do { + let function: [String: Any] = [ + "name": name, + "description": description, + "parameters": parameters + ] + + // 检查是否已存在相同名称的函数 + if let existingIndex = registeredFunctions.firstIndex(where: { ($0["name"] as? String) == name }) { + // 如果已存在,则替换 + registeredFunctions[existingIndex] = function + } else { + // 如果不存在,则添加 + registeredFunctions.append(function) + } + + return true + } catch { + return false + } + } + + /// 发送消息(非流式输出) + public func sendMessage(messages: [[String: Any]], systemPrompt: String) throws -> String { + guard isInitialized, !apiKey.isEmpty else { + throw OpenAIError("OpenAI服务未初始化") + } + + // 构建完整消息,添加系统提示 + var fullMessages: [[String: Any]] = [ + ["role": "system", "content": systemPrompt] + ] + fullMessages.append(contentsOf: messages) + + // 构建请求体 + var requestDict: [String: Any] = [ + "model": model, + "messages": fullMessages, + "temperature": 0.7, + "max_tokens": 2000, + "stream": false + ] + + // 如果有注册的函数,添加到请求中 + if !registeredFunctions.isEmpty { + var tools: [[String: Any]] = [] + for function in registeredFunctions { + let tool: [String: Any] = [ + "type": "function", + "function": function + ] + tools.append(tool) + } + requestDict["tools"] = tools + } + + // 将请求数据转换为JSON数据 + guard let jsonData = try? JSONSerialization.data(withJSONObject: requestDict) else { + throw OpenAIError("无法序列化请求数据") + } + + // 创建URL请求 + guard let url = URL(string: baseUrl) else { + throw OpenAIError("无效的URL") + } + + var request = URLRequest(url: url) + request.httpMethod = "POST" + request.addValue("application/json", forHTTPHeaderField: "Content-Type") + request.addValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization") + request.httpBody = jsonData + + // 创建信号量用于同步请求 + let semaphore = DispatchSemaphore(value: 0) + var responseResult: Result = .failure(OpenAIError("未收到响应")) + + // 执行请求 + let task = session.dataTask(with: request) { data, response, error in + if let error = error { + responseResult = .failure(OpenAIError("请求失败: \(error.localizedDescription)")) + semaphore.signal() + return + } + + guard let httpResponse = response as? HTTPURLResponse else { + responseResult = .failure(OpenAIError("无效的HTTP响应")) + semaphore.signal() + return + } + + guard httpResponse.statusCode == 200 else { + responseResult = .failure(OpenAIError("API调用失败: \(httpResponse.statusCode)")) + semaphore.signal() + return + } + + guard let data = data else { + responseResult = .failure(OpenAIError("响应数据为空")) + semaphore.signal() + return + } + + do { + // 解析JSON响应 + guard let jsonResponse = try JSONSerialization.jsonObject(with: data) as? [String: Any] else { + responseResult = .failure(OpenAIError("无法解析JSON响应")) + semaphore.signal() + return + } + + // 检查是否有函数调用 + if let choices = jsonResponse["choices"] as? [[String: Any]], !choices.isEmpty, + let choice = choices.first, + let message = choice["message"] as? [String: Any] { + + // 检查是否有工具调用 + if let toolCalls = message["tool_calls"] as? [[String: Any]], !toolCalls.isEmpty, + let toolCall = toolCalls.first, + let function = toolCall["function"] as? [String: Any], + let name = function["name"] as? String, + let arguments = function["arguments"] as? String, + let id = toolCall["id"] as? String { + + let functionCallDict: [String: Any] = [ + "name": name, + "arguments": arguments, + "id": id + ] + + // 将函数调用转为JSON字符串 + if let functionCallData = try? JSONSerialization.data(withJSONObject: functionCallDict), + let functionCallString = String(data: functionCallData, encoding: .utf8) { + responseResult = .success(functionCallString) + semaphore.signal() + return + } + } + + // 如果没有工具调用,返回消息内容 + if let content = message["content"] as? String { + responseResult = .success(content) + semaphore.signal() + return + } + } + + responseResult = .failure(OpenAIError("无效的响应格式")) + semaphore.signal() + + } catch { + responseResult = .failure(OpenAIError("解析响应时出错: \(error.localizedDescription)")) + semaphore.signal() + } + } + + task.resume() + + // 等待响应完成 + _ = semaphore.wait(timeout: .distantFuture) + + // 返回结果或抛出错误 + switch responseResult { + case .success(let result): + return result + case .failure(let error): + throw error + } + } + + /// 发送消息(流式输出) + public func sendMessageStream(messages: [[String: Any]], systemPrompt: String, callback: @escaping StreamCallback) { + guard isInitialized, !apiKey.isEmpty else { + callback.onError(OpenAIError("OpenAI服务未初始化")) + return + } + + // 构建完整消息,添加系统提示 + var fullMessages: [[String: Any]] = [ + ["role": "system", "content": systemPrompt] + ] + fullMessages.append(contentsOf: messages) + + // 构建请求体 + var requestDict: [String: Any] = [ + "model": model, + "messages": fullMessages, + "temperature": 0.7, + "max_tokens": 2000, + "stream": true + ] + + // 如果有注册的函数,添加到请求中 + if !registeredFunctions.isEmpty { + var tools: [[String: Any]] = [] + for function in registeredFunctions { + let tool: [String: Any] = [ + "type": "function", + "function": function + ] + tools.append(tool) + } + requestDict["tools"] = tools + } + + // 将请求数据转换为JSON数据 + guard let jsonData = try? JSONSerialization.data(withJSONObject: requestDict) else { + callback.onError(OpenAIError("无法序列化请求数据")) + return + } + + // 创建URL请求 + guard let url = URL(string: baseUrl) else { + callback.onError(OpenAIError("无效的URL")) + return + } + + var request = URLRequest(url: url) + request.httpMethod = "POST" + request.addValue("application/json", forHTTPHeaderField: "Content-Type") + request.addValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization") + request.addValue("text/event-stream", forHTTPHeaderField: "Accept") + request.httpBody = jsonData + + // 用于存储函数调用的各个部分 + var finalToolCalls: [Int: ToolCallInfo] = [:] + + // 创建数据任务 + let task = session.dataTask(with: request) { data, response, error in + if let error = error { + callback.onError(OpenAIError("请求失败: \(error.localizedDescription)")) + return + } + + guard let httpResponse = response as? HTTPURLResponse else { + callback.onError(OpenAIError("无效的HTTP响应")) + return + } + + guard httpResponse.statusCode == 200 else { + callback.onError(OpenAIError("API调用失败: \(httpResponse.statusCode)")) + return + } + + guard let data = data else { + callback.onError(OpenAIError("响应数据为空")) + return + } + + // 处理SSE数据流 + if let text = String(data: data, encoding: .utf8) { + let lines = text.components(separatedBy: "\n") + + for line in lines { + if line.isEmpty { continue } + + if line.hasPrefix("data: ") { + let dataContent = line.dropFirst(6) + + // 处理[DONE]消息 + if dataContent == "[DONE]" { + self.processToolCalls(finalToolCalls, callback: callback) + callback.onComplete() + break + } + + // 解析JSON数据 + do { + if let data = dataContent.data(using: .utf8), + let jsonData = try JSONSerialization.jsonObject(with: data) as? [String: Any] { + + // 处理消息内容 + if let choices = jsonData["choices"] as? [[String: Any]], !choices.isEmpty, + let choice = choices.first { + + if let delta = choice["delta"] as? [String: Any] { + // 处理普通文本内容 + if let content = delta["content"] as? String { + callback.onToken(content) + } + + // 处理工具调用(函数调用) + if let toolCalls = delta["tool_calls"] as? [[String: Any]] { + for toolCall in toolCalls { + if let index = toolCall["index"] as? Int { + // 创建或获取现有的工具调用信息 + let toolCallInfo = finalToolCalls[index] ?? ToolCallInfo() + + // 更新ID + if let id = toolCall["id"] as? String { + toolCallInfo.id = id + } + + // 更新函数信息 + if let function = toolCall["function"] as? [String: Any] { + if let name = function["name"] as? String { + toolCallInfo.name = name + } + + if let arguments = function["arguments"] as? String { + toolCallInfo.arguments += arguments + } + } + + finalToolCalls[index] = toolCallInfo + } + } + } + } + } + } + } catch { + NSLog("解析JSON出错: \(error.localizedDescription)") + // 忽略解析错误,继续处理其他行 + } + } + } + } + } + + task.resume() + } + + /// 发送函数调用结果 + public func sendFunctionCallResult( + messages: [[String: Any]], + systemPrompt: String, + functionCall: [String: Any], + functionResult: String, + callback: @escaping StreamCallback + ) { + do { + // 构建完整消息数组 + var fullMessages: [[String: Any]] = [ + // 添加系统提示 + ["role": "system", "content": systemPrompt] + ] + + // 添加用户消息 + fullMessages.append(contentsOf: messages) + + // 获取函数相关信息 + guard let name = functionCall["name"] as? String, + let arguments = functionCall["arguments"] as? String else { + callback.onError(OpenAIError("函数调用信息不完整")) + return + } + + let id = functionCall["id"] as? String ?? "call_\(Int(Date().timeIntervalSince1970 * 1000))" + + // 添加函数调用消息 + fullMessages.append([ + "role": "assistant", + "content": NSNull(), + "tool_calls": [ + [ + "id": id, + "type": "function", + "function": [ + "name": name, + "arguments": arguments + ] + ] + ] + ]) + + // 添加函数调用结果 + fullMessages.append([ + "role": "tool", + "content": functionResult, + "tool_call_id": id + ]) + + // 发送完整对话 + sendMessageStream(messages: fullMessages, systemPrompt: systemPrompt, callback: callback) + + } catch { + callback.onError(OpenAIError("发送函数调用结果失败: \(error.localizedDescription)")) + } + } + + /// 处理工具调用(函数调用)并回调 + private func processToolCalls(_ toolCalls: [Int: ToolCallInfo], callback: StreamCallback) { + if toolCalls.isEmpty { return } + + // 只处理第一个工具调用 + guard let firstToolCall = toolCalls.values.first, firstToolCall.isValid else { return } + + // 创建函数调用字典 + let functionCall: [String: Any] = [ + "name": firstToolCall.name, + "arguments": firstToolCall.arguments, + "id": firstToolCall.id + ] + + // 回调 + callback.onFunctionCall(functionCall) + } + + /// 流式输出回调协议 + public typealias StreamCallback = (onToken: (String) -> Void, + onComplete: () -> Void, + onError: (Error) -> Void, + onFunctionCall: ([String: Any]) -> Void) +} \ No newline at end of file diff --git a/local_plugins/open_ai_service/ios/Classes/OpenAIServicePlugin.swift b/local_plugins/open_ai_service/ios/Classes/OpenAIServicePlugin.swift new file mode 100644 index 000000000..d919e2584 --- /dev/null +++ b/local_plugins/open_ai_service/ios/Classes/OpenAIServicePlugin.swift @@ -0,0 +1,215 @@ +import Flutter +import UIKit + +public class OpenAIServicePlugin: NSObject, FlutterPlugin, FlutterStreamHandler { + // OpenAI服务实例 + private let openAIService = OpenAIService() + + // 事件接收器 + private var eventSink: FlutterEventSink? + + // 注册插件 + public static func register(with registrar: FlutterPluginRegistrar) { + let methodChannel = FlutterMethodChannel(name: "com.yunqiinnovation.open_ai_service/methods", binaryMessenger: registrar.messenger()) + let eventChannel = FlutterEventChannel(name: "com.yunqiinnovation.open_ai_service/events", binaryMessenger: registrar.messenger()) + + let instance = OpenAIServicePlugin() + registrar.addMethodCallDelegate(instance, channel: methodChannel) + eventChannel.setStreamHandler(instance) + } + + // 处理方法调用 + public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { + switch call.method { + case "initialize": + if let args = call.arguments as? [String: Any], + let apiKey = args["apiKey"] as? String { + let baseUrl = args["baseUrl"] as? String ?? "" + let model = args["model"] as? String ?? "" + let initialized = openAIService.initialize(apiKey: apiKey, baseUrl: baseUrl, model: model) + result(initialized) + } else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "初始化参数无效", details: nil)) + } + + case "registerFunction": + if let args = call.arguments as? [String: Any], + let name = args["name"] as? String, + let description = args["description"] as? String, + let parameters = args["parameters"] as? [String: Any] { + + let registered = openAIService.registerFunction(name: name, description: description, parameters: parameters) + result(registered) + } else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "函数注册参数无效", details: nil)) + } + + case "sendMessage": + guard let args = call.arguments as? [String: Any], + let messagesRaw = args["messages"] as? [[String: Any]], + let systemPrompt = args["systemPrompt"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "发送消息参数无效", details: nil)) + return + } + + // 在后台线程执行 + DispatchQueue.global(qos: .userInitiated).async { + do { + let response = try self.openAIService.sendMessage(messages: messagesRaw, systemPrompt: systemPrompt) + // 在主线程返回结果 + DispatchQueue.main.async { + result(response) + } + } catch { + // 在主线程返回错误 + DispatchQueue.main.async { + result(FlutterError(code: "OPENAI_ERROR", message: error.localizedDescription, details: nil)) + } + } + } + + case "sendMessageStream": + guard let args = call.arguments as? [String: Any], + let messagesRaw = args["messages"] as? [[String: Any]], + let systemPrompt = args["systemPrompt"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "发送消息参数无效", details: nil)) + return + } + + // 检查事件接收器 + guard let eventSink = self.eventSink else { + result(FlutterError(code: "NO_EVENT_SINK", message: "没有可用的事件流接收器", details: nil)) + return + } + + // 在后台线程执行 + DispatchQueue.global(qos: .userInitiated).async { + let callback: OpenAIService.StreamCallback = ( + onToken: { token in + // 发送token事件 + DispatchQueue.main.async { + eventSink(["type": "token", "content": token]) + } + }, + onComplete: { + // 发送完成事件 + DispatchQueue.main.async { + eventSink(["type": "complete"]) + } + }, + onError: { error in + // 发送错误事件 + DispatchQueue.main.async { + eventSink(["type": "error", "content": error.localizedDescription]) + } + }, + onFunctionCall: { functionCall in + // 发送函数调用事件 + DispatchQueue.main.async { + eventSink(["type": "functionCall", "content": functionCall]) + } + } + ) + + self.openAIService.sendMessageStream(messages: messagesRaw, systemPrompt: systemPrompt, callback: callback) + + // 请求已开始 + DispatchQueue.main.async { + result(true) + } + } + + case "sendFunctionCallResult": + guard let args = call.arguments as? [String: Any], + let messagesRaw = args["messages"] as? [[String: Any]], + let systemPrompt = args["systemPrompt"] as? String, + let functionCallRaw = args["functionCall"] as? [String: Any], + let functionResult = args["functionResult"] as? String else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "发送函数调用结果参数无效", details: nil)) + return + } + + // 检查事件接收器 + guard let eventSink = self.eventSink else { + result(FlutterError(code: "NO_EVENT_SINK", message: "没有可用的事件流接收器", details: nil)) + return + } + + // 在后台线程执行 + DispatchQueue.global(qos: .userInitiated).async { + let callback: OpenAIService.StreamCallback = ( + onToken: { token in + // 发送token事件 + DispatchQueue.main.async { + eventSink(["type": "token", "content": token]) + } + }, + onComplete: { + // 发送完成事件 + DispatchQueue.main.async { + eventSink(["type": "complete"]) + } + }, + onError: { error in + // 发送错误事件 + DispatchQueue.main.async { + eventSink(["type": "error", "content": error.localizedDescription]) + } + }, + onFunctionCall: { functionCall in + // 发送函数调用事件 + DispatchQueue.main.async { + eventSink(["type": "functionCall", "content": functionCall]) + } + } + ) + + self.openAIService.sendFunctionCallResult( + messages: messagesRaw, + systemPrompt: systemPrompt, + functionCall: functionCallRaw, + functionResult: functionResult, + callback: callback + ) + + // 请求已开始 + DispatchQueue.main.async { + result(true) + } + } + + case "createUserMessage": + if let args = call.arguments as? [String: Any], + let content = args["content"] as? String { + let message = openAIService.createUserMessage(content: content) + result(message) + } else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "创建用户消息参数无效", details: nil)) + } + + case "createAssistantMessage": + if let args = call.arguments as? [String: Any], + let content = args["content"] as? String { + let message = openAIService.createAssistantMessage(content: content) + result(message) + } else { + result(FlutterError(code: "INVALID_ARGUMENT", message: "创建助手消息参数无效", details: nil)) + } + + default: + result(FlutterMethodNotImplemented) + } + } + + // MARK: - FlutterStreamHandler + + public func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { + self.eventSink = events + return nil + } + + public func onCancel(withArguments arguments: Any?) -> FlutterError? { + self.eventSink = nil + return nil + } +} \ No newline at end of file diff --git a/local_plugins/open_ai_service/lib/open_ai_service.dart b/local_plugins/open_ai_service/lib/open_ai_service.dart new file mode 100644 index 000000000..7c6fe8024 --- /dev/null +++ b/local_plugins/open_ai_service/lib/open_ai_service.dart @@ -0,0 +1,267 @@ +import 'dart:async'; +import 'dart:convert'; + +import 'package:flutter/services.dart'; + +/// OpenAI服务异常 +class OpenAIException implements Exception { + final String message; + + OpenAIException(this.message); + + @override + String toString() => 'OpenAIException: $message'; +} + +/// OpenAI服务事件类型 +enum OpenAIEventType { + token, + complete, + error, + functionCall, +} + +/// OpenAI服务事件 +class OpenAIEvent { + final OpenAIEventType type; + final dynamic content; + + OpenAIEvent({required this.type, this.content}); + + factory OpenAIEvent.fromMap(Map map) { + final typeStr = map['type'] as String; + final content = map['content']; + + return OpenAIEvent( + type: _typeFromString(typeStr), + content: content, + ); + } + + static OpenAIEventType _typeFromString(String typeStr) { + switch (typeStr) { + case 'token': + return OpenAIEventType.token; + case 'complete': + return OpenAIEventType.complete; + case 'error': + return OpenAIEventType.error; + case 'functionCall': + return OpenAIEventType.functionCall; + default: + throw ArgumentError('未知的事件类型: $typeStr'); + } + } +} + +/// OpenAI服务插件 +class OpenAIService { + static const MethodChannel _channel = MethodChannel('com.yunqiinnovation.open_ai_service/methods'); + static const EventChannel _eventChannel = EventChannel('com.yunqiinnovation.open_ai_service/events'); + + /// 事件流控制器 + StreamController? _eventStreamController; + + /// 事件流 + Stream? _eventStream; + + /// 获取事件流 + Stream get eventStream { + if (_eventStream == null) { + _eventStreamController = StreamController.broadcast(); + _eventStream = _eventStreamController!.stream; + + // 监听原生事件 + _eventChannel.receiveBroadcastStream().listen( + (dynamic event) { + if (event is Map) { + final eventMap = Map.from(event); + final openAIEvent = OpenAIEvent.fromMap(eventMap); + _eventStreamController!.add(openAIEvent); + } + }, + onError: (error) { + _eventStreamController!.addError(OpenAIException('事件流错误: $error')); + }, + ); + } + + return _eventStream!; + } + + /// 初始化OpenAI服务 + /// + /// [apiKey] OpenAI API密钥 + /// [baseUrl] 可选,自定义API基础URL + /// [model] 可选,自定义使用的模型 + Future initialize({ + required String apiKey, + String baseUrl = '', + String model = '', + }) async { + try { + final result = await _channel.invokeMethod( + 'initialize', + { + 'apiKey': apiKey, + 'baseUrl': baseUrl, + 'model': model, + }, + ); + + return result ?? false; + } catch (e) { + throw OpenAIException('初始化失败: $e'); + } + } + + /// 注册函数 + /// + /// [name] 函数名称 + /// [description] 函数描述 + /// [parameters] 函数参数 + Future registerFunction({ + required String name, + required String description, + required Map parameters, + }) async { + try { + final result = await _channel.invokeMethod( + 'registerFunction', + { + 'name': name, + 'description': description, + 'parameters': parameters, + }, + ); + + return result ?? false; + } catch (e) { + throw OpenAIException('注册函数失败: $e'); + } + } + + /// 创建用户消息 + /// + /// [content] 消息内容 + Future> createUserMessage(String content) async { + try { + final result = await _channel.invokeMethod>( + 'createUserMessage', + {'content': content}, + ); + + if (result == null) { + throw OpenAIException('创建用户消息失败: 结果为空'); + } + + return Map.from(result); + } catch (e) { + throw OpenAIException('创建用户消息失败: $e'); + } + } + + /// 创建助手消息 + /// + /// [content] 消息内容 + Future> createAssistantMessage(String content) async { + try { + final result = await _channel.invokeMethod>( + 'createAssistantMessage', + {'content': content}, + ); + + if (result == null) { + throw OpenAIException('创建助手消息失败: 结果为空'); + } + + return Map.from(result); + } catch (e) { + throw OpenAIException('创建助手消息失败: $e'); + } + } + + /// 发送消息(非流式输出) + /// + /// [messages] 消息列表 + Future sendMessage({ + required List> messages, + }) async { + try { + final result = await _channel.invokeMethod( + 'sendMessage', + { + 'messages': messages, + }, + ); + + if (result == null) { + throw OpenAIException('发送消息失败: 结果为空'); + } + + return result; + } catch (e) { + throw OpenAIException('发送消息失败: $e'); + } + } + + /// 发送消息(流式输出) + /// + /// [messages] 消息列表 + /// + /// 返回一个布尔值,表示请求是否已开始 + Future sendMessageStream({ + required List> messages, + }) async { + try { + final result = await _channel.invokeMethod( + 'sendMessageStream', + { + 'messages': messages, + }, + ); + + return result ?? false; + } catch (e) { + throw OpenAIException('发送流式消息失败: $e'); + } + } + + /// 发送函数调用结果 + /// + /// [messages] 消息列表 + /// [functionCall] 函数调用信息 + /// [functionResult] 函数调用结果 + /// + /// 返回一个布尔值,表示请求是否已开始 + Future sendFunctionCallResult({ + required List> messages, + required Map functionCall, + required String functionResult, + }) async { + try { + final result = await _channel.invokeMethod( + 'sendFunctionCallResult', + { + 'messages': messages, + 'functionCall': functionCall, + 'functionResult': functionResult, + }, + ); + + return result ?? false; + } catch (e) { + throw OpenAIException('发送函数调用结果失败: $e'); + } + } + + + /// 从JSON字符串解析函数调用 + Map parseFunctionCall(String functionCallJson) { + try { + return json.decode(functionCallJson) as Map; + } catch (e) { + throw OpenAIException('解析函数调用失败: $e'); + } + } +} \ No newline at end of file diff --git a/local_plugins/open_ai_service/pubspec.yaml b/local_plugins/open_ai_service/pubspec.yaml new file mode 100644 index 000000000..8e00ed3ea --- /dev/null +++ b/local_plugins/open_ai_service/pubspec.yaml @@ -0,0 +1,26 @@ +name: open_ai_service +description: 原生OpenAI服务插件,提供与OpenAI API的交互功能,支持流式输出和函数调用。 +version: 0.0.1 +homepage: https://github.com/yunqiinnovation/deep_voice + +environment: + sdk: ">=2.17.0 <4.0.0" + flutter: ">=2.5.0" + +dependencies: + flutter: + sdk: flutter + +dev_dependencies: + flutter_test: + sdk: flutter + flutter_lints: ^2.0.0 + +flutter: + plugin: + platforms: + android: + package: com.yunqiinnovation.open_ai_service + pluginClass: OpenAIServicePlugin + ios: + pluginClass: OpenAIServicePlugin \ No newline at end of file diff --git a/local_plugins/volcano_speech/README.md b/local_plugins/volcano_speech/README.md new file mode 100644 index 000000000..572317b57 --- /dev/null +++ b/local_plugins/volcano_speech/README.md @@ -0,0 +1,206 @@ +# 火山引擎语音服务插件 + +基于火山引擎语音SDK封装的Flutter插件,支持Android平台。 + +## 功能 +- 语音合成(TTS) + - 支持在线/离线/混合合成模式 + - 支持SSML格式文本 + - 支持情感合成和情感预测 + - 支持音量、语速、音高等多种参数调节 + - 复刻音色支持 + - 离线资源管理 +- 语音识别(ASR) + - 一次性识别 + - 连续识别 + - 长按识别 + - 热词优化 + - 多语言支持 + +## 使用说明 + +### TTS 语音合成 + +#### 基础用法 +```dart +import 'package:volcano_speech/volcano_speech.dart'; + +// 初始化 +await VolcanoSpeech.initialize(appId: 'YOUR_APP_ID', apiKey: 'YOUR_API_KEY'); + +// 设置音量、语速等参数 +await VolcanoSpeech.setSpeechParams( + rate: 0, // 语速: -500到500,0为正常速度 + volume: 100, // 音量: 0到100,默认100 + pitch: 0, // 音高: -500到500,0为正常音高 + silenceDuration: 500, // 静音段时长: 毫秒 +); + +// 设置音色 +await VolcanoSpeech.setVoice( + voiceName: 'zh_female_qingxin', + voiceType: 'qingxin' +); + +// 开始合成并播放 +await VolcanoSpeech.speakText(text: '这是一段测试文本'); + +// 暂停播放 +await VolcanoSpeech.pausePlayback(); + +// 继续播放 +await VolcanoSpeech.resumePlayback(); + +// 停止播放 +await VolcanoSpeech.stopSpeaking(); + +// 销毁引擎 +await VolcanoSpeech.dispose(); +``` + +#### 进阶用法 +```dart +// 设置工作模式 +await VolcanoSpeech.setWorkMode(mode: TtsWorkMode.alternate); // 先在线,断网时切换离线 + +// 设置离线发音人 +await VolcanoSpeech.setOfflineVoice( + voiceName: 'xifei', + voiceType: 'qingxin' +); + +// 下载离线资源 +await VolcanoSpeech.downloadOfflineResource( + voiceTypes: ['qingxin', 'zhenjiang'], + languages: ['zh-CN'] +); + +// 使用SSML格式文本 +await VolcanoSpeech.setTextType(type: TtsTextType.ssml); +await VolcanoSpeech.speakText( + text: '这是一段2023-10-01的语音合成' +); + +// 设置情感 +await VolcanoSpeech.setEmotion(emotion: 'happy'); + +// 启用情感预测 +await VolcanoSpeech.setEnableEmotionPredict(enable: true); + +// 启用服务端缓存 +await VolcanoSpeech.setEnableCache(enable: true); + +// 监听TTS进度事件 +VolcanoSpeech.ttsProgressEvents.listen((event) { + print('播放进度: ${(event.progress * 100).toStringAsFixed(1)}%'); +}); + +// 复刻音色支持 +await VolcanoSpeech.setEnableVoiceClone( + enable: true, + backendCluster: 'your_cluster_name' +); +``` + +### ASR 语音识别 +#### 一次性识别 +```dart +import 'package:volcano_speech/volcano_speech.dart'; + +// 初始化 +await VolcanoSpeech.initializeAsr( + appId: 'YOUR_APP_ID', + apiKey: 'YOUR_API_KEY', + supportedLanguages: ['zh-CN'], +); + +// 设置识别语言 +await VolcanoSpeech.setAsrLanguage(language: 'zh-CN'); + +// 设置热词(可选) +await VolcanoSpeech.setAsrHotWords( + hotWords: '{"hotwords":[{"word":"火山引擎","scale":2.0}]}', +); + +// 开始一次性识别 +try { + final result = await VolcanoSpeech.recognizeOnce(); + print('识别结果: ${result['text']}, 语言: ${result['language']}'); +} catch (e) { + print('识别出错: $e'); +} + +// 销毁引擎 +await VolcanoSpeech.disposeAsr(); +``` + +#### 连续识别 +```dart +import 'package:volcano_speech/volcano_speech.dart'; + +// 初始化 +await VolcanoSpeech.initializeAsr( + appId: 'YOUR_APP_ID', + apiKey: 'YOUR_API_KEY', +); + +// 监听识别事件 +VolcanoSpeech.asrEvents.listen((event) { + switch (event.type) { + case AsrEventType.sessionStarted: + print('识别会话开始'); + break; + case AsrEventType.sessionStopped: + print('识别会话结束'); + break; + case AsrEventType.recognizing: + print('正在识别: ${event.text}'); + break; + case AsrEventType.result: + print('识别结果: ${event.text}'); + break; + case AsrEventType.volumeChanged: + print('音量: ${event.volume}'); + break; + case AsrEventType.error: + print('识别错误: ${event.errorMessage}'); + break; + } +}); + +// 开始连续识别 +await VolcanoSpeech.startContinuousRecognition(); + +// 检查是否正在识别 +final isActive = await VolcanoSpeech.isContinuousRecognitionActive(); +print('是否正在识别: $isActive'); + +// 停止连续识别 +await VolcanoSpeech.stopContinuousRecognition(); +``` + +#### 长按识别 +```dart +import 'package:volcano_speech/volcano_speech.dart'; + +// 初始化 +await VolcanoSpeech.initializeAsr( + appId: 'YOUR_APP_ID', + apiKey: 'YOUR_API_KEY', +); + +// 监听识别事件 +VolcanoSpeech.asrEvents.listen((event) { + // 处理事件... +}); + +// 用户按下按钮时开始识别 +onPressed: () async { + await VolcanoSpeech.startListening(); +}, + +// 用户释放按钮时停止识别 +onReleased: () async { + await VolcanoSpeech.stopListening(); +}, +``` \ No newline at end of file diff --git a/local_plugins/volcano_speech/android/build.gradle.kts b/local_plugins/volcano_speech/android/build.gradle.kts new file mode 100644 index 000000000..86b03fd0e --- /dev/null +++ b/local_plugins/volcano_speech/android/build.gradle.kts @@ -0,0 +1,64 @@ +import com.android.build.gradle.LibraryExtension + +buildscript { + repositories { + google() + mavenCentral() + } + dependencies { + classpath("com.android.tools.build:gradle:7.3.0") + classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") + } +} + +allprojects { + repositories { + google() + mavenCentral() + + } +} + +plugins { + id("com.android.library") + kotlin("android") +} + +// 配置android扩展 +configure { + namespace = "com.yunqiinnovation.volcano_speech" + compileSdkVersion(33) + + defaultConfig { + minSdk = 21 + } + + compileOptions { + sourceCompatibility = JavaVersion.VERSION_11 + targetCompatibility = JavaVersion.VERSION_11 + } + + sourceSets { + getByName("main") { + manifest.srcFile("src/main/AndroidManifest.xml") + java.srcDirs("src/main/kotlin") + } + } + + // 添加lint选项 + lintOptions { + isCheckReleaseBuilds = false + } +} + +// 显式设置Kotlin JVM目标版本 +tasks.withType { + kotlinOptions { + jvmTarget = "11" + } +} + +dependencies { + // 添加火山引擎语音合成SDK + implementation("com.bytedance.speechengine:speechengine_tob:0.0.5") +} \ No newline at end of file diff --git a/local_plugins/volcano_speech/android/settings.gradle.kts b/local_plugins/volcano_speech/android/settings.gradle.kts new file mode 100644 index 000000000..7f5342866 --- /dev/null +++ b/local_plugins/volcano_speech/android/settings.gradle.kts @@ -0,0 +1 @@ +rootProject.name = "volcano_speech" \ No newline at end of file diff --git a/local_plugins/volcano_speech/android/src/main/AndroidManifest.xml b/local_plugins/volcano_speech/android/src/main/AndroidManifest.xml new file mode 100644 index 000000000..338e8293c --- /dev/null +++ b/local_plugins/volcano_speech/android/src/main/AndroidManifest.xml @@ -0,0 +1,9 @@ + + + + + + + + \ No newline at end of file diff --git a/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoAsrHelper.kt b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoAsrHelper.kt new file mode 100644 index 000000000..614474d1a --- /dev/null +++ b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoAsrHelper.kt @@ -0,0 +1,529 @@ +package com.yunqiinnovation.volcano_speech + +import android.content.Context +import android.os.Handler +import android.os.Looper +import com.bytedance.speech.speechengine.SpeechEngine +import com.bytedance.speech.speechengine.SpeechEngineDefines +import com.bytedance.speech.speechengine.SpeechEngineGenerator +import com.yunqiinnovation.volcano_speech.utils.FileLogger +import org.json.JSONArray +import org.json.JSONException +import org.json.JSONObject + +/** + * 火山语音识别帮助类 (大模型版本) + */ +class VolcanoAsrHelper(private val context: Context) { + private val TAG = "VolcanoAsrHelper" + private val mainHandler = Handler(Looper.getMainLooper()) + + // 语音引擎相关 + private var engine: SpeechEngine? = null + private var engineHandler: Long = -1 + private var isInitialized = false + + // 当前回调 + private var currentAsrCallback: ASRCallback? = null + private var currentContinuousCallback: ASRContinuousCallback? = null + + // 识别状态 + private var isContinuousRecognitionActive = false + + // 配置参数 + private var language = "zh-CN" + private var enableVolume = false + private var showUtterances = false + + /** + * ASR一次性识别回调 + */ + interface ASRCallback { + fun onSuccess(text: String) + fun onError(error: String) + } + + /** + * ASR连续识别回调 + */ + interface ASRContinuousCallback { + fun onResult(text: String) + fun onRecognizing(text: String) + fun onSessionStarted() + fun onSessionStopped() + fun onVolumeChanged(volume: Int) + fun onError(error: String) + } + + /** + * 初始化语音识别引擎 + */ + fun initialize(appId: String, token: String, resourceId: String): Boolean { + if (isInitialized) { + FileLogger.i(TAG, "引擎已经初始化") + return true + } + + try { + // 准备环境 + SpeechEngineGenerator.PrepareEnvironment(context, null) + + // 创建引擎 + engine = SpeechEngineGenerator.getInstance() + engineHandler = engine?.createEngine() ?: -1 + + if (engineHandler == -1L) { + FileLogger.e(TAG, "创建引擎失败") + return false + } + + // 设置上下文 + engine?.setContext(context) + + // 设置引擎类型为ASR + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, SpeechEngineDefines.ASR_ENGINE) + + // 设置日志级别 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, SpeechEngineDefines.LOG_LEVEL_WARN) + + // 设置用户ID和设备ID (使用静态值,实际项目中应替换为真实值) + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_UID_STRING, "user_id") + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, "device_id") + + // 设置鉴权信息 - 大模型版本不需要Bearer前缀 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, appId) + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, token) + + // 设置资源ID - 大模型必需 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RESOURCE_ID_STRING, resourceId) + + // 设置协议类型为Seed - 大模型必需 + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_PROTOCOL_TYPE_INT, SpeechEngineDefines.PROTOCOL_TYPE_SEED) + + // 设置网络配置 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_ADDRESS_STRING, "wss://openspeech.bytedance.com") + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_URI_STRING, "/api/v3/sauc/bigmodel") + + // 设置超时时间 + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_CONN_TIMEOUT_INT, 12000) + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_RECV_TIMEOUT_INT, 8000) + + // 设置音频来源为录音机 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RECORDER_TYPE_STRING, SpeechEngineDefines.RECORDER_TYPE_RECORDER) + + // 设置最大录音时长 (默认60秒) + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_VAD_MAX_SPEECH_DURATION_INT, 60000) + + // 设置回声消除 (用于ASR识别时不会收到TTS的声音) + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_RECORDER_PRESET_INT, SpeechEngineDefines.RECORDER_PRESET_VOICE_COMMUNICATION) + + // 初始化引擎 + val result = engine?.initEngine(engineHandler) + isInitialized = result == SpeechEngineDefines.ERR_NO_ERROR + + if (isInitialized) { + FileLogger.i(TAG, "引擎初始化成功") + + // 设置回调监听 + engine?.setListener(object : SpeechEngine.SpeechListener { + override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { + val stdData = String(data) + handleEngineEvent(type, stdData) + } + }) + + return true + } else { + FileLogger.e(TAG, "引擎初始化失败: $result") + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "初始化异常: ${e.message}", e) + return false + } + } + + /** + * 设置语言 + */ + fun setLanguage(language: String): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + this.language = language + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_LANGUAGE_STRING, language) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置语言失败: ${e.message}", e) + return false + } + } + + /** + * 设置热词 + */ + fun setHotWords(hotWordsId: String): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + if (hotWordsId.isNotEmpty()) { + // 大模型ASR需要通过请求参数设置热词 + val reqParams = "{\"corpus\":{\"boosting_table_id\":\"$hotWordsId\"}}" + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) + } + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置热词失败: ${e.message}", e) + return false + } + } + + /** + * 设置是否返回音量 + */ + fun setEnableVolume(enable: Boolean): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + this.enableVolume = enable + engine?.setOptionBoolean(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENABLE_GET_VOLUME_BOOL, enable) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置音量返回失败: ${e.message}", e) + return false + } + } + + /** + * 设置是否显示语音停顿、分句、分词信息 + */ + fun setShowUtterances(show: Boolean): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + this.showUtterances = show + engine?.setOptionBoolean(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_SHOW_UTTER_BOOL, show) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置语音信息显示失败: ${e.message}", e) + return false + } + } + + /** + * 设置VAD切句参数 + */ + fun setVadParams(forceToSpeechTime: Int, endWindowSize: Int): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + val reqParams = "{\"force_to_speech_time\":$forceToSpeechTime, \"end_window_size\":$endWindowSize}" + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置VAD参数失败: ${e.message}", e) + return false + } + } + + /** + * 设置纠错词表 + */ + fun setCorrectWords(correctWordsJson: String): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + val reqParams = "{\"context\": \"{\\\"correct_words\\\": $correctWordsJson}\"}" + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置纠错词表失败: ${e.message}", e) + return false + } + } + + /** + * 一次性识别 + */ + fun recognizeOnce(callback: ASRCallback): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + if (isContinuousRecognitionActive) { + FileLogger.e(TAG, "当前正在连续识别中") + return false + } + + try { + this.currentAsrCallback = callback + + // 停止当前引擎 + engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎开始识别 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "启动识别失败: $ret") + return false + } + + return true + } catch (e: Exception) { + FileLogger.e(TAG, "识别异常: ${e.message}", e) + return false + } + } + + /** + * 停止一次性识别 + */ + fun stopRecognize(): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + // 告知引擎音频输入完成 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_FINISH_TALKING, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "停止识别失败: $ret") + return false + } + + return true + } catch (e: Exception) { + FileLogger.e(TAG, "停止识别异常: ${e.message}", e) + return false + } + } + + /** + * 开始连续识别 + */ + fun startContinuousRecognition(callback: ASRContinuousCallback): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + if (isContinuousRecognitionActive) { + FileLogger.e(TAG, "当前已经在连续识别中") + return false + } + + try { + this.currentContinuousCallback = callback + + // 停止当前引擎 + engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎开始识别 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "启动连续识别失败: $ret") + return false + } + + isContinuousRecognitionActive = true + return true + } catch (e: Exception) { + FileLogger.e(TAG, "连续识别异常: ${e.message}", e) + return false + } + } + + /** + * 停止连续识别 + */ + fun stopContinuousRecognition(): Boolean { + if (!isInitialized || !isContinuousRecognitionActive) { + FileLogger.e(TAG, "引擎未初始化或未在连续识别中") + return false + } + + try { + // 停止引擎 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "停止连续识别失败: $ret") + return false + } + + isContinuousRecognitionActive = false + return true + } catch (e: Exception) { + FileLogger.e(TAG, "停止连续识别异常: ${e.message}", e) + return false + } + } + + /** + * 判断是否在连续识别中 + */ + fun isContinuousRecognitionActive(): Boolean { + return isContinuousRecognitionActive + } + + /** + * 释放资源 + */ + fun release() { + if (!isInitialized) { + return + } + + try { + // 停止引擎 + if (isContinuousRecognitionActive) { + stopContinuousRecognition() + } else { + engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + } + + // 销毁引擎 + engine?.destroyEngine(engineHandler) + engineHandler = -1 + engine = null + isInitialized = false + + FileLogger.i(TAG, "引擎已释放") + } catch (e: Exception) { + FileLogger.e(TAG, "释放引擎异常: ${e.message}", e) + } + } + + /** + * 处理引擎事件 + */ + private fun handleEngineEvent(type: Int, data: String) { + when (type) { + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { + // 引擎启动成功 + mainHandler.post { + currentContinuousCallback?.onSessionStarted() + } + } + + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { + // 引擎停止 + isContinuousRecognitionActive = false + mainHandler.post { + currentContinuousCallback?.onSessionStopped() + } + } + + SpeechEngineDefines.MESSAGE_TYPE_PARTIAL_RESULT -> { + // 中间识别结果 + try { + val json = JSONObject(data) + if (!json.has("result")) { + return + } + + val resultArray = json.getJSONArray("result") + if (resultArray.length() > 0) { + val result = resultArray.getJSONObject(0) + val text = result.optString("text", "") + + if (text.isNotEmpty()) { + mainHandler.post { + currentContinuousCallback?.onRecognizing(text) + } + } + } + } catch (e: JSONException) { + FileLogger.e(TAG, "解析中间识别结果异常: ${e.message}", e) + } + } + + SpeechEngineDefines.MESSAGE_TYPE_FINAL_RESULT -> { + // 最终识别结果 + try { + val json = JSONObject(data) + if (!json.has("result")) { + return + } + + val resultArray = json.getJSONArray("result") + if (resultArray.length() > 0) { + val result = resultArray.getJSONObject(0) + val text = result.optString("text", "") + + if (text.isNotEmpty()) { + mainHandler.post { + if (isContinuousRecognitionActive) { + currentContinuousCallback?.onResult(text) + } else { + currentAsrCallback?.onSuccess(text) + } + } + } + } + } catch (e: JSONException) { + FileLogger.e(TAG, "解析最终识别结果异常: ${e.message}", e) + } + } + + SpeechEngineDefines.MESSAGE_TYPE_VOLUME_LEVEL -> { + // 音量回调 + if (enableVolume && isContinuousRecognitionActive) { + try { + val volume = (data.toFloat() * 100).toInt() + mainHandler.post { + currentContinuousCallback?.onVolumeChanged(volume) + } + } catch (e: Exception) { + FileLogger.e(TAG, "解析音量数据异常: ${e.message}", e) + } + } + } + + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { + // 错误信息 + try { + val json = JSONObject(data) + val errCode = json.optInt("err_code", -1) + val errMsg = json.optString("err_msg", "未知错误") + + FileLogger.e(TAG, "引擎错误: $errCode, $errMsg") + + mainHandler.post { + if (isContinuousRecognitionActive) { + currentContinuousCallback?.onError(errMsg) + isContinuousRecognitionActive = false + } else { + currentAsrCallback?.onError(errMsg) + } + } + } catch (e: JSONException) { + FileLogger.e(TAG, "解析错误信息异常: ${e.message}", e) + } + } + } + } +} diff --git a/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoSpeechPlugin.kt b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoSpeechPlugin.kt new file mode 100644 index 000000000..bb6f1e452 --- /dev/null +++ b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoSpeechPlugin.kt @@ -0,0 +1,445 @@ +package com.yunqiinnovation.volcano_speech + +import android.content.Context +import android.os.Handler +import android.os.Looper +import androidx.annotation.NonNull +import com.yunqiinnovation.volcano_speech.utils.FileLogger + +import io.flutter.embedding.engine.plugins.FlutterPlugin +import io.flutter.plugin.common.MethodCall +import io.flutter.plugin.common.MethodChannel +import io.flutter.plugin.common.MethodChannel.MethodCallHandler +import io.flutter.plugin.common.MethodChannel.Result +import io.flutter.plugin.common.EventChannel + +/** VolcanoSpeechPlugin */ +class VolcanoSpeechPlugin: FlutterPlugin { + private val TAG = "VolcanoSpeechPlugin" + private lateinit var context: Context + private val mainHandler = Handler(Looper.getMainLooper()) + + // ASR相关 + private lateinit var asrChannel: MethodChannel + private lateinit var asrEventChannel: EventChannel + private var asrEventSink: EventChannel.EventSink? = null + + // TTS相关 + private lateinit var ttsChannel: MethodChannel + private lateinit var ttsEventChannel: EventChannel + private var ttsEventSink: EventChannel.EventSink? = null + + // TTS帮助类 + private lateinit var volcanoTtsHelper: VolcanoTtsHelper + + // ASR帮助类 + private lateinit var volcanoAsrHelper: VolcanoAsrHelper + + override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { + context = flutterPluginBinding.applicationContext + + // 初始化ASR通道 + asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/asr") + asrChannel.setMethodCallHandler(AsrMethodHandler()) + + // 初始化TTS通道 + ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/tts") + ttsChannel.setMethodCallHandler(TtsMethodHandler()) + + // 初始化ASR事件通道 + asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/asr_events") + asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { + override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { + asrEventSink = events + } + + override fun onCancel(arguments: Any?) { + asrEventSink = null + } + }) + + // 初始化TTS事件通道 + ttsEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/tts_events") + ttsEventChannel.setStreamHandler(object : EventChannel.StreamHandler { + override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { + ttsEventSink = events + } + + override fun onCancel(arguments: Any?) { + ttsEventSink = null + } + }) + + // 初始化TTS帮助类 + volcanoTtsHelper = VolcanoTtsHelper(context) + + // 初始化ASR帮助类 + volcanoAsrHelper = VolcanoAsrHelper(context) + } + + // 发送ASR事件 + private fun sendAsrEvent(event: Map) { + FileLogger.d(TAG, "发送ASR事件: $event") + if (asrEventSink == null) { + FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") + return + } + + mainHandler.post { + try { + asrEventSink?.success(event) + FileLogger.d(TAG, "ASR事件发送成功") + } catch (e: Exception) { + FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") + } + } + } + + // 发送TTS事件 + private fun sendTtsEvent(event: Map) { + FileLogger.d(TAG, "发送TTS事件: $event") + if (ttsEventSink == null) { + FileLogger.w(TAG, "无法发送TTS事件:事件通道未准备好") + return + } + + mainHandler.post { + try { + ttsEventSink?.success(event) + FileLogger.d(TAG, "TTS事件发送成功") + } catch (e: Exception) { + FileLogger.e(TAG, "发送TTS事件失败: ${e.message}") + } + } + } + + // ASR方法处理器 + inner class AsrMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "getPlatformVersion" -> { + result.success("Android ${android.os.Build.VERSION.RELEASE}") + } + "initialize" -> { + val appId = call.argument("appId") ?: "" + val apiKey = call.argument("apiKey") ?: "" + val resourceId = call.argument("resourceId") ?: "" + + if (appId.isEmpty() || apiKey.isEmpty() || resourceId.isEmpty()) { + result.error("INVALID_ARGUMENTS", "appId、apiKey和resourceId不能为空", null) + return + } + + val success = volcanoAsrHelper.initialize(appId, apiKey, resourceId) + result.success(success) + } + "setLanguage" -> { + val language = call.argument("language") ?: "zh-CN" + result.success(volcanoAsrHelper.setLanguage(language)) + } + "setHotWords" -> { + val hotWordsId = call.argument("hotWordsId") ?: "" + result.success(volcanoAsrHelper.setHotWords(hotWordsId)) + } + "setVadParams" -> { + val forceToSpeechTime = call.argument("forceToSpeechTime") ?: 0 + val endWindowSize = call.argument("endWindowSize") ?: 800 + result.success(volcanoAsrHelper.setVadParams(forceToSpeechTime, endWindowSize)) + } + "setCorrectWords" -> { + val correctWordsJson = call.argument("correctWordsJson") ?: "{}" + result.success(volcanoAsrHelper.setCorrectWords(correctWordsJson)) + } + "setEnableVolume" -> { + val enable = call.argument("enable") ?: false + result.success(volcanoAsrHelper.setEnableVolume(enable)) + } + "setShowUtterances" -> { + val enable = call.argument("enable") ?: false + result.success(volcanoAsrHelper.setShowUtterances(enable)) + } + "recognizeOnce" -> { + // 确保当前不在连续识别中 + if (volcanoAsrHelper.isContinuousRecognitionActive()) { + result.error("ASR_BUSY", "当前正在连续识别中", null) + return + } + + volcanoAsrHelper.recognizeOnce(object : VolcanoAsrHelper.ASRCallback { + override fun onSuccess(text: String) { + mainHandler.post { + result.success(mapOf( + "text" to text + )) + } + } + + override fun onError(error: String) { + mainHandler.post { + result.error("ASR_ERROR", error, null) + } + } + }) + } + "startContinuousRecognition" -> { + // 确保事件通道已准备好 + if (asrEventSink == null) { + result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) + return + } + + val success = volcanoAsrHelper.startContinuousRecognition(object : VolcanoAsrHelper.ASRContinuousCallback { + override fun onResult(text: String) { + sendAsrEvent(mapOf( + "type" to "result", + "text" to text + )) + } + + override fun onRecognizing(text: String) { + sendAsrEvent(mapOf( + "type" to "recognizing", + "text" to text + )) + } + + override fun onSessionStarted() { + sendAsrEvent(mapOf("type" to "sessionStarted")) + } + + override fun onSessionStopped() { + sendAsrEvent(mapOf("type" to "sessionStopped")) + } + + override fun onVolumeChanged(volume: Int) { + sendAsrEvent(mapOf( + "type" to "volumeChanged", + "volume" to volume + )) + } + + override fun onError(error: String) { + sendAsrEvent(mapOf( + "type" to "error", + "message" to error + )) + } + }) + + result.success(success) + } + "stopContinuousRecognition" -> { + val success = volcanoAsrHelper.stopContinuousRecognition() + result.success(success) + } + "stopRecognize" -> { + val success = volcanoAsrHelper.stopRecognize() + result.success(success) + } + "isContinuousRecognitionActive" -> { + result.success(volcanoAsrHelper.isContinuousRecognitionActive()) + } + "release" -> { + volcanoAsrHelper.release() + result.success(true) + } + else -> { + result.notImplemented() + } + } + } + } + + // TTS方法处理器 + inner class TtsMethodHandler : MethodCallHandler { + override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { + when (call.method) { + "getPlatformVersion" -> { + result.success("Android ${android.os.Build.VERSION.RELEASE}") + } + "initialize" -> { + val appId = call.argument("appId") ?: "" + val token = call.argument("token") ?: "" + val resourceId = call.argument("resourceId") ?: "" + + if (appId.isEmpty() || token.isEmpty() || resourceId.isEmpty()) { + result.error("INVALID_ARGUMENTS", "appId、token和resourceId不能为空", null) + return + } + + val success = volcanoTtsHelper.initialize(appId, token, resourceId) + result.success(success) + } + "setVoice" -> { + val voice = call.argument("voice") ?: "" + + if (voice.isEmpty()) { + result.error("INVALID_ARGUMENTS", "voice不能为空", null) + return + } + + result.success(volcanoTtsHelper.setVoice(voice)) + } + "setContinuousMode" -> { + val isContinuous = call.argument("isContinuous") ?: false + result.success(volcanoTtsHelper.setContinuousMode(isContinuous)) + } + "speak" -> { + val text = call.argument("text") ?: "" + + if (text.isEmpty()) { + result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) + return + } + + volcanoTtsHelper.speak(text, object : VolcanoTtsHelper.TTSCallback { + override fun onStart(reqId: String) { + sendTtsEvent(mapOf( + "type" to "start", + "reqId" to reqId + )) + } + + override fun onProgress(reqId: String, progress: Double) { + sendTtsEvent(mapOf( + "type" to "progress", + "reqId" to reqId, + "progress" to progress + )) + } + + override fun onComplete(reqId: String) { + sendTtsEvent(mapOf( + "type" to "complete", + "reqId" to reqId + )) + + mainHandler.post { + result.success(true) + } + } + + override fun onError(reqId: String, errorCode: Int, errorMsg: String) { + sendTtsEvent(mapOf( + "type" to "error", + "reqId" to reqId, + "errorCode" to errorCode, + "errorMsg" to errorMsg + )) + + mainHandler.post { + result.error("TTS_ERROR", errorMsg, null) + } + } + }) + } + "synthesisNext" -> { + val text = call.argument("text") ?: "" + + if (text.isEmpty()) { + result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) + return + } + + val success = volcanoTtsHelper.synthesisNext(text) + result.success(success) + } + "pause" -> { + result.success(volcanoTtsHelper.pause()) + } + "resume" -> { + result.success(volcanoTtsHelper.resume()) + } + "stop" -> { + result.success(volcanoTtsHelper.stop()) + } + "release" -> { + volcanoTtsHelper.release() + result.success(true) + } + // 以下方法用于与Azure Speech版本兼容 + "setSpeechSynthesisVoice" -> { + val voiceName = call.argument("voiceName") ?: "" + + if (voiceName.isEmpty()) { + result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) + return + } + + result.success(volcanoTtsHelper.setVoice(voiceName)) + } + "speakText" -> { + val text = call.argument("text") ?: "" + + if (text.isEmpty()) { + result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) + return + } + + volcanoTtsHelper.speak(text, object : VolcanoTtsHelper.TTSCallback { + override fun onStart(reqId: String) { + sendTtsEvent(mapOf( + "type" to "start", + "reqId" to reqId + )) + } + + override fun onProgress(reqId: String, progress: Double) { + sendTtsEvent(mapOf( + "type" to "progress", + "reqId" to reqId, + "progress" to progress + )) + } + + override fun onComplete(reqId: String) { + sendTtsEvent(mapOf( + "type" to "complete", + "reqId" to reqId + )) + + mainHandler.post { + result.success(true) + } + } + + override fun onError(reqId: String, errorCode: Int, errorMsg: String) { + sendTtsEvent(mapOf( + "type" to "error", + "reqId" to reqId, + "errorCode" to errorCode, + "errorMsg" to errorMsg + )) + + mainHandler.post { + result.error("TTS_ERROR", errorMsg, null) + } + } + }) + } + "stopSpeaking" -> { + result.success(volcanoTtsHelper.stop()) + } + "pauseSpeaking" -> { + result.success(volcanoTtsHelper.pause()) + } + "resumeSpeaking" -> { + result.success(volcanoTtsHelper.resume()) + } + else -> { + result.notImplemented() + } + } + } + } + + override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { + asrChannel.setMethodCallHandler(null) + ttsChannel.setMethodCallHandler(null) + asrEventChannel.setStreamHandler(null) + ttsEventChannel.setStreamHandler(null) + + volcanoTtsHelper.release() + volcanoAsrHelper.release() + } +} \ No newline at end of file diff --git a/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoTtsHelper.kt b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoTtsHelper.kt new file mode 100644 index 000000000..e24a1ceb1 --- /dev/null +++ b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/VolcanoTtsHelper.kt @@ -0,0 +1,399 @@ +package com.yunqiinnovation.volcano_speech + +import android.content.Context +import android.os.Handler +import android.os.Looper +import com.bytedance.speech.speechengine.SpeechEngine +import com.bytedance.speech.speechengine.SpeechEngineDefines +import com.bytedance.speech.speechengine.SpeechEngineGenerator +import com.yunqiinnovation.volcano_speech.utils.FileLogger +import org.json.JSONObject + +/** + * 火山语音合成帮助类 (大模型版本) + */ +class VolcanoTtsHelper(private val context: Context) { + private val TAG = "VolcanoTtsHelper" + private val mainHandler = Handler(Looper.getMainLooper()) + + // 语音引擎相关 + private var engine: SpeechEngine? = null + private var engineHandler: Long = -1 + private var isInitialized = false + + // 当前回调 + private var currentTtsCallback: TTSCallback? = null + + // 合成状态 + private var isPlaying = false + private var currentReqId = "" + + // 配置参数 + private var voice = "zh_female_yuxi" + private var ttsText = "" + private var isContinuous = false + + /** + * TTS合成回调接口 + */ + interface TTSCallback { + fun onStart(reqId: String) + fun onProgress(reqId: String, progress: Double) + fun onComplete(reqId: String) + fun onError(reqId: String, errorCode: Int, errorMsg: String) + } + + /** + * 初始化语音合成引擎 + */ + fun initialize(appId: String, token: String, resourceId: String): Boolean { + if (isInitialized) { + FileLogger.i(TAG, "引擎已经初始化") + return true + } + + try { + // 准备环境 + SpeechEngineGenerator.PrepareEnvironment(context, null) + + // 创建引擎 + engine = SpeechEngineGenerator.getInstance() + engineHandler = engine?.createEngine() ?: -1 + + if (engineHandler == -1L) { + FileLogger.e(TAG, "创建引擎失败") + return false + } + + // 设置上下文 + engine?.setContext(context) + + // 设置引擎类型为TTS + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, SpeechEngineDefines.TTS_ENGINE) + + // 设置日志级别 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, SpeechEngineDefines.LOG_LEVEL_WARN) + + // 设置用户ID和设备ID (使用静态值,实际项目中应替换为真实值) + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_UID_STRING, "user_id") + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, "device_id") + + // 设置授权信息 - 大模型版本不需要Bearer前缀 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, appId) + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, token) + + // 设置资源ID - 大模型必需 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RESOURCE_ID_STRING, resourceId) + + // 设置协议类型为Seed - 大模型必需 + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_PROTOCOL_TYPE_INT, SpeechEngineDefines.PROTOCOL_TYPE_SEED) + + // 设置合成策略为在线合成 + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_WORK_MODE_INT, SpeechEngineDefines.TTS_WORK_MODE_ONLINE) + + // 设置在线请求资源配置 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_ADDRESS_STRING, "wss://openspeech.bytedance.com") + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_URI_STRING, "/api/v3/sauc/bigmodel") + + // 设置发音人 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, voice) + + // 设置播放进度回调 + engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_WITH_FRONTEND_INT, 1) + + // 初始化引擎 + val result = engine?.initEngine(engineHandler) + isInitialized = result == SpeechEngineDefines.ERR_NO_ERROR + + if (isInitialized) { + FileLogger.i(TAG, "引擎初始化成功") + + // 设置回调监听 + engine?.setListener(object : SpeechEngine.SpeechListener { + override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { + val stdData = String(data) + handleEngineEvent(type, stdData) + } + }) + + return true + } else { + FileLogger.e(TAG, "引擎初始化失败: $result") + return false + } + } catch (e: Exception) { + FileLogger.e(TAG, "初始化异常: ${e.message}", e) + return false + } + } + + /** + * 设置发音人 + */ + fun setVoice(voice: String): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + this.voice = voice + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, voice) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置发音人失败: ${e.message}", e) + return false + } + } + + /** + * 设置合成场景(单次或连续) + */ + fun setContinuousMode(isContinuous: Boolean): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + this.isContinuous = isContinuous + val scenarioType = if (isContinuous) { + SpeechEngineDefines.TTS_SCENARIO_TYPE_NOVEL + } else { + SpeechEngineDefines.TTS_SCENARIO_TYPE_NORMAL + } + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_SCENARIO_STRING, scenarioType) + return true + } catch (e: Exception) { + FileLogger.e(TAG, "设置合成场景失败: ${e.message}", e) + return false + } + } + + /** + * 开始合成并播放 + */ + fun speak(text: String, callback: TTSCallback? = null): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + if (text.isEmpty()) { + FileLogger.e(TAG, "合成文本为空") + return false + } + + try { + this.ttsText = text + this.currentTtsCallback = callback + + // 设置要合成的文本 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, text) + + // 停止当前引擎 + engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") + + // 启动引擎开始合成 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "启动合成失败: $ret") + return false + } + + return true + } catch (e: Exception) { + FileLogger.e(TAG, "合成异常: ${e.message}", e) + return false + } + } + + /** + * 仅用于连续合成场景:在引擎启动后添加新的文本进行合成 + */ + fun synthesisNext(text: String): Boolean { + if (!isInitialized || !isContinuous) { + FileLogger.e(TAG, "引擎未初始化或非连续合成模式") + return false + } + + if (text.isEmpty()) { + FileLogger.e(TAG, "合成文本为空") + return false + } + + try { + // 设置要合成的文本 + engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, text) + + // 发送合成指令 + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNTHESIS, "") + + if (ret != SpeechEngineDefines.ERR_NO_ERROR) { + FileLogger.e(TAG, "添加合成文本失败: $ret") + return false + } + + return true + } catch (e: Exception) { + FileLogger.e(TAG, "添加合成文本异常: ${e.message}", e) + return false + } + } + + /** + * 暂停播放 + */ + fun pause(): Boolean { + if (!isInitialized || !isPlaying) { + FileLogger.e(TAG, "引擎未初始化或未在播放") + return false + } + + try { + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_PAUSE_PLAYER, "") + return ret == SpeechEngineDefines.ERR_NO_ERROR + } catch (e: Exception) { + FileLogger.e(TAG, "暂停播放异常: ${e.message}", e) + return false + } + } + + /** + * 恢复播放 + */ + fun resume(): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_RESUME_PLAYER, "") + return ret == SpeechEngineDefines.ERR_NO_ERROR + } catch (e: Exception) { + FileLogger.e(TAG, "恢复播放异常: ${e.message}", e) + return false + } + } + + /** + * 停止播放 + */ + fun stop(): Boolean { + if (!isInitialized) { + FileLogger.e(TAG, "引擎未初始化") + return false + } + + try { + val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") + isPlaying = false + return ret == SpeechEngineDefines.ERR_NO_ERROR + } catch (e: Exception) { + FileLogger.e(TAG, "停止播放异常: ${e.message}", e) + return false + } + } + + /** + * 释放资源 + */ + fun release() { + if (!isInitialized) { + return + } + + try { + // 停止引擎 + stop() + + // 销毁引擎 + engine?.destroyEngine(engineHandler) + engineHandler = -1 + engine = null + isInitialized = false + + FileLogger.i(TAG, "引擎已释放") + } catch (e: Exception) { + FileLogger.e(TAG, "释放引擎异常: ${e.message}", e) + } + } + + /** + * 处理引擎事件 + */ + private fun handleEngineEvent(type: Int, data: String) { + when (type) { + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { + // 引擎启动成功,获取请求ID + currentReqId = data + isPlaying = true + + mainHandler.post { + currentTtsCallback?.onStart(currentReqId) + } + + if (isContinuous) { + // 在连续合成模式下,需要单独发送合成指令 + synthesisNext(ttsText) + } + } + + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { + // 引擎停止 + isPlaying = false + } + + SpeechEngineDefines.MESSAGE_TYPE_TTS_START_PLAYING -> { + // 开始播放 + FileLogger.d(TAG, "开始播放: $data") + } + + SpeechEngineDefines.MESSAGE_TYPE_TTS_FINISH_PLAYING -> { + // 播放结束 + isPlaying = false + + mainHandler.post { + currentTtsCallback?.onComplete(currentReqId) + } + } + + SpeechEngineDefines.MESSAGE_TYPE_TTS_PLAYBACK_PROGRESS -> { + // 播放进度 + try { + val json = JSONObject(data) + val progress = json.optDouble("progress", 0.0) + val reqId = json.optString("reqid", "") + + mainHandler.post { + currentTtsCallback?.onProgress(reqId, progress) + } + } catch (e: Exception) { + FileLogger.e(TAG, "解析进度信息异常: ${e.message}", e) + } + } + + SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { + // 错误信息 + try { + val json = JSONObject(data) + val reqId = json.optString("reqid", "") + val errCode = json.optInt("err_code", -1) + val errMsg = json.optString("err_msg", "未知错误") + + FileLogger.e(TAG, "引擎错误: $errCode, $errMsg") + + isPlaying = false + + mainHandler.post { + currentTtsCallback?.onError(reqId, errCode, errMsg) + } + } catch (e: Exception) { + FileLogger.e(TAG, "解析错误信息异常: ${e.message}", e) + } + } + } + } +} diff --git a/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/utils/FileLogger.kt b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/utils/FileLogger.kt new file mode 100644 index 000000000..8fdfb1227 --- /dev/null +++ b/local_plugins/volcano_speech/android/src/main/kotlin/com/yunqiinnovation/volcano_speech/utils/FileLogger.kt @@ -0,0 +1,55 @@ +package com.yunqiinnovation.volcano_speech.utils + +import android.util.Log + +/** + * 文件日志记录工具 + */ +object FileLogger { + private const val TAG = "VolcanoSpeech" + private var isDebugEnabled = true + + /** + * 设置是否启用调试日志 + */ + fun setDebugEnabled(enabled: Boolean) { + isDebugEnabled = enabled + } + + /** + * 记录调试日志 + */ + fun d(tag: String, message: String) { + if (isDebugEnabled) { + Log.d("$TAG-$tag", message) + } + } + + /** + * 记录信息日志 + */ + fun i(tag: String, message: String) { + Log.i("$TAG-$tag", message) + } + + /** + * 记录警告日志 + */ + fun w(tag: String, message: String) { + Log.w("$TAG-$tag", message) + } + + /** + * 记录错误日志 + */ + fun e(tag: String, message: String) { + Log.e("$TAG-$tag", message) + } + + /** + * 记录错误日志,带异常 + */ + fun e(tag: String, message: String, throwable: Throwable) { + Log.e("$TAG-$tag", message, throwable) + } +} \ No newline at end of file diff --git a/local_plugins/volcano_speech/lib/volcano_speech.dart b/local_plugins/volcano_speech/lib/volcano_speech.dart new file mode 100644 index 000000000..99da7d4b7 --- /dev/null +++ b/local_plugins/volcano_speech/lib/volcano_speech.dart @@ -0,0 +1,603 @@ +import 'dart:async'; +import 'package:flutter/services.dart'; + +/// ASR 事件类型 +enum AsrEventType { + /// 会话开始 + sessionStarted, + + /// 会话结束 + sessionStopped, + + /// 正在识别(中间结果) + recognizing, + + /// 识别结果(最终结果) + result, + + /// 音量变化 + volumeChanged, + + /// 错误 + error +} + +/// TTS 工作模式 +enum TtsWorkMode { + /// 在线合成 + online, + + /// 离线合成 + offline, + + /// 同时在线离线 + both, + + /// 先在线再离线(网络不好时自动切换) + alternate, + + /// 文件模式 + file +} + +/// TTS 文本类型 +enum TtsTextType { + /// 纯文本 + plain, + + /// SSML格式 + ssml +} + +/// 协议类型 +enum ProtocolType { + /// 默认协议 + defaultProtocol, + + /// Seed协议(用于大模型) + seed +} + +/// TTS 播放进度事件 +class TtsProgressEvent { + /// 播放进度 0.0-1.0 + final double progress; + + /// 请求ID + final String reqId; + + const TtsProgressEvent({ + required this.progress, + required this.reqId, + }); + + @override + String toString() { + return 'TtsProgressEvent{progress: $progress, reqId: $reqId}'; + } +} + +/// ASR 事件 +class AsrEvent { + /// 事件类型 + final AsrEventType type; + + /// 识别文本(仅在recognizing和result类型时有效) + final String? text; + + /// 识别语言(仅在recognizing和result类型时有效) + final String? language; + + /// 音量值(仅在volumeChanged类型时有效) + final int? volume; + + /// 错误信息(仅在error类型时有效) + final String? errorMessage; + + const AsrEvent({ + required this.type, + this.text, + this.language, + this.volume, + this.errorMessage, + }); + + factory AsrEvent.fromMap(Map map) { + final typeStr = map['type'] as String; + + AsrEventType type; + switch (typeStr) { + case 'sessionStarted': + type = AsrEventType.sessionStarted; + break; + case 'sessionStopped': + type = AsrEventType.sessionStopped; + break; + case 'recognizing': + type = AsrEventType.recognizing; + break; + case 'result': + type = AsrEventType.result; + break; + case 'volumeChanged': + type = AsrEventType.volumeChanged; + break; + case 'error': + type = AsrEventType.error; + break; + default: + throw ArgumentError('未知的事件类型: $typeStr'); + } + + return AsrEvent( + type: type, + text: map['text'] as String?, + language: map['language'] as String?, + volume: map['volume'] as int?, + errorMessage: map['message'] as String?, + ); + } + + @override + String toString() { + return 'AsrEvent{type: $type, text: $text, language: $language, volume: $volume, errorMessage: $errorMessage}'; + } +} + +/// 火山引擎语音服务插件 +class VolcanoSpeech { + static final VolcanoSpeechAsr asr = VolcanoSpeechAsr._(); + static final VolcanoSpeechTts tts = VolcanoSpeechTts._(); + + /// 释放资源 + static Future dispose() async { + await asr.dispose(); + await tts.dispose(); + } +} + +/// 火山引擎语音识别服务 +class VolcanoSpeechAsr { + static const MethodChannel _channel = MethodChannel('volcano_speech/asr'); + static const EventChannel _eventChannel = EventChannel('volcano_speech/asr_events'); + + /// ASR事件流控制器 + static final StreamController _eventStreamController = StreamController.broadcast(); + + /// ASR事件流 + Stream get events => _eventStreamController.stream; + + /// 是否已初始化事件监听 + bool _eventListenerInitialized = false; + + VolcanoSpeechAsr._() { + _initEventListener(); + } + + /// 初始化ASR事件监听 + void _initEventListener() { + if (_eventListenerInitialized) return; + + _eventChannel.receiveBroadcastStream().listen((dynamic event) { + if (event is Map) { + _eventStreamController.add(AsrEvent.fromMap(event)); + } + }); + + _eventListenerInitialized = true; + } + + /// 获取平台版本信息 + Future getPlatformVersion() async { + return await _channel.invokeMethod('getPlatformVersion'); + } + + /// 初始化语音识别引擎 + /// + /// [appId] 火山引擎AppID + /// [apiKey] 火山引擎ApiKey + /// [supportedLanguages] 支持的语言列表 + /// [useBigModel] 是否使用大模型识别 + /// [resourceId] 大模型资源ID(仅当useBigModel为true时有效) + Future initialize({ + required String appId, + required String apiKey, + List supportedLanguages = const ['zh-CN'], + bool useBigModel = false, + String resourceId = '', + }) async { + return await _channel.invokeMethod('initialize', { + 'appId': appId, + 'apiKey': apiKey, + 'supportedLanguages': supportedLanguages, + 'useBigModel': useBigModel, + 'resourceId': resourceId, + }) ?? false; + } + + /// 设置识别语言 + /// + /// [language] 语言代码,例如 zh-CN、en-US + Future setLanguage(String language) async { + return await _channel.invokeMethod('setLanguage', { + 'language': language, + }) ?? false; + } + + /// 设置热词 + /// + /// [hotWords] 热词JSON字符串,例如 {"hotwords":[{"word":"快速入门","scale":2.0}]} + Future setHotWords(String hotWords) async { + return await _channel.invokeMethod('setHotWords', { + 'hotWords': hotWords, + }) ?? false; + } + + /// 设置ASR请求参数 + /// + /// [params] 请求参数JSON字符串 + Future setRequestParams(String params) async { + return await _channel.invokeMethod('setRequestParams', { + 'params': params, + }) ?? false; + } + + /// 启用语音停顿、分句、分词信息输出 + /// + /// [enable] 是否启用 + Future setShowUtterances(bool enable) async { + return await _channel.invokeMethod('setShowUtterances', { + 'enable': enable, + }) ?? false; + } + + /// 一次性识别(直到说话结束) + /// + /// 返回识别结果文本和语言 + Future> recognizeOnce() async { + final result = await _channel.invokeMethod('recognizeOnce'); + return { + 'text': result['text'] ?? '', + 'language': result['language'] ?? 'zh-CN', + }; + } + + /// 开始连续识别 + /// + /// 通过[events]流监听识别结果 + Future startContinuousRecognition() async { + return await _channel.invokeMethod('startContinuousRecognition') ?? false; + } + + /// 停止连续识别 + Future stopContinuousRecognition() async { + return await _channel.invokeMethod('stopContinuousRecognition') ?? false; + } + + /// 检查是否正在连续识别 + Future isContinuousRecognitionActive() async { + return await _channel.invokeMethod('isContinuousRecognitionActive') ?? false; + } + + /// 开始长按识别(按下开始,抬起结束) + /// + /// 通过[events]流监听识别结果 + Future startListening() async { + return await _channel.invokeMethod('startListening') ?? false; + } + + /// 停止长按识别 + Future stopListening() async { + return await _channel.invokeMethod('stopListening') ?? false; + } + + /// 释放ASR资源 + Future dispose() async { + return await _channel.invokeMethod('dispose') ?? false; + } +} + +/// 火山引擎语音合成服务 +class VolcanoSpeechTts { + static const MethodChannel _channel = MethodChannel('volcano_speech/tts'); + static const EventChannel _eventChannel = EventChannel('volcano_speech/tts_events'); + + /// TTS进度事件流控制器 + static final StreamController _progressStreamController = StreamController.broadcast(); + + /// TTS进度事件流 + Stream get progressEvents => _progressStreamController.stream; + + /// 是否已初始化事件监听 + bool _eventListenerInitialized = false; + + VolcanoSpeechTts._() { + _initEventListener(); + } + + /// 初始化TTS事件监听 + void _initEventListener() { + if (_eventListenerInitialized) return; + + _eventChannel.receiveBroadcastStream().listen((dynamic event) { + if (event is Map) { + final progress = event['progress'] as double?; + final reqId = event['reqId'] as String?; + + if (progress != null && reqId != null) { + _progressStreamController.add(TtsProgressEvent( + progress: progress, + reqId: reqId, + )); + } + } + }); + + _eventListenerInitialized = true; + } + + /// 获取平台版本信息 + Future getPlatformVersion() async { + return await _channel.invokeMethod('getPlatformVersion'); + } + + /// 启用大模型TTS + /// + /// [enable] 是否启用大模型TTS + /// [resourceId] 资源ID + Future enableBigModelTts({ + required bool enable, + String resourceId = '', + }) async { + return await _channel.invokeMethod('enableBigModelTts', { + 'enable': enable, + 'resourceId': resourceId, + }) ?? false; + } + + /// 初始化语音合成引擎 + /// + /// [appId] 火山引擎AppID + /// [apiKey] 火山引擎ApiKey + Future initialize({ + required String appId, + required String apiKey, + }) async { + return await _channel.invokeMethod('initialize', { + 'appId': appId, + 'apiKey': apiKey, + }) ?? false; + } + + /// 设置音色 + /// + /// [voiceName] 音色名称 + /// [voiceType] 音色类型,默认 qingxin + Future setVoice({ + required String voiceName, + String voiceType = 'qingxin', + }) async { + return await _channel.invokeMethod('setVoice', { + 'voiceName': voiceName, + 'voiceType': voiceType, + }) ?? false; + } + + /// 设置离线音色 + /// + /// [voiceName] 音色名称 + /// [voiceType] 音色类型,默认 qingxin + Future setOfflineVoice({ + required String voiceName, + String voiceType = 'qingxin', + }) async { + return await _channel.invokeMethod('setOfflineVoice', { + 'voiceName': voiceName, + 'voiceType': voiceType, + }) ?? false; + } + + /// 设置大模型声音ID + /// + /// [voiceId] 声音ID + Future setBigModelVoiceId(String voiceId) async { + return await _channel.invokeMethod('setBigModelVoiceId', { + 'voiceId': voiceId, + }) ?? false; + } + + /// 设置大模型请求参数 + /// + /// [params] 请求参数,JSON字符串 + Future setBigModelRequestParams(String params) async { + return await _channel.invokeMethod('setBigModelRequestParams', { + 'params': params, + }) ?? false; + } + + /// 设置工作模式 + /// + /// [mode] 工作模式 + Future setWorkMode(TtsWorkMode mode) async { + String modeStr; + switch (mode) { + case TtsWorkMode.online: + modeStr = 'online'; + break; + case TtsWorkMode.offline: + modeStr = 'offline'; + break; + case TtsWorkMode.both: + modeStr = 'both'; + break; + case TtsWorkMode.alternate: + modeStr = 'alternate'; + break; + case TtsWorkMode.file: + modeStr = 'file'; + break; + } + + return await _channel.invokeMethod('setWorkMode', { + 'mode': modeStr, + }) ?? false; + } + + /// 设置文本类型 + /// + /// [type] 文本类型,plain或ssml + Future setTextType(TtsTextType type) async { + String typeStr; + switch (type) { + case TtsTextType.plain: + typeStr = 'plain'; + break; + case TtsTextType.ssml: + typeStr = 'ssml'; + break; + } + + return await _channel.invokeMethod('setTextType', { + 'type': typeStr, + }) ?? false; + } + + /// 设置是否启用缓存 + /// + /// [enable] 是否启用 + Future setEnableCache(bool enable) async { + return await _channel.invokeMethod('setEnableCache', { + 'enable': enable, + }) ?? false; + } + + /// 设置情感 + /// + /// [emotion] 情感,例如 neutral、happy、angry、sad等 + Future setEmotion(String emotion) async { + return await _channel.invokeMethod('setEmotion', { + 'emotion': emotion, + }) ?? false; + } + + /// 设置是否启用情感预测 + /// + /// [enable] 是否启用 + Future setEnableEmotionPredict(bool enable) async { + return await _channel.invokeMethod('setEnableEmotionPredict', { + 'enable': enable, + }) ?? false; + } + + /// 设置是否启用声音克隆 + /// + /// [enable] 是否启用 + /// [backendCluster] 后端集群 + Future setEnableVoiceClone({ + required bool enable, + String backendCluster = '', + }) async { + return await _channel.invokeMethod('setEnableVoiceClone', { + 'enable': enable, + 'backendCluster': backendCluster, + }) ?? false; + } + + /// 设置是否启用回声消除 + /// + /// [enable] 是否启用 + Future setEnableAEC(bool enable) async { + return await _channel.invokeMethod('setEnableAEC', { + 'enable': enable, + }) ?? false; + } + + /// 下载离线资源 + /// + /// [voiceTypes] 音色类型列表 + /// [languages] 语言列表 + Future downloadOfflineResource({ + List voiceTypes = const ['qingxin'], + List languages = const ['zh-CN'], + }) async { + return await _channel.invokeMethod('downloadOfflineResource', { + 'voiceTypes': voiceTypes, + 'languages': languages, + }) ?? false; + } + + /// 设置语音参数 + /// + /// [rate] 语速 -500~500 + /// [volume] 音量 0~100 + /// [pitch] 音调 -500~500 + /// [silenceDuration] 静音时长,毫秒 + Future setSpeechParams({ + int rate = 0, + int volume = 100, + int pitch = 0, + int silenceDuration = 0, + }) async { + return await _channel.invokeMethod('setSpeechParams', { + 'rate': rate, + 'volume': volume, + 'pitch': pitch, + 'silenceDuration': silenceDuration, + }) ?? false; + } + + /// 设置音频输出类型 + /// + /// [outputType] 输出类型,speaker(扬声器),earpiece(听筒),auto(自动) + Future setAudioOutputType(String outputType) async { + return await _channel.invokeMethod('setAudioOutputType', { + 'outputType': outputType, + }) ?? false; + } + + /// 合成并播放文本 + /// + /// [text] 待合成的文本 + Future speakText(String text) async { + return await _channel.invokeMethod('speakText', { + 'text': text, + }) ?? false; + } + + /// 使用大模型合成并播放文本 + /// + /// [text] 待合成的文本 + /// [voiceId] 声音ID + /// [params] 额外参数,JSON字符串 + Future speakWithBigModel({ + required String text, + required String voiceId, + String params = '', + }) async { + return await _channel.invokeMethod('speakWithBigModel', { + 'text': text, + 'voiceId': voiceId, + 'params': params, + }) ?? false; + } + + /// 暂停播放 + Future pausePlayback() async { + return await _channel.invokeMethod('pausePlayback') ?? false; + } + + /// 恢复播放 + Future resumePlayback() async { + return await _channel.invokeMethod('resumePlayback') ?? false; + } + + /// 停止播放 + Future stopSpeaking() async { + return await _channel.invokeMethod('stopSpeaking') ?? false; + } + + /// 释放TTS资源 + Future dispose() async { + return await _channel.invokeMethod('dispose') ?? false; + } +} \ No newline at end of file diff --git a/local_plugins/volcano_speech/pubspec.yaml b/local_plugins/volcano_speech/pubspec.yaml new file mode 100644 index 000000000..cd90d8b14 --- /dev/null +++ b/local_plugins/volcano_speech/pubspec.yaml @@ -0,0 +1,28 @@ +name: volcano_speech +description: 火山引擎语音合成服务插件 +version: 0.0.1 +homepage: + +environment: + sdk: ">=2.17.0 <3.0.0" + flutter: ">=2.5.0" + +dependencies: + flutter: + sdk: flutter + +dev_dependencies: + flutter_test: + sdk: flutter + flutter_lints: ^2.0.0 + +# The following section is specific to Flutter packages. +flutter: + # This section identifies this Flutter project as a plugin project. + plugin: + platforms: + android: + package: com.yunqiinnovation.volcano_speech + pluginClass: VolcanoSpeechPlugin + ios: + pluginClass: VolcanoSpeechPlugin \ No newline at end of file diff --git a/pubspec.yaml b/pubspec.yaml index 533f3dd20..d9105183c 100644 --- a/pubspec.yaml +++ b/pubspec.yaml @@ -61,6 +61,12 @@ dependencies: dio: ^5.8.0+1 sqflite: ^2.4.2 path: ^1.9.1 + azure_speech: + path: local_plugins/azure_speech + open_ai_service: + path: local_plugins/open_ai_service + volcano_speech: + path: local_plugins/volcano_speech dev_dependencies: flutter_test: diff --git a/test.json b/test.json new file mode 100644 index 000000000..703948620 --- /dev/null +++ b/test.json @@ -0,0 +1 @@ +curl -v -X POST -H 'Content-Type: application/json' -H 'Authorization: Bearer 168deb3d-fd0c-4912-b9f1-aaee5c6743e6' -H 'Accept: text/event-stream' -d '{"model":"bot-20250405211523-l7c9r","messages":[{"role":"system","content":" 你是一个智能语音助手,能够简洁明了地回答用户的问题。\n时刻关心用户的情绪和需求,主动提供鼓励和温暖。\n\n语言风格活泼、亲切,能够幽默地互动,陪伴用户,缓解压力,增添生活乐趣。\n\n请始终以用户为中心,保持回应的高效性、准确性和温暖体贴,成为用户真正的灵魂伴侣。\n \n 当用户说\"退出\"、\"再见\"、\"结束对话\"等类似意图时,你应该使用exit_interaction函数来结束对话,\n 并在结束前说一句友好的告别语,例如\"再见,有需要随时找我\"。"},{"role":"system","content":" 你是一个智能语音助手,能够简洁明了地回答用户的问题。\n时刻关心用户的情绪和需求,主动提供鼓励和温暖。\n\n语言风格活泼、亲切,能够幽默地互动,陪伴用户,缓解压力,增添生活乐趣。\n\n请始终以用户为中心,保持回应的高效性、准确性和温暖体贴,成为用户真正的灵魂伴侣。\n \n 当用户说\"退出\"、\"再见\"、\"结束对话\"等类似意图时,你应该使用exit_interaction函数来结束对话,\n 并在结束前说一句友好的告别语,例如\"再见,有需要随时找我\"。"},{"role":"user","content":"退下吧。"},{"role":"assistant","content":"","tool_calls":[{"id":"call_8k680azmfc4thqrrnwpwqxah","type":"function","function":{"name":"exit_interaction","arguments":" {}"}}]},{"role":"tool","content":"{\"result\": \"已退出语音交互\"}","tool_call_id":"call_8k680azmfc4thqrrnwpwqxah"}],"temperature":0.7,"max_tokens":2000,"stream":true,"tools":[{"type":"function","function":{"name":"exit_interaction","description":"退出当前语音交互","parameters":{"type":"object","properties":{},"required":[]}}}]}' 'https://ark.cn-beijing.volces.com/api/v3/bots/chat/completions' \ No newline at end of file