Browse Source

init

newdev_shunjiawei
wolfplus 2 years ago
parent
commit
084e27776a
  1. 10
      android/app/build.gradle.kts
  2. 654
      android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt
  3. 344
      android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt
  4. 0
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt
  5. 255
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt
  6. 606
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt
  7. 180
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt
  8. 208
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt
  9. 0
      android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt
  10. 4
      android/settings.gradle.kts
  11. 21
      azure/LICENSE
  12. 66
      azure/README.md
  13. 757
      azure/ios/Classes/AzureAsrHelper.swift
  14. 18
      azure/ios/Classes/AzureSpeechRecognitionPlugin.swift
  15. 427
      azure/ios/Classes/AzureTtsHelper.swift
  16. 259
      azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift
  17. 24
      azure/ios/azure_speech_recognition.podspec
  18. 6
      azure/lib/azure_speech_recognition.dart
  19. 23
      azure/pubspec.yaml
  20. 4
      lib/data/services/speech_impl/azure_asr_service.dart
  21. 2
      lib/data/services/speech_impl/azure_tts_service.dart
  22. 12
      lib/data/services/voice_interaction_service.dart
  23. 136
      local_plugins/azure_speech/README.md
  24. 65
      local_plugins/azure_speech/android/build.gradle.kts
  25. 1
      local_plugins/azure_speech/android/settings.gradle.kts
  26. 6
      local_plugins/azure_speech/android/src/main/AndroidManifest.xml
  27. 593
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt
  28. 297
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt
  29. 12
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt
  30. 45
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt
  31. 461
      local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift
  32. 182
      local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift
  33. 330
      local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift
  34. 32
      local_plugins/azure_speech/pubspec.yaml
  35. 2
      pubspec.yaml

10
android/app/build.gradle.kts

@ -34,6 +34,11 @@ android {
jvmTarget = JavaVersion.VERSION_11.toString() jvmTarget = JavaVersion.VERSION_11.toString()
} }
// 添加lint选项
lintOptions {
isCheckReleaseBuilds = false
}
defaultConfig { defaultConfig {
// TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html). // TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html).
applicationId = "com.yunqiinnovation.deepsound" applicationId = "com.yunqiinnovation.deepsound"
@ -82,6 +87,11 @@ android {
dependencies { dependencies {
implementation("io.modelcontextprotocol:kotlin-sdk:0.4.0")
// 添加本地插件模块依赖
implementation(project(":azure_speech"))
// 添加OkHttp依赖 // 添加OkHttp依赖
implementation("com.squareup.okhttp3:okhttp:4.9.3") implementation("com.squareup.okhttp3:okhttp:4.9.3")

654
android/app/src/main/kotlin/com/example/deep_voice/AzureAsrHelper.kt

@ -1,654 +0,0 @@
package com.yunqiinnovation.deepsound
import android.content.Context
import android.media.AudioAttributes
import android.media.AudioFormat
import android.media.AudioRecord
import android.media.MediaRecorder
import android.media.audiofx.AcousticEchoCanceler
import android.media.audiofx.NoiseSuppressor
import android.media.audiofx.AutomaticGainControl
import android.os.Process
import com.yunqiinnovation.deepsound.core.utils.FileLogger
import com.microsoft.cognitiveservices.speech.*
import com.microsoft.cognitiveservices.speech.audio.*
import com.microsoft.cognitiveservices.speech.util.EventHandler
import java.util.concurrent.ExecutionException
import java.util.concurrent.atomic.AtomicBoolean
// 移除WebRTC相关导入
class AzureAsrHelper(private val context: Context) {
private var recognizer: SpeechRecognizer? = null
private var speechConfig: SpeechConfig? = null
private val TAG = "AzureAsrHelper"
private var isContinuousRecognitionActive = false
private var currentLanguage = "zh-CN"
private var subscriptionKey = ""
private var serviceRegion = ""
private var isAutoDetectLanguage = false
private var supportedLanguages = arrayOf("zh-CN", "en-US")
// 是否使用回音消除 - 内部控制常量
private val useEchoCancellation = true
// 自定义音频处理相关
private var customAudioProcessor: CustomAudioProcessor? = null
private var pushStream: PushAudioInputStream? = null
private var audioConfig: AudioConfig? = null
// 初始化SDK并创建recognizer
fun initialize(subscriptionKey: String, serviceRegion: String,
supportedLanguages: Array<String> = arrayOf("zh-CN", "en-US")): Boolean {
try {
FileLogger.d(TAG, "初始化 Azure 语音服务")
// 检查配置是否为空
if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) {
FileLogger.e(TAG, "Azure 配置信息不完整")
return false
}
// 释放之前的资源
dispose()
this.subscriptionKey = subscriptionKey
this.serviceRegion = serviceRegion
// 设置语言
if (supportedLanguages.isNotEmpty()) {
this.supportedLanguages = supportedLanguages
}
// 根据支持的语言数量决定是否启用自动语言检测
this.isAutoDetectLanguage = supportedLanguages.size >= 2
// 如果只有一种语言,设置为当前语言
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) {
this.currentLanguage = supportedLanguages[0]
}
// 创建语音配置
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion)
// 设置语言配置
if (isAutoDetectLanguage) {
// 设置自动语言检测
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous")
} else {
// 设置指定的识别语言
speechConfig?.speechRecognitionLanguage = currentLanguage
}
// 创建识别器
try {
if (useEchoCancellation) {
// 如果使用回音消除,创建自定义音频输入流
setupCustomAudioProcessing()
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
} else {
recognizer = SpeechRecognizer(speechConfig, audioConfig)
}
} else {
// 使用默认麦克风输入
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
} else {
recognizer = SpeechRecognizer(speechConfig)
}
}
FileLogger.d(TAG, "Azure 语音服务初始化成功")
return true
} catch (e: Exception) {
FileLogger.e(TAG, "创建识别器失败: ${e.message}")
stopCustomAudioProcessing()
return false
}
} catch (e: Exception) {
FileLogger.e(TAG, "初始化失败: ${e.message}")
return false
}
}
// 重置 recognizer
private fun resetRecognizer(): Boolean {
try {
// 释放之前的 recognizer
recognizer?.close()
recognizer = null
// 停止当前的音频处理
stopCustomAudioProcessing()
// 使用现有配置重新创建 recognizer
if (speechConfig != null) {
if (useEchoCancellation) {
// 如果使用回音消除,创建自定义音频输入流
setupCustomAudioProcessing()
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
} else {
recognizer = SpeechRecognizer(speechConfig, audioConfig)
}
} else {
// 使用默认麦克风输入
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
} else {
recognizer = SpeechRecognizer(speechConfig)
}
}
return true
} else {
FileLogger.e(TAG, "语音配置未初始化")
return false
}
} catch (e: Exception) {
FileLogger.e(TAG, "重置识别器失败: ${e.message}")
return false
}
}
// 开始一次性语音识别
fun recognizeOnce(callback: RecognizeCallback) {
if (speechConfig == null) {
callback.onError("语音服务未初始化")
return
}
// 重置 recognizer
if (!resetRecognizer()) {
callback.onError("重置识别器失败")
return
}
try {
// 启动音频处理
startCustomAudioProcessing()
// 执行识别
val result = recognizer?.recognizeOnceAsync()?.get()
// 停止音频处理
stopCustomAudioProcessing()
if (result != null && result.reason == ResultReason.RecognizedSpeech) {
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language
callback.onResult(result.text, detectedLanguage ?: "")
} else {
callback.onError("未能识别语音")
}
} catch (e: Exception) {
// 停止音频处理
stopCustomAudioProcessing()
callback.onError("识别异常: ${e.message}")
}
}
// 开始连续语音识别
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
if (speechConfig == null) {
callback.onError("语音服务未初始化")
return false
}
// 如果已经在进行连续识别,先停止
if (isContinuousRecognitionActive) {
stopContinuousRecognition(callback)
}
// 重置 recognizer
if (!resetRecognizer()) {
callback.onError("重置识别器失败")
return false
}
try {
// 启动音频处理
startCustomAudioProcessing()
// 设置识别事件处理
// 最终识别结果
recognizer?.recognized?.addEventListener(
EventHandler<SpeechRecognitionEventArgs> { _, event ->
if (event.result.reason == ResultReason.RecognizedSpeech) {
val detectedLanguage = if (isAutoDetectLanguage) {
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: ""
} else {
currentLanguage
}
// FileLogger.d(TAG, "最终识别结果: ${event.result.text}")
callback.onResult(event.result.text, detectedLanguage)
}
}
)
// 识别中事件
recognizer?.recognizing?.addEventListener(
EventHandler<SpeechRecognitionEventArgs> { _, event ->
if (event.result.reason == ResultReason.RecognizingSpeech) {
val detectedLanguage = if (isAutoDetectLanguage) {
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: ""
} else {
currentLanguage
}
// FileLogger.d(TAG, "识别中结果: ${event.result.text}")
callback.onRecognizing(event.result.text, detectedLanguage)
}
}
)
// 会话事件
recognizer?.sessionStarted?.addEventListener(
EventHandler<SessionEventArgs> { _, _ ->
isContinuousRecognitionActive = true
callback.onSessionStarted()
}
)
recognizer?.sessionStopped?.addEventListener(
EventHandler<SessionEventArgs> { _, _ ->
isContinuousRecognitionActive = false
callback.onSessionStopped()
}
)
// 取消事件
recognizer?.canceled?.addEventListener(
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event ->
val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else ""
callback.onCanceled(event.reason.toString(), errorDetails)
isContinuousRecognitionActive = false
}
)
// 开始连续识别
recognizer?.startContinuousRecognitionAsync()?.get()
isContinuousRecognitionActive = true
return true
} catch (e: Exception) {
callback.onError("开始连续识别失败: ${e.message}")
isContinuousRecognitionActive = false
stopCustomAudioProcessing()
return false
}
}
// 停止连续语音识别
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
if (!isContinuousRecognitionActive || recognizer == null) {
return true
}
try {
recognizer?.stopContinuousRecognitionAsync()
isContinuousRecognitionActive = false
callback.onSessionStopped()
// 停止音频处理
stopCustomAudioProcessing()
return true
} catch (e: Exception) {
callback.onError("停止连续识别失败: ${e.message}")
isContinuousRecognitionActive = false
stopCustomAudioProcessing()
return false
}
}
// 检查连续识别是否活跃
fun isContinuousRecognitionActive(): Boolean {
return isContinuousRecognitionActive
}
// 设置自定义音频处理
private fun setupCustomAudioProcessing() {
if (!useEchoCancellation) {
return
}
try {
// 1. 创建PushAudioInputStream
pushStream = PushAudioInputStream.create()
// 2. 创建AudioConfig
audioConfig = AudioConfig.fromStreamInput(pushStream)
// 3. 创建自定义音频处理器
customAudioProcessor = CustomAudioProcessor(pushStream)
FileLogger.d(TAG, "自定义音频处理设置完成")
} catch (e: Exception) {
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}")
releaseCustomAudioProcessing()
// 降级处理:如果自定义处理设置失败,尝试使用默认麦克风
try {
FileLogger.d(TAG, "尝试降级到默认麦克风输入")
audioConfig = AudioConfig.fromDefaultMicrophoneInput()
} catch (e2: Exception) {
FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}")
audioConfig = null
}
}
}
// 启动自定义音频处理
private fun startCustomAudioProcessing() {
if (!useEchoCancellation || customAudioProcessor == null) {
return
}
try {
customAudioProcessor?.startRecording()
FileLogger.d(TAG, "自定义音频处理已启动")
} catch (e: Exception) {
FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}")
}
}
// 停止自定义音频处理
private fun stopCustomAudioProcessing() {
if (!useEchoCancellation || customAudioProcessor == null) {
return
}
try {
customAudioProcessor?.stopRecording()
FileLogger.d(TAG, "自定义音频处理已停止")
} catch (e: Exception) {
FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}")
}
}
// 释放自定义音频处理资源
private fun releaseCustomAudioProcessing() {
stopCustomAudioProcessing()
try {
customAudioProcessor = null
pushStream?.close()
pushStream = null
audioConfig?.close()
audioConfig = null
FileLogger.d(TAG, "自定义音频处理资源已释放")
} catch (e: Exception) {
FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}")
}
}
// 释放所有资源
fun dispose() {
try {
// 停止和释放音频处理
releaseCustomAudioProcessing()
recognizer?.close()
recognizer = null
speechConfig?.close()
speechConfig = null
isContinuousRecognitionActive = false
FileLogger.d(TAG, "语音识别资源已释放")
} catch (e: Exception) {
FileLogger.e(TAG, "释放资源出错: ${e.message}")
}
}
// 自定义音频处理器 - 使用Android原生回音消除
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) {
private val SAMPLE_RATE = 16000
private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO
private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT
private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定
private var audioRecord: AudioRecord? = null
private var echoCanceler: AcousticEchoCanceler? = null
private val isRecording = AtomicBoolean(false)
private var recordingThread: Thread? = null
// 启动录音并处理音频数据
fun startRecording() {
if (isRecording.get() || pushStream == null) {
return
}
try {
// 使用Builder模式构建AudioFormat
val audioFormat = AudioFormat.Builder()
.setSampleRate(SAMPLE_RATE)
.setEncoding(AUDIO_FORMAT)
.setChannelMask(CHANNEL_CONFIG)
.build()
// 使用Builder模式创建AudioRecord实例
audioRecord = AudioRecord.Builder()
.setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION)
.setAudioFormat(audioFormat)
.setBufferSizeInBytes(BUFFER_SIZE)
.build()
// 检查AudioRecord初始化状态
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) {
FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}")
// 尝试使用DEFAULT音频源重试一次
audioRecord?.release()
audioRecord = AudioRecord.Builder()
.setAudioSource(MediaRecorder.AudioSource.DEFAULT)
.setAudioFormat(audioFormat)
.setBufferSizeInBytes(BUFFER_SIZE)
.build()
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) {
FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃")
releaseAudioResources()
return
} else {
FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord")
}
}
// 启用音频效果(回音消除、噪声抑制等)
enableAudioEffects()
// 启动录音
audioRecord?.startRecording()
isRecording.set(true)
// 创建录音线程
recordingThread = Thread({
val buffer = ByteArray(BUFFER_SIZE)
while (isRecording.get()) {
try {
val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0
if (readSize > 0) {
try {
// 将处理后的音频数据推送到流
if (readSize == buffer.size) {
// 如果读取的大小等于buffer的大小,直接写入整个buffer
pushStream.write(buffer)
} else {
// 如果只读取了部分数据,创建新的数组只包含有效数据
val validData = buffer.copyOfRange(0, readSize)
pushStream.write(validData)
}
} catch (e: Exception) {
FileLogger.e(TAG, "写入音频数据失败: ${e.message}")
break
}
} else if (readSize == 0) {
// 读取为0,可能是临时的,等待一下继续尝试
Thread.sleep(10)
} else {
// 负值表示错误
FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize")
break
}
} catch (e: Exception) {
FileLogger.e(TAG, "录音线程异常: ${e.message}")
break
}
}
}, "AudioRecordingThread")
// 设置线程优先级并启动
recordingThread?.priority = Thread.MAX_PRIORITY
recordingThread?.start()
FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else ""))
} catch (e: Exception) {
FileLogger.e(TAG, "启动音频录制失败: ${e.message}")
releaseAudioResources()
}
}
// 启用音频效果(回音消除、噪声抑制等)
private fun enableAudioEffects() {
try {
val audioSessionId = audioRecord?.audioSessionId ?: -1
if (audioSessionId != -1) {
// 启用回音消除
if (AcousticEchoCanceler.isAvailable()) {
try {
echoCanceler = AcousticEchoCanceler.create(audioSessionId)
if (echoCanceler != null) {
echoCanceler?.enabled = true
FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId")
} else {
FileLogger.w(TAG, "回音消除器创建返回null")
}
} catch (e: Exception) {
FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}")
}
} else {
FileLogger.d(TAG, "设备不支持回音消除")
}
// 以下功能暂时不启用,可根据需要取消注释
/*
// 启用噪声抑制
if (NoiseSuppressor.isAvailable()) {
try {
val ns = NoiseSuppressor.create(audioSessionId)
ns?.enabled = true
FileLogger.d(TAG, "噪声抑制已启用")
} catch (e: Exception) {
FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}")
}
}
// 启用自动增益控制
if (AutomaticGainControl.isAvailable()) {
try {
val agc = AutomaticGainControl.create(audioSessionId)
agc?.enabled = true
FileLogger.d(TAG, "自动增益控制已启用")
} catch (e: Exception) {
FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}")
}
}
*/
} else {
FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果")
}
} catch (e: Exception) {
FileLogger.e(TAG, "启用音频效果时出错: ${e.message}")
}
}
// 停止录音
fun stopRecording() {
if (!isRecording.get()) {
return
}
isRecording.set(false)
try {
// 等待录音线程结束
recordingThread?.join(1000)
// 释放资源
releaseAudioResources()
FileLogger.d(TAG, "音频录制已停止")
} catch (e: Exception) {
FileLogger.e(TAG, "停止音频录制失败: ${e.message}")
}
}
// 释放音频资源
private fun releaseAudioResources() {
try {
// 停止录音
try {
if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) {
audioRecord?.stop()
}
} catch (e: Exception) {
// 忽略可能的IllegalStateException
FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}")
}
// 释放回音消除器
try {
if (echoCanceler != null) {
echoCanceler?.enabled = false
echoCanceler?.release()
echoCanceler = null
}
} catch (e: Exception) {
FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}")
} finally {
echoCanceler = null
}
// 释放音频记录器
try {
audioRecord?.release()
} catch (e: Exception) {
FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}")
} finally {
audioRecord = null
}
// 重置线程
recordingThread = null
} catch (e: Exception) {
FileLogger.e(TAG, "释放音频资源失败: ${e.message}")
}
}
}
// 一次性识别回调接口
interface RecognizeCallback {
fun onResult(result: String, detectedLanguage: String = "")
fun onError(error: String)
}
// 连续识别回调接口
interface ContinuousRecognizeCallback {
fun onResult(result: String, detectedLanguage: String = "")
fun onRecognizing(recognizing: String, detectedLanguage: String = "")
fun onSessionStarted()
fun onSessionStopped()
fun onCanceled(reason: String, errorDetails: String)
fun onError(error: String)
}
}

344
android/app/src/main/kotlin/com/example/deep_voice/VolcanoAIService.kt

@ -1,344 +0,0 @@
package com.yunqiinnovation.deepsound
import android.util.Log
import okhttp3.*
import okhttp3.MediaType.Companion.toMediaTypeOrNull
import okhttp3.RequestBody.Companion.toRequestBody
import org.json.JSONArray
import org.json.JSONObject
import java.io.IOException
import java.util.concurrent.CountDownLatch
import java.util.concurrent.TimeUnit
/**
* 火山AI服务的原生实现
*
* 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式
*/
class VolcanoAIService() {
private val TAG = "VolcanoAIService"
private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3"
private val chatEndpoint = "/chat/completions"
private val client = OkHttpClient.Builder()
.connectTimeout(30, TimeUnit.SECONDS)
.readTimeout(30, TimeUnit.SECONDS)
.writeTimeout(30, TimeUnit.SECONDS)
.build()
private var apiKey: String = ""
private var isInitialized = false
/**
* 初始化火山AI服务
*
* @param apiKey 火山AI API密钥
* @return 初始化是否成功
*/
fun initialize(apiKey: String): Boolean {
this.apiKey = apiKey
isInitialized = apiKey.isNotEmpty()
if (!isInitialized) {
Log.e(TAG, "初始化失败:API key 不能为空")
} else {
Log.d(TAG, "火山AI服务初始化成功")
}
return isInitialized
}
/**
* 生成个性化问候语
*
* @param agentName 代理名称
* @param systemPrompt 系统提示词
* @param callback 回调函数,返回生成的问候语
*/
fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) {
val messages = JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
put(JSONObject().apply {
put("role", "user")
put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。")
})
}
sendMessageStream(messages, systemPrompt, object : StreamCallback {
val stringBuilder = StringBuilder()
override fun onToken(token: String) {
stringBuilder.append(token)
}
override fun onComplete() {
callback(stringBuilder.toString(), null)
}
override fun onError(e: Exception) {
callback(null, e)
}
})
}
/**
* 发送消息(非流式输出)
*
* @param messages 消息列表
* @param systemPrompt 系统提示词
* @return 返回AI的回复
* @throws VolcanoAIException 如果API调用失败
*/
@Throws(VolcanoAIException::class)
fun sendMessage(messages: JSONArray, systemPrompt: String): String {
// 检查是否已初始化
if (!isInitialized || apiKey.isEmpty()) {
throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")
}
val fullMessages = JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
for (i in 0 until messages.length()) {
put(messages.getJSONObject(i))
}
}
val requestBody = JSONObject().apply {
put("model", "doubao-1-5-lite-32k-250115")
put("messages", fullMessages)
put("temperature", 0.7)
put("max_tokens", 2000)
put("stream", false)
}
val mediaType = "application/json".toMediaTypeOrNull()
val request = Request.Builder()
.url("$baseUrl$chatEndpoint")
.addHeader("Content-Type", "application/json")
.addHeader("Authorization", "Bearer $apiKey")
.post(requestBody.toString().toRequestBody(mediaType))
.build()
try {
client.newCall(request).execute().use { response ->
if (!response.isSuccessful) {
val errorBody = response.body?.string() ?: ""
val errorMessage = try {
JSONObject(errorBody).getJSONObject("error").getString("message")
} catch (e: Exception) {
"Unknown error occurred"
}
throw VolcanoAIException(errorMessage)
}
val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response")
val jsonResponse = JSONObject(responseBody)
if (jsonResponse.has("choices") &&
jsonResponse.getJSONArray("choices").length() > 0 &&
jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) {
return jsonResponse.getJSONArray("choices")
.getJSONObject(0)
.getJSONObject("message")
.getString("content")
}
throw VolcanoAIException("Invalid response format")
}
} catch (e: Exception) {
if (e is VolcanoAIException) throw e
throw VolcanoAIException("Failed to communicate with AI service: ${e.message}")
}
}
/**
* 发送消息(流式输出)
*
* @param messages 消息列表
* @param systemPrompt 系统提示词
* @param callback 回调函数,用于接收流式输出的结果
*/
fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) {
// 检查是否已初始化
if (!isInitialized || apiKey.isEmpty()) {
callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法"))
return
}
val fullMessages = JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
for (i in 0 until messages.length()) {
put(messages.getJSONObject(i))
}
}
val requestBody = JSONObject().apply {
put("model", "doubao-1-5-lite-32k-250115")
put("messages", fullMessages)
put("temperature", 0.7)
put("max_tokens", 2000)
put("stream", true)
}
val mediaType = "application/json".toMediaTypeOrNull()
val request = Request.Builder()
.url("$baseUrl$chatEndpoint")
.addHeader("Content-Type", "application/json")
.addHeader("Authorization", "Bearer $apiKey")
.addHeader("Accept", "text/event-stream")
.post(requestBody.toString().toRequestBody(mediaType))
.build()
client.newCall(request).enqueue(object : Callback {
override fun onFailure(call: Call, e: IOException) {
callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}"))
}
override fun onResponse(call: Call, response: Response) {
if (!response.isSuccessful) {
val errorBody = response.body?.string() ?: ""
val errorMessage = try {
JSONObject(errorBody).getJSONObject("error").getString("message")
} catch (e: Exception) {
"Unknown error occurred"
}
callback.onError(VolcanoAIException(errorMessage))
return
}
val responseBody = response.body ?: return
val source = responseBody.source()
val bufferedSource = source.buffer
try {
while (!bufferedSource.exhausted()) {
val line = bufferedSource.readUtf8Line() ?: continue
if (line.isEmpty()) continue
if (line.startsWith("data: ")) {
val data = line.substring(6)
if (data == "[DONE]") {
callback.onComplete()
break
}
try {
val jsonData = JSONObject(data)
if (jsonData.has("choices") &&
jsonData.getJSONArray("choices").length() > 0 &&
jsonData.getJSONArray("choices").getJSONObject(0).has("delta") &&
jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) {
val content = jsonData.getJSONArray("choices")
.getJSONObject(0)
.getJSONObject("delta")
.getString("content")
callback.onToken(content)
}
} catch (e: Exception) {
// 忽略无效的JSON数据
continue
}
}
}
} catch (e: Exception) {
callback.onError(VolcanoAIException("Error processing stream: ${e.message}"))
} finally {
response.close()
}
}
})
}
/**
* 同步方式发送消息(流式输出)
*
* 注意:此方法会阻塞当前线程,请在后台线程中调用
*
* @param messages 消息列表
* @param systemPrompt 系统提示词
* @return 返回完整的AI回复
* @throws VolcanoAIException 如果API调用失败
*/
@Throws(VolcanoAIException::class)
fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String {
val result = StringBuilder()
val latch = CountDownLatch(1)
var exception: Exception? = null
sendMessageStream(messages, systemPrompt, object : StreamCallback {
override fun onToken(token: String) {
result.append(token)
}
override fun onComplete() {
latch.countDown()
}
override fun onError(e: Exception) {
exception = e
latch.countDown()
}
})
// 等待流式输出完成或出错
latch.await(60, TimeUnit.SECONDS)
if (exception != null) {
throw exception as VolcanoAIException
}
return result.toString()
}
/**
* 创建用户消息
*/
fun createUserMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "user")
put("content", content)
}
}
/**
* 创建系统消息
*/
fun createSystemMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "system")
put("content", content)
}
}
/**
* 创建助手消息
*/
fun createAssistantMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "assistant")
put("content", content)
}
}
/**
* 流式输出回调接口
*/
interface StreamCallback {
fun onToken(token: String)
fun onComplete()
fun onError(e: Exception)
}
}
/**
* 火山AI异常
*/
class VolcanoAIException(message: String) : Exception(message)

0
android/app/src/main/kotlin/com/example/deep_voice/ClassicBluetoothHelper.kt → android/app/src/main/kotlin/com/yunqiinnovation/deepsound/ClassicBluetoothHelper.kt

255
android/app/src/main/kotlin/com/example/deep_voice/MainActivity.kt → android/app/src/main/kotlin/com/yunqiinnovation/deepsound/MainActivity.kt

@ -23,18 +23,12 @@ import com.yunqiinnovation.deepsound.core.utils.FileLogger
class MainActivity: FlutterActivity() { class MainActivity: FlutterActivity() {
private val AZURE_ASR_CHANNEL = "com.deep_voice.azure_asr"
private val AZURE_ASR_EVENT_CHANNEL = "com.deep_voice.azure_asr_events"
private val AZURE_TTS_CHANNEL = "com.deep_voice.azure_tts"
private val VOICE_INTERACTION_CHANNEL = "com.deep_voice.voice_interaction" private val VOICE_INTERACTION_CHANNEL = "com.deep_voice.voice_interaction"
private val VOICE_INTERACTION_EVENT_CHANNEL = "com.deep_voice.voice_interaction_events" private val VOICE_INTERACTION_EVENT_CHANNEL = "com.deep_voice.voice_interaction_events"
private val CLASSIC_BLUETOOTH_CHANNEL = "com.deep_voice.classic_bluetooth" private val CLASSIC_BLUETOOTH_CHANNEL = "com.deep_voice.classic_bluetooth"
private val CLASSIC_BLUETOOTH_EVENT_CHANNEL = "com.deep_voice.classic_bluetooth_events" private val CLASSIC_BLUETOOTH_EVENT_CHANNEL = "com.deep_voice.classic_bluetooth_events"
private val TAG = "MainActivity" private val TAG = "MainActivity"
private lateinit var azureAsrHelper: AzureAsrHelper
private lateinit var azureTtsHelper: AzureTtsHelper
private lateinit var classicBluetoothHelper: ClassicBluetoothHelper private lateinit var classicBluetoothHelper: ClassicBluetoothHelper
private var azureAsrEventSink: EventChannel.EventSink? = null
private var voiceInteractionEventSink: EventChannel.EventSink? = null private var voiceInteractionEventSink: EventChannel.EventSink? = null
private var bluetoothEventSink: EventChannel.EventSink? = null private var bluetoothEventSink: EventChannel.EventSink? = null
@ -58,10 +52,6 @@ class MainActivity: FlutterActivity() {
val assistantMessage = intent.getStringExtra("assistantMessage") ?: "" val assistantMessage = intent.getStringExtra("assistantMessage") ?: ""
val timestamp = intent.getLongExtra("timestamp", System.currentTimeMillis()) val timestamp = intent.getLongExtra("timestamp", System.currentTimeMillis())
Log.d(TAG, "收到聊天记录更新广播: agentId=$agentId, timestamp=$timestamp")
Log.d(TAG, "用户消息: ${userMessage.take(50)}...")
Log.d(TAG, "助手回复: ${assistantMessage.take(50)}...")
sendChatHistoryEvent(agentId, userMessage, assistantMessage, timestamp) sendChatHistoryEvent(agentId, userMessage, assistantMessage, timestamp)
} }
} }
@ -143,13 +133,17 @@ class MainActivity: FlutterActivity() {
var azureSpeechKey: String = "" var azureSpeechKey: String = ""
var azureSpeechRegion: String = "" var azureSpeechRegion: String = ""
var volcanoAiApiKey: String = "" var volcanoAiApiKey: String = ""
var openaiApiKey: String = ""
var openaiBaseUrl: String? = null
// 安全存储相关常量 // 安全存储相关常量
private const val SECURE_PREFS_FILENAME = "deep_voice_secure_prefs" private const val SECURE_PREFS_FILENAME = "deep_voice_secure_prefs"
private const val KEY_AZURE_SPEECH_KEY = "azure_speech_key" private const val KEY_AZURE_SPEECH_KEY = "azure_speech_key"
private const val KEY_AZURE_SPEECH_REGION = "azure_speech_region" private const val KEY_AZURE_SPEECH_REGION = "azure_speech_region"
private const val KEY_VOLCANO_AI_API_KEY = "volcano_ai_api_key" private const val KEY_VOLCANO_AI_API_KEY = "volcano_ai_api_key"
private const val KEY_OPENAI_API_KEY = "openai_api_key"
private const val KEY_OPENAI_BASE_URL = "openai_base_url"
private const val KEY_MCP_SERVER_ENDPOINT = "mcp_server_endpoint"
// 会话管理 // 会话管理
private const val KEY_SESSION_ID = "session_id" private const val KEY_SESSION_ID = "session_id"
private var currentSessionId = "" private var currentSessionId = ""
@ -183,6 +177,8 @@ class MainActivity: FlutterActivity() {
.putString(KEY_AZURE_SPEECH_KEY, azureSpeechKey) .putString(KEY_AZURE_SPEECH_KEY, azureSpeechKey)
.putString(KEY_AZURE_SPEECH_REGION, azureSpeechRegion) .putString(KEY_AZURE_SPEECH_REGION, azureSpeechRegion)
.putString(KEY_VOLCANO_AI_API_KEY, volcanoAiApiKey) .putString(KEY_VOLCANO_AI_API_KEY, volcanoAiApiKey)
.putString(KEY_OPENAI_API_KEY, openaiApiKey)
.putString(KEY_OPENAI_BASE_URL, openaiBaseUrl)
.putString(KEY_SESSION_ID, currentSessionId) .putString(KEY_SESSION_ID, currentSessionId)
.apply() .apply()
@ -227,11 +223,13 @@ class MainActivity: FlutterActivity() {
azureSpeechKey = sharedPreferences.getString(KEY_AZURE_SPEECH_KEY, "") ?: "" azureSpeechKey = sharedPreferences.getString(KEY_AZURE_SPEECH_KEY, "") ?: ""
azureSpeechRegion = sharedPreferences.getString(KEY_AZURE_SPEECH_REGION, "") ?: "" azureSpeechRegion = sharedPreferences.getString(KEY_AZURE_SPEECH_REGION, "") ?: ""
volcanoAiApiKey = sharedPreferences.getString(KEY_VOLCANO_AI_API_KEY, "") ?: "" volcanoAiApiKey = sharedPreferences.getString(KEY_VOLCANO_AI_API_KEY, "") ?: ""
openaiApiKey = sharedPreferences.getString(KEY_OPENAI_API_KEY, "") ?: ""
openaiBaseUrl = sharedPreferences.getString(KEY_OPENAI_BASE_URL, null)
FileLogger.d("MainActivity", "已从加密存储加载密钥") FileLogger.d("MainActivity", "已从加密存储加载密钥")
// 检查是否成功获取所有密钥 // 检查是否成功获取所有必要密钥
return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() && volcanoAiApiKey.isNotEmpty() return azureSpeechKey.isNotEmpty() && azureSpeechRegion.isNotEmpty() &&
(openaiApiKey.isNotEmpty() || volcanoAiApiKey.isNotEmpty())
} catch (e: Exception) { } catch (e: Exception) {
FileLogger.e("MainActivity", "从加密存储加载密钥失败: ${e.message}") FileLogger.e("MainActivity", "从加密存储加载密钥失败: ${e.message}")
e.printStackTrace() e.printStackTrace()
@ -247,8 +245,6 @@ class MainActivity: FlutterActivity() {
FileLogger.init(applicationContext) FileLogger.init(applicationContext)
// 初始化 Azure 语音服务 // 初始化 Azure 语音服务
azureAsrHelper = AzureAsrHelper(applicationContext)
azureTtsHelper = AzureTtsHelper(applicationContext)
classicBluetoothHelper = ClassicBluetoothHelper(applicationContext) classicBluetoothHelper = ClassicBluetoothHelper(applicationContext)
// 注册广播接收器 // 注册广播接收器
@ -300,8 +296,7 @@ class MainActivity: FlutterActivity() {
setupMethodChannels(flutterEngine) setupMethodChannels(flutterEngine)
// 初始化 Azure 语音服务 // 初始化 Azure 语音服务
azureAsrHelper = AzureAsrHelper(this) classicBluetoothHelper = ClassicBluetoothHelper(this)
azureTtsHelper = AzureTtsHelper(this)
Log.d(TAG, "Flutter 引擎配置完成") Log.d(TAG, "Flutter 引擎配置完成")
} }
@ -327,20 +322,6 @@ class MainActivity: FlutterActivity() {
} }
) )
// Azure ASR 事件通道
EventChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_EVENT_CHANNEL).setStreamHandler(
object : EventChannel.StreamHandler {
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) {
Log.d(TAG, "ASR事件通道开始监听")
azureAsrEventSink = events
}
override fun onCancel(arguments: Any?) {
azureAsrEventSink = null
}
}
)
// 设置蓝牙事件通道 // 设置蓝牙事件通道
EventChannel(flutterEngine.dartExecutor.binaryMessenger, CLASSIC_BLUETOOTH_EVENT_CHANNEL).setStreamHandler( EventChannel(flutterEngine.dartExecutor.binaryMessenger, CLASSIC_BLUETOOTH_EVENT_CHANNEL).setStreamHandler(
object : EventChannel.StreamHandler { object : EventChannel.StreamHandler {
@ -367,195 +348,6 @@ class MainActivity: FlutterActivity() {
private fun setupMethodChannels(flutterEngine: FlutterEngine) { private fun setupMethodChannels(flutterEngine: FlutterEngine) {
Log.d(TAG, "开始设置方法通道") Log.d(TAG, "开始设置方法通道")
// 设置 Azure ASR 方法通道
MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_ASR_CHANNEL).setMethodCallHandler { call, result ->
when (call.method) {
"initialize" -> {
val subscriptionKey = call.argument<String>("subscriptionKey")
val region = call.argument<String>("region")
val supportedLanguages = call.argument<List<String>>("supportedLanguages")?.toTypedArray() ?: arrayOf("zh-CN", "en-US")
if (subscriptionKey == null || region == null) {
result.error("INVALID_ARGUMENTS", "订阅密钥和区域不能为空", null)
return@setMethodCallHandler
}
try {
val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages)
result.success(success)
} catch (e: Exception) {
result.error("INITIALIZATION_ERROR", e.message, null)
}
}
"recognizeOnce" -> {
azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {
result.success(mapOf(
"text" to text,
"detectedLanguage" to detectedLanguage
))
}
override fun onError(error: String) {
result.error("RECOGNITION_ERROR", error, null)
}
})
}
"startContinuousRecognition" -> {
// 确保事件通道已准备好
if (azureAsrEventSink == null) {
result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null)
return@setMethodCallHandler
}
val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {
sendAsrEvent(mapOf(
"type" to "result",
"text" to text,
"detectedLanguage" to detectedLanguage
))
}
override fun onRecognizing(recognizing: String, detectedLanguage: String) {
sendAsrEvent(mapOf(
"type" to "recognizing",
"text" to recognizing,
"detectedLanguage" to detectedLanguage
))
}
override fun onSessionStarted() {
sendAsrEvent(mapOf("type" to "sessionStarted"))
}
override fun onSessionStopped() {
sendAsrEvent(mapOf("type" to "sessionStopped"))
}
override fun onCanceled(reason: String, errorDetails: String) {
sendAsrEvent(mapOf(
"type" to "canceled",
"reason" to reason,
"errorDetails" to errorDetails
))
}
override fun onError(error: String) {
sendAsrEvent(mapOf("type" to "error", "message" to error))
}
})
result.success(success)
}
"stopContinuousRecognition" -> {
try {
if (!azureAsrHelper.isContinuousRecognitionActive()) {
result.success(true)
return@setMethodCallHandler
}
val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {}
override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
override fun onSessionStarted() {}
override fun onSessionStopped() {}
override fun onCanceled(reason: String, errorDetails: String) {}
override fun onError(error: String) {
result.error("STOP_ERROR", error, null)
}
})
result.success(success)
} catch (e: Exception) {
result.error("STOP_ERROR", e.message, null)
}
}
"isContinuousRecognitionActive" -> {
result.success(azureAsrHelper.isContinuousRecognitionActive())
}
"dispose" -> {
azureAsrHelper.dispose()
result.success(true)
}
else -> {
result.notImplemented()
}
}
}
// 设置 Azure TTS 方法通道
MethodChannel(flutterEngine.dartExecutor.binaryMessenger, AZURE_TTS_CHANNEL).setMethodCallHandler { call, result ->
when (call.method) {
"initialize" -> {
val subscriptionKey = call.argument<String>("subscriptionKey") ?: ""
val region = call.argument<String>("region") ?: ""
val language = call.argument<String>("language") ?: "zh-CN"
val success = azureTtsHelper.initialize(subscriptionKey, region, language)
result.success(success)
}
"setVoice" -> {
val voiceName = call.argument<String>("voiceName") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "语音名称不能为空", null)
result.success(azureTtsHelper.setVoice(voiceName))
}
"setSpeechParams" -> {
val rate = call.argument<Int>("rate") ?: 0
val pitch = call.argument<Int>("pitch") ?: 0
val volume = call.argument<Int>("volume") ?: 100
result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume))
}
"setAudioOutputType" -> {
val outputTypeStr = call.argument<String>("outputType") ?: "speaker"
val outputType = when (outputTypeStr.lowercase()) {
"speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER
"earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE
"auto" -> AzureTtsHelper.AudioOutputType.AUTO
else -> AzureTtsHelper.AudioOutputType.SPEAKER
}
result.success(azureTtsHelper.setAudioOutputType(outputType))
}
"speakText" -> {
val text = call.argument<String>("text") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "文本不能为空", null)
azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback {
override fun onSuccess(message: String) {
runOnUiThread { result.success(message) }
}
override fun onError(error: String) {
runOnUiThread { result.error("SPEAK_ERROR", error, null) }
}
})
}
"speakSsml" -> {
val ssml = call.argument<String>("ssml") ?: return@setMethodCallHandler result.error("INVALID_ARGUMENTS", "SSML不能为空", null)
azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback {
override fun onSuccess(message: String) {
runOnUiThread { result.success(message) }
}
override fun onError(error: String) {
runOnUiThread { result.error("SPEAK_ERROR", error, null) }
}
})
}
"stopSpeaking" -> {
result.success(azureTtsHelper.stopSpeaking())
}
"isSpeaking" -> {
result.success(azureTtsHelper.isSpeaking())
}
"dispose" -> {
azureTtsHelper.dispose()
result.success(true)
}
else -> {
result.notImplemented()
}
}
}
// 设置语音交互方法通道 // 设置语音交互方法通道
MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOICE_INTERACTION_CHANNEL).setMethodCallHandler { call, result -> MethodChannel(flutterEngine.dartExecutor.binaryMessenger, VOICE_INTERACTION_CHANNEL).setMethodCallHandler { call, result ->
when (call.method) { when (call.method) {
@ -657,15 +449,6 @@ class MainActivity: FlutterActivity() {
} }
} }
// ASR 事件发送方法
private fun sendAsrEvent(event: Map<String, Any>) {
if (azureAsrEventSink == null) return
runOnUiThread {
azureAsrEventSink?.success(event)
}
}
/** /**
* 启动语音交互服务(带配置参数) * 启动语音交互服务(带配置参数)
*/ */
@ -675,12 +458,16 @@ class MainActivity: FlutterActivity() {
// 获取配置参数 // 获取配置参数
val key = call.argument<String>("azure_speech_key") ?: "" val key = call.argument<String>("azure_speech_key") ?: ""
val region = call.argument<String>("azure_speech_region") ?: "" val region = call.argument<String>("azure_speech_region") ?: ""
val aiKey = call.argument<String>("volcano_ai_api_key") ?: "" val openaiKey = call.argument<String>("openai_api_key") ?: ""
val baseUrl = call.argument<String>("openai_base_url")
// 设置 Azure Speech 配置 // 设置 Azure Speech 和 AI 配置
azureSpeechKey = key azureSpeechKey = key
azureSpeechRegion = region azureSpeechRegion = region
volcanoAiApiKey = aiKey openaiApiKey = openaiKey
if (baseUrl != null) {
openaiBaseUrl = baseUrl
}
// 保存密钥到安全存储 // 保存密钥到安全存储
saveKeysToSecureStorage(applicationContext) saveKeysToSecureStorage(applicationContext)
@ -799,8 +586,6 @@ class MainActivity: FlutterActivity() {
} }
// 释放资源 // 释放资源
azureAsrHelper.dispose()
azureTtsHelper.dispose()
classicBluetoothHelper.dispose() classicBluetoothHelper.dispose()
super.onDestroy() super.onDestroy()

606
android/app/src/main/kotlin/com/yunqiinnovation/deepsound/OpenAIService.kt

@ -0,0 +1,606 @@
package com.yunqiinnovation.deepsound
import android.util.Log
import okhttp3.*
import okhttp3.MediaType.Companion.toMediaTypeOrNull
import okhttp3.RequestBody.Companion.toRequestBody
import org.json.JSONArray
import org.json.JSONObject
import java.io.IOException
import java.util.concurrent.TimeUnit
import com.yunqiinnovation.deepsound.core.utils.FileLogger
/**
* OpenAI服务的原生实现
*/
class OpenAIService() {
private val TAG = "OpenAIService"
private var baseUrl = "https://api.openai.com/v1/chat/completions"
private val client = OkHttpClient.Builder()
.connectTimeout(30, TimeUnit.SECONDS)
.readTimeout(30, TimeUnit.SECONDS)
.writeTimeout(30, TimeUnit.SECONDS)
.build()
private var apiKey: String = ""
private var isInitialized = false
private var model: String = "doubao-1-5-lite-32k-250115" // 默认模型
// 用于存储注册的函数
private val registeredFunctions = mutableListOf<JSONObject>()
/**
* 初始化OpenAI服务
*/
fun initialize(apiKey: String, baseUrl: String = ""): Boolean {
this.apiKey = apiKey
if (baseUrl.isNotEmpty()) {
this.baseUrl = baseUrl
}
isInitialized = apiKey.isNotEmpty()
return isInitialized
}
/**
* 注册函数
*/
fun registerFunction(name: String, description: String, parameters: JSONObject): Boolean {
try {
val function = JSONObject().apply {
put("name", name)
put("description", description)
put("parameters", parameters)
}
// 检查是否已存在相同名称的函数
val existingIndex = registeredFunctions.indexOfFirst {
it.getString("name") == name
}
if (existingIndex >= 0) {
// 如果已存在,则替换
registeredFunctions[existingIndex] = function
} else {
// 如果不存在,则添加
registeredFunctions.add(function)
}
return true
} catch (e: Exception) {
return false
}
}
/**
* 发送消息(非流式输出)
*/
@Throws(OpenAIException::class)
fun sendMessage(messages: JSONArray, systemPrompt: String): String {
if (!isInitialized || apiKey.isEmpty()) {
throw OpenAIException("OpenAI服务未初始化")
}
val fullMessages = JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
for (i in 0 until messages.length()) {
put(messages.getJSONObject(i))
}
}
val requestBody = JSONObject().apply {
put("model", model)
put("messages", fullMessages)
put("temperature", 0.7)
put("max_tokens", 2000)
put("stream", false)
// 如果有注册的函数,则添加到请求中
if (registeredFunctions.isNotEmpty()) {
val tools = JSONArray()
for (function in registeredFunctions) {
val tool = JSONObject().apply {
put("type", "function")
put("function", function)
}
tools.put(tool)
}
put("tools", tools)
}
}
val mediaType = "application/json".toMediaTypeOrNull()
val request = Request.Builder()
.url(baseUrl)
.addHeader("Content-Type", "application/json")
.addHeader("Authorization", "Bearer $apiKey")
.post(requestBody.toString().toRequestBody(mediaType))
.build()
try {
client.newCall(request).execute().use { response ->
if (!response.isSuccessful) {
throw OpenAIException("API调用失败: ${response.code}")
}
val responseBody = response.body?.string() ?: throw OpenAIException("Empty response")
val jsonResponse = JSONObject(responseBody)
// 检查是否有函数调用
if (jsonResponse.has("choices") &&
jsonResponse.getJSONArray("choices").length() > 0) {
val choice = jsonResponse.getJSONArray("choices").getJSONObject(0)
// 检查是否是函数调用
if (choice.has("message")) {
val message = choice.getJSONObject("message")
// 检查是否有工具调用
if (message.has("tool_calls")) {
val toolCalls = message.getJSONArray("tool_calls")
if (toolCalls.length() > 0) {
val toolCall = toolCalls.getJSONObject(0)
if (toolCall.has("function")) {
val function = toolCall.getJSONObject("function")
val functionCall = JSONObject().apply {
put("name", function.getString("name"))
put("arguments", function.getString("arguments"))
put("id", toolCall.getString("id"))
}
return functionCall.toString()
}
}
}
// 如果没有工具调用,返回消息内容
if (message.has("content")) {
return message.getString("content")
}
}
}
throw OpenAIException("Invalid response format")
}
} catch (e: Exception) {
if (e is OpenAIException) throw e
throw OpenAIException("Failed to communicate with AI service: ${e.message}")
}
}
/**
* 发送消息(流式输出)
*/
fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) {
if (!isInitialized || apiKey.isEmpty()) {
callback.onError(OpenAIException("OpenAI服务未初始化"))
return
}
val fullMessages = JSONArray().apply {
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
for (i in 0 until messages.length()) {
put(messages.getJSONObject(i))
}
}
val requestBody = JSONObject().apply {
put("model", model)
put("messages", fullMessages)
put("temperature", 0.7)
put("max_tokens", 2000)
put("stream", true)
// 如果有注册的函数,则添加到请求中
if (registeredFunctions.isNotEmpty()) {
val tools = JSONArray()
for (function in registeredFunctions) {
val tool = JSONObject().apply {
put("type", "function")
put("function", function)
}
tools.put(tool)
}
put("tools", tools)
}
}
val mediaType = "application/json".toMediaTypeOrNull()
val request = Request.Builder()
.url(baseUrl)
.addHeader("Content-Type", "application/json")
.addHeader("Authorization", "Bearer $apiKey")
.addHeader("Accept", "text/event-stream")
.post(requestBody.toString().toRequestBody(mediaType))
.build()
client.newCall(request).enqueue(object : Callback {
override fun onFailure(call: Call, e: IOException) {
callback.onError(OpenAIException(e.message ?: "请求失败"))
}
override fun onResponse(call: Call, response: Response) {
if (!response.isSuccessful) {
callback.onError(OpenAIException("API调用失败: ${response.code}"))
return
}
val responseBody = response.body ?: return
val source = responseBody.source()
try {
// 预取数据到缓冲区
source.request(Long.MAX_VALUE)
val bufferedSource = source.buffer
// 用于存储函数调用的各个部分
val finalToolCalls = mutableMapOf<Int, ToolCallInfo>()
while (!bufferedSource.exhausted()) {
val line = bufferedSource.readUtf8Line() ?: continue
if (line.isEmpty()) continue
if (line.startsWith("data: ")) {
val data = line.substring(6)
if (data == "[DONE]") {
callback.onComplete()
break
}
try {
val jsonData = JSONObject(data)
if (jsonData.has("choices") &&
jsonData.getJSONArray("choices").length() > 0) {
val choice = jsonData.getJSONArray("choices").getJSONObject(0)
// 检查是否有delta
if (choice.has("delta")) {
val delta = choice.getJSONObject("delta")
// 检查是否有工具调用
if (delta.has("tool_calls")) {
val toolCalls = delta.getJSONArray("tool_calls")
for (i in 0 until toolCalls.length()) {
val toolCall = toolCalls.getJSONObject(i)
val index = toolCall.optInt("index", i)
// 如果是新的工具调用,初始化
if (!finalToolCalls.containsKey(index)) {
finalToolCalls[index] = ToolCallInfo()
}
// 获取ID
if (toolCall.has("id")) {
finalToolCalls[index]?.id = toolCall.getString("id")
}
// 处理函数信息
if (toolCall.has("function")) {
val function = toolCall.getJSONObject("function")
if (function.has("name")) {
finalToolCalls[index]?.name = function.getString("name")
}
if (function.has("arguments")) {
finalToolCalls[index]?.arguments += function.getString("arguments")
}
}
}
continue
}
// 如果有内容,发送给回调
if (delta.has("content") && !delta.isNull("content")) {
val content = delta.getString("content")
callback.onToken(content)
}
}
}
} catch (e: Exception) {
// 忽略无效的JSON
continue
}
}
}
// 处理完整的函数调用
for ((_, toolCallInfo) in finalToolCalls) {
if (toolCallInfo.name.isNotEmpty()) {
try {
// 创建函数调用对象
val functionCall = JSONObject().apply {
put("id", toolCallInfo.id)
put("name", toolCallInfo.name)
put("arguments", toolCallInfo.arguments.trim())
}
callback.onFunctionCall(functionCall)
} catch (e: Exception) {
// 出错时使用空参数
val functionCall = JSONObject().apply {
put("id", toolCallInfo.id)
put("name", toolCallInfo.name)
put("arguments", "{}")
}
callback.onFunctionCall(functionCall)
}
}
}
} catch (e: Exception) {
callback.onError(OpenAIException("处理流式响应出错: ${e.message}"))
} finally {
response.close()
}
}
})
}
/**
* 发送函数调用结果
*/
fun sendFunctionCallResult(
messages: JSONArray,
systemPrompt: String,
functionCall: JSONObject,
functionResult: String,
callback: StreamCallback
) {
if (!isInitialized || apiKey.isEmpty()) {
callback.onError(OpenAIException("OpenAI服务未初始化"))
return
}
// 构建完整的消息历史
val fullMessages = JSONArray().apply {
// 添加系统提示
put(JSONObject().apply {
put("role", "system")
put("content", systemPrompt)
})
// 添加历史消息
for (i in 0 until messages.length()) {
put(messages.getJSONObject(i))
}
// 添加函数调用信息
put(JSONObject().apply {
put("role", "assistant")
put("content", null)
put("tool_calls", JSONArray().apply {
put(JSONObject().apply {
put("id", functionCall.optString("id", "call_${System.currentTimeMillis()}"))
put("type", "function")
put("function", JSONObject().apply {
put("name", functionCall.getString("name"))
put("arguments", functionCall.getString("arguments"))
})
})
})
})
// 添加函数返回结果
put(JSONObject().apply {
put("role", "tool")
put("tool_call_id", functionCall.optString("id", "call_${System.currentTimeMillis()}"))
put("content", functionResult)
})
}
// 构建请求
val requestBody = JSONObject().apply {
put("model", model)
put("messages", fullMessages)
put("temperature", 0.7)
put("max_tokens", 2000)
put("stream", true)
// 如果有注册的函数,则添加到请求中
if (registeredFunctions.isNotEmpty()) {
val tools = JSONArray()
for (function in registeredFunctions) {
val tool = JSONObject().apply {
put("type", "function")
put("function", function)
}
tools.put(tool)
}
put("tools", tools)
}
}
val mediaType = "application/json".toMediaTypeOrNull()
val request = Request.Builder()
.url(baseUrl)
.addHeader("Content-Type", "application/json")
.addHeader("Authorization", "Bearer $apiKey")
.addHeader("Accept", "text/event-stream")
.post(requestBody.toString().toRequestBody(mediaType))
.build()
// 发送请求
client.newCall(request).enqueue(object : Callback {
override fun onFailure(call: Call, e: IOException) {
callback.onError(OpenAIException(e.message ?: "请求失败"))
}
override fun onResponse(call: Call, response: Response) {
if (!response.isSuccessful) {
callback.onError(OpenAIException("API调用失败: ${response.code}"))
return
}
val responseBody = response.body ?: return
val source = responseBody.source()
try {
// 预取数据到缓冲区
source.request(Long.MAX_VALUE)
val bufferedSource = source.buffer
// 用于存储函数调用的各个部分
val finalToolCalls = mutableMapOf<Int, ToolCallInfo>()
while (!bufferedSource.exhausted()) {
val line = bufferedSource.readUtf8Line() ?: continue
if (line.isEmpty()) continue
if (line.startsWith("data: ")) {
val data = line.substring(6)
if (data == "[DONE]") {
callback.onComplete()
break
}
try {
val jsonData = JSONObject(data)
if (jsonData.has("choices") &&
jsonData.getJSONArray("choices").length() > 0) {
val choice = jsonData.getJSONArray("choices").getJSONObject(0)
// 检查是否有delta
if (choice.has("delta")) {
val delta = choice.getJSONObject("delta")
// 检查是否有工具调用
if (delta.has("tool_calls")) {
val toolCalls = delta.getJSONArray("tool_calls")
for (i in 0 until toolCalls.length()) {
val toolCall = toolCalls.getJSONObject(i)
val index = toolCall.optInt("index", i)
// 如果是新的工具调用,初始化
if (!finalToolCalls.containsKey(index)) {
finalToolCalls[index] = ToolCallInfo()
}
// 获取ID
if (toolCall.has("id")) {
finalToolCalls[index]?.id = toolCall.getString("id")
}
// 处理函数信息
if (toolCall.has("function")) {
val function = toolCall.getJSONObject("function")
if (function.has("name")) {
finalToolCalls[index]?.name = function.getString("name")
}
if (function.has("arguments")) {
finalToolCalls[index]?.arguments += function.getString("arguments")
}
}
}
continue
}
// 如果有内容,发送给回调
if (delta.has("content") && !delta.isNull("content")) {
val content = delta.getString("content")
callback.onToken(content)
}
}
}
} catch (e: Exception) {
// 忽略无效的JSON
continue
}
}
}
// 处理完整的函数调用
for ((_, toolCallInfo) in finalToolCalls) {
if (toolCallInfo.name.isNotEmpty()) {
try {
// 创建函数调用对象
val functionCall = JSONObject().apply {
put("id", toolCallInfo.id)
put("name", toolCallInfo.name)
put("arguments", toolCallInfo.arguments.trim())
}
callback.onFunctionCall(functionCall)
} catch (e: Exception) {
// 出错时使用空参数
val functionCall = JSONObject().apply {
put("id", toolCallInfo.id)
put("name", toolCallInfo.name)
put("arguments", "{}")
}
callback.onFunctionCall(functionCall)
}
}
}
} catch (e: Exception) {
callback.onError(OpenAIException("处理流式响应出错: ${e.message}"))
} finally {
response.close()
}
}
})
}
/**
* 创建用户消息
*/
fun createUserMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "user")
put("content", content)
}
}
/**
* 创建系统消息
*/
fun createSystemMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "system")
put("content", content)
}
}
/**
* 创建助手消息
*/
fun createAssistantMessage(content: String): JSONObject {
return JSONObject().apply {
put("role", "assistant")
put("content", content)
}
}
/**
* 流式输出回调接口
*/
interface StreamCallback {
fun onToken(token: String)
fun onComplete()
fun onError(e: Exception)
fun onFunctionCall(functionCall: JSONObject) {}
}
/**
* 用于存储工具调用信息的辅助类
*/
private class ToolCallInfo {
var id: String = ""
var name: String = ""
var arguments: String = ""
}
}
/**
* OpenAI异常
*/
class OpenAIException(message: String) : Exception(message)

180
android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceFunctionHandler.kt

@ -0,0 +1,180 @@
package com.yunqiinnovation.deepsound
import org.json.JSONArray
import org.json.JSONObject
import com.yunqiinnovation.deepsound.OpenAIService
import com.yunqiinnovation.deepsound.core.utils.FileLogger
/**
* 语音功能处理器 - 处理AI函数调用
*/
class VoiceFunctionHandler(
private val openAIService: OpenAIService,
private val systemPrompt: String
) {
companion object {
private const val TAG = "VoiceFunctionHandler"
}
/**
* 初始化并注册所有可用的函数
*/
fun initialize() {
try {
// 注册退出交互函数
registerExitInteractionFunction()
// 在这里可以注册更多函数
} catch (e: Exception) {
FileLogger.e(TAG, "初始化函数处理器失败: ${e.message}", e)
}
}
/**
* 注册退出交互函数
*/
private fun registerExitInteractionFunction() {
try {
openAIService.registerFunction(
"exit_interaction",
"退出当前语音交互",
JSONObject("""
{
"type": "object",
"properties": {},
"required": []
}
""")
)
FileLogger.d(TAG, "退出交互功能已注册")
} catch (e: Exception) {
FileLogger.e(TAG, "注册退出交互函数失败: ${e.message}", e)
}
}
/**
* 处理函数调用
*
* @param functionCall 函数调用信息
* @param messages 消息历史
* @param callback 回调,处理退出等操作
* @return 是否已处理函数调用
*/
fun handleFunctionCall(
functionCall: JSONObject,
messages: JSONArray,
callback: FunctionCallCallback
): Boolean {
val functionName = functionCall.getString("name")
FileLogger.d(TAG, "处理函数调用: $functionName")
return when (functionName) {
"exit_interaction" -> {
handleExitInteraction(functionCall, messages, callback)
true
}
else -> {
// 未知函数,返回默认结果
handleUnknownFunction(functionCall, messages, callback)
false
}
}
}
/**
* 处理退出交互函数
*/
private fun handleExitInteraction(
functionCall: JSONObject,
messages: JSONArray,
callback: FunctionCallCallback
) {
FileLogger.d(TAG, "处理退出交互函数")
val responseBuilder = StringBuilder()
openAIService.sendFunctionCallResult(
messages = messages,
systemPrompt = systemPrompt,
functionCall = functionCall,
functionResult = "{\"result\": \"已退出语音交互\"}",
callback = object : OpenAIService.StreamCallback {
override fun onToken(token: String) {
responseBuilder.append(token)
}
override fun onComplete() {
FileLogger.d(TAG, "handleExitInteraction onComplete: ${responseBuilder.toString()}")
val response = responseBuilder.toString()
if (response.isNotEmpty()) {
callback.onExitWithMessage(response)
} else {
callback.onExitWithMessage("已退出语音交互")
}
}
override fun onError(e: Exception) {
FileLogger.e(TAG, "处理退出交互函数调用出错: ${e.message}")
callback.onError("退出交互时出错")
}
override fun onFunctionCall(nestedCall: JSONObject) {
FileLogger.e(TAG, "意外收到嵌套函数调用: ${nestedCall.getString("name")}")
}
}
)
}
/**
* 处理未知函数调用
*/
private fun handleUnknownFunction(
functionCall: JSONObject,
messages: JSONArray,
callback: FunctionCallCallback
) {
FileLogger.d(TAG, "处理未知函数: ${functionCall.getString("name")}")
try {
openAIService.sendFunctionCallResult(
messages = messages,
systemPrompt = systemPrompt,
functionCall = functionCall,
functionResult = "{\"result\": \"处理函数调用中\"}",
callback = object : OpenAIService.StreamCallback {
override fun onToken(token: String) {
callback.onTokenReceived(token)
}
override fun onComplete() {
callback.onComplete()
}
override fun onError(e: Exception) {
FileLogger.e(TAG, "处理函数调用失败: ${e.message}")
callback.onError("处理函数调用失败: ${e.message}")
}
override fun onFunctionCall(nestedCall: JSONObject) {
callback.onFunctionCall(nestedCall)
}
}
)
} catch (e: Exception) {
FileLogger.e(TAG, "处理函数调用失败: ${e.message}")
callback.onError("处理函数调用失败: ${e.message}")
}
}
/**
* 函数调用回调接口
*/
interface FunctionCallCallback {
fun onTokenReceived(token: String)
fun onComplete()
fun onError(message: String)
fun onFunctionCall(functionCall: JSONObject)
fun onExitWithMessage(farewell: String)
}
}

208
android/app/src/main/kotlin/com/example/deep_voice/VoiceInteractionService.kt → android/app/src/main/kotlin/com/yunqiinnovation/deepsound/VoiceInteractionService.kt

@ -22,10 +22,15 @@ import android.os.Handler
import android.os.Looper import android.os.Looper
import java.util.concurrent.atomic.AtomicBoolean import java.util.concurrent.atomic.AtomicBoolean
import org.json.JSONArray import org.json.JSONArray
import org.json.JSONObject
import android.media.MediaPlayer import android.media.MediaPlayer
import android.media.AudioAttributes import android.media.AudioAttributes
import android.net.Uri import android.net.Uri
import com.yunqiinnovation.deepsound.core.utils.FileLogger import com.yunqiinnovation.deepsound.core.utils.FileLogger
import com.yunqiinnovation.azure_speech.AzureAsrHelper
import com.yunqiinnovation.azure_speech.AzureTtsHelper
import com.yunqiinnovation.deepsound.OpenAIService
/** /**
* 后台语音交互 Service: * 后台语音交互 Service:
@ -78,7 +83,8 @@ class VoiceInteractionService : Service() {
private lateinit var audioManager: AudioManager private lateinit var audioManager: AudioManager
private lateinit var azureAsrHelper: AzureAsrHelper private lateinit var azureAsrHelper: AzureAsrHelper
private lateinit var azureTtsHelper: AzureTtsHelper private lateinit var azureTtsHelper: AzureTtsHelper
private lateinit var volcanoAIService: VolcanoAIService private lateinit var openAIService: OpenAIService
private lateinit var functionHandler: VoiceFunctionHandler
// 定时器 // 定时器
private val handler = Handler(Looper.getMainLooper()) private val handler = Handler(Looper.getMainLooper())
@ -95,7 +101,10 @@ class VoiceInteractionService : Service() {
请保持回答简短、准确,避免过长的解释。 请保持回答简短、准确,避免过长的解释。
如果用户的问题不清楚,请礼貌地请求澄清。 如果用户的问题不清楚,请礼貌地请求澄清。
不要使用复杂的术语,除非用户明确要求。 不要使用复杂的术语,除非用户明确要求。
用户用语音和你交互. 用户用语音和你交互。
当用户说"退出"、"再见"、"结束对话"等类似意图时,你应该使用exit_interaction函数来结束对话,
并在结束前说一句友好的告别语,例如"再见,有需要随时找我"。
""".trimIndent() """.trimIndent()
// 添加媒体播放器 // 添加媒体播放器
@ -157,10 +166,11 @@ class VoiceInteractionService : Service() {
// 尝试从静态变量获取配置 // 尝试从静态变量获取配置
var subscriptionKey = MainActivity.azureSpeechKey var subscriptionKey = MainActivity.azureSpeechKey
var serviceRegion = MainActivity.azureSpeechRegion var serviceRegion = MainActivity.azureSpeechRegion
var volcanoKey = MainActivity.volcanoAiApiKey var openaiKey = MainActivity.openaiApiKey
var openaiBaseUrl = MainActivity.openaiBaseUrl ?: "" // OpenAI API基本URL
// 如果静态变量中没有配置,尝试从加密存储中加载 // 如果静态变量中没有配置,尝试从加密存储中加载
if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || volcanoKey.isEmpty()) { if (subscriptionKey.isEmpty() || serviceRegion.isEmpty() || openaiKey.isEmpty()) {
FileLogger.d(TAG, "静态变量中的配置信息不完整,尝试从加密存储加载") FileLogger.d(TAG, "静态变量中的配置信息不完整,尝试从加密存储加载")
// 从加密存储加载密钥 // 从加密存储加载密钥
@ -170,7 +180,8 @@ class VoiceInteractionService : Service() {
// 更新本地变量 // 更新本地变量
subscriptionKey = MainActivity.azureSpeechKey subscriptionKey = MainActivity.azureSpeechKey
serviceRegion = MainActivity.azureSpeechRegion serviceRegion = MainActivity.azureSpeechRegion
volcanoKey = MainActivity.volcanoAiApiKey openaiKey = MainActivity.openaiApiKey
openaiBaseUrl = MainActivity.openaiBaseUrl ?: ""
FileLogger.d(TAG, "已从加密存储加载配置信息") FileLogger.d(TAG, "已从加密存储加载配置信息")
} else { } else {
@ -191,15 +202,29 @@ class VoiceInteractionService : Service() {
FileLogger.e(TAG, "Azure配置信息不完整,无法初始化Azure服务") FileLogger.e(TAG, "Azure配置信息不完整,无法初始化Azure服务")
} }
// 初始化火山AI服务 // 初始化OpenAI服务
volcanoAIService = VolcanoAIService() openAIService = OpenAIService()
// 初始化OpenAI服务
if (openaiKey.isNotEmpty()) {
val initialized = if (openaiBaseUrl.isNotEmpty()) {
openAIService.initialize(openaiKey, openaiBaseUrl)
} else {
openAIService.initialize(openaiKey)
}
if (initialized) {
FileLogger.d(TAG, "OpenAI服务已初始化")
// 初始化函数处理器
functionHandler = VoiceFunctionHandler(openAIService, systemPrompt)
functionHandler.initialize()
// 初始化火山AI服务 } else {
if (volcanoKey.isNotEmpty()) { FileLogger.e(TAG, "OpenAI服务初始化失败")
volcanoAIService.initialize(volcanoKey) }
FileLogger.d(TAG, "火山AI服务已初始化")
} else { } else {
FileLogger.e(TAG, "火山AI配置信息不完整,无法初始化火山AI服务") FileLogger.e(TAG, "OpenAI配置信息不完整,无法初始化OpenAI服务")
} }
} }
@ -404,7 +429,7 @@ class VoiceInteractionService : Service() {
override fun onResult(result: String, detectedLanguage: String) { override fun onResult(result: String, detectedLanguage: String) {
if (result.isNotEmpty()) { if (result.isNotEmpty()) {
processWithVolcanoAI(result) processWithOpenAI(result)
} }
// 重置状态,继续识别 // 重置状态,继续识别
@ -483,27 +508,114 @@ class VoiceInteractionService : Service() {
} }
/** /**
* 使用VolcanoAI处理语音识别结果 * 使用OpenAI处理语音识别结果
*/ */
private fun processWithVolcanoAI(text: String) { private fun processWithOpenAI(text: String) {
// 保存当前用户输入,用于后续同步聊天记录 // 保存当前用户输入,用于后续同步聊天记录
currentUserInput = text currentUserInput = text
Thread { Thread {
try { try {
val messages = JSONArray().apply { val messages = JSONArray().apply {
put(volcanoAIService.createUserMessage(text)) put(openAIService.createUserMessage(text))
} }
val response = volcanoAIService.sendMessage(messages, systemPrompt) // 创建响应构建器
val responseBuilder = StringBuilder()
openAIService.sendMessageStream(
messages = messages,
systemPrompt = systemPrompt,
callback = object : OpenAIService.StreamCallback {
override fun onToken(token: String) {
// 累加响应内容
responseBuilder.append(token)
}
override fun onComplete() {
// 处理完整响应
val response = responseBuilder.toString()
if (response.isNotEmpty()) {
// 播放AI回复
Log.d(TAG, "AI 回复: $response")
speakAIResponse(response)
// 同步聊天记录到Flutter端
notifyChatHistoryUpdated("personal_assistant", text, response)
}
}
// 播放AI回复 override fun onError(e: Exception) {
speakAIResponse(response) FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e)
playNotification("AI处理出错")
}
override fun onFunctionCall(call: JSONObject) {
FileLogger.d(TAG, "收到函数调用请求: ${call.getString("name")}")
// 使用函数处理器处理函数调用
val handled = functionHandler.handleFunctionCall(
functionCall = call,
messages = messages,
callback = object : VoiceFunctionHandler.FunctionCallCallback {
override fun onTokenReceived(token: String) {
responseBuilder.append(token)
}
override fun onComplete() {
val response = responseBuilder.toString()
if (response.isNotEmpty()) {
// 播放AI回复
Log.d(TAG, "AI Function Call 回复: $response")
speakAIResponse(response)
// 同步聊天记录到Flutter端
notifyChatHistoryUpdated("personal_assistant", text, response)
}
updateLastActivityTime()
}
override fun onError(message: String) {
FileLogger.e(TAG, "函数处理出错: $message")
playNotification(message)
}
override fun onFunctionCall(nestedCall: JSONObject) {
FileLogger.d(TAG, "收到嵌套函数调用: ${nestedCall.getString("name")}")
// 递归处理嵌套函数调用
functionHandler.handleFunctionCall(
functionCall = nestedCall,
messages = messages,
callback = this
)
}
override fun onExitWithMessage(farewell: String) {
// 播放退出消息
speakAIResponse(farewell)
// 同步聊天记录
notifyChatHistoryUpdated("personal_assistant", text, farewell)
// 停止语音识别
stopVoiceRecognition()
}
}
)
if (!handled) {
// 如果函数没有被处理,作为普通文本处理
FileLogger.d(TAG, "函数未处理,作为普通文本处理")
speakAIResponse("我无法处理这个请求")
notifyChatHistoryUpdated("personal_assistant", text, "我无法处理这个请求")
}
}
}
)
// 同步聊天记录到Flutter端
notifyChatHistoryUpdated("personal_assistant", text, response)
} catch (e: Exception) { } catch (e: Exception) {
FileLogger.e(TAG, "AI处理出错: ${e.message}") FileLogger.e(TAG, "AI处理出错: ${e.message}", e)
playNotification("AI处理出错") playNotification("AI处理出错")
} }
}.start() }.start()
@ -746,38 +858,45 @@ class VoiceInteractionService : Service() {
override fun onBind(intent: Intent?): IBinder? = null override fun onBind(intent: Intent?): IBinder? = null
override fun onDestroy() { override fun onDestroy() {
FileLogger.d(TAG, "onDestroy - 语音交互服务正在销毁") super.onDestroy()
FileLogger.d(TAG, "onDestroy - 语音交互服务即将销毁")
// 释放音频播放器
audioPlayer?.release()
audioPlayer = null
// 停止监控 // 停止服务监控
stopMonitoring() stopMonitoring()
// 停止语音识别 // 停止语音识别
if (isRecognitionActive) { stopVoiceRecognition()
stopVoiceRecognition()
} // 停止媒体会话
mediaSession.release()
FileLogger.d(TAG, "媒体会话已释放")
// 停止TTS // 停止TTS
stopCurrentTTS() stopCurrentTTS()
// 释放 MediaSession // 关闭Azure语音服务
mediaSession.release() if (::azureAsrHelper.isInitialized) {
FileLogger.d(TAG, "关闭Azure语音服务")
azureAsrHelper.dispose()
}
// 释放Azure服务实例 if (::azureTtsHelper.isInitialized) {
FileLogger.d(TAG, "释放Azure服务实例") FileLogger.d(TAG, "关闭Azure TTS服务")
azureAsrHelper.dispose() azureTtsHelper.dispose()
azureTtsHelper.dispose() }
// 更新服务状态 // 关闭OpenAI服务
isRunning.set(false) if (::openAIService.isInitialized) {
FileLogger.d(TAG, "关闭OpenAI服务")
}
// 关闭日志系统 // 关闭音频播放器
FileLogger.shutdown() audioPlayer?.release()
super.onDestroy() // 设置服务状态
isRunning.set(false)
FileLogger.d(TAG, "语音交互服务已销毁")
} }
/** /**
@ -804,9 +923,7 @@ class VoiceInteractionService : Service() {
* 通知 Flutter 端聊天记录已更新 * 通知 Flutter 端聊天记录已更新
*/ */
private fun notifyChatHistoryUpdated(agentId: String, userMessage: String, assistantMessage: String) { private fun notifyChatHistoryUpdated(agentId: String, userMessage: String, assistantMessage: String) {
FileLogger.d(TAG, "通知Flutter聊天记录已更新: agentId=$agentId")
FileLogger.d(TAG, "用户消息: ${userMessage.take(50)}...")
FileLogger.d(TAG, "助手回复: ${assistantMessage.take(50)}...")
// 创建广播 Intent // 创建广播 Intent
val intent = Intent(ACTION_CHAT_HISTORY_UPDATED).apply { val intent = Intent(ACTION_CHAT_HISTORY_UPDATED).apply {
@ -818,7 +935,6 @@ class VoiceInteractionService : Service() {
// 发送广播 // 发送广播
sendBroadcast(intent) sendBroadcast(intent)
FileLogger.d(TAG, "已发送聊天记录广播")
} }
/** /**

0
android/app/src/main/kotlin/com/example/deep_voice/core/utils/FileLogger.kt → android/app/src/main/kotlin/com/yunqiinnovation/deepsound/core/utils/FileLogger.kt

4
android/settings.gradle.kts

@ -29,4 +29,8 @@ plugins {
} }
include(":app") include(":app")
include(":azure_speech")
// 设置azure_speech项目的路径
project(":azure_speech").projectDir = file("../local_plugins/azure_speech/android")

21
azure/LICENSE

@ -1,21 +0,0 @@
MIT License
Copyright (c) 2024 Your Company
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.

66
azure/README.md

@ -1,66 +0,0 @@
# Azure Speech Recognition
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities.
## Features
- Speech-to-text (Azure Speech Recognition)
- Text-to-speech (Azure Speech Synthesis)
- Support for multiple languages
- Language detection
- Continuous recognition
- Streaming synthesis
## Getting Started
### Prerequisites
- Azure Speech service subscription key
- Azure Speech service region
### Installation
Add this to your package's `pubspec.yaml` file:
```yaml
dependencies:
azure_speech_recognition:
path: ./azure
```
### Usage
```dart
import 'package:azure_speech_recognition/azure_speech_recognition.dart';
// Initialize the service
await AzureSpeechRecognition.initialize(
subscriptionKey: 'your_subscription_key',
region: 'your_region',
supportedLanguages: ['zh-CN', 'en-US'],
);
// Start continuous recognition
await AzureSpeechRecognition.startContinuousRecognition();
// Listen for recognition events
AzureSpeechRecognition.onRecognitionEvent.listen((event) {
if (event['type'] == 'result') {
print('Recognized: ${event['text']}');
print('Detected language: ${event['detectedLanguage']}');
}
});
// Stop recognition when done
await AzureSpeechRecognition.stopContinuousRecognition();
// Speak text
await AzureSpeechRecognition.speakText('Hello, world!');
// Clean up
await AzureSpeechRecognition.dispose();
```
## License
This project is licensed under the MIT License - see the LICENSE file for details.

757
azure/ios/Classes/AzureAsrHelper.swift

@ -1,757 +0,0 @@
import Foundation
import MicrosoftCognitiveServicesSpeech
import AVFoundation
import AudioToolbox
/// Azure ASR工具类,负责实现语音识别服务接口
@available(iOS 13.0, *)
class AzureAsrHelper: NSObject {
// MARK: - 属性
/// 事件处理回调
private var eventHandler: (String, [String: Any]) -> Void
/// 语音配置信息
private var speechSubscriptionKey: String = ""
private var serviceRegion: String = ""
/// 语音识别相关
private var speechConfig: SPXSpeechConfiguration?
private var recognizer: SPXSpeechRecognizer?
private var audioConfig: SPXAudioConfiguration?
private var pushStream: SPXPushAudioInputStream?
/// 音频处理相关
private var audioProcessor: CustomAudioProcessor?
private var isProcessingAudio = false
private var audioProcessingTimer: Timer?
/// 状态标志
private var isInitialized = false
private var _isContinuousRecognitionActive = false
/// 当前语言和支持的语言
private var currentLanguage = "zh-CN"
private var supportedLanguages: [String] = ["zh-CN", "en-US"]
private var isAutoDetectLanguage = false
// MARK: - 初始化
init(eventHandler: @escaping (String, [String: Any]) -> Void) {
self.eventHandler = eventHandler
super.init()
}
deinit {
dispose()
}
// MARK: - ASR Service 接口实现
/// 初始化语音识别服务
/// - Parameters:
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
/// - serviceRegion: Azure 服务区域 (如 eastasia)
/// - supportedLanguages: 支持的语言代码数组 (可选)
/// - Returns: 初始化是否成功
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool {
print("[AzureAsrHelper] 初始化 Azure 语音服务")
// 检查配置是否为空
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
eventHandler("error", ["message": "Azure 配置信息不完整"])
return false
}
// 释放之前的资源
dispose()
// 记录配置信息
self.speechSubscriptionKey = speechSubscriptionKey
self.serviceRegion = serviceRegion
// 设置语言
if let languages = supportedLanguages, !languages.isEmpty {
self.supportedLanguages = languages
}
// 根据支持的语言数量决定是否启用自动语言检测
isAutoDetectLanguage = self.supportedLanguages.count >= 2
// 如果只有一种语言,设置为当前语言
if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty {
currentLanguage = self.supportedLanguages[0]
}
// 创建识别器和设置回调
if !createRecognizerAndSetupCallbacks() {
return false
}
print("[AzureAsrHelper] Azure 语音服务初始化成功")
isInitialized = true
return true
}
/// 创建识别器并设置回调
private func createRecognizerAndSetupCallbacks() -> Bool {
// 释放之前的 recognizer
recognizer = nil
audioConfig = nil
do {
// 创建语音配置
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置音频输入参数
try setupAudioSession()
// 创建自定义推送流,替代默认的麦克风输入
pushStream = try SPXPushAudioInputStream()
audioConfig = try SPXAudioConfiguration(streamInput: pushStream!)
// 初始化自定义音频处理器
audioProcessor = CustomAudioProcessor()
// 设置语言配置
if isAutoDetectLanguage {
// 设置自动语言检测
speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode)
// 创建自动语言检测配置
let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages)
// 创建识别器
recognizer = try SPXSpeechRecognizer(
speechConfiguration: speechConfig!,
autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig,
audioConfiguration: audioConfig!
)
} else {
// 设置指定的识别语言
speechConfig?.speechRecognitionLanguage = currentLanguage
// 创建识别器
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
}
// 设置所有回调
setupAllCallbacks()
return true
} catch {
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)")
eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"])
return false
}
}
/// 设置音频会话
private func setupAudioSession() throws {
let audioSession = AVAudioSession.sharedInstance()
// 使用playAndRecord类别允许同时录音和播放
try audioSession.setCategory(.playAndRecord,
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers])
// 设置首选的输入和输出
let currentRoute = audioSession.currentRoute
// 获取当前是否连接了耳机或外部麦克风
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
}
// 如果没有耳机,明确启用内置麦克风和扬声器的回音消除
if !hasHeadphones {
try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除
// 启用回音消除和噪声抑制
try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性
} else {
// 耳机模式,可以使用不同的设置
try audioSession.setMode(.voiceChat)
try audioSession.setInputGain(1.0)
}
// 设置合适的采样率
try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率
try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟
// 激活音频会话
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除")
}
/// 设置所有回调
private func setupAllCallbacks() {
guard let recognizer = recognizer else { return }
// 最终识别结果
recognizer.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
self.eventHandler("result", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
// 识别中事件
recognizer.addRecognizingEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == SPXResultReason.recognizingSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
self.eventHandler("recognizing", [
"text": event.result.text ?? "",
"detectedLanguage": detectedLanguage
])
}
}
// 会话事件
recognizer.addSessionStartedEventHandler { [weak self] _, _ in
guard let self = self else { return }
print("[AzureAsrHelper] 识别会话已开始")
self._isContinuousRecognitionActive = true
self.eventHandler("sessionStarted", [:])
}
recognizer.addSessionStoppedEventHandler { [weak self] _, _ in
guard let self = self else { return }
print("[AzureAsrHelper] 识别会话已结束")
self._isContinuousRecognitionActive = false
self.eventHandler("sessionStopped", [:])
}
// 取消事件
recognizer.addCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
let reason = event.reason.rawValue
let errorDetails = event.errorDetails ?? "未知错误"
print("[AzureAsrHelper] 识别取消: \(errorDetails)")
self.eventHandler("canceled", [
"reason": reason,
"errorDetails": errorDetails
])
self._isContinuousRecognitionActive = false
}
}
/// 执行一次性语音识别
/// - Returns: 是否成功启动识别
func recognizeOnce() -> Bool {
if !isInitialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
eventHandler("error", ["message": "语音服务未初始化"])
return false
}
// 如果正在连续识别,先停止
if _isContinuousRecognitionActive {
stopContinuousRecognition()
}
// 确保识别器已创建
if recognizer == nil && !createRecognizerAndSetupCallbacks() {
return false
}
do {
// 启动音频处理
startAudioProcessing()
// 通知会话开始
eventHandler("sessionStarted", [:])
// 执行识别
try recognizer?.recognizeOnceAsync { [weak self] result in
guard let self = self else { return }
// 停止音频处理
self.stopAudioProcessing()
if result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: result)
self.eventHandler("result", [
"text": result.text ?? "",
"detectedLanguage": detectedLanguage
])
} else if result.reason == SPXResultReason.noMatch {
print("[AzureAsrHelper] 无匹配结果")
self.eventHandler("noMatch", [:])
} else if result.reason == SPXResultReason.canceled {
do {
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result)
let errorDetails = details.errorDetails ?? "未知错误"
self.eventHandler("error", ["message": "识别取消: \(errorDetails)"])
} catch {
print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)")
self.eventHandler("error", ["message": "识别取消,无法获取详细原因"])
}
}
}
return true
} catch {
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)")
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"])
stopAudioProcessing()
return false
}
}
/// 开始连续语音识别
/// - Returns: 是否成功启动识别
func startContinuousRecognition() -> Bool {
if !isInitialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
eventHandler("error", ["message": "语音服务未初始化"])
return false
}
// 如果已经在进行连续识别,先停止
if _isContinuousRecognitionActive {
stopContinuousRecognition()
}
// 确保识别器已创建
if recognizer == nil && !createRecognizerAndSetupCallbacks() {
return false
}
// 重新确保音频设置正确
do {
try setupAudioSession()
} catch {
print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)")
}
do {
// 启动音频处理
startAudioProcessing()
// 启动连续识别
try recognizer?.startContinuousRecognition()
_isContinuousRecognitionActive = true
print("[AzureAsrHelper] 连续识别开始")
return true
} catch {
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"])
_isContinuousRecognitionActive = false
stopAudioProcessing()
return false
}
}
/// 停止连续语音识别
/// - Returns: 是否成功停止识别
func stopContinuousRecognition() -> Bool {
// 停止音频处理
stopAudioProcessing()
if !_isContinuousRecognitionActive || recognizer == nil {
return true
}
do {
try recognizer?.stopContinuousRecognition()
_isContinuousRecognitionActive = false
print("[AzureAsrHelper] 连续识别已停止")
return true
} catch {
print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)")
eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"])
_isContinuousRecognitionActive = false
return false
}
}
/// 检查连续识别是否活跃
/// - Returns: 连续识别是否处于活跃状态
func isContinuousRecognitionActive() -> Bool {
return _isContinuousRecognitionActive
}
/// 释放资源
func dispose() {
print("[AzureAsrHelper] 释放资源")
// 停止音频处理
stopAudioProcessing()
// 停止连续识别
if _isContinuousRecognitionActive {
stopContinuousRecognition()
}
// 释放音频会话
do {
try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation)
} catch {
print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)")
}
// 释放资源
recognizer = nil
speechConfig = nil
audioConfig = nil
pushStream = nil
audioProcessor = nil
// 重置状态
_isContinuousRecognitionActive = false
isInitialized = false
}
/// 从结果中获取检测到的语言
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
if isAutoDetectLanguage {
do {
let langResult = try SPXAutoDetectSourceLanguageResult(result)
return langResult.language ?? currentLanguage
} catch {
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)")
return currentLanguage
}
} else {
return currentLanguage
}
}
// MARK: - 音频处理
/// 开始音频处理
private func startAudioProcessing() {
guard !isProcessingAudio, let audioProcessor = audioProcessor else { return }
isProcessingAudio = true
// 启动音频处理器
if !audioProcessor.startRecord() {
print("[AzureAsrHelper] 错误: 启动音频处理器失败")
eventHandler("error", ["message": "启动音频处理器失败"])
return
}
// 启动音频处理定时器
audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in
guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else {
return
}
// 读取处理后的音频数据
var bytes = [UInt8](repeating: 0, count: 2560)
let bytesRead = processor.read(bytes: &bytes)
if bytesRead > 0 {
// 推送数据到Azure语音服务
let data = Data(bytes: bytes, count: bytesRead)
stream.write(data)
// 通知音频数据可用
self.eventHandler("audioData", ["data": bytes])
}
}
print("[AzureAsrHelper] 音频处理已启动")
}
/// 停止音频处理
private func stopAudioProcessing() {
// 停止定时器
audioProcessingTimer?.invalidate()
audioProcessingTimer = nil
// 停止音频处理器
audioProcessor?.stopRecord()
isProcessingAudio = false
print("[AzureAsrHelper] 音频处理已停止")
}
}
// MARK: - 自定义音频处理器
@available(iOS 13.0, *)
class CustomAudioProcessor: NSObject {
// 音频单元
private var ioUnit: AudioUnit?
// 音频格式
private var audioFormat: AudioStreamBasicDescription
// 音频缓冲
private var audioBufferList: AudioBufferList
private var audioList: [Float] = []
private let audioListQueue = DispatchQueue(label: "audioListQueue")
// 回音消除状态
private var isEchoCancellationEnabled = true
override init() {
// 设置音频格式 - 16kHz, 16位, 单声道
audioFormat = AudioStreamBasicDescription(
mSampleRate: 16000.0,
mFormatID: kAudioFormatLinearPCM,
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked,
mBytesPerPacket: 2,
mFramesPerPacket: 1,
mBytesPerFrame: 2,
mChannelsPerFrame: 1,
mBitsPerChannel: 16,
mReserved: 0
)
// 初始化音频缓冲
audioBufferList = AudioBufferList(
mNumberBuffers: 1,
mBuffers: AudioBuffer(
mNumberChannels: 1,
mDataByteSize: 4096,
mData: malloc(4096)
)
)
super.init()
}
deinit {
stopRecord()
free(audioBufferList.mBuffers.mData)
}
/// 启动音频处理
/// - Returns: 是否成功启动
func startRecord() -> Bool {
print("[CustomAudioProcessor] 配置音频单元")
// 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除
var ioUnitDescription = AudioComponentDescription(
componentType: kAudioUnitType_Output,
componentSubType: kAudioUnitSubType_VoiceProcessingIO,
componentManufacturer: kAudioUnitManufacturer_Apple,
componentFlags: 0,
componentFlagsMask: 0
)
// 查找音频组件
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else {
print("[CustomAudioProcessor] 错误: 未找到音频组件")
return false
}
// 创建音频单元实例
if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") {
ioUnit = nil
return false
}
// 启用输入端口
var enableInput: UInt32 = 1
let kInputBus: AudioUnitElement = 1
let kOutputBus: AudioUnitElement = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Input, kInputBus, &enableInput,
UInt32(MemoryLayout<UInt32>.size)), "启用输入端口") {
return false
}
// 禁用输出端口 (我们只需要输入)
var enableOutput: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO,
kAudioUnitScope_Output, kOutputBus,
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "禁用输出端口") {
return false
}
// 设置缓冲区分配标志
var flag: UInt32 = 0
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer,
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置缓冲区分配标志") {
return false
}
// 设置音频格式
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size)
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") {
return false
}
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat,
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") {
return false
}
// 启用回音消除
if isEchoCancellationEnabled {
var echoCancellation: UInt32 = 1
AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing,
kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout<UInt32>.size))
}
// 设置输入回调 - 当有新音频数据时调用
var inputCallback = AURenderCallbackStruct(
inputProc: CustomAudioProcessor.onAudioDataAvailable,
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque())
)
if checkError(AudioUnitSetProperty(ioUnit!,
kAudioOutputUnitProperty_SetInputCallback,
kAudioUnitScope_Global, kInputBus,
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") {
return false
}
// 初始化音频单元
var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
while hasError {
Thread.sleep(forTimeInterval: 0.1)
hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元")
}
// 启动音频单元
hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元")
print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")")
return !hasError
}
/// 停止音频处理
func stopRecord() {
print("[CustomAudioProcessor] 停止音频处理器")
if let ioUnit = ioUnit {
// 停止音频单元
_ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元")
// 关闭音频单元
_ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元")
_ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元")
self.ioUnit = nil
}
// 清空音频数据缓冲
audioListQueue.sync {
audioList.removeAll()
}
}
/// 音频数据回调 - 当有新的音频数据可用时调用
private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in
// 获取实例
let processor = Unmanaged<CustomAudioProcessor>.fromOpaque(inRefCon).takeUnretainedValue()
// 计算预期数据大小
let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame
// 确保缓冲区足够大
if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize {
processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize))
processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize
}
// 渲染音频数据
let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp,
inBusNumber, inNumberFrames, &processor.audioBufferList),
"渲染音频数据")
// 将Int16数据转换为浮点数据进行处理
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames))
let buffer = processor.audioBufferList.mBuffers
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self)
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) {
// 归一化到[-1.0, 1.0]范围
audioDataFloat[j] = Float(bufferData[j]) / 32768.0
}
// 应用附加处理 (如有需要)
// processor.applyAdditionalProcessing(&audioDataFloat)
// 保存处理后的数据
if status == noErr {
processor.audioListQueue.async {
processor.audioList.append(contentsOf: audioDataFloat)
}
}
return status
}
/// 读取处理后的音频数据
/// - Parameter bytes: 输出字节数组
/// - Returns: 读取的字节数
func read(bytes: inout [UInt8]) -> Int {
return audioListQueue.sync {
// 如果没有数据,返回0
if audioList.isEmpty {
return 0
}
// 确保有足够的数据 (至少1280个样本)
if audioList.count < 1280 {
return 0
}
// 读取一帧数据 (1280个样本)
let frameLength = 1280
let buffer = Array(audioList.prefix(frameLength))
audioList.removeFirst(frameLength)
// 将浮点数据转回Int16格式
var int16Data = buffer.map { Int16($0 * 32767) }
// 转换为字节数组
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count))
bytes = [UInt8](data)
// 每个样本2字节 (16位PCM)
return frameLength * 2
}
}
/// 检查错误并打印日志
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 是否发生错误
private func checkError(_ status: OSStatus, _ operation: String) -> Bool {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
return true
}
return false
}
/// 检查OSStatus并返回状态
/// - Parameters:
/// - status: 操作状态
/// - operation: 操作描述
/// - Returns: 原始状态
private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus {
if status != noErr {
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)")
}
return status
}
}

18
azure/ios/Classes/AzureSpeechRecognitionPlugin.swift

@ -1,18 +0,0 @@
import Flutter
import UIKit
public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
public static func register(with registrar: FlutterPluginRegistrar) {
if #available(iOS 13.0, *) {
SwiftAzureSpeechRecognitionPlugin.register(with: registrar)
} else {
// 如果低于iOS 13.0,返回不支持的错误
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger())
channel.setMethodCallHandler { (call, result) in
result(FlutterError(code: "UNSUPPORTED",
message: "需要iOS 13.0及以上系统",
details: nil))
}
}
}
}

427
azure/ios/Classes/AzureTtsHelper.swift

@ -1,427 +0,0 @@
import Foundation
import MicrosoftCognitiveServicesSpeech
import AVFoundation
/// Azure TTS工具类,负责实现TTS服务接口
@available(iOS 13.0, *)
class AzureTtsHelper: NSObject {
// MARK: - 属性
/// 事件处理回调
private var eventHandler: (String, [String: Any]) -> Void
/// 语音配置信息
private var speechSubscriptionKey: String = ""
private var serviceRegion: String = ""
/// 语音合成配置
private var speechConfig: SPXSpeechConfiguration?
/// 语音合成器
private var synthesizer: SPXSpeechSynthesizer?
/// 是否初始化成功
private var isInitialized = false
/// 当前是否正在播放
private var _isSpeaking = false
/// 音频会话配置
private var isAudioSessionConfigured = false
// MARK: - 语音设置
/// 当前语音
private var currentVoice = "zh-CN-XiaoxiaoNeural"
/// 支持的语音映射
private var voiceMap: [String: String] = [
"zh-CN": "zh-CN-XiaoxiaoNeural",
"en-US": "en-US-JennyNeural",
"ja-JP": "ja-JP-NanamiNeural",
"ko-KR": "ko-KR-SunHiNeural",
"zh-TW": "zh-TW-HsiaoChenNeural",
"zh-HK": "zh-HK-HiuMaanNeural"
]
/// 当前语音合成参数
private var currentSpeechRate = "0%"
private var currentPitch = "0%"
private var currentVolume = "100%"
// MARK: - 初始化
init(eventHandler: @escaping (String, [String: Any]) -> Void) {
self.eventHandler = eventHandler
super.init()
}
deinit {
dispose()
}
// MARK: - TTS 接口实现
/// 初始化语音合成服务
/// - Parameters:
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
/// - serviceRegion: Azure 服务区域 (如 eastasia)
/// - language: 语言代码 (默认 zh-CN)
/// - Returns: 初始化是否成功
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool {
print("[AzureTtsHelper] 初始化语音合成服务")
// 检查配置是否为空
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureTtsHelper] 错误: Azure 配置信息不完整")
eventHandler("error", ["error": "Azure 配置信息不完整"])
return false
}
// 释放之前的资源
dispose()
// 记录配置信息
self.speechSubscriptionKey = speechSubscriptionKey
self.serviceRegion = serviceRegion
// 配置音频会话
if !configureAudioSession() {
print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化")
}
do {
// 创建语音配置
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置默认语音
let defaultVoice = getDefaultVoiceForLanguage(language)
currentVoice = defaultVoice
speechConfig?.speechSynthesisVoiceName = defaultVoice
// 创建语音合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig!)
// 设置事件处理器
setupSynthesizerEvents()
isInitialized = true
print("[AzureTtsHelper] TTS 引擎初始化成功")
return true
} catch {
print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)")
eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"])
return false
}
}
/// 配置音频会话
private func configureAudioSession() -> Bool {
let audioSession = AVAudioSession.sharedInstance()
do {
// 使用playback类别,但支持混合和空中播放
try audioSession.setCategory(.playback,
mode: .spokenAudio,
options: [.mixWithOthers, .allowAirPlay, .duckOthers])
// 根据设备类型选择最佳配置
let currentRoute = audioSession.currentRoute
let hasHeadphones = currentRoute.outputs.contains {
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP
}
// 优化音频路由
if hasHeadphones {
// 耳机模式,使用默认设置
try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟
} else {
// 扬声器模式
try audioSession.setPreferredIOBufferDuration(0.005)
}
// 避免完全激活音频会话,因为ASR可能已经激活
// 这里使用setActive(false)是为了不与ASR冲突
if !audioSession.isOtherAudioPlaying {
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
}
isAudioSessionConfigured = true
print("[AzureTtsHelper] 音频会话配置成功")
return true
} catch {
print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)")
isAudioSessionConfigured = false
return false
}
}
/// 设置语音
/// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural")
/// - Returns: 设置是否成功
func setVoice(voiceName: String) -> Bool {
if !isInitialized {
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
eventHandler("error", ["error": "TTS 引擎尚未初始化"])
return false
}
if voiceName.isEmpty {
print("[AzureTtsHelper] 错误: 声音名称为空")
eventHandler("error", ["error": "声音名称不能为空"])
return false
}
if voiceName == currentVoice {
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
return true
}
print("[AzureTtsHelper] 设置声音: \(voiceName)")
currentVoice = voiceName
// 更新语音配置
if let speechConfig = speechConfig {
speechConfig.speechSynthesisVoiceName = voiceName
return true
}
return false
}
/// 设置语音合成参数
/// - Parameters:
/// - rate: 语速,范围 -100 到 100,默认为 0
/// - pitch: 音调,范围 -100 到 100,默认为 0
/// - volume: 音量,范围 0 到 100,默认为 100
/// - Returns: 是否设置成功
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
if !isInitialized {
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
eventHandler("error", ["error": "TTS 引擎尚未初始化"])
return false
}
// 转换参数格式
currentSpeechRate = formatRateParam(rate)
currentPitch = formatPitchParam(pitch)
currentVolume = formatVolumeParam(volume)
print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)")
return true
}
/// 合成文本为语音并播放
/// - Parameter text: 要合成的文本
/// - Returns: 操作是否成功启动
func speakText(text: String) -> Bool {
if !isInitialized {
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化")
eventHandler("error", ["error": "TTS 引擎尚未初始化"])
return false
}
if text.isEmpty {
print("[AzureTtsHelper] 警告: 要播放的文本为空")
return true
}
// 确保音频会话已配置
if !isAudioSessionConfigured {
_ = configureAudioSession()
}
print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...")
// 生成SSML
let ssml = generateSsml(text: text)
// 直接进行SSML合成
return speakSsmlInternal(text: ssml)
}
/// 内部SSML合成和播放
private func speakSsmlInternal(text: String) -> Bool {
guard let synthesizer = synthesizer else {
print("[AzureTtsHelper] 错误: 合成器未初始化")
eventHandler("error", ["error": "合成器未初始化"])
return false
}
_isSpeaking = true
eventHandler("started", [:])
Task {
do {
// 使用异步方法进行合成并直接播放
_ = try await synthesizer.startSpeakingSsml(text)
} catch {
print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)")
DispatchQueue.main.async {
self._isSpeaking = false
self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"])
}
}
}
return true
}
/// 停止当前语音合成
/// - Returns: 操作是否成功
func stopSpeaking() -> Bool {
if !isInitialized || !_isSpeaking {
return true
}
// 停止合成
do {
try synthesizer?.stopSpeaking()
_isSpeaking = false
eventHandler("canceled", [:])
print("[AzureTtsHelper] 已停止语音合成")
return true
} catch {
print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)")
eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"])
return false
}
}
/// 检查是否正在播放
/// - Returns: 当前是否正在播放语音
func isSpeaking() -> Bool {
return _isSpeaking
}
/// 释放资源
func dispose() {
try? stopSpeaking()
// 释放合成器和配置
synthesizer = nil
speechConfig = nil
isInitialized = false
_isSpeaking = false
isAudioSessionConfigured = false
print("[AzureTtsHelper] TTS 引擎已释放")
}
// MARK: - 私有辅助方法
/// 设置合成器事件处理
private func setupSynthesizerEvents() {
guard let synthesizer = synthesizer else { return }
// 添加书签到达事件处理
synthesizer.addBookmarkReachedEventHandler { _, e in
print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"")
}
// 合成完成事件
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in
guard let self = self else { return }
print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)")
DispatchQueue.main.async {
self._isSpeaking = false
self.eventHandler("completed", [:])
}
}
// 合成取消事件
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in
guard let self = self else { return }
let result = e.result
do {
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result)
print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)")
if cancellationDetails.reason == SPXCancellationReason.error {
print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)")
print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")")
}
DispatchQueue.main.async {
self._isSpeaking = false
self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"])
}
} catch {
print("[AzureTtsHelper] 获取取消详情时出错: \(error)")
DispatchQueue.main.async {
self._isSpeaking = false
self.eventHandler("error", ["error": "语音合成被取消"])
}
}
}
// 合成开始事件
synthesizer.addSynthesisStartedEventHandler { _, _ in
// print("[AzureTtsHelper] 语音合成开始")
}
// 合成中事件
synthesizer.addSynthesizingEventHandler { _, _ in
// print("[AzureTtsHelper] 语音合成中")
}
}
/// 生成 SSML 文本
private func generateSsml(text: String) -> String {
return """
<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis' xml:lang='zh-CN'>
<voice name='\(currentVoice)'>
<prosody rate='\(currentSpeechRate)' pitch='\(currentPitch)' volume='\(currentVolume)'>
\(text)
</prosody>
</voice>
</speak>
"""
}
/// 格式化语速参数
private func formatRateParam(_ rate: Int) -> String {
let clampedRate = rate.clamp(min: -100, max: 100)
if clampedRate == 0 {
return "0%"
} else if clampedRate < 0 {
return "\(Int(Double(clampedRate) * 0.9))%"
} else {
return "+\(clampedRate)%"
}
}
/// 格式化音调参数
private func formatPitchParam(_ pitch: Int) -> String {
let clampedPitch = pitch.clamp(min: -100, max: 100)
if clampedPitch == 0 {
return "0%"
} else {
return "\(Int(Double(clampedPitch) * 0.5))%"
}
}
/// 格式化音量参数
private func formatVolumeParam(_ volume: Int) -> String {
let clampedVolume = volume.clamp(min: 0, max: 100)
return "\(clampedVolume)%"
}
/// 获取指定语言的默认语音
private func getDefaultVoiceForLanguage(_ language: String) -> String {
return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural"
}
}
// MARK: - 扩展
extension Int {
func clamp(min: Int, max: Int) -> Int {
if self < min { return min }
if self > max { return max }
return self
}
}

259
azure/ios/Classes/SwiftAzureSpeechRecognitionPlugin.swift

@ -1,259 +0,0 @@
import Flutter
import UIKit
import MicrosoftCognitiveServicesSpeech
import AVFoundation
@available(iOS 13.0, *)
public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin {
private var azureChannel: FlutterMethodChannel
private var ttsChannel: FlutterMethodChannel
private var asrHelper: AzureAsrHelper
private var ttsHelper: AzureTtsHelper
private static var eventStreamHandler: AzureEventStreamHandler?
// 创建方法到通道的映射
private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]()
private static var asrMethodHandlers = [String: FlutterMethodCallHandler]()
public static func register(with registrar: FlutterPluginRegistrar) {
// ASR通道
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger())
// TTS通道
let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger())
// 设置ASR事件通道
let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger())
eventStreamHandler = AzureEventStreamHandler()
eventChannel.setStreamHandler(eventStreamHandler)
let instance = SwiftAzureSpeechRecognitionPlugin(
azureChannel: channel,
ttsChannel: ttsChannel,
eventStreamHandler: eventStreamHandler!
)
// 直接设置各自通道的处理器
channel.setMethodCallHandler(instance.handleAsrMethodCalls)
ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls)
}
// 新增直接处理方法调用的函数
private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
handleTtsMethod(call, result)
}
private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
handleAsrMethod(call, result)
}
init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) {
self.azureChannel = azureChannel
self.ttsChannel = ttsChannel
// 创建辅助类实例,使用自定义事件回调处理器
let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in
DispatchQueue.main.async {
if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink {
var eventData = arguments
eventData["type"] = eventName
eventSink(eventData)
}
}
}
asrHelper = AzureAsrHelper(eventHandler: eventHandler)
ttsHelper = AzureTtsHelper(eventHandler: eventHandler)
super.init()
}
private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
let args = call.arguments as? Dictionary<String, Any>
switch call.method {
case "initialize":
// 仅在初始化时读取必要参数
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else {
let errorMsg = "语音订阅密钥不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil))
return
}
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else {
let errorMsg = "服务区域不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil))
return
}
let supportedLanguages = args?["supportedLanguages"] as? [String] ?? []
let success = asrHelper.initialize(
speechSubscriptionKey: speechSubscriptionKey,
serviceRegion: serviceRegion,
supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages
)
result(success)
case "startContinuousRecognition":
// 只有使用参数时才验证
let success = asrHelper.startContinuousRecognition()
result(success)
case "stopContinuousRecognition":
// 不需要额外参数
let success = asrHelper.stopContinuousRecognition()
result(success)
case "recognizeOnce":
// 只有使用参数时才验证
let success = asrHelper.recognizeOnce()
result(success)
case "isContinuousRecognitionActive":
// 不需要额外参数
result(asrHelper.isContinuousRecognitionActive())
case "dispose":
// 不需要额外参数
print("[AzurePlugin] 释放ASR资源")
asrHelper.dispose()
result(true)
default:
print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)")
result(FlutterMethodNotImplemented)
}
}
private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) {
let args = call.arguments as? Dictionary<String, Any>
switch call.method {
case "initialize":
// 仅在初始化时验证参数
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else {
let errorMsg = "语音订阅密钥不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil))
return
}
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else {
let errorMsg = "服务区域不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil))
return
}
let language = args?["language"] as? String ?? "zh-CN"
print("[AzurePlugin] 初始化TTS,语言: \(language)")
let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language)
result(success)
case "setVoice":
// 仅获取voice参数
guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else {
let errorMsg = "声音名称不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil))
return
}
print("[AzurePlugin] 设置声音: \(voiceName)")
let success = ttsHelper.setVoice(voiceName: voiceName)
result(success)
case "speakText":
// 仅获取text参数
let text = args?["text"] as? String ?? ""
if text.isEmpty {
print("[AzurePlugin] 警告: 要播放的文本为空")
result("OK")
return
}
print("[AzurePlugin] 播放文本: \(text.prefix(50))...")
let success = ttsHelper.speakText(text: text)
result(success ? "OK" : "ERROR")
case "speakSsml":
// 仅获取ssml参数
guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else {
let errorMsg = "SSML内容不能为空"
print("[AzurePlugin] 错误: \(errorMsg)")
result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil))
return
}
print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...")
// 由于我们移除了speakSsml方法,这里改用speakText方法
// Azure SDK内部会自动检测是普通文本还是SSML
let success = ttsHelper.speakText(text: ssml)
result(success)
case "stopSpeaking":
// 不需要参数
print("[AzurePlugin] 停止播放")
let success = ttsHelper.stopSpeaking()
result(success)
case "isSpeaking":
// 不需要参数
result(ttsHelper.isSpeaking())
case "setSpeechParams":
// 仅获取语音参数
let rate = args?["rate"] as? Int ?? 0
let pitch = args?["pitch"] as? Int ?? 0
let volume = args?["volume"] as? Int ?? 100
print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)")
let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume)
result(success)
case "dispose":
// 释放TTS资源
print("[AzurePlugin] 释放TTS资源")
ttsHelper.dispose()
result(true)
default:
print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)")
result(FlutterMethodNotImplemented)
}
}
}
// 用于处理事件流的辅助类
@available(iOS 13.0, *)
class AzureEventStreamHandler: NSObject, FlutterStreamHandler {
var eventSink: FlutterEventSink?
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? {
self.eventSink = events
// 通知Flutter端事件通道已准备好
DispatchQueue.main.async {
events(["type": "channelReady"])
}
return nil
}
func onCancel(withArguments arguments: Any?) -> FlutterError? {
self.eventSink = nil
return nil
}
}

24
azure/ios/azure_speech_recognition.podspec

@ -1,24 +0,0 @@
#
# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html.
# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing.
#
Pod::Spec.new do |s|
s.name = 'azure_speech_recognition'
s.version = '0.1.0'
s.summary = 'Azure Speech Recognition plugin for Flutter'
s.description = <<-DESC
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities.
DESC
s.homepage = 'https://github.com/yourusername/azure_speech_recognition'
s.license = { :type => 'MIT', :file => '../LICENSE' }
s.author = { 'Your Company' => 'your-email@example.com' }
s.source = { :path => '.' }
s.source_files = 'Classes/**/*'
s.dependency 'Flutter'
s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0'
s.platform = :ios, '12.0'
# Flutter.framework does not contain a i386 slice.
s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' }
s.swift_version = '5.0'
end

6
azure/lib/azure_speech_recognition.dart

@ -1,6 +0,0 @@
// This is a placeholder file that exports nothing.
// The actual implementation is in the app's services folder.
// This file exists just to satisfy the Flutter plugin structure requirements.
// Empty library to satisfy plugin structure
library azure_speech_recognition;

23
azure/pubspec.yaml

@ -1,23 +0,0 @@
name: azure_speech_recognition
description: Azure Speech Recognition and Text-to-Speech services Flutter plugin
version: 0.1.0
homepage: https://github.com/yourusername/azure_speech_recognition
environment:
sdk: '>=2.12.0 <3.0.0'
flutter: ">=2.0.0"
dependencies:
flutter:
sdk: flutter
dev_dependencies:
flutter_test:
sdk: flutter
flutter_lints: ^1.0.0
flutter:
plugin:
platforms:
ios:
pluginClass: AzureSpeechRecognitionPlugin

4
lib/data/services/speech_impl/azure_asr_service.dart

@ -10,8 +10,8 @@ import '../asr_service.dart';
/// 该服务提供了通过平台通道与 Android 上的 Microsoft Speech SDK 交互的接口 /// 该服务提供了通过平台通道与 Android 上的 Microsoft Speech SDK 交互的接口
class AzureAsrService extends GetxService implements AsrService { class AzureAsrService extends GetxService implements AsrService {
static final AzureAsrService to = Get.put(AzureAsrService()); static final AzureAsrService to = Get.put(AzureAsrService());
static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_asr'); static const MethodChannel _channel = MethodChannel('azure_speech/asr');
static const EventChannel _eventChannel = EventChannel('com.deep_voice.azure_asr_events'); static const EventChannel _eventChannel = EventChannel('azure_speech/asr_events');
bool _isInitialized = false; bool _isInitialized = false;
late final String _subscriptionKey; late final String _subscriptionKey;

2
lib/data/services/speech_impl/azure_tts_service.dart

@ -12,7 +12,7 @@ import '../tts_service.dart';
/// 提供文本转语音功能。 /// 提供文本转语音功能。
class AzureTtsService extends GetxService implements TtsService { class AzureTtsService extends GetxService implements TtsService {
static final AzureTtsService to = Get.put(AzureTtsService()); static final AzureTtsService to = Get.put(AzureTtsService());
static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_tts'); static const MethodChannel _channel = MethodChannel('azure_speech/tts');
bool _isInitialized = false; bool _isInitialized = false;
late final String _subscriptionKey; late final String _subscriptionKey;

12
lib/data/services/voice_interaction_service.dart

@ -38,7 +38,8 @@ class VoiceInteractionService extends GetxService {
// 配置信息 // 配置信息
late String _azureSpeechKey; late String _azureSpeechKey;
late String _azureSpeechRegion; late String _azureSpeechRegion;
late String _volcanoAiKey; late String _openaiApiKey;
late String _openaiBaseUrl;
// 聊天历史服务 // 聊天历史服务
late final ChatHistoryService _chatHistoryService; late final ChatHistoryService _chatHistoryService;
@ -57,15 +58,13 @@ class VoiceInteractionService extends GetxService {
void _loadConfig() { void _loadConfig() {
_azureSpeechKey = dotenv.env['AZURE_SPEECH_KEY'] ?? ''; _azureSpeechKey = dotenv.env['AZURE_SPEECH_KEY'] ?? '';
_azureSpeechRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? ''; _azureSpeechRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? '';
_volcanoAiKey = dotenv.env['VOLCANO_AI_API_KEY'] ?? ''; _openaiApiKey = dotenv.env['OPENAI_API_KEY'] ?? '';
_openaiBaseUrl = dotenv.env['OPENAI_BASE_URL'] ?? '';
if (_azureSpeechKey.isEmpty || _azureSpeechRegion.isEmpty) { if (_azureSpeechKey.isEmpty || _azureSpeechRegion.isEmpty) {
Logger.warning('未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION'); Logger.warning('未找到 Azure 语音服务配置。请在 .env 文件中设置 AZURE_SPEECH_KEY 和 AZURE_SPEECH_REGION');
} }
if (_volcanoAiKey.isEmpty) {
Logger.warning('未找到火山 AI API 密钥。请在 .env 文件中设置 VOLCANO_AI_API_KEY');
}
} }
/// 处理来自原生层的事件 /// 处理来自原生层的事件
@ -195,7 +194,8 @@ class VoiceInteractionService extends GetxService {
final result = await _channel.invokeMethod<bool>('startService', { final result = await _channel.invokeMethod<bool>('startService', {
'azure_speech_key': _azureSpeechKey, 'azure_speech_key': _azureSpeechKey,
'azure_speech_region': _azureSpeechRegion, 'azure_speech_region': _azureSpeechRegion,
'volcano_ai_api_key': _volcanoAiKey, 'openai_api_key': _openaiApiKey,
'openai_base_url': _openaiBaseUrl,
}) ?? false; }) ?? false;
if (result) { if (result) {

136
local_plugins/azure_speech/README.md

@ -0,0 +1,136 @@
# Azure Speech 插件
本插件为Flutter提供了Azure语音服务的集成,包括:
- 语音合成(TTS)
- 语音识别(ASR)
## 功能
### 语音合成(TTS)
- 支持多种语音(如中文、英文等)
- 语音参数调整(语速、音调、音量)
- 音频输出设备选择(扬声器、听筒、自动)
- SSML支持
### 语音识别(ASR)
- 一次性语音识别
- 连续语音识别
- 自动语言检测
- 音频处理优化(回音消除、噪声抑制等)
## 平台支持
- Android
- iOS
## 如何使用
### 初始化
```dart
import 'package:azure_speech/azure_speech.dart';
// 初始化TTS
await AzureSpeech.initializeTts(
'your_subscription_key',
'your_service_region',
language: 'zh-CN',
);
// 初始化ASR
await AzureSpeech.initializeAsr(
'your_subscription_key',
'your_service_region',
['zh-CN', 'en-US'],
);
```
### 语音合成
```dart
// 设置语音
await AzureSpeech.setTtsVoice('zh-CN-XiaoxiaoNeural');
// 设置语音参数
await AzureSpeech.setTtsSpeechParams(
rate: 0, // 语速 -100~100
pitch: 0, // 音调 -100~100
volume: 100, // 音量 0~100
);
// 设置音频输出设备
await AzureSpeech.setTtsAudioOutputType('AUTO'); // 'SPEAKER', 'EARPIECE', 'AUTO'
// 播放文本
await AzureSpeech.speakText('你好,世界!');
// 停止播放
await AzureSpeech.stopSpeaking();
// 检查是否正在播放
bool isSpeaking = await AzureSpeech.isSpeaking();
```
### 语音识别
```dart
// 一次性识别
final result = await AzureSpeech.recognizeOnce();
if (result['success']) {
print('识别文本: ${result['text']}');
print('识别语言: ${result['language']}');
} else {
print('识别失败: ${result['error']}');
}
// 连续识别
// 监听识别结果
AzureSpeech.asrResultStream.listen((event) {
switch (event['eventType']) {
case 'recognizing':
// 实时识别中的结果
print('识别中: ${event['text']}');
break;
case 'finalResult':
// 最终识别结果
print('最终结果: ${event['text']}');
break;
case 'error':
// 错误
print('错误: ${event['error']}');
break;
}
});
// 开始连续识别
await AzureSpeech.startContinuousRecognition();
// 停止连续识别
await AzureSpeech.stopContinuousRecognition();
// 检查连续识别是否活跃
bool isActive = await AzureSpeech.isContinuousRecognitionActive();
```
### 释放资源
```dart
// 释放所有资源
await AzureSpeech.dispose();
```
## 依赖项
本插件依赖于:
- Microsoft Cognitive Services Speech SDK
- Flutter
## 注意事项
- 使用前需要在Azure门户中创建语音服务资源,并获取订阅密钥和区域
- Android需要相关权限:RECORD_AUDIO, INTERNET等
- iOS需要在Info.plist中添加麦克风使用权限描述

65
local_plugins/azure_speech/android/build.gradle.kts

@ -0,0 +1,65 @@
import com.android.build.gradle.LibraryExtension
buildscript {
repositories {
google()
mavenCentral()
}
dependencies {
classpath("com.android.tools.build:gradle:7.3.0")
classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10")
}
}
allprojects {
repositories {
google()
mavenCentral()
}
}
plugins {
id("com.android.library")
kotlin("android")
}
// 配置android扩展
configure<LibraryExtension> {
namespace = "com.yunqiinnovation.azure_speech"
compileSdkVersion(33)
defaultConfig {
minSdk = 21
}
compileOptions {
sourceCompatibility = JavaVersion.VERSION_1_8
targetCompatibility = JavaVersion.VERSION_1_8
}
sourceSets {
getByName("main") {
manifest.srcFile("src/main/AndroidManifest.xml")
java.srcDirs("src/main/kotlin")
}
}
// 添加lint选项
lintOptions {
isCheckReleaseBuilds = false
}
}
// 显式设置Kotlin JVM目标版本
tasks.withType<org.jetbrains.kotlin.gradle.tasks.KotlinCompile> {
kotlinOptions {
jvmTarget = "1.8"
}
}
dependencies {
// 直接通过本地依赖方式添加Flutter
implementation(fileTree(mapOf("dir" to "libs", "include" to listOf("*.jar"))))
// 添加Microsoft语音SDK
implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.30.0")
}

1
local_plugins/azure_speech/android/settings.gradle.kts

@ -0,0 +1 @@
rootProject.name = "azure_speech"

6
local_plugins/azure_speech/android/src/main/AndroidManifest.xml

@ -0,0 +1,6 @@
<?xml version="1.0" encoding="utf-8"?>
<manifest xmlns:android="http://schemas.android.com/apk/res/android"
package="com.yunqiinnovation.azure_speech">
<uses-permission android:name="android.permission.INTERNET" />
<uses-permission android:name="android.permission.RECORD_AUDIO" />
</manifest>

593
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt

@ -0,0 +1,593 @@
package com.yunqiinnovation.azure_speech
import android.content.Context
import android.media.AudioAttributes
import android.media.AudioFormat
import android.media.AudioRecord
import android.media.MediaRecorder
import android.media.audiofx.AcousticEchoCanceler
import android.media.audiofx.NoiseSuppressor
import android.media.audiofx.AutomaticGainControl
import android.os.Process
import com.yunqiinnovation.azure_speech.utils.FileLogger
import com.microsoft.cognitiveservices.speech.*
import com.microsoft.cognitiveservices.speech.audio.*
import com.microsoft.cognitiveservices.speech.util.EventHandler
import java.util.concurrent.ExecutionException
import java.util.concurrent.atomic.AtomicBoolean
class AzureAsrHelper(private val context: Context) {
private var recognizer: SpeechRecognizer? = null
private var speechConfig: SpeechConfig? = null
private val TAG = "AzureAsrHelper"
private var isContinuousRecognitionActive = false
private var currentLanguage = "zh-CN"
private var subscriptionKey = ""
private var region = ""
private var isAutoDetectLanguage = false
private var supportedLanguages = arrayOf("zh-CN", "en-US")
// 是否使用回音消除 - 内部控制常量
private val useEchoCancellation = false
// 自定义音频处理相关
private var customAudioProcessor: CustomAudioProcessor? = null
private var pushStream: PushAudioInputStream? = null
private var audioConfig: AudioConfig? = null
// 初始化SDK并创建recognizer
fun initialize(subscriptionKey: String, region: String,
supportedLanguages: Array<String> = arrayOf("zh-CN", "en-US")): Boolean {
try {
FileLogger.d(TAG, "初始化 Azure 语音服务")
// 检查配置是否为空
if (subscriptionKey.isEmpty() || region.isEmpty()) {
FileLogger.e(TAG, "Azure 配置信息不完整")
return false
}
// 释放之前的资源
dispose()
this.subscriptionKey = subscriptionKey
this.region = region
// 设置语言
if (supportedLanguages.isNotEmpty()) {
this.supportedLanguages = supportedLanguages
}
// 根据支持的语言数量决定是否启用自动语言检测
this.isAutoDetectLanguage = supportedLanguages.size >= 2
// 如果只有一种语言,设置为当前语言
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) {
this.currentLanguage = supportedLanguages[0]
}
// 创建语音配置
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region)
// 设置语言配置
if (isAutoDetectLanguage) {
// 设置自动语言检测
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous")
} else {
// 设置指定的识别语言
speechConfig?.speechRecognitionLanguage = currentLanguage
}
// 创建识别器
try {
if (useEchoCancellation) {
// 如果使用回音消除,创建自定义音频输入流
setupCustomAudioProcessing()
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
} else {
recognizer = SpeechRecognizer(speechConfig, audioConfig)
}
} else {
// 使用默认麦克风输入
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
} else {
recognizer = SpeechRecognizer(speechConfig)
}
}
FileLogger.d(TAG, "Azure 语音服务初始化成功")
return true
} catch (e: Exception) {
FileLogger.e(TAG, "创建识别器失败: ${e.message}")
stopCustomAudioProcessing()
return false
}
} catch (e: Exception) {
FileLogger.e(TAG, "初始化失败: ${e.message}")
return false
}
}
// 重置 recognizer
private fun resetRecognizer(): Boolean {
try {
// 释放之前的 recognizer
recognizer?.close()
recognizer = null
// 停止当前的音频处理
stopCustomAudioProcessing()
// 使用现有配置重新创建 recognizer
if (speechConfig != null) {
if (useEchoCancellation) {
// 如果使用回音消除,创建自定义音频输入流
setupCustomAudioProcessing()
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig)
} else {
recognizer = SpeechRecognizer(speechConfig, audioConfig)
}
} else {
// 使用默认麦克风输入
if (isAutoDetectLanguage) {
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList())
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig)
} else {
recognizer = SpeechRecognizer(speechConfig)
}
}
return true
} else {
FileLogger.e(TAG, "语音配置未初始化")
return false
}
} catch (e: Exception) {
FileLogger.e(TAG, "重置识别器失败: ${e.message}")
return false
}
}
// 开始一次性语音识别
fun recognizeOnce(callback: RecognizeCallback) {
if (speechConfig == null) {
callback.onError("语音服务未初始化")
return
}
// 重置 recognizer
if (!resetRecognizer()) {
callback.onError("重置识别器失败")
return
}
try {
// 启动音频处理
startCustomAudioProcessing()
// 执行识别
val result = recognizer?.recognizeOnceAsync()?.get()
// 停止音频处理
stopCustomAudioProcessing()
if (result != null && result.reason == ResultReason.RecognizedSpeech) {
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language
callback.onResult(result.text, detectedLanguage ?: "")
} else {
callback.onError("未能识别语音")
}
} catch (e: Exception) {
// 停止音频处理
stopCustomAudioProcessing()
callback.onError("识别异常: ${e.message}")
}
}
// 开始连续语音识别
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
if (speechConfig == null) {
callback.onError("语音服务未初始化")
return false
}
if (isContinuousRecognitionActive) {
FileLogger.w(TAG, "已在进行连续识别,忽略请求")
return true
}
// 重置 recognizer
if (!resetRecognizer()) {
callback.onError("重置识别器失败")
return false
}
try {
// 启动音频处理
startCustomAudioProcessing()
// 识别中事件
recognizer?.recognizing?.addEventListener(
EventHandler<SpeechRecognitionEventArgs> { _, event ->
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language
FileLogger.d(TAG, "识别中: ${event.result.text}, 语言: $detectedLanguage")
callback.onRecognizing(event.result.text, detectedLanguage ?: "")
}
)
// 识别完成事件
recognizer?.recognized?.addEventListener(
EventHandler<SpeechRecognitionEventArgs> { _, event ->
if (event.result.reason == ResultReason.RecognizedSpeech) {
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language
FileLogger.d(TAG, "识别完成: ${event.result.text}, 语言: $detectedLanguage")
callback.onResult(event.result.text, detectedLanguage ?: "")
}
}
)
// 会话开始事件
recognizer?.sessionStarted?.addEventListener(
EventHandler<SessionEventArgs> { _, _ ->
FileLogger.d(TAG, "识别会话已开始")
callback.onSessionStarted()
callback.onSuccess("开始识别") // 兼容旧接口
}
)
// 会话结束事件
recognizer?.sessionStopped?.addEventListener(
EventHandler<SessionEventArgs> { _, _ ->
FileLogger.d(TAG, "识别会话已结束")
isContinuousRecognitionActive = false
stopCustomAudioProcessing()
callback.onSessionStopped()
}
)
// 取消事件
recognizer?.canceled?.addEventListener(
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event ->
val errorDetails = try {
event.errorDetails ?: "未知错误"
} catch (e: Exception) {
"未知错误"
}
val reason = event.reason.toString()
FileLogger.e(TAG, "识别取消: $errorDetails")
isContinuousRecognitionActive = false
stopCustomAudioProcessing()
callback.onCanceled(reason, errorDetails)
callback.onError("识别取消: $errorDetails") // 兼容旧接口
}
)
// 开始连续识别
recognizer?.startContinuousRecognitionAsync()
isContinuousRecognitionActive = true
FileLogger.d(TAG, "连续识别已启动")
return true
} catch (e: Exception) {
// 停止音频处理
stopCustomAudioProcessing()
FileLogger.e(TAG, "开始连续识别失败: ${e.message}")
e.printStackTrace()
callback.onError("开始连续识别失败: ${e.message}")
return false
}
}
// 停止连续语音识别
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean {
if (speechConfig == null) {
callback.onError("语音服务未初始化")
return false
}
if (!isContinuousRecognitionActive) {
FileLogger.d(TAG, "未进行连续识别,忽略停止请求")
return true
}
try {
FileLogger.d(TAG, "停止连续语音识别")
if (recognizer == null) {
if (isContinuousRecognitionActive) {
FileLogger.w(TAG, "识别器为空,但状态显示活跃")
}
isContinuousRecognitionActive = false
return true
}
// 停止连续识别
val future = recognizer?.stopContinuousRecognitionAsync()
future?.get()
// 停止音频处理
stopCustomAudioProcessing()
isContinuousRecognitionActive = false
FileLogger.d(TAG, "连续识别已停止")
callback.onSuccess("连续识别已停止")
return true
} catch (e: Exception) {
// 强制重置状态
isContinuousRecognitionActive = false
FileLogger.e(TAG, "停止连续识别失败: ${e.message}")
e.printStackTrace()
callback.onError("停止连续识别失败: ${e.message}")
// 停止音频处理
stopCustomAudioProcessing()
// 尝试强制关闭识别器
try {
recognizer?.close()
recognizer = null
} catch (ex: Exception) {
FileLogger.e(TAG, "关闭识别器失败: ${ex.message}")
}
return false
}
}
// 释放资源
fun dispose() {
try {
// 如果正在进行连续识别,先停止
if (isContinuousRecognitionActive) {
recognizer?.stopContinuousRecognitionAsync()?.get()
isContinuousRecognitionActive = false
}
// 停止音频处理
stopCustomAudioProcessing()
// 释放recognizer
recognizer?.close()
recognizer = null
// 释放speechConfig
speechConfig?.close()
speechConfig = null
FileLogger.d(TAG, "资源已释放")
} catch (e: Exception) {
FileLogger.e(TAG, "释放资源失败: ${e.message}")
// 确保状态被重置
isContinuousRecognitionActive = false
customAudioProcessor = null
pushStream = null
audioConfig = null
recognizer = null
speechConfig = null
}
}
// 检查连续识别是否处于活跃状态
fun isContinuousRecognitionActive(): Boolean {
return isContinuousRecognitionActive
}
// 设置自定义音频处理
private fun setupCustomAudioProcessing() {
try {
// 创建音频推送流
pushStream = PushAudioInputStream.create()
// 创建音频配置
audioConfig = AudioConfig.fromStreamInput(pushStream)
// 创建自定义音频处理器
customAudioProcessor = CustomAudioProcessor(pushStream!!)
} catch (e: Exception) {
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}")
e.printStackTrace()
}
}
// 启动自定义音频处理
private fun startCustomAudioProcessing() {
if (useEchoCancellation && customAudioProcessor != null) {
try {
customAudioProcessor?.startProcessing()
} catch (e: Exception) {
FileLogger.e(TAG, "启动音频处理器失败")
e.printStackTrace()
}
}
}
// 停止自定义音频处理
private fun stopCustomAudioProcessing() {
if (customAudioProcessor != null) {
try {
customAudioProcessor?.stopProcessing()
customAudioProcessor = null
} catch (e: Exception) {
FileLogger.e(TAG, "停止音频处理器失败: ${e.message}")
e.printStackTrace()
}
}
}
// 修改认证取消事件处理代码
private fun setupCancelledEventHandler(callback: ContinuousRecognizeCallback) {
recognizer?.canceled?.addEventListener(
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event ->
val errorDetails = try {
event.errorDetails ?: "未知错误"
} catch (e: Exception) {
"未知错误"
}
val reason = event.reason.toString()
FileLogger.e(TAG, "识别取消: $errorDetails")
isContinuousRecognitionActive = false
stopCustomAudioProcessing()
callback.onCanceled(reason, errorDetails)
callback.onError("识别取消: $errorDetails") // 兼容旧接口
}
)
}
// 自定义音频处理器
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream) {
private val isProcessing = AtomicBoolean(false)
private var audioRecord: AudioRecord? = null
private var echoCanceler: AcousticEchoCanceler? = null
private var noiseSuppressor: NoiseSuppressor? = null
private var automaticGainControl: AutomaticGainControl? = null
// 音频配置
private val sampleRate = 16000 // 16kHz,适合语音识别
private val channelConfig = AudioFormat.CHANNEL_IN_MONO
private val audioFormat = AudioFormat.ENCODING_PCM_16BIT
// 计算最小 buffer 大小
private val bufferSize = AudioRecord.getMinBufferSize(
sampleRate, channelConfig, audioFormat
)
// 启动音频处理
fun startProcessing() {
if (isProcessing.get()) return
// 创建录音对象
try {
audioRecord = AudioRecord(
MediaRecorder.AudioSource.VOICE_RECOGNITION,
sampleRate,
channelConfig,
audioFormat,
bufferSize * 2 // 使用更大的缓冲区以确保不会丢失数据
)
// 创建音频处理效果
if (AcousticEchoCanceler.isAvailable()) {
echoCanceler = AcousticEchoCanceler.create(audioRecord!!.audioSessionId)
echoCanceler?.enabled = true
}
if (NoiseSuppressor.isAvailable()) {
noiseSuppressor = NoiseSuppressor.create(audioRecord!!.audioSessionId)
noiseSuppressor?.enabled = true
}
if (AutomaticGainControl.isAvailable()) {
automaticGainControl = AutomaticGainControl.create(audioRecord!!.audioSessionId)
automaticGainControl?.enabled = true
}
// 开始录音
audioRecord?.startRecording()
// 处理线程
Thread {
android.os.Process.setThreadPriority(Process.THREAD_PRIORITY_AUDIO)
processAudio()
}.start()
isProcessing.set(true)
FileLogger.d(TAG, "音频处理已启动")
} catch (e: Exception) {
FileLogger.e(TAG, "创建音频处理器失败: ${e.message}")
releaseAudioResources()
throw e
}
}
// 停止音频处理
fun stopProcessing() {
if (!isProcessing.get()) return
isProcessing.set(false)
releaseAudioResources()
FileLogger.d(TAG, "音频处理已停止")
}
// 释放音频资源
private fun releaseAudioResources() {
try {
audioRecord?.stop()
echoCanceler?.release()
echoCanceler = null
noiseSuppressor?.release()
noiseSuppressor = null
automaticGainControl?.release()
automaticGainControl = null
audioRecord?.release()
audioRecord = null
} catch (e: Exception) {
FileLogger.e(TAG, "释放音频资源失败: ${e.message}")
}
}
// 音频处理线程
private fun processAudio() {
// 设置线程优先级
try {
Process.setThreadPriority(Process.THREAD_PRIORITY_URGENT_AUDIO)
} catch (e: Exception) {
FileLogger.e(TAG, "设置线程优先级失败")
}
val buffer = ByteArray(bufferSize)
while (isProcessing.get()) {
try {
val readSize = audioRecord?.read(buffer, 0, buffer.size) ?: -1
if (readSize > 0) {
// 修复: 只传入buffer,不传readSize
// 创建新的byte数组,只包含读取到的数据
val audioData = buffer.copyOfRange(0, readSize)
pushStream.write(audioData)
}
// 适当休眠,避免占用过多 CPU
Thread.sleep(5)
} catch (e: Exception) {
if (isProcessing.get()) {
FileLogger.e(TAG, "处理音频数据异常: ${e.message}")
}
break
}
}
}
}
// 一次性识别回调接口
interface RecognizeCallback {
fun onResult(text: String, detectedLanguage: String)
fun onError(error: String)
}
// 连续识别回调接口
interface ContinuousRecognizeCallback {
fun onResult(text: String, detectedLanguage: String)
fun onRecognizing(recognizing: String, detectedLanguage: String)
fun onSessionStarted()
fun onSessionStopped()
fun onCanceled(reason: String, errorDetails: String)
fun onError(error: String)
// 兼容旧版本的接口
fun onSuccess(message: String) {}
}
}

297
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureSpeechPlugin.kt

@ -0,0 +1,297 @@
package com.yunqiinnovation.azure_speech
import android.content.Context
import android.app.Activity
import android.os.Handler
import android.os.Looper
import androidx.annotation.NonNull
import com.yunqiinnovation.azure_speech.utils.FileLogger
import io.flutter.embedding.engine.plugins.FlutterPlugin
import io.flutter.plugin.common.MethodCall
import io.flutter.plugin.common.MethodChannel
import io.flutter.plugin.common.MethodChannel.MethodCallHandler
import io.flutter.plugin.common.MethodChannel.Result
import io.flutter.plugin.common.EventChannel
/** AzureSpeechPlugin */
class AzureSpeechPlugin: FlutterPlugin {
private val TAG = "AzureSpeechPlugin"
private lateinit var context: Context
private val mainHandler = Handler(Looper.getMainLooper())
// ASR相关
private lateinit var asrChannel : MethodChannel
private lateinit var asrEventChannel: EventChannel
private var asrEventSink: EventChannel.EventSink? = null
private lateinit var azureAsrHelper: AzureAsrHelper
// TTS相关
private lateinit var ttsChannel : MethodChannel
private lateinit var azureTtsHelper: AzureTtsHelper
// ASR 事件发送方法
private fun sendAsrEvent(event: Map<String, Any>) {
FileLogger.d(TAG, "发送ASR事件: $event")
if (asrEventSink == null) {
FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好")
return
}
mainHandler.post {
try {
asrEventSink?.success(event)
FileLogger.d(TAG, "ASR事件发送成功")
} catch (e: Exception) {
FileLogger.e(TAG, "发送ASR事件失败: ${e.message}")
}
}
}
override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) {
context = flutterPluginBinding.applicationContext
// 初始化ASR通道
asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr")
asrChannel.setMethodCallHandler(AsrMethodHandler())
// 初始化TTS通道
ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/tts")
ttsChannel.setMethodCallHandler(TtsMethodHandler())
// 初始化ASR事件通道
asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events")
asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler {
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) {
asrEventSink = events
}
override fun onCancel(arguments: Any?) {
asrEventSink = null
}
})
// 初始化Azure语音服务
azureTtsHelper = AzureTtsHelper(context)
azureAsrHelper = AzureAsrHelper(context)
}
// ASR方法处理器
inner class AsrMethodHandler : MethodCallHandler {
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) {
when (call.method) {
"initialize" -> {
val subscriptionKey = call.argument<String>("subscriptionKey") ?: ""
val region = call.argument<String>("region") ?: ""
val supportedLanguages = call.argument<List<String>>("supportedLanguages") ?: listOf("zh-CN")
try {
val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages.toTypedArray())
result.success(success)
} catch (e: Exception) {
result.error("INITIALIZATION_ERROR", e.message, null)
}
}
"recognizeOnce" -> {
azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {
mainHandler.post {
result.success(mapOf(
"text" to text,
"detectedLanguage" to detectedLanguage
))
}
}
override fun onError(error: String) {
mainHandler.post {
result.error("RECOGNITION_ERROR", error, null)
}
}
})
}
"startContinuousRecognition" -> {
// 确保事件通道已准备好
if (asrEventSink == null) {
result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null)
return
}
val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {
sendAsrEvent(mapOf(
"type" to "result",
"text" to text,
"detectedLanguage" to detectedLanguage
))
}
override fun onRecognizing(recognizing: String, detectedLanguage: String) {
sendAsrEvent(mapOf(
"type" to "recognizing",
"text" to recognizing,
"detectedLanguage" to detectedLanguage
))
}
override fun onSessionStarted() {
sendAsrEvent(mapOf("type" to "sessionStarted"))
}
override fun onSessionStopped() {
sendAsrEvent(mapOf("type" to "sessionStopped"))
}
override fun onCanceled(reason: String, errorDetails: String) {
sendAsrEvent(mapOf(
"type" to "canceled",
"reason" to reason,
"errorDetails" to errorDetails
))
}
override fun onError(error: String) {
sendAsrEvent(mapOf("type" to "error", "message" to error))
}
override fun onSuccess(message: String) {
sendAsrEvent(mapOf("type" to "success", "message" to message))
}
})
result.success(success)
}
"stopContinuousRecognition" -> {
try {
if (!azureAsrHelper.isContinuousRecognitionActive()) {
result.success(true)
return
}
val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback {
override fun onResult(text: String, detectedLanguage: String) {}
override fun onRecognizing(recognizing: String, detectedLanguage: String) {}
override fun onSessionStarted() {}
override fun onSessionStopped() {}
override fun onCanceled(reason: String, errorDetails: String) {}
override fun onError(error: String) {
mainHandler.post {
result.error("STOP_ERROR", error, null)
}
}
override fun onSuccess(message: String) {}
})
result.success(success)
} catch (e: Exception) {
result.error("STOP_ERROR", e.message, null)
}
}
"isContinuousRecognitionActive" -> {
result.success(azureAsrHelper.isContinuousRecognitionActive())
}
"dispose" -> {
azureAsrHelper.dispose()
result.success(true)
}
else -> {
result.notImplemented()
}
}
}
}
// TTS方法处理器
inner class TtsMethodHandler : MethodCallHandler {
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) {
when (call.method) {
"initialize" -> {
val subscriptionKey = call.argument<String>("subscriptionKey") ?: ""
val region = call.argument<String>("region") ?: ""
val language = call.argument<String>("language") ?: "zh-CN"
val success = azureTtsHelper.initialize(subscriptionKey, region, language)
result.success(success)
}
"setVoice" -> {
val voiceName = call.argument<String>("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null)
result.success(azureTtsHelper.setVoice(voiceName))
}
"setSpeechParams" -> {
val rate = call.argument<Int>("rate") ?: 0
val pitch = call.argument<Int>("pitch") ?: 0
val volume = call.argument<Int>("volume") ?: 100
result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume))
}
"setAudioOutputType" -> {
val outputTypeStr = call.argument<String>("outputType") ?: "speaker"
val outputType = when (outputTypeStr.lowercase()) {
"speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER
"earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE
"auto" -> AzureTtsHelper.AudioOutputType.AUTO
else -> AzureTtsHelper.AudioOutputType.SPEAKER
}
result.success(azureTtsHelper.setAudioOutputType(outputType))
}
"speakText" -> {
val text = call.argument<String>("text") ?: return result.error("INVALID_ARGUMENTS", "文本不能为空", null)
azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback {
override fun onSuccess(message: String) {
mainHandler.post {
result.success(true)
}
}
override fun onError(error: String) {
mainHandler.post {
result.error("SPEAK_ERROR", error, null)
}
}
})
}
"speakSsml" -> {
val ssml = call.argument<String>("ssml") ?: return result.error("INVALID_ARGUMENTS", "SSML不能为空", null)
azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback {
override fun onSuccess(message: String) {
mainHandler.post {
result.success(true)
}
}
override fun onError(error: String) {
mainHandler.post {
result.error("SPEAK_ERROR", error, null)
}
}
})
}
"stopSpeaking" -> {
result.success(azureTtsHelper.stopSpeaking())
}
"isSpeaking" -> {
result.success(azureTtsHelper.isSpeaking())
}
"dispose" -> {
azureTtsHelper.dispose()
result.success(true)
}
else -> {
result.notImplemented()
}
}
}
}
override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) {
asrChannel.setMethodCallHandler(null)
ttsChannel.setMethodCallHandler(null)
asrEventChannel.setStreamHandler(null)
try {
azureTtsHelper.dispose()
azureAsrHelper.dispose()
} catch (e: Exception) {
FileLogger.e(TAG, "Dispose resources error: ${e.message}")
}
}
}

12
android/app/src/main/kotlin/com/example/deep_voice/AzureTtsHelper.kt → local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureTtsHelper.kt

@ -1,11 +1,11 @@
package com.yunqiinnovation.deepsound package com.yunqiinnovation.azure_speech
import android.content.Context import android.content.Context
import android.media.AudioAttributes import android.media.AudioAttributes
import android.media.AudioDeviceInfo import android.media.AudioDeviceInfo
import android.media.AudioManager import android.media.AudioManager
import android.os.Build import android.os.Build
import com.yunqiinnovation.deepsound.core.utils.FileLogger import com.yunqiinnovation.azure_speech.utils.FileLogger
import com.microsoft.cognitiveservices.speech.* import com.microsoft.cognitiveservices.speech.*
import com.microsoft.cognitiveservices.speech.audio.* import com.microsoft.cognitiveservices.speech.audio.*
import java.util.concurrent.Future import java.util.concurrent.Future
@ -50,16 +50,16 @@ class AzureTtsHelper(private val context: Context) {
* 初始化 TTS 引擎 * 初始化 TTS 引擎
* *
* @param subscriptionKey Azure 语音服务订阅密钥 * @param subscriptionKey Azure 语音服务订阅密钥
* @param serviceRegion Azure 语音服务区域 * @param region Azure 语音服务区域
* @param language 可选,默认语言,默认为 "zh-CN" * @param language 可选,默认语言,默认为 "zh-CN"
*/ */
fun initialize(subscriptionKey: String, serviceRegion: String, language: String = "zh-CN"): Boolean { fun initialize(subscriptionKey: String, region: String, language: String = "zh-CN"): Boolean {
try { try {
audioManager = context.getSystemService(Context.AUDIO_SERVICE) as AudioManager audioManager = context.getSystemService(Context.AUDIO_SERVICE) as AudioManager
// 创建语音配置 // 创建语音配置
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region)
// 设置语音合成输出格式为高质量音频 // 设置语音合成输出格式为高质量音频
speechConfig?.setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff24Khz16BitMonoPcm) speechConfig?.setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff24Khz16BitMonoPcm)
@ -96,7 +96,7 @@ class AzureTtsHelper(private val context: Context) {
return true return true
} catch (e: Exception) { } catch (e: Exception) {
FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${serviceRegion}") FileLogger.e(TAG, "TTS 引擎初始化失败: ${e.message}, ${region}")
e.printStackTrace() e.printStackTrace()
return false return false
} }

45
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/utils/FileLogger.kt

@ -0,0 +1,45 @@
package com.yunqiinnovation.azure_speech.utils
import android.util.Log
/**
* 简单的文件日志工具类
*/
object FileLogger {
private const val TAG_PREFIX = "AzureSpeech_"
/**
* 记录调试信息
*/
fun d(tag: String, message: String) {
Log.d("$TAG_PREFIX$tag", message)
}
/**
* 记录信息
*/
fun i(tag: String, message: String) {
Log.i("$TAG_PREFIX$tag", message)
}
/**
* 记录警告信息
*/
fun w(tag: String, message: String) {
Log.w("$TAG_PREFIX$tag", message)
}
/**
* 记录错误信息
*/
fun e(tag: String, message: String) {
Log.e("$TAG_PREFIX$tag", message)
}
/**
* 记录异常
*/
fun e(tag: String, message: String, throwable: Throwable) {
Log.e("$TAG_PREFIX$tag", message, throwable)
}
}

461
local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift

@ -0,0 +1,461 @@
import Foundation
import AVFoundation
import MicrosoftCognitiveServicesSpeech
/// Azure 语音识别辅助类
class AzureAsrHelper: NSObject {
private var recognizer: SPXSpeechRecognizer?
private var speechConfig: SPXSpeechConfig?
private var audioConfig: SPXAudioConfig?
private var initialized = false
private var isContinuousRecognitionActive = false
private var currentLanguage = "zh-CN"
private var subscriptionKey = ""
private var serviceRegion = ""
private var isAutoDetectLanguage = false
private var supportedLanguages = ["zh-CN", "en-US"]
// 音频会话管理
private let audioSession = AVAudioSession.sharedInstance()
// 事件回调
private var eventHandler: (([String: Any]) -> Void)?
/// 设置事件处理器
///
/// - Parameter handler: 事件处理回调
func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) {
self.eventHandler = handler
}
/// 初始化语音识别服务
///
/// - Parameters:
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
/// - serviceRegion: Azure 语音服务区域
/// - supportedLanguages: 支持的语言列表,默认为 ["zh-CN", "en-US"]
/// - Returns: 是否初始化成功
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String] = ["zh-CN", "en-US"]) -> Bool {
print("[AzureAsrHelper] 初始化 Azure 语音服务")
// 检查配置是否为空
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
return false
}
// 释放之前的资源
dispose()
// 保存配置
self.subscriptionKey = speechSubscriptionKey
self.serviceRegion = serviceRegion
// 设置语言
if supportedLanguages.isEmpty {
print("[AzureAsrHelper] 警告: 传入的支持语言列表为空,将使用默认语言")
} else {
self.supportedLanguages = supportedLanguages
}
// 根据支持的语言数量决定是否启用自动语言检测
self.isAutoDetectLanguage = supportedLanguages.count >= 2
// 如果只有一种语言,设置为当前语言
if !isAutoDetectLanguage && !supportedLanguages.isEmpty {
self.currentLanguage = supportedLanguages[0]
}
// 创建语音配置
do {
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置语言配置
if isAutoDetectLanguage {
// 设置自动语言检测
try speechConfig?.setPropertyTo("Continuous", byId: SPXPropertyId.SpeechServiceConnection_LanguageIdMode)
} else {
// 设置指定的识别语言
speechConfig?.speechRecognitionLanguage = currentLanguage
}
// 创建音频配置 - 使用默认麦克风
audioConfig = SPXAudioConfig.default()
// 创建识别器
if isAutoDetectLanguage {
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages)
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!)
} else {
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
}
// 配置音频会话
try configureAudioSession()
initialized = true
print("[AzureAsrHelper] Azure 语音服务初始化成功")
return true
} catch {
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)")
return false
}
}
/// 配置音频会话
private func configureAudioSession() throws {
print("[AzureAsrHelper] 开始配置音频会话...")
do {
// 设置音频会话类别和模式
try audioSession.setCategory(.record, mode: .measurement, options: [.duckOthers, .allowBluetooth])
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
} catch {
print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败")
throw error
}
}
/// 一次性语音识别
///
/// - Parameter completion: 完成回调,返回是否成功、识别文本、识别语言和可能的错误信息
func recognizeOnce(completion: @escaping (Bool, String?, String?, String?) -> Void) {
if !initialized {
completion(false, nil, nil, "语音服务未初始化")
return
}
// 重置 recognizer
if !resetRecognizer() {
completion(false, nil, nil, "重置识别器失败")
return
}
do {
// 激活音频会话
try audioSession.setActive(true)
// 添加识别事件处理
recognizer?.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == .recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
completion(true, event.result.text, detectedLanguage, nil)
}
}
recognizer?.addRecognizingEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == .recognizingSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)")
}
}
// 添加会话事件处理
recognizer?.addSessionStartedEventHandler { _, _ in
print("[AzureAsrHelper] 识别会话已开始")
}
recognizer?.addSessionStoppedEventHandler { _, _ in
print("[AzureAsrHelper] 识别会话已结束")
}
// 添加取消事件处理
recognizer?.addCanceledEventHandler { _, event in
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) {
let errorDetails = cancellationDetails.errorDetails ?? "未知错误"
print("[AzureAsrHelper] 识别取消: \(errorDetails)")
completion(false, nil, nil, "识别取消: \(errorDetails)")
}
}
// 执行识别
let result = try recognizer?.recognizeOnceAsync().get()
if result?.reason != .recognizedSpeech {
completion(false, nil, nil, "未能识别语音")
}
} catch {
completion(false, nil, nil, "识别异常: \(error.localizedDescription)")
}
}
/// 重置识别器
///
/// - Returns: 是否重置成功
private func resetRecognizer() -> Bool {
if !initialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
return false
}
// 检查配置是否为空
if subscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
return false
}
do {
// 释放之前的 recognizer
recognizer = nil
// 创建音频配置 - 使用默认麦克风
audioConfig = SPXAudioConfig.default()
// 重新创建识别器
if isAutoDetectLanguage {
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages)
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!)
} else {
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!)
}
return true
} catch {
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)")
return false
}
}
/// 获取检测到的语言
///
/// - Parameter result: 识别结果
/// - Returns: 检测到的语言代码
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
if isAutoDetectLanguage {
do {
if let autoDetectResult = try SPXAutoDetectSourceLanguageResult(fromRecognitionResult: result) {
return autoDetectResult.language
}
return ""
} catch {
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)")
return ""
}
} else {
return currentLanguage
}
}
/// 开始连续语音识别
///
/// - Returns: 是否成功启动连续识别
func startContinuousRecognition() -> Bool {
if !initialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
return false
}
// 检查配置是否为空
if subscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureAsrHelper] 错误: Azure 配置信息不完整")
return false
}
// 如果已经在进行连续识别,直接返回
if isContinuousRecognitionActive {
print("[AzureAsrHelper] 已经在进行连续识别中,忽略请求")
return true
}
// 重置 recognizer
if !resetRecognizer() {
print("[AzureAsrHelper] 尝试重新创建识别器...")
return false
}
do {
// 激活音频会话
try audioSession.setActive(true)
// 添加识别事件处理
recognizer?.addRecognizedEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == .recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
let eventData: [String: Any] = [
"eventType": "finalResult",
"text": event.result.text ?? "",
"language": detectedLanguage
]
self.eventHandler?(eventData)
}
}
// 识别中事件
recognizer?.addRecognizingEventHandler { [weak self] _, event in
guard let self = self else { return }
if event.result.reason == .recognizingSpeech {
let detectedLanguage = self.getDetectedLanguage(from: event.result)
let eventData: [String: Any] = [
"eventType": "recognizing",
"text": event.result.text ?? "",
"language": detectedLanguage
]
self.eventHandler?(eventData)
}
}
// 会话事件
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in
guard let self = self else { return }
let eventData: [String: Any] = [
"eventType": "sessionStarted"
]
self.eventHandler?(eventData)
}
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in
guard let self = self else { return }
self.isContinuousRecognitionActive = false
let eventData: [String: Any] = [
"eventType": "sessionStopped"
]
self.eventHandler?(eventData)
}
// 取消事件
recognizer?.addCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
self.isContinuousRecognitionActive = false
var errorMessage = "未知错误"
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) {
errorMessage = cancellationDetails.errorDetails ?? "未知错误"
}
let eventData: [String: Any] = [
"eventType": "error",
"error": "识别取消: \(errorMessage)"
]
self.eventHandler?(eventData)
}
// 开始连续识别
try recognizer?.startContinuousRecognition()
isContinuousRecognitionActive = true
print("[AzureAsrHelper] 连续识别已启动")
return true
} catch {
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)")
return false
}
}
/// 停止连续语音识别
///
/// - Returns: 是否成功停止连续识别
func stopContinuousRecognition() -> Bool {
if !initialized {
print("[AzureAsrHelper] 错误: 语音服务未初始化")
return false
}
if !isContinuousRecognitionActive {
print("[AzureAsrHelper] 未进行连续识别,忽略停止请求")
return true
}
do {
print("[AzureAsrHelper] 停止连续语音识别")
if recognizer == nil {
if isContinuousRecognitionActive {
print("[AzureAsrHelper] 警告: 识别器为空,但状态显示活跃")
}
isContinuousRecognitionActive = false
// 通知停止成功
let eventData: [String: Any] = [
"eventType": "success",
"message": "连续识别已停止"
]
eventHandler?(eventData)
return true
}
// 停止连续识别
try recognizer?.stopContinuousRecognition()
// 延迟一点时间确保处理完成
DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { [weak self] in
guard let self = self else { return }
// 重置状态
self.isContinuousRecognitionActive = false
// 恢复音频会话
do {
try self.audioSession.setActive(false, options: .notifyOthersOnDeactivation)
} catch {
// 忽略错误
}
print("[AzureAsrHelper] 连续识别已停止")
// 通知停止成功
let eventData: [String: Any] = [
"eventType": "success",
"message": "连续识别已停止"
]
self.eventHandler?(eventData)
}
return true
} catch {
// 强制重置状态
isContinuousRecognitionActive = false
print("[AzureAsrHelper] 警告: 停止连续识别失败: \(error.localizedDescription)")
// 通知停止失败,但仍然视为处理完成
let eventData: [String: Any] = [
"eventType": "success",
"message": "连续识别已停止(但有错误)"
]
eventHandler?(eventData)
return false
}
}
/// 释放资源
func dispose() {
// 如果正在进行连续识别,先停止
if isContinuousRecognitionActive {
_ = stopContinuousRecognition()
}
// 恢复音频会话
do {
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
} catch {
// 忽略错误
}
// 释放资源
recognizer = nil
speechConfig = nil
audioConfig = nil
initialized = false
isContinuousRecognitionActive = false
print("[AzureAsrHelper] 资源已释放")
}
/// 检查连续识别是否处于活跃状态
///
/// - Returns: 是否正在进行连续识别
func isContinuousRecognitionActive() -> Bool {
return isContinuousRecognitionActive
}
}

182
local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift

@ -0,0 +1,182 @@
import Flutter
import UIKit
public class AzureSpeechPlugin: NSObject, FlutterPlugin {
private var ttsHelper: AzureTtsHelper?
private var asrHelper: AzureAsrHelper?
private var eventSink: FlutterEventSink?
public static func register(with registrar: FlutterPluginRegistrar) {
let channel = FlutterMethodChannel(name: "azure_speech", binaryMessenger: registrar.messenger())
let instance = AzureSpeechPlugin()
registrar.addMethodCallDelegate(instance, channel: channel)
// 初始化事件通道
let eventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger())
eventChannel.setStreamHandler(AsrStreamHandler(instance: instance))
}
override init() {
super.init()
ttsHelper = AzureTtsHelper()
asrHelper = AzureAsrHelper()
// 设置ASR事件处理
asrHelper?.setEventHandler { [weak self] event in
self?.handleAsrEvent(event)
}
}
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
switch call.method {
// TTS相关方法
case "initializeTts":
guard let args = call.arguments as? [String: Any],
let subscriptionKey = args["subscriptionKey"] as? String,
let serviceRegion = args["serviceRegion"] as? String else {
result(false)
return
}
let language = args["language"] as? String ?? "zh-CN"
let success = ttsHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, language: language) ?? false
result(success)
case "setTtsVoice":
guard let args = call.arguments as? [String: Any],
let voiceName = args["voiceName"] as? String else {
result(false)
return
}
let success = ttsHelper?.setVoice(voiceName: voiceName) ?? false
result(success)
case "setTtsSpeechParams":
guard let args = call.arguments as? [String: Any],
let rate = args["rate"] as? Int,
let pitch = args["pitch"] as? Int,
let volume = args["volume"] as? Int else {
result(false)
return
}
let success = ttsHelper?.setSpeechParams(rate: rate, pitch: pitch, volume: volume) ?? false
result(success)
case "speakText":
guard let args = call.arguments as? [String: Any],
let text = args["text"] as? String else {
result(false)
return
}
ttsHelper?.speakText(text: text) { success, _ in
result(success)
}
case "stopSpeaking":
let success = ttsHelper?.stopSpeaking() ?? false
result(success)
case "isSpeaking":
let speaking = ttsHelper?.isSpeaking() ?? false
result(speaking)
case "setTtsAudioOutputType":
guard let args = call.arguments as? [String: Any],
let outputType = args["outputType"] as? String else {
result(false)
return
}
var type: AzureTtsHelper.AudioOutputType = .auto
switch outputType.uppercased() {
case "SPEAKER":
type = .speaker
case "EARPIECE":
type = .earpiece
default:
type = .auto
}
let success = ttsHelper?.setAudioOutputType(outputType: type) ?? false
result(success)
// ASR相关方法
case "initializeAsr":
guard let args = call.arguments as? [String: Any],
let subscriptionKey = args["subscriptionKey"] as? String,
let serviceRegion = args["serviceRegion"] as? String,
let supportedLanguages = args["supportedLanguages"] as? [String] else {
result(false)
return
}
let success = asrHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, supportedLanguages: supportedLanguages) ?? false
result(success)
case "recognizeOnce":
asrHelper?.recognizeOnce { success, text, language, error in
var resultMap: [String: Any] = ["success": success]
if success {
resultMap["text"] = text
resultMap["language"] = language
} else {
resultMap["error"] = error
}
result(resultMap)
}
case "startContinuousRecognition":
let success = asrHelper?.startContinuousRecognition() ?? false
result(success)
case "stopContinuousRecognition":
let success = asrHelper?.stopContinuousRecognition() ?? false
result(success)
case "isContinuousRecognitionActive":
let isActive = asrHelper?.isContinuousRecognitionActive() ?? false
result(isActive)
case "dispose":
ttsHelper?.dispose()
asrHelper?.dispose()
result(nil)
default:
result(FlutterMethodNotImplemented)
}
}
// 设置事件接收器
func setEventSink(_ sink: FlutterEventSink?) {
self.eventSink = sink
}
// 处理ASR事件
private func handleAsrEvent(_ event: [String: Any]) {
self.eventSink?(event)
}
}
// ASR事件流处理器
class AsrStreamHandler: NSObject, FlutterStreamHandler {
private weak var plugin: AzureSpeechPlugin?
init(instance: AzureSpeechPlugin) {
self.plugin = instance
super.init()
}
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? {
plugin?.setEventSink(events)
return nil
}
func onCancel(withArguments arguments: Any?) -> FlutterError? {
plugin?.setEventSink(nil)
return nil
}
}

330
local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift

@ -0,0 +1,330 @@
import Foundation
import AVFoundation
import MicrosoftCognitiveServicesSpeech
/// Azure 语音合成辅助类
class AzureTtsHelper: NSObject {
private var synthesizer: SPXSpeechSynthesizer?
private var speechConfig: SPXSpeechConfig?
private var audioConfig: SPXAudioConfig?
private var initialized = false
private var speaking = false
// 音频输出类型
enum AudioOutputType {
case speaker // 扬声器
case earpiece // 听筒
case auto // 自动选择
}
// 当前设置
private var currentVoiceName = "zh-CN-XiaoxiaoNeural"
private var currentSpeechRate = 0
private var currentPitch = 0
private var currentVolume = 100
private var currentAudioOutputType: AudioOutputType = .auto
// 音频会话管理
private let audioSession = AVAudioSession.sharedInstance()
/// 初始化 TTS 引擎
///
/// - Parameters:
/// - speechSubscriptionKey: Azure 语音服务订阅密钥
/// - serviceRegion: Azure 语音服务区域
/// - language: 可选,默认语言,默认为 "zh-CN"
/// - Returns: 是否初始化成功
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool {
print("[AzureTtsHelper] 初始化 Azure 语音服务")
// 检查配置是否为空
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty {
print("[AzureTtsHelper] 错误: Azure 配置信息不完整")
return false
}
// 释放之前的资源
dispose()
do {
// 创建语音配置
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion)
// 设置语音合成输出格式为高质量音频
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm)
// 设置默认语言
speechConfig?.setSpeechSynthesisLanguage(language)
// 设置默认语音
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName)
// 创建音频配置 - 使用默认扬声器
audioConfig = SPXAudioConfig.default()
// 创建语音合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
initialized = true
// 设置默认音频输出类型为自动
setAudioOutputType(outputType: .auto)
print("[AzureTtsHelper] TTS 引擎初始化成功")
return true
} catch {
print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)")
return false
}
}
/// 设置音频输出设备类型
///
/// - Parameter outputType: 音频输出设备类型
/// - Returns: 是否设置成功
func setAudioOutputType(outputType: AudioOutputType) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
do {
currentAudioOutputType = outputType
switch outputType {
case .speaker:
// 使用扬声器
try audioSession.setCategory(.playback, mode: .default)
try audioSession.overrideOutputAudioPort(.speaker)
print("[AzureTtsHelper] 已设置音频输出设备为扬声器")
case .earpiece:
// 使用听筒
try audioSession.setCategory(.playback, mode: .voiceChat)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
case .auto:
// 检查是否有耳机连接
let outputs = audioSession.currentRoute.outputs
let hasHeadphones = outputs.contains { output in
return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP
}
if hasHeadphones {
// 有耳机,使用耳机
try audioSession.setCategory(.playback, mode: .default)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为耳机")
} else {
// 无耳机,使用听筒
try audioSession.setCategory(.playback, mode: .voiceChat)
try audioSession.overrideOutputAudioPort(.none)
print("[AzureTtsHelper] 已设置音频输出设备为听筒")
}
}
try audioSession.setActive(true)
return true
} catch {
print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)")
return false
}
}
/// 设置语音
///
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural"
/// - Returns: 是否设置成功
func setVoice(voiceName: String) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
if voiceName == currentVoiceName {
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
return true
}
do {
currentVoiceName = voiceName
speechConfig?.setSpeechSynthesisVoiceName(voiceName)
// 重新创建合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!)
print("[AzureTtsHelper] 已设置语音: \(voiceName)")
return true
} catch {
print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)")
return false
}
}
/// 设置语音合成参数
///
/// - Parameters:
/// - rate: 语速,范围 -100 到 100,默认为 0
/// - pitch: 音调,范围 -100 到 100,默认为 0
/// - volume: 音量,范围 0 到 100,默认为 100
/// - Returns: 是否设置成功
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
currentSpeechRate = rate
currentPitch = pitch
currentVolume = volume
print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)")
return true
}
/// 合成文本为语音并播放
///
/// - Parameters:
/// - text: 要合成的文本
/// - completion: 完成回调,返回是否成功和可能的错误信息
func speakText(text: String, completion: @escaping (Bool, String?) -> Void) {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
completion(false, "TTS 引擎尚未初始化")
return
}
do {
print("[AzureTtsHelper] 开始合成文本: \(text)")
// 生成 SSML
let ssml = generateSsml(text: text)
// 使用 SSML 合成语音
speakSsml(ssml: ssml, completion: completion)
} catch {
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
completion(false, "语音合成异常: \(error.localizedDescription)")
}
}
/// 生成 SSML 文本
///
/// - Parameter text: 要转换的文本
/// - Returns: SSML 格式的文本
private func generateSsml(text: String) -> String {
// 计算 SSML 参数
let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%")
let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%"
let volumeParam = "\(min(max(currentVolume, 0), 100))%"
return """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN">
<voice name="\(currentVoiceName)">
<prosody rate="\(rateParam)" pitch="\(pitchParam)" volume="\(volumeParam)">
\(text)
</prosody>
</voice>
</speak>
"""
}
/// 合成 SSML 为语音并播放
///
/// - Parameters:
/// - ssml: SSML 格式的文本
/// - completion: 完成回调,返回是否成功和可能的错误信息
private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
completion(false, "TTS 引擎尚未初始化")
return
}
do {
print("[AzureTtsHelper] 开始合成 SSML")
// 标记为正在播放
speaking = true
// 激活音频会话
try audioSession.setActive(true)
// 异步合成语音
let result = try synthesizer!.speakSsml(ssml)
switch result.reason {
case .synthesizingAudioCompleted:
print("[AzureTtsHelper] 语音合成完成")
speaking = false
completion(true, "语音合成完成")
case .canceled:
if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) {
print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
speaking = false
completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")")
} else {
print("[AzureTtsHelper] 语音合成取消")
speaking = false
completion(false, "语音合成取消")
}
default:
print("[AzureTtsHelper] 语音合成失败: \(result.reason)")
speaking = false
completion(false, "语音合成失败: \(result.reason)")
}
} catch {
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)")
speaking = false
completion(false, "语音合成异常: \(error.localizedDescription)")
}
}
/// 停止当前语音合成
///
/// - Returns: 是否停止成功
func stopSpeaking() -> Bool {
if !initialized {
print("[AzureTtsHelper] TTS 引擎尚未初始化")
return false
}
do {
try synthesizer?.stopSpeaking()
speaking = false
print("[AzureTtsHelper] 已停止语音合成")
return true
} catch {
print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)")
return false
}
}
/// 释放资源
func dispose() {
do {
stopSpeaking()
// 恢复音频会话
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
synthesizer = nil
speechConfig = nil
audioConfig = nil
initialized = false
speaking = false
print("[AzureTtsHelper] TTS 引擎已释放")
} catch {
print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)")
}
}
/// 检查当前是否正在播放语音
///
/// - Returns: 是否正在播放语音
func isSpeaking() -> Bool {
return speaking
}
}

32
local_plugins/azure_speech/pubspec.yaml

@ -0,0 +1,32 @@
name: azure_speech
description: Azure语音服务插件,包含TTS和ASR服务
version: 0.0.1
homepage:
environment:
sdk: ">=2.17.0 <3.0.0"
flutter: ">=2.5.0"
dependencies:
flutter:
sdk: flutter
dev_dependencies:
flutter_test:
sdk: flutter
flutter_lints: ^2.0.0
# For information on the generic Dart part of this file, see the
# following page: https://dart.dev/tools/pub/pubspec
# The following section is specific to Flutter packages.
flutter:
# This section identifies this Flutter project as a plugin project.
plugin:
platforms:
android:
package: com.yunqiinnovation.azure_speech
pluginClass: AzureSpeechPlugin
ios:
pluginClass: AzureSpeechPlugin

2
pubspec.yaml

@ -60,6 +60,8 @@ dependencies:
dio: ^5.8.0+1 dio: ^5.8.0+1
sqflite: ^2.4.2 sqflite: ^2.4.2
path: ^1.9.1 path: ^1.9.1
azure_speech:
path: local_plugins/azure_speech
dev_dependencies: dev_dependencies:
flutter_test: flutter_test:

Loading…
Cancel
Save