63 changed files with 8045 additions and 3156 deletions
@ -1,654 +0,0 @@ |
|||||
package com.yunqiinnovation.deepsound |
|
||||
|
|
||||
import android.content.Context |
|
||||
import android.media.AudioAttributes |
|
||||
import android.media.AudioFormat |
|
||||
import android.media.AudioRecord |
|
||||
import android.media.MediaRecorder |
|
||||
import android.media.audiofx.AcousticEchoCanceler |
|
||||
import android.media.audiofx.NoiseSuppressor |
|
||||
import android.media.audiofx.AutomaticGainControl |
|
||||
import android.os.Process |
|
||||
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
|
||||
import com.microsoft.cognitiveservices.speech.* |
|
||||
import com.microsoft.cognitiveservices.speech.audio.* |
|
||||
import com.microsoft.cognitiveservices.speech.util.EventHandler |
|
||||
import java.util.concurrent.ExecutionException |
|
||||
import java.util.concurrent.atomic.AtomicBoolean |
|
||||
// 移除WebRTC相关导入 |
|
||||
|
|
||||
class AzureAsrHelper(private val context: Context) { |
|
||||
private var recognizer: SpeechRecognizer? = null |
|
||||
private var speechConfig: SpeechConfig? = null |
|
||||
private val TAG = "AzureAsrHelper" |
|
||||
private var isContinuousRecognitionActive = false |
|
||||
private var currentLanguage = "zh-CN" |
|
||||
private var subscriptionKey = "" |
|
||||
private var serviceRegion = "" |
|
||||
private var isAutoDetectLanguage = false |
|
||||
private var supportedLanguages = arrayOf("zh-CN", "en-US") |
|
||||
|
|
||||
// 是否使用回音消除 - 内部控制常量 |
|
||||
private val useEchoCancellation = true |
|
||||
|
|
||||
// 自定义音频处理相关 |
|
||||
private var customAudioProcessor: CustomAudioProcessor? = null |
|
||||
private var pushStream: PushAudioInputStream? = null |
|
||||
private var audioConfig: AudioConfig? = null |
|
||||
|
|
||||
// 初始化SDK并创建recognizer |
|
||||
fun initialize(subscriptionKey: String, serviceRegion: String, |
|
||||
supportedLanguages: Array<String> = arrayOf("zh-CN", "en-US")): Boolean { |
|
||||
try { |
|
||||
FileLogger.d(TAG, "初始化 Azure 语音服务") |
|
||||
|
|
||||
// 检查配置是否为空 |
|
||||
if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) { |
|
||||
FileLogger.e(TAG, "Azure 配置信息不完整") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 释放之前的资源 |
|
||||
dispose() |
|
||||
|
|
||||
this.subscriptionKey = subscriptionKey |
|
||||
this.serviceRegion = serviceRegion |
|
||||
|
|
||||
// 设置语言 |
|
||||
if (supportedLanguages.isNotEmpty()) { |
|
||||
this.supportedLanguages = supportedLanguages |
|
||||
} |
|
||||
|
|
||||
// 根据支持的语言数量决定是否启用自动语言检测 |
|
||||
this.isAutoDetectLanguage = supportedLanguages.size >= 2 |
|
||||
|
|
||||
// 如果只有一种语言,设置为当前语言 |
|
||||
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { |
|
||||
this.currentLanguage = supportedLanguages[0] |
|
||||
} |
|
||||
|
|
||||
// 创建语音配置 |
|
||||
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) |
|
||||
|
|
||||
// 设置语言配置 |
|
||||
if (isAutoDetectLanguage) { |
|
||||
// 设置自动语言检测 |
|
||||
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") |
|
||||
} else { |
|
||||
// 设置指定的识别语言 |
|
||||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|
||||
} |
|
||||
|
|
||||
// 创建识别器 |
|
||||
try { |
|
||||
if (useEchoCancellation) { |
|
||||
// 如果使用回音消除,创建自定义音频输入流 |
|
||||
setupCustomAudioProcessing() |
|
||||
|
|
||||
if (isAutoDetectLanguage) { |
|
||||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
||||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|
||||
} else { |
|
||||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|
||||
} |
|
||||
} else { |
|
||||
// 使用默认麦克风输入 |
|
||||
if (isAutoDetectLanguage) { |
|
||||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
||||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|
||||
} else { |
|
||||
recognizer = SpeechRecognizer(speechConfig) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
FileLogger.d(TAG, "Azure 语音服务初始化成功") |
|
||||
return true |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "创建识别器失败: ${e.message}") |
|
||||
stopCustomAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "初始化失败: ${e.message}") |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 重置 recognizer |
|
||||
private fun resetRecognizer(): Boolean { |
|
||||
try { |
|
||||
// 释放之前的 recognizer |
|
||||
recognizer?.close() |
|
||||
recognizer = null |
|
||||
|
|
||||
// 停止当前的音频处理 |
|
||||
stopCustomAudioProcessing() |
|
||||
|
|
||||
// 使用现有配置重新创建 recognizer |
|
||||
if (speechConfig != null) { |
|
||||
if (useEchoCancellation) { |
|
||||
// 如果使用回音消除,创建自定义音频输入流 |
|
||||
setupCustomAudioProcessing() |
|
||||
|
|
||||
if (isAutoDetectLanguage) { |
|
||||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
||||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|
||||
} else { |
|
||||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|
||||
} |
|
||||
} else { |
|
||||
// 使用默认麦克风输入 |
|
||||
if (isAutoDetectLanguage) { |
|
||||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
||||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|
||||
} else { |
|
||||
recognizer = SpeechRecognizer(speechConfig) |
|
||||
} |
|
||||
} |
|
||||
return true |
|
||||
} else { |
|
||||
FileLogger.e(TAG, "语音配置未初始化") |
|
||||
return false |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "重置识别器失败: ${e.message}") |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 开始一次性语音识别 |
|
||||
fun recognizeOnce(callback: RecognizeCallback) { |
|
||||
if (speechConfig == null) { |
|
||||
callback.onError("语音服务未初始化") |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 重置 recognizer |
|
||||
if (!resetRecognizer()) { |
|
||||
callback.onError("重置识别器失败") |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
// 启动音频处理 |
|
||||
startCustomAudioProcessing() |
|
||||
|
|
||||
// 执行识别 |
|
||||
val result = recognizer?.recognizeOnceAsync()?.get() |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopCustomAudioProcessing() |
|
||||
|
|
||||
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
|
||||
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language |
|
||||
callback.onResult(result.text, detectedLanguage ?: "") |
|
||||
} else { |
|
||||
callback.onError("未能识别语音") |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
// 停止音频处理 |
|
||||
stopCustomAudioProcessing() |
|
||||
callback.onError("识别异常: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 开始连续语音识别 |
|
||||
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|
||||
if (speechConfig == null) { |
|
||||
callback.onError("语音服务未初始化") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 如果已经在进行连续识别,先停止 |
|
||||
if (isContinuousRecognitionActive) { |
|
||||
stopContinuousRecognition(callback) |
|
||||
} |
|
||||
|
|
||||
// 重置 recognizer |
|
||||
if (!resetRecognizer()) { |
|
||||
callback.onError("重置识别器失败") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
// 启动音频处理 |
|
||||
startCustomAudioProcessing() |
|
||||
|
|
||||
// 设置识别事件处理 |
|
||||
// 最终识别结果 |
|
||||
recognizer?.recognized?.addEventListener( |
|
||||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|
||||
if (event.result.reason == ResultReason.RecognizedSpeech) { |
|
||||
val detectedLanguage = if (isAutoDetectLanguage) { |
|
||||
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
||||
} else { |
|
||||
currentLanguage |
|
||||
} |
|
||||
// FileLogger.d(TAG, "最终识别结果: ${event.result.text}") |
|
||||
callback.onResult(event.result.text, detectedLanguage) |
|
||||
} |
|
||||
} |
|
||||
) |
|
||||
|
|
||||
// 识别中事件 |
|
||||
recognizer?.recognizing?.addEventListener( |
|
||||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|
||||
if (event.result.reason == ResultReason.RecognizingSpeech) { |
|
||||
val detectedLanguage = if (isAutoDetectLanguage) { |
|
||||
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
||||
} else { |
|
||||
currentLanguage |
|
||||
} |
|
||||
// FileLogger.d(TAG, "识别中结果: ${event.result.text}") |
|
||||
callback.onRecognizing(event.result.text, detectedLanguage) |
|
||||
} |
|
||||
} |
|
||||
) |
|
||||
|
|
||||
// 会话事件 |
|
||||
recognizer?.sessionStarted?.addEventListener( |
|
||||
EventHandler<SessionEventArgs> { _, _ -> |
|
||||
isContinuousRecognitionActive = true |
|
||||
callback.onSessionStarted() |
|
||||
} |
|
||||
) |
|
||||
|
|
||||
recognizer?.sessionStopped?.addEventListener( |
|
||||
EventHandler<SessionEventArgs> { _, _ -> |
|
||||
isContinuousRecognitionActive = false |
|
||||
callback.onSessionStopped() |
|
||||
} |
|
||||
) |
|
||||
|
|
||||
// 取消事件 |
|
||||
recognizer?.canceled?.addEventListener( |
|
||||
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
|
||||
val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else "" |
|
||||
callback.onCanceled(event.reason.toString(), errorDetails) |
|
||||
isContinuousRecognitionActive = false |
|
||||
} |
|
||||
) |
|
||||
|
|
||||
// 开始连续识别 |
|
||||
recognizer?.startContinuousRecognitionAsync()?.get() |
|
||||
isContinuousRecognitionActive = true |
|
||||
|
|
||||
return true |
|
||||
} catch (e: Exception) { |
|
||||
callback.onError("开始连续识别失败: ${e.message}") |
|
||||
isContinuousRecognitionActive = false |
|
||||
stopCustomAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 停止连续语音识别 |
|
||||
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|
||||
if (!isContinuousRecognitionActive || recognizer == null) { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
recognizer?.stopContinuousRecognitionAsync() |
|
||||
isContinuousRecognitionActive = false |
|
||||
callback.onSessionStopped() |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopCustomAudioProcessing() |
|
||||
|
|
||||
return true |
|
||||
} catch (e: Exception) { |
|
||||
callback.onError("停止连续识别失败: ${e.message}") |
|
||||
isContinuousRecognitionActive = false |
|
||||
stopCustomAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 检查连续识别是否活跃 |
|
||||
fun isContinuousRecognitionActive(): Boolean { |
|
||||
return isContinuousRecognitionActive |
|
||||
} |
|
||||
|
|
||||
// 设置自定义音频处理 |
|
||||
private fun setupCustomAudioProcessing() { |
|
||||
if (!useEchoCancellation) { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
// 1. 创建PushAudioInputStream |
|
||||
pushStream = PushAudioInputStream.create() |
|
||||
|
|
||||
// 2. 创建AudioConfig |
|
||||
audioConfig = AudioConfig.fromStreamInput(pushStream) |
|
||||
|
|
||||
// 3. 创建自定义音频处理器 |
|
||||
customAudioProcessor = CustomAudioProcessor(pushStream) |
|
||||
|
|
||||
FileLogger.d(TAG, "自定义音频处理设置完成") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") |
|
||||
releaseCustomAudioProcessing() |
|
||||
|
|
||||
// 降级处理:如果自定义处理设置失败,尝试使用默认麦克风 |
|
||||
try { |
|
||||
FileLogger.d(TAG, "尝试降级到默认麦克风输入") |
|
||||
audioConfig = AudioConfig.fromDefaultMicrophoneInput() |
|
||||
} catch (e2: Exception) { |
|
||||
FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}") |
|
||||
audioConfig = null |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 启动自定义音频处理 |
|
||||
private fun startCustomAudioProcessing() { |
|
||||
if (!useEchoCancellation || customAudioProcessor == null) { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
customAudioProcessor?.startRecording() |
|
||||
FileLogger.d(TAG, "自定义音频处理已启动") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 停止自定义音频处理 |
|
||||
private fun stopCustomAudioProcessing() { |
|
||||
if (!useEchoCancellation || customAudioProcessor == null) { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
customAudioProcessor?.stopRecording() |
|
||||
FileLogger.d(TAG, "自定义音频处理已停止") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 释放自定义音频处理资源 |
|
||||
private fun releaseCustomAudioProcessing() { |
|
||||
stopCustomAudioProcessing() |
|
||||
|
|
||||
try { |
|
||||
customAudioProcessor = null |
|
||||
pushStream?.close() |
|
||||
pushStream = null |
|
||||
audioConfig?.close() |
|
||||
audioConfig = null |
|
||||
|
|
||||
FileLogger.d(TAG, "自定义音频处理资源已释放") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 释放所有资源 |
|
||||
fun dispose() { |
|
||||
try { |
|
||||
// 停止和释放音频处理 |
|
||||
releaseCustomAudioProcessing() |
|
||||
|
|
||||
recognizer?.close() |
|
||||
recognizer = null |
|
||||
|
|
||||
speechConfig?.close() |
|
||||
speechConfig = null |
|
||||
|
|
||||
isContinuousRecognitionActive = false |
|
||||
|
|
||||
FileLogger.d(TAG, "语音识别资源已释放") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "释放资源出错: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 自定义音频处理器 - 使用Android原生回音消除 |
|
||||
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) { |
|
||||
private val SAMPLE_RATE = 16000 |
|
||||
private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO |
|
||||
private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT |
|
||||
private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定 |
|
||||
|
|
||||
private var audioRecord: AudioRecord? = null |
|
||||
private var echoCanceler: AcousticEchoCanceler? = null |
|
||||
private val isRecording = AtomicBoolean(false) |
|
||||
private var recordingThread: Thread? = null |
|
||||
|
|
||||
// 启动录音并处理音频数据 |
|
||||
fun startRecording() { |
|
||||
if (isRecording.get() || pushStream == null) { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
// 使用Builder模式构建AudioFormat |
|
||||
val audioFormat = AudioFormat.Builder() |
|
||||
.setSampleRate(SAMPLE_RATE) |
|
||||
.setEncoding(AUDIO_FORMAT) |
|
||||
.setChannelMask(CHANNEL_CONFIG) |
|
||||
.build() |
|
||||
|
|
||||
// 使用Builder模式创建AudioRecord实例 |
|
||||
audioRecord = AudioRecord.Builder() |
|
||||
.setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION) |
|
||||
.setAudioFormat(audioFormat) |
|
||||
.setBufferSizeInBytes(BUFFER_SIZE) |
|
||||
.build() |
|
||||
|
|
||||
// 检查AudioRecord初始化状态 |
|
||||
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { |
|
||||
FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}") |
|
||||
// 尝试使用DEFAULT音频源重试一次 |
|
||||
audioRecord?.release() |
|
||||
audioRecord = AudioRecord.Builder() |
|
||||
.setAudioSource(MediaRecorder.AudioSource.DEFAULT) |
|
||||
.setAudioFormat(audioFormat) |
|
||||
.setBufferSizeInBytes(BUFFER_SIZE) |
|
||||
.build() |
|
||||
|
|
||||
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { |
|
||||
FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃") |
|
||||
releaseAudioResources() |
|
||||
return |
|
||||
} else { |
|
||||
FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 启用音频效果(回音消除、噪声抑制等) |
|
||||
enableAudioEffects() |
|
||||
|
|
||||
// 启动录音 |
|
||||
audioRecord?.startRecording() |
|
||||
isRecording.set(true) |
|
||||
|
|
||||
// 创建录音线程 |
|
||||
recordingThread = Thread({ |
|
||||
val buffer = ByteArray(BUFFER_SIZE) |
|
||||
|
|
||||
while (isRecording.get()) { |
|
||||
try { |
|
||||
val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0 |
|
||||
|
|
||||
if (readSize > 0) { |
|
||||
try { |
|
||||
// 将处理后的音频数据推送到流 |
|
||||
if (readSize == buffer.size) { |
|
||||
// 如果读取的大小等于buffer的大小,直接写入整个buffer |
|
||||
pushStream.write(buffer) |
|
||||
} else { |
|
||||
// 如果只读取了部分数据,创建新的数组只包含有效数据 |
|
||||
val validData = buffer.copyOfRange(0, readSize) |
|
||||
pushStream.write(validData) |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "写入音频数据失败: ${e.message}") |
|
||||
break |
|
||||
} |
|
||||
} else if (readSize == 0) { |
|
||||
// 读取为0,可能是临时的,等待一下继续尝试 |
|
||||
Thread.sleep(10) |
|
||||
} else { |
|
||||
// 负值表示错误 |
|
||||
FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize") |
|
||||
break |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "录音线程异常: ${e.message}") |
|
||||
break |
|
||||
} |
|
||||
} |
|
||||
}, "AudioRecordingThread") |
|
||||
|
|
||||
// 设置线程优先级并启动 |
|
||||
recordingThread?.priority = Thread.MAX_PRIORITY |
|
||||
recordingThread?.start() |
|
||||
|
|
||||
FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else "")) |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "启动音频录制失败: ${e.message}") |
|
||||
releaseAudioResources() |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 启用音频效果(回音消除、噪声抑制等) |
|
||||
private fun enableAudioEffects() { |
|
||||
try { |
|
||||
val audioSessionId = audioRecord?.audioSessionId ?: -1 |
|
||||
|
|
||||
if (audioSessionId != -1) { |
|
||||
// 启用回音消除 |
|
||||
if (AcousticEchoCanceler.isAvailable()) { |
|
||||
try { |
|
||||
echoCanceler = AcousticEchoCanceler.create(audioSessionId) |
|
||||
if (echoCanceler != null) { |
|
||||
echoCanceler?.enabled = true |
|
||||
FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId") |
|
||||
} else { |
|
||||
FileLogger.w(TAG, "回音消除器创建返回null") |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}") |
|
||||
} |
|
||||
} else { |
|
||||
FileLogger.d(TAG, "设备不支持回音消除") |
|
||||
} |
|
||||
|
|
||||
// 以下功能暂时不启用,可根据需要取消注释 |
|
||||
/* |
|
||||
// 启用噪声抑制 |
|
||||
if (NoiseSuppressor.isAvailable()) { |
|
||||
try { |
|
||||
val ns = NoiseSuppressor.create(audioSessionId) |
|
||||
ns?.enabled = true |
|
||||
FileLogger.d(TAG, "噪声抑制已启用") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 启用自动增益控制 |
|
||||
if (AutomaticGainControl.isAvailable()) { |
|
||||
try { |
|
||||
val agc = AutomaticGainControl.create(audioSessionId) |
|
||||
agc?.enabled = true |
|
||||
FileLogger.d(TAG, "自动增益控制已启用") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
*/ |
|
||||
} else { |
|
||||
FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果") |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "启用音频效果时出错: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 停止录音 |
|
||||
fun stopRecording() { |
|
||||
if (!isRecording.get()) { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
isRecording.set(false) |
|
||||
|
|
||||
try { |
|
||||
// 等待录音线程结束 |
|
||||
recordingThread?.join(1000) |
|
||||
|
|
||||
// 释放资源 |
|
||||
releaseAudioResources() |
|
||||
|
|
||||
FileLogger.d(TAG, "音频录制已停止") |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "停止音频录制失败: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 释放音频资源 |
|
||||
private fun releaseAudioResources() { |
|
||||
try { |
|
||||
// 停止录音 |
|
||||
try { |
|
||||
if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) { |
|
||||
audioRecord?.stop() |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
// 忽略可能的IllegalStateException |
|
||||
FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}") |
|
||||
} |
|
||||
|
|
||||
// 释放回音消除器 |
|
||||
try { |
|
||||
if (echoCanceler != null) { |
|
||||
echoCanceler?.enabled = false |
|
||||
echoCanceler?.release() |
|
||||
echoCanceler = null |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}") |
|
||||
} finally { |
|
||||
echoCanceler = null |
|
||||
} |
|
||||
|
|
||||
// 释放音频记录器 |
|
||||
try { |
|
||||
audioRecord?.release() |
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}") |
|
||||
} finally { |
|
||||
audioRecord = null |
|
||||
} |
|
||||
|
|
||||
// 重置线程 |
|
||||
recordingThread = null |
|
||||
|
|
||||
} catch (e: Exception) { |
|
||||
FileLogger.e(TAG, "释放音频资源失败: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 一次性识别回调接口 |
|
||||
interface RecognizeCallback { |
|
||||
fun onResult(result: String, detectedLanguage: String = "") |
|
||||
fun onError(error: String) |
|
||||
} |
|
||||
|
|
||||
// 连续识别回调接口 |
|
||||
interface ContinuousRecognizeCallback { |
|
||||
fun onResult(result: String, detectedLanguage: String = "") |
|
||||
fun onRecognizing(recognizing: String, detectedLanguage: String = "") |
|
||||
fun onSessionStarted() |
|
||||
fun onSessionStopped() |
|
||||
fun onCanceled(reason: String, errorDetails: String) |
|
||||
fun onError(error: String) |
|
||||
} |
|
||||
} |
|
||||
@ -1,344 +0,0 @@ |
|||||
package com.yunqiinnovation.deepsound |
|
||||
|
|
||||
import android.util.Log |
|
||||
import okhttp3.* |
|
||||
import okhttp3.MediaType.Companion.toMediaTypeOrNull |
|
||||
import okhttp3.RequestBody.Companion.toRequestBody |
|
||||
import org.json.JSONArray |
|
||||
import org.json.JSONObject |
|
||||
import java.io.IOException |
|
||||
import java.util.concurrent.CountDownLatch |
|
||||
import java.util.concurrent.TimeUnit |
|
||||
|
|
||||
/** |
|
||||
* 火山AI服务的原生实现 |
|
||||
* |
|
||||
* 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式 |
|
||||
*/ |
|
||||
class VolcanoAIService() { |
|
||||
private val TAG = "VolcanoAIService" |
|
||||
private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3" |
|
||||
private val chatEndpoint = "/chat/completions" |
|
||||
private val client = OkHttpClient.Builder() |
|
||||
.connectTimeout(30, TimeUnit.SECONDS) |
|
||||
.readTimeout(30, TimeUnit.SECONDS) |
|
||||
.writeTimeout(30, TimeUnit.SECONDS) |
|
||||
.build() |
|
||||
|
|
||||
private var apiKey: String = "" |
|
||||
private var isInitialized = false |
|
||||
|
|
||||
/** |
|
||||
* 初始化火山AI服务 |
|
||||
* |
|
||||
* @param apiKey 火山AI API密钥 |
|
||||
* @return 初始化是否成功 |
|
||||
*/ |
|
||||
fun initialize(apiKey: String): Boolean { |
|
||||
this.apiKey = apiKey |
|
||||
isInitialized = apiKey.isNotEmpty() |
|
||||
|
|
||||
if (!isInitialized) { |
|
||||
Log.e(TAG, "初始化失败:API key 不能为空") |
|
||||
} else { |
|
||||
Log.d(TAG, "火山AI服务初始化成功") |
|
||||
} |
|
||||
|
|
||||
return isInitialized |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 生成个性化问候语 |
|
||||
* |
|
||||
* @param agentName 代理名称 |
|
||||
* @param systemPrompt 系统提示词 |
|
||||
* @param callback 回调函数,返回生成的问候语 |
|
||||
*/ |
|
||||
fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) { |
|
||||
val messages = JSONArray().apply { |
|
||||
put(JSONObject().apply { |
|
||||
put("role", "system") |
|
||||
put("content", systemPrompt) |
|
||||
}) |
|
||||
put(JSONObject().apply { |
|
||||
put("role", "user") |
|
||||
put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。") |
|
||||
}) |
|
||||
} |
|
||||
|
|
||||
sendMessageStream(messages, systemPrompt, object : StreamCallback { |
|
||||
val stringBuilder = StringBuilder() |
|
||||
|
|
||||
override fun onToken(token: String) { |
|
||||
stringBuilder.append(token) |
|
||||
} |
|
||||
|
|
||||
override fun onComplete() { |
|
||||
callback(stringBuilder.toString(), null) |
|
||||
} |
|
||||
|
|
||||
override fun onError(e: Exception) { |
|
||||
callback(null, e) |
|
||||
} |
|
||||
}) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 发送消息(非流式输出) |
|
||||
* |
|
||||
* @param messages 消息列表 |
|
||||
* @param systemPrompt 系统提示词 |
|
||||
* @return 返回AI的回复 |
|
||||
* @throws VolcanoAIException 如果API调用失败 |
|
||||
*/ |
|
||||
@Throws(VolcanoAIException::class) |
|
||||
fun sendMessage(messages: JSONArray, systemPrompt: String): String { |
|
||||
// 检查是否已初始化 |
|
||||
if (!isInitialized || apiKey.isEmpty()) { |
|
||||
throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法") |
|
||||
} |
|
||||
|
|
||||
val fullMessages = JSONArray().apply { |
|
||||
put(JSONObject().apply { |
|
||||
put("role", "system") |
|
||||
put("content", systemPrompt) |
|
||||
}) |
|
||||
for (i in 0 until messages.length()) { |
|
||||
put(messages.getJSONObject(i)) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
val requestBody = JSONObject().apply { |
|
||||
put("model", "doubao-1-5-lite-32k-250115") |
|
||||
put("messages", fullMessages) |
|
||||
put("temperature", 0.7) |
|
||||
put("max_tokens", 2000) |
|
||||
put("stream", false) |
|
||||
} |
|
||||
|
|
||||
val mediaType = "application/json".toMediaTypeOrNull() |
|
||||
val request = Request.Builder() |
|
||||
.url("$baseUrl$chatEndpoint") |
|
||||
.addHeader("Content-Type", "application/json") |
|
||||
.addHeader("Authorization", "Bearer $apiKey") |
|
||||
.post(requestBody.toString().toRequestBody(mediaType)) |
|
||||
.build() |
|
||||
|
|
||||
try { |
|
||||
client.newCall(request).execute().use { response -> |
|
||||
if (!response.isSuccessful) { |
|
||||
val errorBody = response.body?.string() ?: "" |
|
||||
val errorMessage = try { |
|
||||
JSONObject(errorBody).getJSONObject("error").getString("message") |
|
||||
} catch (e: Exception) { |
|
||||
"Unknown error occurred" |
|
||||
} |
|
||||
throw VolcanoAIException(errorMessage) |
|
||||
} |
|
||||
|
|
||||
val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response") |
|
||||
val jsonResponse = JSONObject(responseBody) |
|
||||
|
|
||||
if (jsonResponse.has("choices") && |
|
||||
jsonResponse.getJSONArray("choices").length() > 0 && |
|
||||
jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) { |
|
||||
return jsonResponse.getJSONArray("choices") |
|
||||
.getJSONObject(0) |
|
||||
.getJSONObject("message") |
|
||||
.getString("content") |
|
||||
} |
|
||||
|
|
||||
throw VolcanoAIException("Invalid response format") |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
if (e is VolcanoAIException) throw e |
|
||||
throw VolcanoAIException("Failed to communicate with AI service: ${e.message}") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 发送消息(流式输出) |
|
||||
* |
|
||||
* @param messages 消息列表 |
|
||||
* @param systemPrompt 系统提示词 |
|
||||
* @param callback 回调函数,用于接收流式输出的结果 |
|
||||
*/ |
|
||||
fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { |
|
||||
// 检查是否已初始化 |
|
||||
if (!isInitialized || apiKey.isEmpty()) { |
|
||||
callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
val fullMessages = JSONArray().apply { |
|
||||
put(JSONObject().apply { |
|
||||
put("role", "system") |
|
||||
put("content", systemPrompt) |
|
||||
}) |
|
||||
for (i in 0 until messages.length()) { |
|
||||
put(messages.getJSONObject(i)) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
val requestBody = JSONObject().apply { |
|
||||
put("model", "doubao-1-5-lite-32k-250115") |
|
||||
put("messages", fullMessages) |
|
||||
put("temperature", 0.7) |
|
||||
put("max_tokens", 2000) |
|
||||
put("stream", true) |
|
||||
} |
|
||||
|
|
||||
val mediaType = "application/json".toMediaTypeOrNull() |
|
||||
val request = Request.Builder() |
|
||||
.url("$baseUrl$chatEndpoint") |
|
||||
.addHeader("Content-Type", "application/json") |
|
||||
.addHeader("Authorization", "Bearer $apiKey") |
|
||||
.addHeader("Accept", "text/event-stream") |
|
||||
.post(requestBody.toString().toRequestBody(mediaType)) |
|
||||
.build() |
|
||||
|
|
||||
client.newCall(request).enqueue(object : Callback { |
|
||||
override fun onFailure(call: Call, e: IOException) { |
|
||||
callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}")) |
|
||||
} |
|
||||
|
|
||||
override fun onResponse(call: Call, response: Response) { |
|
||||
if (!response.isSuccessful) { |
|
||||
val errorBody = response.body?.string() ?: "" |
|
||||
val errorMessage = try { |
|
||||
JSONObject(errorBody).getJSONObject("error").getString("message") |
|
||||
} catch (e: Exception) { |
|
||||
"Unknown error occurred" |
|
||||
} |
|
||||
callback.onError(VolcanoAIException(errorMessage)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
val responseBody = response.body ?: return |
|
||||
val source = responseBody.source() |
|
||||
val bufferedSource = source.buffer |
|
||||
|
|
||||
try { |
|
||||
while (!bufferedSource.exhausted()) { |
|
||||
val line = bufferedSource.readUtf8Line() ?: continue |
|
||||
|
|
||||
if (line.isEmpty()) continue |
|
||||
if (line.startsWith("data: ")) { |
|
||||
val data = line.substring(6) |
|
||||
if (data == "[DONE]") { |
|
||||
callback.onComplete() |
|
||||
break |
|
||||
} |
|
||||
|
|
||||
try { |
|
||||
val jsonData = JSONObject(data) |
|
||||
if (jsonData.has("choices") && |
|
||||
jsonData.getJSONArray("choices").length() > 0 && |
|
||||
jsonData.getJSONArray("choices").getJSONObject(0).has("delta") && |
|
||||
jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) { |
|
||||
val content = jsonData.getJSONArray("choices") |
|
||||
.getJSONObject(0) |
|
||||
.getJSONObject("delta") |
|
||||
.getString("content") |
|
||||
callback.onToken(content) |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
// 忽略无效的JSON数据 |
|
||||
continue |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
} catch (e: Exception) { |
|
||||
callback.onError(VolcanoAIException("Error processing stream: ${e.message}")) |
|
||||
} finally { |
|
||||
response.close() |
|
||||
} |
|
||||
} |
|
||||
}) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 同步方式发送消息(流式输出) |
|
||||
* |
|
||||
* 注意:此方法会阻塞当前线程,请在后台线程中调用 |
|
||||
* |
|
||||
* @param messages 消息列表 |
|
||||
* @param systemPrompt 系统提示词 |
|
||||
* @return 返回完整的AI回复 |
|
||||
* @throws VolcanoAIException 如果API调用失败 |
|
||||
*/ |
|
||||
@Throws(VolcanoAIException::class) |
|
||||
fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String { |
|
||||
val result = StringBuilder() |
|
||||
val latch = CountDownLatch(1) |
|
||||
var exception: Exception? = null |
|
||||
|
|
||||
sendMessageStream(messages, systemPrompt, object : StreamCallback { |
|
||||
override fun onToken(token: String) { |
|
||||
result.append(token) |
|
||||
} |
|
||||
|
|
||||
override fun onComplete() { |
|
||||
latch.countDown() |
|
||||
} |
|
||||
|
|
||||
override fun onError(e: Exception) { |
|
||||
exception = e |
|
||||
latch.countDown() |
|
||||
} |
|
||||
}) |
|
||||
|
|
||||
// 等待流式输出完成或出错 |
|
||||
latch.await(60, TimeUnit.SECONDS) |
|
||||
|
|
||||
if (exception != null) { |
|
||||
throw exception as VolcanoAIException |
|
||||
} |
|
||||
|
|
||||
return result.toString() |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 创建用户消息 |
|
||||
*/ |
|
||||
fun createUserMessage(content: String): JSONObject { |
|
||||
return JSONObject().apply { |
|
||||
put("role", "user") |
|
||||
put("content", content) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 创建系统消息 |
|
||||
*/ |
|
||||
fun createSystemMessage(content: String): JSONObject { |
|
||||
return JSONObject().apply { |
|
||||
put("role", "system") |
|
||||
put("content", content) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 创建助手消息 |
|
||||
*/ |
|
||||
fun createAssistantMessage(content: String): JSONObject { |
|
||||
return JSONObject().apply { |
|
||||
put("role", "assistant") |
|
||||
put("content", content) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 流式输出回调接口 |
|
||||
*/ |
|
||||
interface StreamCallback { |
|
||||
fun onToken(token: String) |
|
||||
fun onComplete() |
|
||||
fun onError(e: Exception) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 火山AI异常 |
|
||||
*/ |
|
||||
class VolcanoAIException(message: String) : Exception(message) |
|
||||
@ -0,0 +1,456 @@ |
|||||
|
package com.yunqiinnovation.deepsound |
||||
|
|
||||
|
import android.content.Context |
||||
|
import org.json.JSONArray |
||||
|
import org.json.JSONObject |
||||
|
import android.util.Log |
||||
|
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
||||
|
import com.yunqiinnovation.azure_speech.AzureAsrHelper |
||||
|
import com.yunqiinnovation.volcano_speech.VolcanoTtsHelper |
||||
|
import com.yunqiinnovation.open_ai_service.OpenAIService |
||||
|
|
||||
|
|
||||
|
/** |
||||
|
* 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑 |
||||
|
*/ |
||||
|
class VoiceInteractionHandler( |
||||
|
private val context: Context, |
||||
|
private val azureSpeechKey: String, |
||||
|
private val azureSpeechRegion: String, |
||||
|
private val openaiApiKey: String, |
||||
|
private val openaiBaseUrl: String = "", |
||||
|
private val openaiModel: String = "", |
||||
|
private val volcanoSpeechAppId: String, |
||||
|
private val volcanoSpeechAppToken: String |
||||
|
) { |
||||
|
private val TAG = "VoiceInteractionHandler" |
||||
|
|
||||
|
// Azure服务 |
||||
|
private var azureAsrHelper: AzureAsrHelper? = null |
||||
|
private var volcanoTtsHelper: VolcanoTtsHelper? = null |
||||
|
|
||||
|
// OpenAI服务 |
||||
|
private val openAIService = OpenAIService() |
||||
|
|
||||
|
|
||||
|
// 语音功能处理 |
||||
|
private val voiceFunctionHandler = VoiceFunctionHandler(openAIService, context) |
||||
|
|
||||
|
// 当前用户输入 |
||||
|
private var currentUserInput = "" |
||||
|
|
||||
|
// 状态 |
||||
|
private var isInitialized = false |
||||
|
var isRecognitionActive = false |
||||
|
private set |
||||
|
var isTtsSpeaking = false |
||||
|
private set |
||||
|
var hasSpeechDetected = false |
||||
|
private set |
||||
|
|
||||
|
// 回调 |
||||
|
private var callback: InteractionCallback? = null |
||||
|
|
||||
|
/** |
||||
|
* 初始化 |
||||
|
*/ |
||||
|
fun initialize(): Boolean { |
||||
|
if (isInitialized) return true |
||||
|
|
||||
|
try { |
||||
|
// 初始化Azure ASR |
||||
|
azureAsrHelper = AzureAsrHelper(context).apply { |
||||
|
initialize(azureSpeechKey, azureSpeechRegion) |
||||
|
} |
||||
|
|
||||
|
// 初始化Volcano TTS (使用大模型TTS) |
||||
|
volcanoTtsHelper = VolcanoTtsHelper(context).apply { |
||||
|
// 这里需要替换为实际的Volcano SDK初始化参数 |
||||
|
// 暂时使用假参数,实际使用时需要替换为真实值 |
||||
|
initialize(volcanoSpeechAppId, volcanoSpeechAppToken, "volc.bigasr.sauc.duration") |
||||
|
|
||||
|
} |
||||
|
|
||||
|
// 初始化OpenAI服务 |
||||
|
openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel) |
||||
|
|
||||
|
// 初始化语音功能处理器 |
||||
|
voiceFunctionHandler.initialize() |
||||
|
|
||||
|
isInitialized = true |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "初始化失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置回调 |
||||
|
*/ |
||||
|
fun setCallback(callback: InteractionCallback) { |
||||
|
this.callback = callback |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 开始语音识别 |
||||
|
*/ |
||||
|
fun startRecognition() { |
||||
|
if (isRecognitionActive) return |
||||
|
|
||||
|
// 检查录音权限 |
||||
|
if (!checkRecordAudioPermission()) { |
||||
|
callback?.onError("需要录音权限,请在设置中授予权限") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
isRecognitionActive = true |
||||
|
hasSpeechDetected = false |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
try { |
||||
|
azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) { |
||||
|
if (recognizing.isNotEmpty()) { |
||||
|
hasSpeechDetected = true |
||||
|
stopTts() |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onResult(result: String, detectedLanguage: String) { |
||||
|
if (result.isNotEmpty()) { |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
processWithOpenAI(result) |
||||
|
|
||||
|
} |
||||
|
|
||||
|
// 重置状态,继续识别 |
||||
|
hasSpeechDetected = false |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStarted() { |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStopped() { |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onCanceled(reason: String, errorDetails: String) { |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
isRecognitionActive = false |
||||
|
callback?.onError("语音识别出错") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onSuccess(message: String) { |
||||
|
// 处理成功事件 |
||||
|
} |
||||
|
}) |
||||
|
} catch (e: Exception) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e) |
||||
|
callback?.onError("启动语音识别失败") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止语音识别 |
||||
|
*/ |
||||
|
fun stopRecognition() { |
||||
|
if (!isRecognitionActive) return |
||||
|
|
||||
|
FileLogger.d(TAG, "停止语音识别") |
||||
|
|
||||
|
try { |
||||
|
azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onResult(result: String, detectedLanguage: String) {} |
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) {} |
||||
|
override fun onSessionStarted() {} |
||||
|
override fun onSessionStopped() { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别会话已停止") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onCanceled(reason: String, errorDetails: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别已取消: $reason") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onError(error: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.e(TAG, "停止语音识别时出错: $error") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onSuccess(message: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别已停止: $message") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
}) |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e) |
||||
|
// 确保状态一致性 |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 使用OpenAI处理语音识别结果 |
||||
|
*/ |
||||
|
private fun processWithOpenAI(text: String) { |
||||
|
// 保存当前用户输入,用于后续同步聊天记录 |
||||
|
currentUserInput = text |
||||
|
|
||||
|
Thread { |
||||
|
try { |
||||
|
val messages = JSONArray().apply { |
||||
|
put(openAIService.createUserMessage(text)) |
||||
|
} |
||||
|
|
||||
|
// 创建响应构建器 |
||||
|
val responseBuilder = StringBuilder() |
||||
|
|
||||
|
openAIService.sendMessageStream( |
||||
|
messages = messages, |
||||
|
callback = object : OpenAIService.StreamCallback { |
||||
|
override fun onToken(token: String) { |
||||
|
// 累加响应内容 |
||||
|
responseBuilder.append(token) |
||||
|
} |
||||
|
|
||||
|
override fun onComplete() { |
||||
|
// 处理完整响应 |
||||
|
val response = responseBuilder.toString() |
||||
|
if (response.isNotEmpty()) { |
||||
|
// 播放AI回复 |
||||
|
Log.d(TAG, "AI 回复: $response") |
||||
|
|
||||
|
speakAIResponse(response) |
||||
|
|
||||
|
// 同步聊天记录到Flutter端 |
||||
|
sendChatHistoryUpdate("personal_assistant", text, response) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(e: Exception) { |
||||
|
FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e) |
||||
|
callback?.onError("AI处理出错") |
||||
|
} |
||||
|
|
||||
|
override fun onFunctionCall(call: JSONObject) { |
||||
|
FileLogger.d(TAG, "收到函数调用请求: ${call.getString("name")}") |
||||
|
|
||||
|
// 使用函数处理器处理函数调用 |
||||
|
val handled = voiceFunctionHandler.handleFunctionCall( |
||||
|
functionCall = call, |
||||
|
messages = messages, |
||||
|
callback = object : VoiceFunctionHandler.FunctionCallCallback { |
||||
|
override fun onTokenReceived(token: String) { |
||||
|
responseBuilder.append(token) |
||||
|
} |
||||
|
|
||||
|
override fun onComplete() { |
||||
|
val response = responseBuilder.toString() |
||||
|
if (response.isNotEmpty()) { |
||||
|
// 播放AI回复 |
||||
|
Log.d(TAG, "AI Function Call 回复: $response") |
||||
|
speakAIResponse(response) |
||||
|
|
||||
|
// 同步聊天记录到Flutter端 |
||||
|
sendChatHistoryUpdate("personal_assistant", text, response) |
||||
|
} |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onError(message: String) { |
||||
|
FileLogger.e(TAG, "函数处理出错: $message") |
||||
|
callback?.onError(message) |
||||
|
} |
||||
|
|
||||
|
override fun onFunctionCall(nestedCall: JSONObject) { |
||||
|
FileLogger.d(TAG, "收到嵌套函数调用: ${nestedCall.getString("name")}") |
||||
|
// 不处理嵌套函数调用,直接返回错误提示 |
||||
|
// speakAIResponse("抱歉,暂不支持嵌套函数调用") |
||||
|
} |
||||
|
|
||||
|
override fun onExitWithMessage(farewell: String) { |
||||
|
// 播放退出消息 |
||||
|
speakAIResponse(farewell) |
||||
|
|
||||
|
// 同步聊天记录 |
||||
|
sendChatHistoryUpdate("personal_assistant", text, farewell) |
||||
|
|
||||
|
// 停止语音识别 |
||||
|
stopRecognition() |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
if (!handled) { |
||||
|
// 如果函数没有被处理,作为普通文本处理 |
||||
|
FileLogger.d(TAG, "函数未处理,作为普通文本处理") |
||||
|
speakAIResponse("我无法处理这个请求") |
||||
|
sendChatHistoryUpdate("personal_assistant", text, "我无法处理这个请求") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "AI处理出错: ${e.message}", e) |
||||
|
callback?.onError("AI处理出错") |
||||
|
} |
||||
|
}.start() |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 播放TTS |
||||
|
*/ |
||||
|
fun playTts(text: String, callback: TtsCallback? = null) { |
||||
|
isTtsSpeaking = true |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback { |
||||
|
override fun onStart(reqId: String) { |
||||
|
// TTS开始播放事件 |
||||
|
} |
||||
|
|
||||
|
override fun onProgress(reqId: String, progress: Double) { |
||||
|
// TTS播放进度事件 |
||||
|
} |
||||
|
|
||||
|
override fun onComplete(reqId: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
callback?.onComplete() |
||||
|
} |
||||
|
|
||||
|
override fun onError(reqId: String, errorCode: Int, errorMsg: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
callback?.onError(errorMsg) |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止TTS播放 |
||||
|
*/ |
||||
|
fun stopTts() { |
||||
|
if (isTtsSpeaking) { |
||||
|
volcanoTtsHelper?.stop() |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 播放AI回复 |
||||
|
*/ |
||||
|
private fun speakAIResponse(text: String) { |
||||
|
isTtsSpeaking = true |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
volcanoTtsHelper?.speak(text, object : VolcanoTtsHelper.TTSCallback { |
||||
|
override fun onStart(reqId: String) { |
||||
|
// TTS开始播放事件 |
||||
|
} |
||||
|
|
||||
|
override fun onProgress(reqId: String, progress: Double) { |
||||
|
// TTS播放进度事件 |
||||
|
} |
||||
|
|
||||
|
override fun onComplete(reqId: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onError(reqId: String, errorCode: Int, errorMsg: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 释放资源 |
||||
|
*/ |
||||
|
fun dispose() { |
||||
|
// 停止语音识别 |
||||
|
stopRecognition() |
||||
|
|
||||
|
// 停止TTS播放 |
||||
|
stopTts() |
||||
|
|
||||
|
// 释放Azure资源 |
||||
|
azureAsrHelper?.let { |
||||
|
FileLogger.d(TAG, "关闭Azure ASR服务") |
||||
|
it.dispose() |
||||
|
} |
||||
|
|
||||
|
volcanoTtsHelper?.let { |
||||
|
FileLogger.d(TAG, "关闭Volcano TTS服务") |
||||
|
it.release() |
||||
|
} |
||||
|
|
||||
|
FileLogger.d(TAG, "语音交互处理器资源已释放") |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 检查录音权限 |
||||
|
*/ |
||||
|
private fun checkRecordAudioPermission(): Boolean { |
||||
|
val permission = android.Manifest.permission.RECORD_AUDIO |
||||
|
val result = context.checkCallingOrSelfPermission(permission) |
||||
|
return result == android.content.pm.PackageManager.PERMISSION_GRANTED |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 通知状态变化 |
||||
|
*/ |
||||
|
private fun notifyStateChanged() { |
||||
|
callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 发送聊天历史更新 |
||||
|
*/ |
||||
|
private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) { |
||||
|
val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply { |
||||
|
putExtra("agentId", agentId) |
||||
|
putExtra("userMessage", userMessage) |
||||
|
putExtra("assistantMessage", assistantMessage) |
||||
|
putExtra("timestamp", System.currentTimeMillis()) |
||||
|
} |
||||
|
|
||||
|
// 发送广播 |
||||
|
context.sendBroadcast(intent) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 交互回调接口 |
||||
|
*/ |
||||
|
interface InteractionCallback { |
||||
|
fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) |
||||
|
fun onError(message: String) |
||||
|
fun onPromptRequest(message: String) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* TTS回调接口 |
||||
|
*/ |
||||
|
interface TtsCallback { |
||||
|
fun onComplete() |
||||
|
fun onError(error: String) |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,416 @@ |
|||||
|
package com.yunqiinnovation.deepsound |
||||
|
|
||||
|
import android.content.BroadcastReceiver |
||||
|
import android.content.Context |
||||
|
import android.content.Intent |
||||
|
import android.content.IntentFilter |
||||
|
import org.json.JSONArray |
||||
|
import org.json.JSONObject |
||||
|
import android.util.Log |
||||
|
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
||||
|
import com.yunqiinnovation.azure_speech.AzureAsrHelper |
||||
|
import com.yunqiinnovation.azure_speech.AzureTtsHelper |
||||
|
import com.yunqiinnovation.open_ai_service.OpenAIService |
||||
|
import com.yunqiinnovation.open_ai_service.SystemFunctionHandler |
||||
|
|
||||
|
|
||||
|
/** |
||||
|
* 语音交互处理器 - 处理语音识别、TTS和AI对话相关逻辑 |
||||
|
*/ |
||||
|
class VoiceInteractionHandler( |
||||
|
private val context: Context, |
||||
|
private val azureSpeechKey: String, |
||||
|
private val azureSpeechRegion: String, |
||||
|
private val openaiApiKey: String, |
||||
|
private val openaiBaseUrl: String = "", |
||||
|
private val openaiModel: String = "", |
||||
|
private val volcanoSpeechAppId: String, |
||||
|
private val volcanoSpeechAppToken: String |
||||
|
) { |
||||
|
private val TAG = "VoiceInteractionHandler" |
||||
|
|
||||
|
// Azure服务 |
||||
|
private var azureAsrHelper: AzureAsrHelper? = null |
||||
|
private var azureTtsHelper: AzureTtsHelper? = null |
||||
|
|
||||
|
// OpenAI服务 |
||||
|
private val openAIService = OpenAIService(context.applicationContext) |
||||
|
|
||||
|
// 当前用户输入 |
||||
|
private var currentUserInput = "" |
||||
|
|
||||
|
// 状态 |
||||
|
private var isInitialized = false |
||||
|
var isRecognitionActive = false |
||||
|
private set |
||||
|
var isTtsSpeaking = false |
||||
|
private set |
||||
|
var hasSpeechDetected = false |
||||
|
private set |
||||
|
|
||||
|
// 回调 |
||||
|
private var callback: InteractionCallback? = null |
||||
|
|
||||
|
// 广播接收器 |
||||
|
private val exitInteractionReceiver = object : BroadcastReceiver() { |
||||
|
override fun onReceive(context: Context, intent: Intent) { |
||||
|
if (intent.action == SystemFunctionHandler.ACTION_EXIT_INTERACTION) { |
||||
|
Log.d(TAG, "收到退出交互广播") |
||||
|
stopRecognition() |
||||
|
|
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 初始化 |
||||
|
*/ |
||||
|
fun initialize(): Boolean { |
||||
|
if (isInitialized) return true |
||||
|
|
||||
|
try { |
||||
|
// 初始化Azure ASR |
||||
|
azureAsrHelper = AzureAsrHelper(context).apply { |
||||
|
initialize(azureSpeechKey, azureSpeechRegion) |
||||
|
} |
||||
|
|
||||
|
// 初始化Azure TTS |
||||
|
azureTtsHelper = AzureTtsHelper(context).apply { |
||||
|
initialize(azureSpeechKey, azureSpeechRegion) |
||||
|
} |
||||
|
|
||||
|
// 初始化OpenAI服务 |
||||
|
openAIService.initialize(openaiApiKey, openaiBaseUrl, openaiModel) |
||||
|
|
||||
|
// 注册广播接收器 |
||||
|
try { |
||||
|
Log.d(TAG, "注册退出交互广播接收器,包名=${context.packageName}, action=${SystemFunctionHandler.ACTION_EXIT_INTERACTION}") |
||||
|
context.registerReceiver( |
||||
|
exitInteractionReceiver, |
||||
|
IntentFilter(SystemFunctionHandler.ACTION_EXIT_INTERACTION), |
||||
|
Context.RECEIVER_NOT_EXPORTED |
||||
|
) |
||||
|
Log.d(TAG, "退出交互广播接收器注册成功") |
||||
|
} catch (e: Exception) { |
||||
|
// 广播注册失败不应该影响整个应用初始化 |
||||
|
Log.e(TAG, "注册退出交互广播接收器失败: ${e.message}", e) |
||||
|
} |
||||
|
|
||||
|
isInitialized = true |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "初始化失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置回调 |
||||
|
*/ |
||||
|
fun setCallback(callback: InteractionCallback) { |
||||
|
this.callback = callback |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 开始语音识别 |
||||
|
*/ |
||||
|
fun startRecognition() { |
||||
|
if (isRecognitionActive) return |
||||
|
|
||||
|
// 检查录音权限 |
||||
|
if (!checkRecordAudioPermission()) { |
||||
|
callback?.onError("需要录音权限,请在设置中授予权限") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
isRecognitionActive = true |
||||
|
hasSpeechDetected = false |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
try { |
||||
|
azureAsrHelper?.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) { |
||||
|
if (recognizing.isNotEmpty()) { |
||||
|
hasSpeechDetected = true |
||||
|
stopTts() |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onResult(result: String, detectedLanguage: String) { |
||||
|
if (result.isNotEmpty()) { |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
processWithOpenAI(result) |
||||
|
|
||||
|
} |
||||
|
|
||||
|
// 重置状态,继续识别 |
||||
|
hasSpeechDetected = false |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStarted() { |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStopped() { |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onCanceled(reason: String, errorDetails: String) { |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
isRecognitionActive = false |
||||
|
callback?.onError("语音识别出错") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onSuccess(message: String) { |
||||
|
// 处理成功事件 |
||||
|
} |
||||
|
}) |
||||
|
} catch (e: Exception) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.e(TAG, "启动语音识别失败: ${e.message}", e) |
||||
|
callback?.onError("启动语音识别失败") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止语音识别 |
||||
|
*/ |
||||
|
fun stopRecognition() { |
||||
|
if (!isRecognitionActive) return |
||||
|
|
||||
|
FileLogger.d(TAG, "停止语音识别") |
||||
|
|
||||
|
try { |
||||
|
azureAsrHelper?.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onResult(result: String, detectedLanguage: String) {} |
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) {} |
||||
|
override fun onSessionStarted() {} |
||||
|
override fun onSessionStopped() { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别会话已停止") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onCanceled(reason: String, errorDetails: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别已取消: $reason") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onError(error: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.e(TAG, "停止语音识别时出错: $error") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
override fun onSuccess(message: String) { |
||||
|
isRecognitionActive = false |
||||
|
FileLogger.d(TAG, "语音识别已停止: $message") |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
}) |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止语音识别异常: ${e.message}", e) |
||||
|
// 确保状态一致性 |
||||
|
isRecognitionActive = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 使用OpenAI处理语音识别结果 |
||||
|
*/ |
||||
|
private fun processWithOpenAI(text: String) { |
||||
|
// 保存当前用户输入,用于后续同步聊天记录 |
||||
|
currentUserInput = text |
||||
|
|
||||
|
Thread { |
||||
|
try { |
||||
|
val messages = JSONArray().apply { |
||||
|
put(openAIService.createUserMessage(text)) |
||||
|
} |
||||
|
|
||||
|
// 创建响应构建器 |
||||
|
val responseBuilder = StringBuilder() |
||||
|
|
||||
|
openAIService.sendMessageStream( |
||||
|
messages = messages, |
||||
|
callback = object : OpenAIService.StreamCallback { |
||||
|
override fun onToken(token: String) { |
||||
|
// 累加响应内容 |
||||
|
responseBuilder.append(token) |
||||
|
} |
||||
|
|
||||
|
override fun onComplete() { |
||||
|
// 处理完整响应 |
||||
|
val response = responseBuilder.toString() |
||||
|
if (response.isNotEmpty()) { |
||||
|
// 播放AI回复 |
||||
|
Log.d(TAG, "AI 回复: $response") |
||||
|
|
||||
|
speakAIResponse(response) |
||||
|
|
||||
|
// 同步聊天记录到Flutter端 |
||||
|
sendChatHistoryUpdate("personal_assistant", text, response) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(e: Exception) { |
||||
|
FileLogger.e(TAG, "AI流式处理出错: ${e.message}", e) |
||||
|
callback?.onError("AI处理出错") |
||||
|
} |
||||
|
|
||||
|
override fun onFunctionCall(call: JSONObject) { |
||||
|
FileLogger.d(TAG, "processWithOpenAI 收到函数调用请求: ${call.getString("name")}") |
||||
|
|
||||
|
|
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "AI处理出错: ${e.message}", e) |
||||
|
callback?.onError("AI处理出错") |
||||
|
} |
||||
|
}.start() |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 播放TTS |
||||
|
*/ |
||||
|
fun playTts(text: String, callback: TtsCallback? = null) { |
||||
|
isTtsSpeaking = true |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback { |
||||
|
override fun onSuccess(message: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
callback?.onComplete() |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
callback?.onError(error) |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止TTS播放 |
||||
|
*/ |
||||
|
fun stopTts() { |
||||
|
if (isTtsSpeaking) { |
||||
|
azureTtsHelper?.stopSpeaking() |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 播放AI回复 |
||||
|
*/ |
||||
|
private fun speakAIResponse(text: String) { |
||||
|
isTtsSpeaking = true |
||||
|
notifyStateChanged() |
||||
|
|
||||
|
azureTtsHelper?.speakText(text, object : AzureTtsHelper.TTSCallback { |
||||
|
override fun onSuccess(message: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
isTtsSpeaking = false |
||||
|
notifyStateChanged() |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 释放资源 |
||||
|
*/ |
||||
|
fun dispose() { |
||||
|
// 停止语音识别 |
||||
|
stopRecognition() |
||||
|
|
||||
|
// 停止TTS播放 |
||||
|
stopTts() |
||||
|
|
||||
|
// 注销广播接收器 |
||||
|
try { |
||||
|
context.unregisterReceiver(exitInteractionReceiver) |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "注销广播接收器失败: ${e.message}", e) |
||||
|
} |
||||
|
|
||||
|
// 释放Azure资源 |
||||
|
azureAsrHelper?.let { |
||||
|
FileLogger.d(TAG, "关闭Azure ASR服务") |
||||
|
it.dispose() |
||||
|
} |
||||
|
|
||||
|
azureTtsHelper?.let { |
||||
|
FileLogger.d(TAG, "关闭Azure TTS服务") |
||||
|
it.dispose() |
||||
|
} |
||||
|
|
||||
|
FileLogger.d(TAG, "语音交互处理器资源已释放") |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 检查录音权限 |
||||
|
*/ |
||||
|
private fun checkRecordAudioPermission(): Boolean { |
||||
|
val permission = android.Manifest.permission.RECORD_AUDIO |
||||
|
val result = context.checkCallingOrSelfPermission(permission) |
||||
|
return result == android.content.pm.PackageManager.PERMISSION_GRANTED |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 通知状态变化 |
||||
|
*/ |
||||
|
private fun notifyStateChanged() { |
||||
|
callback?.onStateChanged(isRecognitionActive, isTtsSpeaking, hasSpeechDetected) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 发送聊天历史更新 |
||||
|
*/ |
||||
|
private fun sendChatHistoryUpdate(agentId: String, userMessage: String, assistantMessage: String) { |
||||
|
val intent = android.content.Intent(VoiceInteractionService.ACTION_CHAT_HISTORY_UPDATED).apply { |
||||
|
putExtra("agentId", agentId) |
||||
|
putExtra("userMessage", userMessage) |
||||
|
putExtra("assistantMessage", assistantMessage) |
||||
|
putExtra("timestamp", System.currentTimeMillis()) |
||||
|
setPackage(context.packageName) |
||||
|
} |
||||
|
|
||||
|
// 发送广播 |
||||
|
context.sendBroadcast(intent) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 交互回调接口 |
||||
|
*/ |
||||
|
interface InteractionCallback { |
||||
|
fun onStateChanged(isRecognitionActive: Boolean, isTtsSpeaking: Boolean, hasSpeechDetected: Boolean) |
||||
|
fun onError(message: String) |
||||
|
fun onPromptRequest(message: String) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* TTS回调接口 |
||||
|
*/ |
||||
|
interface TtsCallback { |
||||
|
fun onComplete() |
||||
|
fun onError(error: String) |
||||
|
} |
||||
|
} |
||||
@ -1,21 +0,0 @@ |
|||||
MIT License |
|
||||
|
|
||||
Copyright (c) 2024 Your Company |
|
||||
|
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy |
|
||||
of this software and associated documentation files (the "Software"), to deal |
|
||||
in the Software without restriction, including without limitation the rights |
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell |
|
||||
copies of the Software, and to permit persons to whom the Software is |
|
||||
furnished to do so, subject to the following conditions: |
|
||||
|
|
||||
The above copyright notice and this permission notice shall be included in all |
|
||||
copies or substantial portions of the Software. |
|
||||
|
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR |
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, |
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE |
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER |
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, |
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE |
|
||||
SOFTWARE. |
|
||||
@ -1,66 +0,0 @@ |
|||||
# Azure Speech Recognition |
|
||||
|
|
||||
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. |
|
||||
|
|
||||
## Features |
|
||||
|
|
||||
- Speech-to-text (Azure Speech Recognition) |
|
||||
- Text-to-speech (Azure Speech Synthesis) |
|
||||
- Support for multiple languages |
|
||||
- Language detection |
|
||||
- Continuous recognition |
|
||||
- Streaming synthesis |
|
||||
|
|
||||
## Getting Started |
|
||||
|
|
||||
### Prerequisites |
|
||||
|
|
||||
- Azure Speech service subscription key |
|
||||
- Azure Speech service region |
|
||||
|
|
||||
### Installation |
|
||||
|
|
||||
Add this to your package's `pubspec.yaml` file: |
|
||||
|
|
||||
```yaml |
|
||||
dependencies: |
|
||||
azure_speech_recognition: |
|
||||
path: ./azure |
|
||||
``` |
|
||||
|
|
||||
### Usage |
|
||||
|
|
||||
```dart |
|
||||
import 'package:azure_speech_recognition/azure_speech_recognition.dart'; |
|
||||
|
|
||||
// Initialize the service |
|
||||
await AzureSpeechRecognition.initialize( |
|
||||
subscriptionKey: 'your_subscription_key', |
|
||||
region: 'your_region', |
|
||||
supportedLanguages: ['zh-CN', 'en-US'], |
|
||||
); |
|
||||
|
|
||||
// Start continuous recognition |
|
||||
await AzureSpeechRecognition.startContinuousRecognition(); |
|
||||
|
|
||||
// Listen for recognition events |
|
||||
AzureSpeechRecognition.onRecognitionEvent.listen((event) { |
|
||||
if (event['type'] == 'result') { |
|
||||
print('Recognized: ${event['text']}'); |
|
||||
print('Detected language: ${event['detectedLanguage']}'); |
|
||||
} |
|
||||
}); |
|
||||
|
|
||||
// Stop recognition when done |
|
||||
await AzureSpeechRecognition.stopContinuousRecognition(); |
|
||||
|
|
||||
// Speak text |
|
||||
await AzureSpeechRecognition.speakText('Hello, world!'); |
|
||||
|
|
||||
// Clean up |
|
||||
await AzureSpeechRecognition.dispose(); |
|
||||
``` |
|
||||
|
|
||||
## License |
|
||||
|
|
||||
This project is licensed under the MIT License - see the LICENSE file for details. |
|
||||
@ -1,757 +0,0 @@ |
|||||
import Foundation |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
import AVFoundation |
|
||||
import AudioToolbox |
|
||||
|
|
||||
/// Azure ASR工具类,负责实现语音识别服务接口 |
|
||||
@available(iOS 13.0, *) |
|
||||
class AzureAsrHelper: NSObject { |
|
||||
// MARK: - 属性 |
|
||||
|
|
||||
/// 事件处理回调 |
|
||||
private var eventHandler: (String, [String: Any]) -> Void |
|
||||
|
|
||||
/// 语音配置信息 |
|
||||
private var speechSubscriptionKey: String = "" |
|
||||
private var serviceRegion: String = "" |
|
||||
|
|
||||
/// 语音识别相关 |
|
||||
private var speechConfig: SPXSpeechConfiguration? |
|
||||
private var recognizer: SPXSpeechRecognizer? |
|
||||
private var audioConfig: SPXAudioConfiguration? |
|
||||
private var pushStream: SPXPushAudioInputStream? |
|
||||
|
|
||||
/// 音频处理相关 |
|
||||
private var audioProcessor: CustomAudioProcessor? |
|
||||
private var isProcessingAudio = false |
|
||||
private var audioProcessingTimer: Timer? |
|
||||
|
|
||||
/// 状态标志 |
|
||||
private var isInitialized = false |
|
||||
private var _isContinuousRecognitionActive = false |
|
||||
|
|
||||
/// 当前语言和支持的语言 |
|
||||
private var currentLanguage = "zh-CN" |
|
||||
private var supportedLanguages: [String] = ["zh-CN", "en-US"] |
|
||||
private var isAutoDetectLanguage = false |
|
||||
|
|
||||
// MARK: - 初始化 |
|
||||
|
|
||||
init(eventHandler: @escaping (String, [String: Any]) -> Void) { |
|
||||
self.eventHandler = eventHandler |
|
||||
super.init() |
|
||||
} |
|
||||
|
|
||||
deinit { |
|
||||
dispose() |
|
||||
} |
|
||||
|
|
||||
// MARK: - ASR Service 接口实现 |
|
||||
|
|
||||
/// 初始化语音识别服务 |
|
||||
/// - Parameters: |
|
||||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|
||||
/// - serviceRegion: Azure 服务区域 (如 eastasia) |
|
||||
/// - supportedLanguages: 支持的语言代码数组 (可选) |
|
||||
/// - Returns: 初始化是否成功 |
|
||||
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool { |
|
||||
print("[AzureAsrHelper] 初始化 Azure 语音服务") |
|
||||
|
|
||||
// 检查配置是否为空 |
|
||||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|
||||
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
|
||||
eventHandler("error", ["message": "Azure 配置信息不完整"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 释放之前的资源 |
|
||||
dispose() |
|
||||
|
|
||||
// 记录配置信息 |
|
||||
self.speechSubscriptionKey = speechSubscriptionKey |
|
||||
self.serviceRegion = serviceRegion |
|
||||
|
|
||||
// 设置语言 |
|
||||
if let languages = supportedLanguages, !languages.isEmpty { |
|
||||
self.supportedLanguages = languages |
|
||||
} |
|
||||
|
|
||||
// 根据支持的语言数量决定是否启用自动语言检测 |
|
||||
isAutoDetectLanguage = self.supportedLanguages.count >= 2 |
|
||||
|
|
||||
// 如果只有一种语言,设置为当前语言 |
|
||||
if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty { |
|
||||
currentLanguage = self.supportedLanguages[0] |
|
||||
} |
|
||||
|
|
||||
// 创建识别器和设置回调 |
|
||||
if !createRecognizerAndSetupCallbacks() { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
|
||||
isInitialized = true |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
/// 创建识别器并设置回调 |
|
||||
private func createRecognizerAndSetupCallbacks() -> Bool { |
|
||||
// 释放之前的 recognizer |
|
||||
recognizer = nil |
|
||||
audioConfig = nil |
|
||||
|
|
||||
do { |
|
||||
// 创建语音配置 |
|
||||
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|
||||
|
|
||||
// 设置音频输入参数 |
|
||||
try setupAudioSession() |
|
||||
|
|
||||
// 创建自定义推送流,替代默认的麦克风输入 |
|
||||
pushStream = try SPXPushAudioInputStream() |
|
||||
audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) |
|
||||
|
|
||||
// 初始化自定义音频处理器 |
|
||||
audioProcessor = CustomAudioProcessor() |
|
||||
|
|
||||
// 设置语言配置 |
|
||||
if isAutoDetectLanguage { |
|
||||
// 设置自动语言检测 |
|
||||
speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode) |
|
||||
|
|
||||
// 创建自动语言检测配置 |
|
||||
let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) |
|
||||
|
|
||||
// 创建识别器 |
|
||||
recognizer = try SPXSpeechRecognizer( |
|
||||
speechConfiguration: speechConfig!, |
|
||||
autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig, |
|
||||
audioConfiguration: audioConfig! |
|
||||
) |
|
||||
} else { |
|
||||
// 设置指定的识别语言 |
|
||||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|
||||
|
|
||||
// 创建识别器 |
|
||||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|
||||
} |
|
||||
|
|
||||
// 设置所有回调 |
|
||||
setupAllCallbacks() |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 设置音频会话 |
|
||||
private func setupAudioSession() throws { |
|
||||
let audioSession = AVAudioSession.sharedInstance() |
|
||||
|
|
||||
// 使用playAndRecord类别允许同时录音和播放 |
|
||||
try audioSession.setCategory(.playAndRecord, |
|
||||
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 |
|
||||
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) |
|
||||
|
|
||||
// 设置首选的输入和输出 |
|
||||
let currentRoute = audioSession.currentRoute |
|
||||
|
|
||||
// 获取当前是否连接了耳机或外部麦克风 |
|
||||
let hasHeadphones = currentRoute.outputs.contains { |
|
||||
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP |
|
||||
} |
|
||||
|
|
||||
// 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 |
|
||||
if !hasHeadphones { |
|
||||
try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 |
|
||||
|
|
||||
// 启用回音消除和噪声抑制 |
|
||||
try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 |
|
||||
} else { |
|
||||
// 耳机模式,可以使用不同的设置 |
|
||||
try audioSession.setMode(.voiceChat) |
|
||||
try audioSession.setInputGain(1.0) |
|
||||
} |
|
||||
|
|
||||
// 设置合适的采样率 |
|
||||
try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 |
|
||||
try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 |
|
||||
|
|
||||
// 激活音频会话 |
|
||||
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|
||||
|
|
||||
print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") |
|
||||
} |
|
||||
|
|
||||
/// 设置所有回调 |
|
||||
private func setupAllCallbacks() { |
|
||||
guard let recognizer = recognizer else { return } |
|
||||
|
|
||||
// 最终识别结果 |
|
||||
recognizer.addRecognizedEventHandler { [weak self] _, event in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
if event.result.reason == SPXResultReason.recognizedSpeech { |
|
||||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
||||
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
||||
self.eventHandler("result", [ |
|
||||
"text": event.result.text ?? "", |
|
||||
"detectedLanguage": detectedLanguage |
|
||||
]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 识别中事件 |
|
||||
recognizer.addRecognizingEventHandler { [weak self] _, event in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
if event.result.reason == SPXResultReason.recognizingSpeech { |
|
||||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|
||||
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|
||||
self.eventHandler("recognizing", [ |
|
||||
"text": event.result.text ?? "", |
|
||||
"detectedLanguage": detectedLanguage |
|
||||
]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 会话事件 |
|
||||
recognizer.addSessionStartedEventHandler { [weak self] _, _ in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
print("[AzureAsrHelper] 识别会话已开始") |
|
||||
self._isContinuousRecognitionActive = true |
|
||||
self.eventHandler("sessionStarted", [:]) |
|
||||
} |
|
||||
|
|
||||
recognizer.addSessionStoppedEventHandler { [weak self] _, _ in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
print("[AzureAsrHelper] 识别会话已结束") |
|
||||
self._isContinuousRecognitionActive = false |
|
||||
self.eventHandler("sessionStopped", [:]) |
|
||||
} |
|
||||
|
|
||||
// 取消事件 |
|
||||
recognizer.addCanceledEventHandler { [weak self] _, event in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
let reason = event.reason.rawValue |
|
||||
let errorDetails = event.errorDetails ?? "未知错误" |
|
||||
|
|
||||
print("[AzureAsrHelper] 识别取消: \(errorDetails)") |
|
||||
|
|
||||
self.eventHandler("canceled", [ |
|
||||
"reason": reason, |
|
||||
"errorDetails": errorDetails |
|
||||
]) |
|
||||
|
|
||||
self._isContinuousRecognitionActive = false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 执行一次性语音识别 |
|
||||
/// - Returns: 是否成功启动识别 |
|
||||
func recognizeOnce() -> Bool { |
|
||||
if !isInitialized { |
|
||||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|
||||
eventHandler("error", ["message": "语音服务未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 如果正在连续识别,先停止 |
|
||||
if _isContinuousRecognitionActive { |
|
||||
stopContinuousRecognition() |
|
||||
} |
|
||||
|
|
||||
// 确保识别器已创建 |
|
||||
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 启动音频处理 |
|
||||
startAudioProcessing() |
|
||||
|
|
||||
// 通知会话开始 |
|
||||
eventHandler("sessionStarted", [:]) |
|
||||
|
|
||||
// 执行识别 |
|
||||
try recognizer?.recognizeOnceAsync { [weak self] result in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
self.stopAudioProcessing() |
|
||||
|
|
||||
if result.reason == SPXResultReason.recognizedSpeech { |
|
||||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
||||
self.eventHandler("result", [ |
|
||||
"text": result.text ?? "", |
|
||||
"detectedLanguage": detectedLanguage |
|
||||
]) |
|
||||
} else if result.reason == SPXResultReason.noMatch { |
|
||||
print("[AzureAsrHelper] 无匹配结果") |
|
||||
self.eventHandler("noMatch", [:]) |
|
||||
} else if result.reason == SPXResultReason.canceled { |
|
||||
do { |
|
||||
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) |
|
||||
let errorDetails = details.errorDetails ?? "未知错误" |
|
||||
self.eventHandler("error", ["message": "识别取消: \(errorDetails)"]) |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)") |
|
||||
self.eventHandler("error", ["message": "识别取消,无法获取详细原因"]) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) |
|
||||
stopAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 开始连续语音识别 |
|
||||
/// - Returns: 是否成功启动识别 |
|
||||
func startContinuousRecognition() -> Bool { |
|
||||
if !isInitialized { |
|
||||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|
||||
eventHandler("error", ["message": "语音服务未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 如果已经在进行连续识别,先停止 |
|
||||
if _isContinuousRecognitionActive { |
|
||||
stopContinuousRecognition() |
|
||||
} |
|
||||
|
|
||||
// 确保识别器已创建 |
|
||||
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 重新确保音频设置正确 |
|
||||
do { |
|
||||
try setupAudioSession() |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 启动音频处理 |
|
||||
startAudioProcessing() |
|
||||
|
|
||||
// 启动连续识别 |
|
||||
try recognizer?.startContinuousRecognition() |
|
||||
_isContinuousRecognitionActive = true |
|
||||
|
|
||||
print("[AzureAsrHelper] 连续识别开始") |
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) |
|
||||
_isContinuousRecognitionActive = false |
|
||||
stopAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 停止连续语音识别 |
|
||||
/// - Returns: 是否成功停止识别 |
|
||||
func stopContinuousRecognition() -> Bool { |
|
||||
// 停止音频处理 |
|
||||
stopAudioProcessing() |
|
||||
|
|
||||
if !_isContinuousRecognitionActive || recognizer == nil { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
try recognizer?.stopContinuousRecognition() |
|
||||
_isContinuousRecognitionActive = false |
|
||||
print("[AzureAsrHelper] 连续识别已停止") |
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"]) |
|
||||
_isContinuousRecognitionActive = false |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 检查连续识别是否活跃 |
|
||||
/// - Returns: 连续识别是否处于活跃状态 |
|
||||
func isContinuousRecognitionActive() -> Bool { |
|
||||
return _isContinuousRecognitionActive |
|
||||
} |
|
||||
|
|
||||
/// 释放资源 |
|
||||
func dispose() { |
|
||||
print("[AzureAsrHelper] 释放资源") |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopAudioProcessing() |
|
||||
|
|
||||
// 停止连续识别 |
|
||||
if _isContinuousRecognitionActive { |
|
||||
stopContinuousRecognition() |
|
||||
} |
|
||||
|
|
||||
// 释放音频会话 |
|
||||
do { |
|
||||
try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") |
|
||||
} |
|
||||
|
|
||||
// 释放资源 |
|
||||
recognizer = nil |
|
||||
speechConfig = nil |
|
||||
audioConfig = nil |
|
||||
pushStream = nil |
|
||||
audioProcessor = nil |
|
||||
|
|
||||
// 重置状态 |
|
||||
_isContinuousRecognitionActive = false |
|
||||
isInitialized = false |
|
||||
} |
|
||||
|
|
||||
/// 从结果中获取检测到的语言 |
|
||||
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
||||
if isAutoDetectLanguage { |
|
||||
do { |
|
||||
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
||||
return langResult.language ?? currentLanguage |
|
||||
} catch { |
|
||||
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") |
|
||||
return currentLanguage |
|
||||
} |
|
||||
} else { |
|
||||
return currentLanguage |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 音频处理 |
|
||||
|
|
||||
/// 开始音频处理 |
|
||||
private func startAudioProcessing() { |
|
||||
guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } |
|
||||
|
|
||||
isProcessingAudio = true |
|
||||
|
|
||||
// 启动音频处理器 |
|
||||
if !audioProcessor.startRecord() { |
|
||||
print("[AzureAsrHelper] 错误: 启动音频处理器失败") |
|
||||
eventHandler("error", ["message": "启动音频处理器失败"]) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 启动音频处理定时器 |
|
||||
audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in |
|
||||
guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 读取处理后的音频数据 |
|
||||
var bytes = [UInt8](repeating: 0, count: 2560) |
|
||||
let bytesRead = processor.read(bytes: &bytes) |
|
||||
|
|
||||
if bytesRead > 0 { |
|
||||
// 推送数据到Azure语音服务 |
|
||||
let data = Data(bytes: bytes, count: bytesRead) |
|
||||
stream.write(data) |
|
||||
|
|
||||
// 通知音频数据可用 |
|
||||
self.eventHandler("audioData", ["data": bytes]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
print("[AzureAsrHelper] 音频处理已启动") |
|
||||
} |
|
||||
|
|
||||
/// 停止音频处理 |
|
||||
private func stopAudioProcessing() { |
|
||||
// 停止定时器 |
|
||||
audioProcessingTimer?.invalidate() |
|
||||
audioProcessingTimer = nil |
|
||||
|
|
||||
// 停止音频处理器 |
|
||||
audioProcessor?.stopRecord() |
|
||||
|
|
||||
isProcessingAudio = false |
|
||||
print("[AzureAsrHelper] 音频处理已停止") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 自定义音频处理器 |
|
||||
|
|
||||
@available(iOS 13.0, *) |
|
||||
class CustomAudioProcessor: NSObject { |
|
||||
// 音频单元 |
|
||||
private var ioUnit: AudioUnit? |
|
||||
|
|
||||
// 音频格式 |
|
||||
private var audioFormat: AudioStreamBasicDescription |
|
||||
|
|
||||
// 音频缓冲 |
|
||||
private var audioBufferList: AudioBufferList |
|
||||
private var audioList: [Float] = [] |
|
||||
private let audioListQueue = DispatchQueue(label: "audioListQueue") |
|
||||
|
|
||||
// 回音消除状态 |
|
||||
private var isEchoCancellationEnabled = true |
|
||||
|
|
||||
override init() { |
|
||||
// 设置音频格式 - 16kHz, 16位, 单声道 |
|
||||
audioFormat = AudioStreamBasicDescription( |
|
||||
mSampleRate: 16000.0, |
|
||||
mFormatID: kAudioFormatLinearPCM, |
|
||||
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, |
|
||||
mBytesPerPacket: 2, |
|
||||
mFramesPerPacket: 1, |
|
||||
mBytesPerFrame: 2, |
|
||||
mChannelsPerFrame: 1, |
|
||||
mBitsPerChannel: 16, |
|
||||
mReserved: 0 |
|
||||
) |
|
||||
|
|
||||
// 初始化音频缓冲 |
|
||||
audioBufferList = AudioBufferList( |
|
||||
mNumberBuffers: 1, |
|
||||
mBuffers: AudioBuffer( |
|
||||
mNumberChannels: 1, |
|
||||
mDataByteSize: 4096, |
|
||||
mData: malloc(4096) |
|
||||
) |
|
||||
) |
|
||||
|
|
||||
super.init() |
|
||||
} |
|
||||
|
|
||||
deinit { |
|
||||
stopRecord() |
|
||||
free(audioBufferList.mBuffers.mData) |
|
||||
} |
|
||||
|
|
||||
/// 启动音频处理 |
|
||||
/// - Returns: 是否成功启动 |
|
||||
func startRecord() -> Bool { |
|
||||
print("[CustomAudioProcessor] 配置音频单元") |
|
||||
|
|
||||
// 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 |
|
||||
var ioUnitDescription = AudioComponentDescription( |
|
||||
componentType: kAudioUnitType_Output, |
|
||||
componentSubType: kAudioUnitSubType_VoiceProcessingIO, |
|
||||
componentManufacturer: kAudioUnitManufacturer_Apple, |
|
||||
componentFlags: 0, |
|
||||
componentFlagsMask: 0 |
|
||||
) |
|
||||
|
|
||||
// 查找音频组件 |
|
||||
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { |
|
||||
print("[CustomAudioProcessor] 错误: 未找到音频组件") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 创建音频单元实例 |
|
||||
if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { |
|
||||
ioUnit = nil |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 启用输入端口 |
|
||||
var enableInput: UInt32 = 1 |
|
||||
let kInputBus: AudioUnitElement = 1 |
|
||||
let kOutputBus: AudioUnitElement = 0 |
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|
||||
kAudioUnitScope_Input, kInputBus, &enableInput, |
|
||||
UInt32(MemoryLayout<UInt32>.size)), "启用输入端口") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 禁用输出端口 (我们只需要输入) |
|
||||
var enableOutput: UInt32 = 0 |
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|
||||
kAudioUnitScope_Output, kOutputBus, |
|
||||
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "禁用输出端口") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 设置缓冲区分配标志 |
|
||||
var flag: UInt32 = 0 |
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, |
|
||||
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置缓冲区分配标志") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 设置音频格式 |
|
||||
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size) |
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|
||||
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|
||||
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 启用回音消除 |
|
||||
if isEchoCancellationEnabled { |
|
||||
var echoCancellation: UInt32 = 1 |
|
||||
AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, |
|
||||
kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout<UInt32>.size)) |
|
||||
} |
|
||||
|
|
||||
// 设置输入回调 - 当有新音频数据时调用 |
|
||||
var inputCallback = AURenderCallbackStruct( |
|
||||
inputProc: CustomAudioProcessor.onAudioDataAvailable, |
|
||||
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) |
|
||||
) |
|
||||
|
|
||||
if checkError(AudioUnitSetProperty(ioUnit!, |
|
||||
kAudioOutputUnitProperty_SetInputCallback, |
|
||||
kAudioUnitScope_Global, kInputBus, |
|
||||
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") { |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 初始化音频单元 |
|
||||
var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") |
|
||||
while hasError { |
|
||||
Thread.sleep(forTimeInterval: 0.1) |
|
||||
hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") |
|
||||
} |
|
||||
|
|
||||
// 启动音频单元 |
|
||||
hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") |
|
||||
|
|
||||
print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") |
|
||||
return !hasError |
|
||||
} |
|
||||
|
|
||||
/// 停止音频处理 |
|
||||
func stopRecord() { |
|
||||
print("[CustomAudioProcessor] 停止音频处理器") |
|
||||
|
|
||||
if let ioUnit = ioUnit { |
|
||||
// 停止音频单元 |
|
||||
_ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") |
|
||||
|
|
||||
// 关闭音频单元 |
|
||||
_ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") |
|
||||
_ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") |
|
||||
|
|
||||
self.ioUnit = nil |
|
||||
} |
|
||||
|
|
||||
// 清空音频数据缓冲 |
|
||||
audioListQueue.sync { |
|
||||
audioList.removeAll() |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 音频数据回调 - 当有新的音频数据可用时调用 |
|
||||
private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in |
|
||||
// 获取实例 |
|
||||
let processor = Unmanaged<CustomAudioProcessor>.fromOpaque(inRefCon).takeUnretainedValue() |
|
||||
|
|
||||
// 计算预期数据大小 |
|
||||
let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame |
|
||||
|
|
||||
// 确保缓冲区足够大 |
|
||||
if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { |
|
||||
processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) |
|
||||
processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize |
|
||||
} |
|
||||
|
|
||||
// 渲染音频数据 |
|
||||
let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, |
|
||||
inBusNumber, inNumberFrames, &processor.audioBufferList), |
|
||||
"渲染音频数据") |
|
||||
|
|
||||
// 将Int16数据转换为浮点数据进行处理 |
|
||||
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) |
|
||||
let buffer = processor.audioBufferList.mBuffers |
|
||||
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) |
|
||||
|
|
||||
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) { |
|
||||
// 归一化到[-1.0, 1.0]范围 |
|
||||
audioDataFloat[j] = Float(bufferData[j]) / 32768.0 |
|
||||
} |
|
||||
|
|
||||
// 应用附加处理 (如有需要) |
|
||||
// processor.applyAdditionalProcessing(&audioDataFloat) |
|
||||
|
|
||||
// 保存处理后的数据 |
|
||||
if status == noErr { |
|
||||
processor.audioListQueue.async { |
|
||||
processor.audioList.append(contentsOf: audioDataFloat) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return status |
|
||||
} |
|
||||
|
|
||||
/// 读取处理后的音频数据 |
|
||||
/// - Parameter bytes: 输出字节数组 |
|
||||
/// - Returns: 读取的字节数 |
|
||||
func read(bytes: inout [UInt8]) -> Int { |
|
||||
return audioListQueue.sync { |
|
||||
// 如果没有数据,返回0 |
|
||||
if audioList.isEmpty { |
|
||||
return 0 |
|
||||
} |
|
||||
|
|
||||
// 确保有足够的数据 (至少1280个样本) |
|
||||
if audioList.count < 1280 { |
|
||||
return 0 |
|
||||
} |
|
||||
|
|
||||
// 读取一帧数据 (1280个样本) |
|
||||
let frameLength = 1280 |
|
||||
let buffer = Array(audioList.prefix(frameLength)) |
|
||||
audioList.removeFirst(frameLength) |
|
||||
|
|
||||
// 将浮点数据转回Int16格式 |
|
||||
var int16Data = buffer.map { Int16($0 * 32767) } |
|
||||
|
|
||||
// 转换为字节数组 |
|
||||
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) |
|
||||
bytes = [UInt8](data) |
|
||||
|
|
||||
// 每个样本2字节 (16位PCM) |
|
||||
return frameLength * 2 |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 检查错误并打印日志 |
|
||||
/// - Parameters: |
|
||||
/// - status: 操作状态 |
|
||||
/// - operation: 操作描述 |
|
||||
/// - Returns: 是否发生错误 |
|
||||
private func checkError(_ status: OSStatus, _ operation: String) -> Bool { |
|
||||
if status != noErr { |
|
||||
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") |
|
||||
return true |
|
||||
} |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
/// 检查OSStatus并返回状态 |
|
||||
/// - Parameters: |
|
||||
/// - status: 操作状态 |
|
||||
/// - operation: 操作描述 |
|
||||
/// - Returns: 原始状态 |
|
||||
private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { |
|
||||
if status != noErr { |
|
||||
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") |
|
||||
} |
|
||||
return status |
|
||||
} |
|
||||
} |
|
||||
@ -1,18 +0,0 @@ |
|||||
import Flutter |
|
||||
import UIKit |
|
||||
|
|
||||
public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { |
|
||||
public static func register(with registrar: FlutterPluginRegistrar) { |
|
||||
if #available(iOS 13.0, *) { |
|
||||
SwiftAzureSpeechRecognitionPlugin.register(with: registrar) |
|
||||
} else { |
|
||||
// 如果低于iOS 13.0,返回不支持的错误 |
|
||||
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) |
|
||||
channel.setMethodCallHandler { (call, result) in |
|
||||
result(FlutterError(code: "UNSUPPORTED", |
|
||||
message: "需要iOS 13.0及以上系统", |
|
||||
details: nil)) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
@ -1,427 +0,0 @@ |
|||||
import Foundation |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
import AVFoundation |
|
||||
|
|
||||
/// Azure TTS工具类,负责实现TTS服务接口 |
|
||||
@available(iOS 13.0, *) |
|
||||
class AzureTtsHelper: NSObject { |
|
||||
// MARK: - 属性 |
|
||||
|
|
||||
/// 事件处理回调 |
|
||||
private var eventHandler: (String, [String: Any]) -> Void |
|
||||
|
|
||||
/// 语音配置信息 |
|
||||
private var speechSubscriptionKey: String = "" |
|
||||
private var serviceRegion: String = "" |
|
||||
|
|
||||
/// 语音合成配置 |
|
||||
private var speechConfig: SPXSpeechConfiguration? |
|
||||
|
|
||||
/// 语音合成器 |
|
||||
private var synthesizer: SPXSpeechSynthesizer? |
|
||||
|
|
||||
/// 是否初始化成功 |
|
||||
private var isInitialized = false |
|
||||
|
|
||||
/// 当前是否正在播放 |
|
||||
private var _isSpeaking = false |
|
||||
|
|
||||
/// 音频会话配置 |
|
||||
private var isAudioSessionConfigured = false |
|
||||
|
|
||||
// MARK: - 语音设置 |
|
||||
|
|
||||
/// 当前语音 |
|
||||
private var currentVoice = "zh-CN-XiaoxiaoNeural" |
|
||||
|
|
||||
/// 支持的语音映射 |
|
||||
private var voiceMap: [String: String] = [ |
|
||||
"zh-CN": "zh-CN-XiaoxiaoNeural", |
|
||||
"en-US": "en-US-JennyNeural", |
|
||||
"ja-JP": "ja-JP-NanamiNeural", |
|
||||
"ko-KR": "ko-KR-SunHiNeural", |
|
||||
"zh-TW": "zh-TW-HsiaoChenNeural", |
|
||||
"zh-HK": "zh-HK-HiuMaanNeural" |
|
||||
] |
|
||||
|
|
||||
/// 当前语音合成参数 |
|
||||
private var currentSpeechRate = "0%" |
|
||||
private var currentPitch = "0%" |
|
||||
private var currentVolume = "100%" |
|
||||
|
|
||||
// MARK: - 初始化 |
|
||||
|
|
||||
init(eventHandler: @escaping (String, [String: Any]) -> Void) { |
|
||||
self.eventHandler = eventHandler |
|
||||
super.init() |
|
||||
} |
|
||||
|
|
||||
deinit { |
|
||||
dispose() |
|
||||
} |
|
||||
|
|
||||
// MARK: - TTS 接口实现 |
|
||||
|
|
||||
/// 初始化语音合成服务 |
|
||||
/// - Parameters: |
|
||||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|
||||
/// - serviceRegion: Azure 服务区域 (如 eastasia) |
|
||||
/// - language: 语言代码 (默认 zh-CN) |
|
||||
/// - Returns: 初始化是否成功 |
|
||||
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { |
|
||||
print("[AzureTtsHelper] 初始化语音合成服务") |
|
||||
|
|
||||
// 检查配置是否为空 |
|
||||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|
||||
print("[AzureTtsHelper] 错误: Azure 配置信息不完整") |
|
||||
eventHandler("error", ["error": "Azure 配置信息不完整"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 释放之前的资源 |
|
||||
dispose() |
|
||||
|
|
||||
// 记录配置信息 |
|
||||
self.speechSubscriptionKey = speechSubscriptionKey |
|
||||
self.serviceRegion = serviceRegion |
|
||||
|
|
||||
// 配置音频会话 |
|
||||
if !configureAudioSession() { |
|
||||
print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化") |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 创建语音配置 |
|
||||
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|
||||
|
|
||||
// 设置默认语音 |
|
||||
let defaultVoice = getDefaultVoiceForLanguage(language) |
|
||||
currentVoice = defaultVoice |
|
||||
speechConfig?.speechSynthesisVoiceName = defaultVoice |
|
||||
|
|
||||
// 创建语音合成器 |
|
||||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
||||
|
|
||||
// 设置事件处理器 |
|
||||
setupSynthesizerEvents() |
|
||||
|
|
||||
isInitialized = true |
|
||||
print("[AzureTtsHelper] TTS 引擎初始化成功") |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 配置音频会话 |
|
||||
private func configureAudioSession() -> Bool { |
|
||||
let audioSession = AVAudioSession.sharedInstance() |
|
||||
do { |
|
||||
// 使用playback类别,但支持混合和空中播放 |
|
||||
try audioSession.setCategory(.playback, |
|
||||
mode: .spokenAudio, |
|
||||
options: [.mixWithOthers, .allowAirPlay, .duckOthers]) |
|
||||
|
|
||||
// 根据设备类型选择最佳配置 |
|
||||
let currentRoute = audioSession.currentRoute |
|
||||
let hasHeadphones = currentRoute.outputs.contains { |
|
||||
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP |
|
||||
} |
|
||||
|
|
||||
// 优化音频路由 |
|
||||
if hasHeadphones { |
|
||||
// 耳机模式,使用默认设置 |
|
||||
try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟 |
|
||||
} else { |
|
||||
// 扬声器模式 |
|
||||
try audioSession.setPreferredIOBufferDuration(0.005) |
|
||||
} |
|
||||
|
|
||||
// 避免完全激活音频会话,因为ASR可能已经激活 |
|
||||
// 这里使用setActive(false)是为了不与ASR冲突 |
|
||||
if !audioSession.isOtherAudioPlaying { |
|
||||
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|
||||
} |
|
||||
|
|
||||
isAudioSessionConfigured = true |
|
||||
print("[AzureTtsHelper] 音频会话配置成功") |
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)") |
|
||||
isAudioSessionConfigured = false |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 设置语音 |
|
||||
/// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") |
|
||||
/// - Returns: 设置是否成功 |
|
||||
func setVoice(voiceName: String) -> Bool { |
|
||||
if !isInitialized { |
|
||||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|
||||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if voiceName.isEmpty { |
|
||||
print("[AzureTtsHelper] 错误: 声音名称为空") |
|
||||
eventHandler("error", ["error": "声音名称不能为空"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if voiceName == currentVoice { |
|
||||
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
print("[AzureTtsHelper] 设置声音: \(voiceName)") |
|
||||
currentVoice = voiceName |
|
||||
|
|
||||
// 更新语音配置 |
|
||||
if let speechConfig = speechConfig { |
|
||||
speechConfig.speechSynthesisVoiceName = voiceName |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
/// 设置语音合成参数 |
|
||||
/// - Parameters: |
|
||||
/// - rate: 语速,范围 -100 到 100,默认为 0 |
|
||||
/// - pitch: 音调,范围 -100 到 100,默认为 0 |
|
||||
/// - volume: 音量,范围 0 到 100,默认为 100 |
|
||||
/// - Returns: 是否设置成功 |
|
||||
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|
||||
if !isInitialized { |
|
||||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|
||||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 转换参数格式 |
|
||||
currentSpeechRate = formatRateParam(rate) |
|
||||
currentPitch = formatPitchParam(pitch) |
|
||||
currentVolume = formatVolumeParam(volume) |
|
||||
|
|
||||
print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)") |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
/// 合成文本为语音并播放 |
|
||||
/// - Parameter text: 要合成的文本 |
|
||||
/// - Returns: 操作是否成功启动 |
|
||||
func speakText(text: String) -> Bool { |
|
||||
if !isInitialized { |
|
||||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|
||||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if text.isEmpty { |
|
||||
print("[AzureTtsHelper] 警告: 要播放的文本为空") |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
// 确保音频会话已配置 |
|
||||
if !isAudioSessionConfigured { |
|
||||
_ = configureAudioSession() |
|
||||
} |
|
||||
|
|
||||
print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") |
|
||||
|
|
||||
// 生成SSML |
|
||||
let ssml = generateSsml(text: text) |
|
||||
|
|
||||
// 直接进行SSML合成 |
|
||||
return speakSsmlInternal(text: ssml) |
|
||||
} |
|
||||
|
|
||||
/// 内部SSML合成和播放 |
|
||||
private func speakSsmlInternal(text: String) -> Bool { |
|
||||
guard let synthesizer = synthesizer else { |
|
||||
print("[AzureTtsHelper] 错误: 合成器未初始化") |
|
||||
eventHandler("error", ["error": "合成器未初始化"]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
_isSpeaking = true |
|
||||
eventHandler("started", [:]) |
|
||||
|
|
||||
Task { |
|
||||
do { |
|
||||
// 使用异步方法进行合成并直接播放 |
|
||||
_ = try await synthesizer.startSpeakingSsml(text) |
|
||||
|
|
||||
} catch { |
|
||||
print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)") |
|
||||
DispatchQueue.main.async { |
|
||||
self._isSpeaking = false |
|
||||
self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"]) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
/// 停止当前语音合成 |
|
||||
/// - Returns: 操作是否成功 |
|
||||
func stopSpeaking() -> Bool { |
|
||||
if !isInitialized || !_isSpeaking { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
// 停止合成 |
|
||||
do { |
|
||||
try synthesizer?.stopSpeaking() |
|
||||
_isSpeaking = false |
|
||||
eventHandler("canceled", [:]) |
|
||||
print("[AzureTtsHelper] 已停止语音合成") |
|
||||
return true |
|
||||
} catch { |
|
||||
print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)") |
|
||||
eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 检查是否正在播放 |
|
||||
/// - Returns: 当前是否正在播放语音 |
|
||||
func isSpeaking() -> Bool { |
|
||||
return _isSpeaking |
|
||||
} |
|
||||
|
|
||||
/// 释放资源 |
|
||||
func dispose() { |
|
||||
try? stopSpeaking() |
|
||||
|
|
||||
// 释放合成器和配置 |
|
||||
synthesizer = nil |
|
||||
speechConfig = nil |
|
||||
|
|
||||
isInitialized = false |
|
||||
_isSpeaking = false |
|
||||
isAudioSessionConfigured = false |
|
||||
print("[AzureTtsHelper] TTS 引擎已释放") |
|
||||
} |
|
||||
|
|
||||
// MARK: - 私有辅助方法 |
|
||||
|
|
||||
/// 设置合成器事件处理 |
|
||||
private func setupSynthesizerEvents() { |
|
||||
guard let synthesizer = synthesizer else { return } |
|
||||
|
|
||||
// 添加书签到达事件处理 |
|
||||
synthesizer.addBookmarkReachedEventHandler { _, e in |
|
||||
print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"") |
|
||||
} |
|
||||
|
|
||||
// 合成完成事件 |
|
||||
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in |
|
||||
guard let self = self else { return } |
|
||||
print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)") |
|
||||
DispatchQueue.main.async { |
|
||||
self._isSpeaking = false |
|
||||
self.eventHandler("completed", [:]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 合成取消事件 |
|
||||
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
let result = e.result |
|
||||
do { |
|
||||
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result) |
|
||||
print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)") |
|
||||
|
|
||||
if cancellationDetails.reason == SPXCancellationReason.error { |
|
||||
print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)") |
|
||||
print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")") |
|
||||
} |
|
||||
|
|
||||
DispatchQueue.main.async { |
|
||||
self._isSpeaking = false |
|
||||
self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"]) |
|
||||
} |
|
||||
} catch { |
|
||||
print("[AzureTtsHelper] 获取取消详情时出错: \(error)") |
|
||||
|
|
||||
DispatchQueue.main.async { |
|
||||
self._isSpeaking = false |
|
||||
self.eventHandler("error", ["error": "语音合成被取消"]) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 合成开始事件 |
|
||||
synthesizer.addSynthesisStartedEventHandler { _, _ in |
|
||||
// print("[AzureTtsHelper] 语音合成开始") |
|
||||
} |
|
||||
|
|
||||
// 合成中事件 |
|
||||
synthesizer.addSynthesizingEventHandler { _, _ in |
|
||||
// print("[AzureTtsHelper] 语音合成中") |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 生成 SSML 文本 |
|
||||
private func generateSsml(text: String) -> String { |
|
||||
return """ |
|
||||
<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis' xml:lang='zh-CN'> |
|
||||
<voice name='\(currentVoice)'> |
|
||||
<prosody rate='\(currentSpeechRate)' pitch='\(currentPitch)' volume='\(currentVolume)'> |
|
||||
\(text) |
|
||||
</prosody> |
|
||||
</voice> |
|
||||
</speak> |
|
||||
""" |
|
||||
} |
|
||||
|
|
||||
/// 格式化语速参数 |
|
||||
private func formatRateParam(_ rate: Int) -> String { |
|
||||
let clampedRate = rate.clamp(min: -100, max: 100) |
|
||||
if clampedRate == 0 { |
|
||||
return "0%" |
|
||||
} else if clampedRate < 0 { |
|
||||
return "\(Int(Double(clampedRate) * 0.9))%" |
|
||||
} else { |
|
||||
return "+\(clampedRate)%" |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 格式化音调参数 |
|
||||
private func formatPitchParam(_ pitch: Int) -> String { |
|
||||
let clampedPitch = pitch.clamp(min: -100, max: 100) |
|
||||
if clampedPitch == 0 { |
|
||||
return "0%" |
|
||||
} else { |
|
||||
return "\(Int(Double(clampedPitch) * 0.5))%" |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/// 格式化音量参数 |
|
||||
private func formatVolumeParam(_ volume: Int) -> String { |
|
||||
let clampedVolume = volume.clamp(min: 0, max: 100) |
|
||||
return "\(clampedVolume)%" |
|
||||
} |
|
||||
|
|
||||
/// 获取指定语言的默认语音 |
|
||||
private func getDefaultVoiceForLanguage(_ language: String) -> String { |
|
||||
return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural" |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 扩展 |
|
||||
|
|
||||
extension Int { |
|
||||
func clamp(min: Int, max: Int) -> Int { |
|
||||
if self < min { return min } |
|
||||
if self > max { return max } |
|
||||
return self |
|
||||
} |
|
||||
} |
|
||||
@ -1,259 +0,0 @@ |
|||||
import Flutter |
|
||||
import UIKit |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
import AVFoundation |
|
||||
|
|
||||
@available(iOS 13.0, *) |
|
||||
public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { |
|
||||
private var azureChannel: FlutterMethodChannel |
|
||||
private var ttsChannel: FlutterMethodChannel |
|
||||
private var asrHelper: AzureAsrHelper |
|
||||
private var ttsHelper: AzureTtsHelper |
|
||||
private static var eventStreamHandler: AzureEventStreamHandler? |
|
||||
|
|
||||
// 创建方法到通道的映射 |
|
||||
private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]() |
|
||||
private static var asrMethodHandlers = [String: FlutterMethodCallHandler]() |
|
||||
|
|
||||
public static func register(with registrar: FlutterPluginRegistrar) { |
|
||||
// ASR通道 |
|
||||
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) |
|
||||
|
|
||||
// TTS通道 |
|
||||
let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger()) |
|
||||
|
|
||||
// 设置ASR事件通道 |
|
||||
let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger()) |
|
||||
eventStreamHandler = AzureEventStreamHandler() |
|
||||
eventChannel.setStreamHandler(eventStreamHandler) |
|
||||
|
|
||||
let instance = SwiftAzureSpeechRecognitionPlugin( |
|
||||
azureChannel: channel, |
|
||||
ttsChannel: ttsChannel, |
|
||||
eventStreamHandler: eventStreamHandler! |
|
||||
) |
|
||||
|
|
||||
// 直接设置各自通道的处理器 |
|
||||
channel.setMethodCallHandler(instance.handleAsrMethodCalls) |
|
||||
ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls) |
|
||||
} |
|
||||
|
|
||||
|
|
||||
|
|
||||
// 新增直接处理方法调用的函数 |
|
||||
private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|
||||
handleTtsMethod(call, result) |
|
||||
} |
|
||||
|
|
||||
private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|
||||
handleAsrMethod(call, result) |
|
||||
} |
|
||||
|
|
||||
|
|
||||
init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) { |
|
||||
self.azureChannel = azureChannel |
|
||||
self.ttsChannel = ttsChannel |
|
||||
|
|
||||
// 创建辅助类实例,使用自定义事件回调处理器 |
|
||||
let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in |
|
||||
DispatchQueue.main.async { |
|
||||
if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink { |
|
||||
var eventData = arguments |
|
||||
eventData["type"] = eventName |
|
||||
eventSink(eventData) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
asrHelper = AzureAsrHelper(eventHandler: eventHandler) |
|
||||
ttsHelper = AzureTtsHelper(eventHandler: eventHandler) |
|
||||
|
|
||||
super.init() |
|
||||
} |
|
||||
|
|
||||
private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|
||||
|
|
||||
let args = call.arguments as? Dictionary<String, Any> |
|
||||
|
|
||||
switch call.method { |
|
||||
case "initialize": |
|
||||
// 仅在初始化时读取必要参数 |
|
||||
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { |
|
||||
let errorMsg = "语音订阅密钥不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { |
|
||||
let errorMsg = "服务区域不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let supportedLanguages = args?["supportedLanguages"] as? [String] ?? [] |
|
||||
|
|
||||
let success = asrHelper.initialize( |
|
||||
speechSubscriptionKey: speechSubscriptionKey, |
|
||||
serviceRegion: serviceRegion, |
|
||||
supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages |
|
||||
) |
|
||||
result(success) |
|
||||
|
|
||||
case "startContinuousRecognition": |
|
||||
// 只有使用参数时才验证 |
|
||||
let success = asrHelper.startContinuousRecognition() |
|
||||
result(success) |
|
||||
|
|
||||
case "stopContinuousRecognition": |
|
||||
// 不需要额外参数 |
|
||||
let success = asrHelper.stopContinuousRecognition() |
|
||||
result(success) |
|
||||
|
|
||||
case "recognizeOnce": |
|
||||
// 只有使用参数时才验证 |
|
||||
let success = asrHelper.recognizeOnce() |
|
||||
result(success) |
|
||||
|
|
||||
case "isContinuousRecognitionActive": |
|
||||
// 不需要额外参数 |
|
||||
result(asrHelper.isContinuousRecognitionActive()) |
|
||||
|
|
||||
case "dispose": |
|
||||
// 不需要额外参数 |
|
||||
print("[AzurePlugin] 释放ASR资源") |
|
||||
asrHelper.dispose() |
|
||||
result(true) |
|
||||
|
|
||||
default: |
|
||||
print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)") |
|
||||
result(FlutterMethodNotImplemented) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|
||||
|
|
||||
let args = call.arguments as? Dictionary<String, Any> |
|
||||
|
|
||||
switch call.method { |
|
||||
case "initialize": |
|
||||
// 仅在初始化时验证参数 |
|
||||
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { |
|
||||
let errorMsg = "语音订阅密钥不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { |
|
||||
let errorMsg = "服务区域不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let language = args?["language"] as? String ?? "zh-CN" |
|
||||
|
|
||||
print("[AzurePlugin] 初始化TTS,语言: \(language)") |
|
||||
|
|
||||
let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language) |
|
||||
result(success) |
|
||||
|
|
||||
case "setVoice": |
|
||||
// 仅获取voice参数 |
|
||||
guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else { |
|
||||
let errorMsg = "声音名称不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
print("[AzurePlugin] 设置声音: \(voiceName)") |
|
||||
|
|
||||
let success = ttsHelper.setVoice(voiceName: voiceName) |
|
||||
result(success) |
|
||||
|
|
||||
case "speakText": |
|
||||
// 仅获取text参数 |
|
||||
let text = args?["text"] as? String ?? "" |
|
||||
|
|
||||
if text.isEmpty { |
|
||||
print("[AzurePlugin] 警告: 要播放的文本为空") |
|
||||
result("OK") |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
print("[AzurePlugin] 播放文本: \(text.prefix(50))...") |
|
||||
|
|
||||
let success = ttsHelper.speakText(text: text) |
|
||||
result(success ? "OK" : "ERROR") |
|
||||
|
|
||||
case "speakSsml": |
|
||||
// 仅获取ssml参数 |
|
||||
guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else { |
|
||||
let errorMsg = "SSML内容不能为空" |
|
||||
print("[AzurePlugin] 错误: \(errorMsg)") |
|
||||
result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...") |
|
||||
|
|
||||
// 由于我们移除了speakSsml方法,这里改用speakText方法 |
|
||||
// Azure SDK内部会自动检测是普通文本还是SSML |
|
||||
let success = ttsHelper.speakText(text: ssml) |
|
||||
result(success) |
|
||||
|
|
||||
case "stopSpeaking": |
|
||||
// 不需要参数 |
|
||||
print("[AzurePlugin] 停止播放") |
|
||||
let success = ttsHelper.stopSpeaking() |
|
||||
result(success) |
|
||||
|
|
||||
case "isSpeaking": |
|
||||
// 不需要参数 |
|
||||
result(ttsHelper.isSpeaking()) |
|
||||
|
|
||||
case "setSpeechParams": |
|
||||
// 仅获取语音参数 |
|
||||
let rate = args?["rate"] as? Int ?? 0 |
|
||||
let pitch = args?["pitch"] as? Int ?? 0 |
|
||||
let volume = args?["volume"] as? Int ?? 100 |
|
||||
|
|
||||
print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)") |
|
||||
let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume) |
|
||||
result(success) |
|
||||
|
|
||||
case "dispose": |
|
||||
// 释放TTS资源 |
|
||||
print("[AzurePlugin] 释放TTS资源") |
|
||||
ttsHelper.dispose() |
|
||||
result(true) |
|
||||
|
|
||||
default: |
|
||||
print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)") |
|
||||
result(FlutterMethodNotImplemented) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 用于处理事件流的辅助类 |
|
||||
@available(iOS 13.0, *) |
|
||||
class AzureEventStreamHandler: NSObject, FlutterStreamHandler { |
|
||||
var eventSink: FlutterEventSink? |
|
||||
|
|
||||
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
|
||||
self.eventSink = events |
|
||||
// 通知Flutter端事件通道已准备好 |
|
||||
DispatchQueue.main.async { |
|
||||
events(["type": "channelReady"]) |
|
||||
} |
|
||||
return nil |
|
||||
} |
|
||||
|
|
||||
func onCancel(withArguments arguments: Any?) -> FlutterError? { |
|
||||
self.eventSink = nil |
|
||||
return nil |
|
||||
} |
|
||||
} |
|
||||
@ -1,24 +0,0 @@ |
|||||
# |
|
||||
# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html. |
|
||||
# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing. |
|
||||
# |
|
||||
Pod::Spec.new do |s| |
|
||||
s.name = 'azure_speech_recognition' |
|
||||
s.version = '0.1.0' |
|
||||
s.summary = 'Azure Speech Recognition plugin for Flutter' |
|
||||
s.description = <<-DESC |
|
||||
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. |
|
||||
DESC |
|
||||
s.homepage = 'https://github.com/yourusername/azure_speech_recognition' |
|
||||
s.license = { :type => 'MIT', :file => '../LICENSE' } |
|
||||
s.author = { 'Your Company' => 'your-email@example.com' } |
|
||||
s.source = { :path => '.' } |
|
||||
s.source_files = 'Classes/**/*' |
|
||||
s.dependency 'Flutter' |
|
||||
s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0' |
|
||||
s.platform = :ios, '12.0' |
|
||||
|
|
||||
# Flutter.framework does not contain a i386 slice. |
|
||||
s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' } |
|
||||
s.swift_version = '5.0' |
|
||||
end |
|
||||
@ -1,6 +0,0 @@ |
|||||
// This is a placeholder file that exports nothing. |
|
||||
// The actual implementation is in the app's services folder. |
|
||||
// This file exists just to satisfy the Flutter plugin structure requirements. |
|
||||
|
|
||||
// Empty library to satisfy plugin structure |
|
||||
library azure_speech_recognition; |
|
||||
@ -1,23 +0,0 @@ |
|||||
name: azure_speech_recognition |
|
||||
description: Azure Speech Recognition and Text-to-Speech services Flutter plugin |
|
||||
version: 0.1.0 |
|
||||
homepage: https://github.com/yourusername/azure_speech_recognition |
|
||||
|
|
||||
environment: |
|
||||
sdk: '>=2.12.0 <3.0.0' |
|
||||
flutter: ">=2.0.0" |
|
||||
|
|
||||
dependencies: |
|
||||
flutter: |
|
||||
sdk: flutter |
|
||||
|
|
||||
dev_dependencies: |
|
||||
flutter_test: |
|
||||
sdk: flutter |
|
||||
flutter_lints: ^1.0.0 |
|
||||
|
|
||||
flutter: |
|
||||
plugin: |
|
||||
platforms: |
|
||||
ios: |
|
||||
pluginClass: AzureSpeechRecognitionPlugin |
|
||||
@ -0,0 +1,247 @@ |
|||||
|
import 'dart:async'; |
||||
|
import 'package:flutter_dotenv/flutter_dotenv.dart'; |
||||
|
import 'package:open_ai_service/open_ai_service.dart'; |
||||
|
import 'package:get/get.dart'; |
||||
|
import 'ai_service.dart'; |
||||
|
|
||||
|
/// OpenAI服务适配器 - 连接AiService接口与OpenAIService插件 |
||||
|
class OpenAIServiceAdapter implements AiService { |
||||
|
final OpenAIService _openAIService = OpenAIService(); |
||||
|
StreamSubscription<OpenAIEvent>? _eventSubscription; |
||||
|
final StreamController<String> _tokenStreamController = StreamController<String>.broadcast(); |
||||
|
bool _isProcessingStream = false; |
||||
|
|
||||
|
/// 构造函数 |
||||
|
OpenAIServiceAdapter() { |
||||
|
printInfo(info: '创建OpenAIServiceAdapter实例'); |
||||
|
_setupEventListener(); |
||||
|
} |
||||
|
|
||||
|
/// 设置事件监听器 |
||||
|
void _setupEventListener() { |
||||
|
try { |
||||
|
// 首先确保访问eventStream以初始化底层事件通道 |
||||
|
_openAIService.eventStream; |
||||
|
|
||||
|
// 设置事件处理 |
||||
|
_eventSubscription = _openAIService.eventStream.listen( |
||||
|
(event) { |
||||
|
if (!_isProcessingStream) return; |
||||
|
|
||||
|
try { |
||||
|
switch (event.type) { |
||||
|
case OpenAIEventType.token: |
||||
|
if (event.content is String) { |
||||
|
_tokenStreamController.add(event.content as String); |
||||
|
} else { |
||||
|
printInfo(info: '收到非字符串类型的token: ${event.content}'); |
||||
|
} |
||||
|
break; |
||||
|
case OpenAIEventType.complete: |
||||
|
_isProcessingStream = false; |
||||
|
break; |
||||
|
case OpenAIEventType.error: |
||||
|
if (event.content is String) { |
||||
|
_tokenStreamController.addError(event.content as String); |
||||
|
} else { |
||||
|
_tokenStreamController.addError('未知错误: ${event.content}'); |
||||
|
} |
||||
|
_isProcessingStream = false; |
||||
|
break; |
||||
|
case OpenAIEventType.functionCall: |
||||
|
try { |
||||
|
if (event.content is Map) { |
||||
|
_tokenStreamController.addError('收到函数调用,该流仅支持文本响应'); |
||||
|
} else { |
||||
|
_tokenStreamController.addError('收到未知格式的函数调用'); |
||||
|
printError(info: '函数调用格式错误: ${event.content}'); |
||||
|
} |
||||
|
} catch (e) { |
||||
|
printError(info: '处理函数调用事件出错: $e'); |
||||
|
_tokenStreamController.addError('处理函数调用失败: $e'); |
||||
|
} |
||||
|
_isProcessingStream = false; |
||||
|
break; |
||||
|
} |
||||
|
} catch (e) { |
||||
|
printError(info: '处理事件出错: $e'); |
||||
|
_tokenStreamController.addError('处理事件失败: $e'); |
||||
|
_isProcessingStream = false; |
||||
|
} |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
printError(info: '事件流错误: $error'); |
||||
|
_tokenStreamController.addError('事件流错误: $error'); |
||||
|
_isProcessingStream = false; |
||||
|
}, |
||||
|
onDone: () { |
||||
|
printInfo(info: '事件流已关闭'); |
||||
|
_isProcessingStream = false; |
||||
|
}, |
||||
|
); |
||||
|
} catch (e) { |
||||
|
printError(info: '设置事件监听器失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 初始化OpenAI服务 |
||||
|
Future<bool> initialize() async { |
||||
|
try { |
||||
|
// 从.env文件中读取配置 |
||||
|
final apiKey = dotenv.env['OPENAI_API_KEY'] ?? ''; |
||||
|
final baseUrl = dotenv.env['OPENAI_BASE_URL'] ?? ''; |
||||
|
final model = dotenv.env['OPENAI_MODEL'] ?? ''; |
||||
|
|
||||
|
printInfo(info: '从.env读取OpenAI配置'); |
||||
|
printInfo(info: '基础URL: $baseUrl'); |
||||
|
printInfo(info: '模型名称: $model'); |
||||
|
|
||||
|
if (apiKey.isEmpty) { |
||||
|
printError(info: '错误: OpenAI API密钥未配置,请在.env文件中设置OPENAI_API_KEY'); |
||||
|
return false; |
||||
|
} |
||||
|
|
||||
|
// 初始化OpenAI服务 |
||||
|
final result = await _openAIService.initialize( |
||||
|
apiKey: apiKey, |
||||
|
baseUrl: baseUrl, |
||||
|
model: model, |
||||
|
); |
||||
|
|
||||
|
if (result) { |
||||
|
printInfo(info: 'OpenAI服务初始化成功'); |
||||
|
} else { |
||||
|
printError(info: 'OpenAI服务初始化失败'); |
||||
|
} |
||||
|
|
||||
|
return result; |
||||
|
} catch (e) { |
||||
|
printError(info: 'OpenAI服务初始化异常: $e'); |
||||
|
return false; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息并获取回复 |
||||
|
@override |
||||
|
Future<String> sendMessage({ |
||||
|
required List<Map<String, String>> messages, |
||||
|
required String systemPrompt, |
||||
|
}) async { |
||||
|
try { |
||||
|
// 在方法内直接转换 |
||||
|
final convertedMessages = messages.map((m) => |
||||
|
Map<String, dynamic>.from(m)).toList(); |
||||
|
|
||||
|
// 发送消息并获取回复 |
||||
|
final response = await _openAIService.sendMessage( |
||||
|
messages: convertedMessages, |
||||
|
); |
||||
|
|
||||
|
return response; |
||||
|
} catch (e) { |
||||
|
printError(info: 'OpenAI发送消息失败: $e'); |
||||
|
throw '发送消息失败: $e'; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息并获取流式回复 |
||||
|
@override |
||||
|
Stream<String> sendMessageStream({ |
||||
|
required List<Map<String, String>> messages, |
||||
|
required String systemPrompt, |
||||
|
}) async* { |
||||
|
try { |
||||
|
// 在方法内直接转换 |
||||
|
final convertedMessages = messages.map((m) => |
||||
|
Map<String, dynamic>.from(m)).toList(); |
||||
|
|
||||
|
// 创建用于接收token的控制器 |
||||
|
final localController = StreamController<String>(); |
||||
|
|
||||
|
// 标记开始处理流 |
||||
|
_isProcessingStream = true; |
||||
|
|
||||
|
// 添加从广播流到本地流的订阅 |
||||
|
final subscription = _tokenStreamController.stream.listen( |
||||
|
(token) => localController.add(token), |
||||
|
onError: (error) { |
||||
|
printError(info: '令牌流错误: $error'); |
||||
|
localController.addError(error); |
||||
|
localController.close(); |
||||
|
}, |
||||
|
onDone: () { |
||||
|
printInfo(info: '令牌流已完成'); |
||||
|
localController.close(); |
||||
|
} |
||||
|
); |
||||
|
|
||||
|
// 当本地控制器关闭时,取消订阅 |
||||
|
localController.onCancel = () { |
||||
|
subscription.cancel(); |
||||
|
}; |
||||
|
|
||||
|
// 启动流式消息请求 |
||||
|
bool started = false; |
||||
|
try { |
||||
|
started = await _openAIService.sendMessageStream( |
||||
|
messages: convertedMessages, |
||||
|
); |
||||
|
} catch (e) { |
||||
|
printError(info: '启动消息流失败: $e'); |
||||
|
localController.addError('启动消息流失败: $e'); |
||||
|
localController.close(); |
||||
|
_isProcessingStream = false; |
||||
|
throw '启动消息流失败: $e'; |
||||
|
} |
||||
|
|
||||
|
if (!started) { |
||||
|
printError(info: '无法启动消息流'); |
||||
|
localController.addError('无法启动消息流'); |
||||
|
localController.close(); |
||||
|
_isProcessingStream = false; |
||||
|
throw '无法启动消息流'; |
||||
|
} |
||||
|
|
||||
|
// 通过yield*将controller的流转发 |
||||
|
yield* localController.stream; |
||||
|
} catch (e) { |
||||
|
printError(info: 'OpenAI流式消息处理失败: $e'); |
||||
|
throw '流式消息处理失败: $e'; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 注册函数 |
||||
|
Future<bool> registerFunction(String name, String description, Map<String, dynamic> parameters) async { |
||||
|
try { |
||||
|
final result = await _openAIService.registerFunction( |
||||
|
name: name, |
||||
|
description: description, |
||||
|
parameters: parameters, |
||||
|
); |
||||
|
|
||||
|
if (result) { |
||||
|
printInfo(info: '函数 "$name" 注册成功'); |
||||
|
} else { |
||||
|
printError(info: '函数 "$name" 注册失败'); |
||||
|
} |
||||
|
|
||||
|
return result; |
||||
|
} catch (e) { |
||||
|
printError(info: '注册函数失败: $e'); |
||||
|
return false; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 释放资源 |
||||
|
void dispose() { |
||||
|
try { |
||||
|
_isProcessingStream = false; |
||||
|
_eventSubscription?.cancel(); |
||||
|
_tokenStreamController.close(); |
||||
|
printInfo(info: 'OpenAIServiceAdapter资源已释放'); |
||||
|
} catch (e) { |
||||
|
printError(info: '释放资源时出错: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
} |
||||
@ -0,0 +1,136 @@ |
|||||
|
# Azure Speech 插件 |
||||
|
|
||||
|
本插件为Flutter提供了Azure语音服务的集成,包括: |
||||
|
|
||||
|
- 语音合成(TTS) |
||||
|
- 语音识别(ASR) |
||||
|
|
||||
|
## 功能 |
||||
|
|
||||
|
### 语音合成(TTS) |
||||
|
|
||||
|
- 支持多种语音(如中文、英文等) |
||||
|
- 语音参数调整(语速、音调、音量) |
||||
|
- 音频输出设备选择(扬声器、听筒、自动) |
||||
|
- SSML支持 |
||||
|
|
||||
|
### 语音识别(ASR) |
||||
|
|
||||
|
- 一次性语音识别 |
||||
|
- 连续语音识别 |
||||
|
- 自动语言检测 |
||||
|
- 音频处理优化(回音消除、噪声抑制等) |
||||
|
|
||||
|
## 平台支持 |
||||
|
|
||||
|
- Android |
||||
|
- iOS |
||||
|
|
||||
|
## 如何使用 |
||||
|
|
||||
|
### 初始化 |
||||
|
|
||||
|
```dart |
||||
|
import 'package:azure_speech/azure_speech.dart'; |
||||
|
|
||||
|
// 初始化TTS |
||||
|
await AzureSpeech.initializeTts( |
||||
|
'your_subscription_key', |
||||
|
'your_service_region', |
||||
|
language: 'zh-CN', |
||||
|
); |
||||
|
|
||||
|
// 初始化ASR |
||||
|
await AzureSpeech.initializeAsr( |
||||
|
'your_subscription_key', |
||||
|
'your_service_region', |
||||
|
['zh-CN', 'en-US'], |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### 语音合成 |
||||
|
|
||||
|
```dart |
||||
|
// 设置语音 |
||||
|
await AzureSpeech.setTtsVoice('zh-CN-XiaoxiaoNeural'); |
||||
|
|
||||
|
// 设置语音参数 |
||||
|
await AzureSpeech.setTtsSpeechParams( |
||||
|
rate: 0, // 语速 -100~100 |
||||
|
pitch: 0, // 音调 -100~100 |
||||
|
volume: 100, // 音量 0~100 |
||||
|
); |
||||
|
|
||||
|
// 设置音频输出设备 |
||||
|
await AzureSpeech.setTtsAudioOutputType('AUTO'); // 'SPEAKER', 'EARPIECE', 'AUTO' |
||||
|
|
||||
|
// 播放文本 |
||||
|
await AzureSpeech.speakText('你好,世界!'); |
||||
|
|
||||
|
// 停止播放 |
||||
|
await AzureSpeech.stopSpeaking(); |
||||
|
|
||||
|
// 检查是否正在播放 |
||||
|
bool isSpeaking = await AzureSpeech.isSpeaking(); |
||||
|
``` |
||||
|
|
||||
|
### 语音识别 |
||||
|
|
||||
|
```dart |
||||
|
// 一次性识别 |
||||
|
final result = await AzureSpeech.recognizeOnce(); |
||||
|
if (result['success']) { |
||||
|
print('识别文本: ${result['text']}'); |
||||
|
print('识别语言: ${result['language']}'); |
||||
|
} else { |
||||
|
print('识别失败: ${result['error']}'); |
||||
|
} |
||||
|
|
||||
|
// 连续识别 |
||||
|
// 监听识别结果 |
||||
|
AzureSpeech.asrResultStream.listen((event) { |
||||
|
switch (event['eventType']) { |
||||
|
case 'recognizing': |
||||
|
// 实时识别中的结果 |
||||
|
print('识别中: ${event['text']}'); |
||||
|
break; |
||||
|
case 'finalResult': |
||||
|
// 最终识别结果 |
||||
|
print('最终结果: ${event['text']}'); |
||||
|
break; |
||||
|
case 'error': |
||||
|
// 错误 |
||||
|
print('错误: ${event['error']}'); |
||||
|
break; |
||||
|
} |
||||
|
}); |
||||
|
|
||||
|
// 开始连续识别 |
||||
|
await AzureSpeech.startContinuousRecognition(); |
||||
|
|
||||
|
// 停止连续识别 |
||||
|
await AzureSpeech.stopContinuousRecognition(); |
||||
|
|
||||
|
// 检查连续识别是否活跃 |
||||
|
bool isActive = await AzureSpeech.isContinuousRecognitionActive(); |
||||
|
``` |
||||
|
|
||||
|
### 释放资源 |
||||
|
|
||||
|
```dart |
||||
|
// 释放所有资源 |
||||
|
await AzureSpeech.dispose(); |
||||
|
``` |
||||
|
|
||||
|
## 依赖项 |
||||
|
|
||||
|
本插件依赖于: |
||||
|
|
||||
|
- Microsoft Cognitive Services Speech SDK |
||||
|
- Flutter |
||||
|
|
||||
|
## 注意事项 |
||||
|
|
||||
|
- 使用前需要在Azure门户中创建语音服务资源,并获取订阅密钥和区域 |
||||
|
- Android需要相关权限:RECORD_AUDIO, INTERNET等 |
||||
|
- iOS需要在Info.plist中添加麦克风使用权限描述 |
||||
@ -0,0 +1,64 @@ |
|||||
|
import com.android.build.gradle.LibraryExtension |
||||
|
|
||||
|
buildscript { |
||||
|
repositories { |
||||
|
google() |
||||
|
mavenCentral() |
||||
|
} |
||||
|
dependencies { |
||||
|
classpath("com.android.tools.build:gradle:7.3.0") |
||||
|
classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
allprojects { |
||||
|
repositories { |
||||
|
google() |
||||
|
mavenCentral() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
plugins { |
||||
|
id("com.android.library") |
||||
|
kotlin("android") |
||||
|
} |
||||
|
|
||||
|
// 配置android扩展 |
||||
|
configure<LibraryExtension> { |
||||
|
namespace = "com.yunqiinnovation.azure_speech" |
||||
|
compileSdkVersion(33) |
||||
|
|
||||
|
defaultConfig { |
||||
|
minSdk = 21 |
||||
|
} |
||||
|
|
||||
|
compileOptions { |
||||
|
sourceCompatibility = JavaVersion.VERSION_11 |
||||
|
targetCompatibility = JavaVersion.VERSION_11 |
||||
|
} |
||||
|
|
||||
|
sourceSets { |
||||
|
getByName("main") { |
||||
|
manifest.srcFile("src/main/AndroidManifest.xml") |
||||
|
java.srcDirs("src/main/kotlin") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 添加lint选项 |
||||
|
lintOptions { |
||||
|
isCheckReleaseBuilds = false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 显式设置Kotlin JVM目标版本 |
||||
|
tasks.withType<org.jetbrains.kotlin.gradle.tasks.KotlinCompile> { |
||||
|
kotlinOptions { |
||||
|
jvmTarget = "11" |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
dependencies { |
||||
|
|
||||
|
// 添加Microsoft语音SDK |
||||
|
implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.30.0") |
||||
|
} |
||||
@ -0,0 +1 @@ |
|||||
|
rootProject.name = "azure_speech" |
||||
@ -0,0 +1,6 @@ |
|||||
|
<?xml version="1.0" encoding="utf-8"?> |
||||
|
<manifest xmlns:android="http://schemas.android.com/apk/res/android" |
||||
|
package="com.yunqiinnovation.azure_speech"> |
||||
|
<uses-permission android:name="android.permission.INTERNET" /> |
||||
|
<uses-permission android:name="android.permission.RECORD_AUDIO" /> |
||||
|
</manifest> |
||||
@ -0,0 +1,593 @@ |
|||||
|
package com.yunqiinnovation.azure_speech |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.media.AudioAttributes |
||||
|
import android.media.AudioFormat |
||||
|
import android.media.AudioRecord |
||||
|
import android.media.MediaRecorder |
||||
|
import android.media.audiofx.AcousticEchoCanceler |
||||
|
import android.media.audiofx.NoiseSuppressor |
||||
|
import android.media.audiofx.AutomaticGainControl |
||||
|
import android.os.Process |
||||
|
import com.yunqiinnovation.azure_speech.utils.FileLogger |
||||
|
import com.microsoft.cognitiveservices.speech.* |
||||
|
import com.microsoft.cognitiveservices.speech.audio.* |
||||
|
import com.microsoft.cognitiveservices.speech.util.EventHandler |
||||
|
import java.util.concurrent.ExecutionException |
||||
|
import java.util.concurrent.atomic.AtomicBoolean |
||||
|
|
||||
|
class AzureAsrHelper(private val context: Context) { |
||||
|
private var recognizer: SpeechRecognizer? = null |
||||
|
private var speechConfig: SpeechConfig? = null |
||||
|
private val TAG = "AzureAsrHelper" |
||||
|
private var isContinuousRecognitionActive = false |
||||
|
private var currentLanguage = "zh-CN" |
||||
|
private var subscriptionKey = "" |
||||
|
private var region = "" |
||||
|
private var isAutoDetectLanguage = false |
||||
|
private var supportedLanguages = arrayOf("zh-CN") |
||||
|
|
||||
|
// 是否使用回音消除 - 内部控制常量 |
||||
|
private val useEchoCancellation = false |
||||
|
|
||||
|
// 自定义音频处理相关 |
||||
|
private var customAudioProcessor: CustomAudioProcessor? = null |
||||
|
private var pushStream: PushAudioInputStream? = null |
||||
|
private var audioConfig: AudioConfig? = null |
||||
|
|
||||
|
// 初始化SDK并创建recognizer |
||||
|
fun initialize(subscriptionKey: String, region: String, |
||||
|
supportedLanguages: Array<String> = arrayOf("zh-CN")): Boolean { |
||||
|
try { |
||||
|
FileLogger.d(TAG, "初始化 Azure 语音服务") |
||||
|
|
||||
|
// 检查配置是否为空 |
||||
|
if (subscriptionKey.isEmpty() || region.isEmpty()) { |
||||
|
FileLogger.e(TAG, "Azure 配置信息不完整") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 释放之前的资源 |
||||
|
dispose() |
||||
|
|
||||
|
this.subscriptionKey = subscriptionKey |
||||
|
this.region = region |
||||
|
|
||||
|
// 设置语言 |
||||
|
if (supportedLanguages.isNotEmpty()) { |
||||
|
this.supportedLanguages = supportedLanguages |
||||
|
} |
||||
|
|
||||
|
// 根据支持的语言数量决定是否启用自动语言检测 |
||||
|
this.isAutoDetectLanguage = supportedLanguages.size >= 2 |
||||
|
|
||||
|
// 如果只有一种语言,设置为当前语言 |
||||
|
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { |
||||
|
this.currentLanguage = supportedLanguages[0] |
||||
|
} |
||||
|
|
||||
|
// 创建语音配置 |
||||
|
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) |
||||
|
|
||||
|
// 设置语言配置 |
||||
|
if (isAutoDetectLanguage) { |
||||
|
// 设置自动语言检测 |
||||
|
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") |
||||
|
} else { |
||||
|
// 设置指定的识别语言 |
||||
|
speechConfig?.speechRecognitionLanguage = currentLanguage |
||||
|
} |
||||
|
|
||||
|
// 创建识别器 |
||||
|
try { |
||||
|
if (useEchoCancellation) { |
||||
|
// 如果使用回音消除,创建自定义音频输入流 |
||||
|
setupCustomAudioProcessing() |
||||
|
|
||||
|
if (isAutoDetectLanguage) { |
||||
|
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
||||
|
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
||||
|
} else { |
||||
|
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
||||
|
} |
||||
|
} else { |
||||
|
// 使用默认麦克风输入 |
||||
|
if (isAutoDetectLanguage) { |
||||
|
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
||||
|
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
||||
|
} else { |
||||
|
recognizer = SpeechRecognizer(speechConfig) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
FileLogger.d(TAG, "Azure 语音服务初始化成功") |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "创建识别器失败: ${e.message}") |
||||
|
stopCustomAudioProcessing() |
||||
|
return false |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "初始化失败: ${e.message}") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 重置 recognizer |
||||
|
private fun resetRecognizer(): Boolean { |
||||
|
try { |
||||
|
// 释放之前的 recognizer |
||||
|
recognizer?.close() |
||||
|
recognizer = null |
||||
|
|
||||
|
// 停止当前的音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
|
||||
|
// 使用现有配置重新创建 recognizer |
||||
|
if (speechConfig != null) { |
||||
|
if (useEchoCancellation) { |
||||
|
// 如果使用回音消除,创建自定义音频输入流 |
||||
|
setupCustomAudioProcessing() |
||||
|
|
||||
|
if (isAutoDetectLanguage) { |
||||
|
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
||||
|
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
||||
|
} else { |
||||
|
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
||||
|
} |
||||
|
} else { |
||||
|
// 使用默认麦克风输入 |
||||
|
if (isAutoDetectLanguage) { |
||||
|
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
||||
|
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
||||
|
} else { |
||||
|
recognizer = SpeechRecognizer(speechConfig) |
||||
|
} |
||||
|
} |
||||
|
return true |
||||
|
} else { |
||||
|
FileLogger.e(TAG, "语音配置未初始化") |
||||
|
return false |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "重置识别器失败: ${e.message}") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 开始一次性语音识别 |
||||
|
fun recognizeOnce(callback: RecognizeCallback) { |
||||
|
if (speechConfig == null) { |
||||
|
callback.onError("语音服务未初始化") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 重置 recognizer |
||||
|
if (!resetRecognizer()) { |
||||
|
callback.onError("重置识别器失败") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 启动音频处理 |
||||
|
startCustomAudioProcessing() |
||||
|
|
||||
|
// 执行识别 |
||||
|
val result = recognizer?.recognizeOnceAsync()?.get() |
||||
|
|
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
|
||||
|
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
||||
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language |
||||
|
callback.onResult(result.text, detectedLanguage ?: "") |
||||
|
} else { |
||||
|
callback.onError("未能识别语音") |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
callback.onError("识别异常: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 开始连续语音识别 |
||||
|
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
||||
|
if (speechConfig == null) { |
||||
|
callback.onError("语音服务未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (isContinuousRecognitionActive) { |
||||
|
FileLogger.w(TAG, "已在进行连续识别,忽略请求") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
// 重置 recognizer |
||||
|
if (!resetRecognizer()) { |
||||
|
callback.onError("重置识别器失败") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 启动音频处理 |
||||
|
startCustomAudioProcessing() |
||||
|
|
||||
|
// 识别中事件 |
||||
|
recognizer?.recognizing?.addEventListener( |
||||
|
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
||||
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language |
||||
|
FileLogger.d(TAG, "识别中: ${event.result.text}, 语言: $detectedLanguage") |
||||
|
callback.onRecognizing(event.result.text, detectedLanguage ?: "") |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 识别完成事件 |
||||
|
recognizer?.recognized?.addEventListener( |
||||
|
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
||||
|
if (event.result.reason == ResultReason.RecognizedSpeech) { |
||||
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language |
||||
|
FileLogger.d(TAG, "识别完成: ${event.result.text}, 语言: $detectedLanguage") |
||||
|
callback.onResult(event.result.text, detectedLanguage ?: "") |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 会话开始事件 |
||||
|
recognizer?.sessionStarted?.addEventListener( |
||||
|
EventHandler<SessionEventArgs> { _, _ -> |
||||
|
FileLogger.d(TAG, "识别会话已开始") |
||||
|
callback.onSessionStarted() |
||||
|
callback.onSuccess("开始识别") // 兼容旧接口 |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 会话结束事件 |
||||
|
recognizer?.sessionStopped?.addEventListener( |
||||
|
EventHandler<SessionEventArgs> { _, _ -> |
||||
|
FileLogger.d(TAG, "识别会话已结束") |
||||
|
isContinuousRecognitionActive = false |
||||
|
stopCustomAudioProcessing() |
||||
|
callback.onSessionStopped() |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 取消事件 |
||||
|
recognizer?.canceled?.addEventListener( |
||||
|
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
||||
|
val errorDetails = try { |
||||
|
event.errorDetails ?: "未知错误" |
||||
|
} catch (e: Exception) { |
||||
|
"未知错误" |
||||
|
} |
||||
|
val reason = event.reason.toString() |
||||
|
FileLogger.e(TAG, "识别取消: $errorDetails") |
||||
|
isContinuousRecognitionActive = false |
||||
|
stopCustomAudioProcessing() |
||||
|
callback.onCanceled(reason, errorDetails) |
||||
|
callback.onError("识别取消: $errorDetails") // 兼容旧接口 |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 开始连续识别 |
||||
|
recognizer?.startContinuousRecognitionAsync() |
||||
|
isContinuousRecognitionActive = true |
||||
|
FileLogger.d(TAG, "连续识别已启动") |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
FileLogger.e(TAG, "开始连续识别失败: ${e.message}") |
||||
|
e.printStackTrace() |
||||
|
callback.onError("开始连续识别失败: ${e.message}") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 停止连续语音识别 |
||||
|
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
||||
|
if (speechConfig == null) { |
||||
|
callback.onError("语音服务未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (!isContinuousRecognitionActive) { |
||||
|
FileLogger.d(TAG, "未进行连续识别,忽略停止请求") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
FileLogger.d(TAG, "停止连续语音识别") |
||||
|
|
||||
|
if (recognizer == null) { |
||||
|
if (isContinuousRecognitionActive) { |
||||
|
FileLogger.w(TAG, "识别器为空,但状态显示活跃") |
||||
|
} |
||||
|
isContinuousRecognitionActive = false |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
// 停止连续识别 |
||||
|
val future = recognizer?.stopContinuousRecognitionAsync() |
||||
|
future?.get() |
||||
|
|
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
|
||||
|
isContinuousRecognitionActive = false |
||||
|
FileLogger.d(TAG, "连续识别已停止") |
||||
|
callback.onSuccess("连续识别已停止") |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
// 强制重置状态 |
||||
|
isContinuousRecognitionActive = false |
||||
|
FileLogger.e(TAG, "停止连续识别失败: ${e.message}") |
||||
|
e.printStackTrace() |
||||
|
callback.onError("停止连续识别失败: ${e.message}") |
||||
|
|
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
|
||||
|
// 尝试强制关闭识别器 |
||||
|
try { |
||||
|
recognizer?.close() |
||||
|
recognizer = null |
||||
|
} catch (ex: Exception) { |
||||
|
FileLogger.e(TAG, "关闭识别器失败: ${ex.message}") |
||||
|
} |
||||
|
|
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 释放资源 |
||||
|
fun dispose() { |
||||
|
try { |
||||
|
// 如果正在进行连续识别,先停止 |
||||
|
if (isContinuousRecognitionActive) { |
||||
|
recognizer?.stopContinuousRecognitionAsync()?.get() |
||||
|
isContinuousRecognitionActive = false |
||||
|
} |
||||
|
|
||||
|
// 停止音频处理 |
||||
|
stopCustomAudioProcessing() |
||||
|
|
||||
|
// 释放recognizer |
||||
|
recognizer?.close() |
||||
|
recognizer = null |
||||
|
|
||||
|
// 释放speechConfig |
||||
|
speechConfig?.close() |
||||
|
speechConfig = null |
||||
|
|
||||
|
FileLogger.d(TAG, "资源已释放") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "释放资源失败: ${e.message}") |
||||
|
|
||||
|
// 确保状态被重置 |
||||
|
isContinuousRecognitionActive = false |
||||
|
customAudioProcessor = null |
||||
|
pushStream = null |
||||
|
audioConfig = null |
||||
|
recognizer = null |
||||
|
speechConfig = null |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 检查连续识别是否处于活跃状态 |
||||
|
fun isContinuousRecognitionActive(): Boolean { |
||||
|
return isContinuousRecognitionActive |
||||
|
} |
||||
|
|
||||
|
// 设置自定义音频处理 |
||||
|
private fun setupCustomAudioProcessing() { |
||||
|
try { |
||||
|
// 创建音频推送流 |
||||
|
pushStream = PushAudioInputStream.create() |
||||
|
|
||||
|
// 创建音频配置 |
||||
|
audioConfig = AudioConfig.fromStreamInput(pushStream) |
||||
|
|
||||
|
// 创建自定义音频处理器 |
||||
|
customAudioProcessor = CustomAudioProcessor(pushStream!!) |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") |
||||
|
e.printStackTrace() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 启动自定义音频处理 |
||||
|
private fun startCustomAudioProcessing() { |
||||
|
if (useEchoCancellation && customAudioProcessor != null) { |
||||
|
try { |
||||
|
customAudioProcessor?.startProcessing() |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "启动音频处理器失败") |
||||
|
e.printStackTrace() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 停止自定义音频处理 |
||||
|
private fun stopCustomAudioProcessing() { |
||||
|
if (customAudioProcessor != null) { |
||||
|
try { |
||||
|
customAudioProcessor?.stopProcessing() |
||||
|
customAudioProcessor = null |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止音频处理器失败: ${e.message}") |
||||
|
e.printStackTrace() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 修改认证取消事件处理代码 |
||||
|
private fun setupCancelledEventHandler(callback: ContinuousRecognizeCallback) { |
||||
|
recognizer?.canceled?.addEventListener( |
||||
|
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
||||
|
val errorDetails = try { |
||||
|
event.errorDetails ?: "未知错误" |
||||
|
} catch (e: Exception) { |
||||
|
"未知错误" |
||||
|
} |
||||
|
val reason = event.reason.toString() |
||||
|
FileLogger.e(TAG, "识别取消: $errorDetails") |
||||
|
isContinuousRecognitionActive = false |
||||
|
stopCustomAudioProcessing() |
||||
|
callback.onCanceled(reason, errorDetails) |
||||
|
callback.onError("识别取消: $errorDetails") // 兼容旧接口 |
||||
|
} |
||||
|
) |
||||
|
} |
||||
|
|
||||
|
// 自定义音频处理器 |
||||
|
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream) { |
||||
|
private val isProcessing = AtomicBoolean(false) |
||||
|
private var audioRecord: AudioRecord? = null |
||||
|
private var echoCanceler: AcousticEchoCanceler? = null |
||||
|
private var noiseSuppressor: NoiseSuppressor? = null |
||||
|
private var automaticGainControl: AutomaticGainControl? = null |
||||
|
|
||||
|
// 音频配置 |
||||
|
private val sampleRate = 16000 // 16kHz,适合语音识别 |
||||
|
private val channelConfig = AudioFormat.CHANNEL_IN_MONO |
||||
|
private val audioFormat = AudioFormat.ENCODING_PCM_16BIT |
||||
|
|
||||
|
// 计算最小 buffer 大小 |
||||
|
private val bufferSize = AudioRecord.getMinBufferSize( |
||||
|
sampleRate, channelConfig, audioFormat |
||||
|
) |
||||
|
|
||||
|
// 启动音频处理 |
||||
|
fun startProcessing() { |
||||
|
if (isProcessing.get()) return |
||||
|
|
||||
|
// 创建录音对象 |
||||
|
try { |
||||
|
audioRecord = AudioRecord( |
||||
|
MediaRecorder.AudioSource.VOICE_RECOGNITION, |
||||
|
sampleRate, |
||||
|
channelConfig, |
||||
|
audioFormat, |
||||
|
bufferSize * 2 // 使用更大的缓冲区以确保不会丢失数据 |
||||
|
) |
||||
|
|
||||
|
// 创建音频处理效果 |
||||
|
if (AcousticEchoCanceler.isAvailable()) { |
||||
|
echoCanceler = AcousticEchoCanceler.create(audioRecord!!.audioSessionId) |
||||
|
echoCanceler?.enabled = true |
||||
|
} |
||||
|
|
||||
|
if (NoiseSuppressor.isAvailable()) { |
||||
|
noiseSuppressor = NoiseSuppressor.create(audioRecord!!.audioSessionId) |
||||
|
noiseSuppressor?.enabled = true |
||||
|
} |
||||
|
|
||||
|
if (AutomaticGainControl.isAvailable()) { |
||||
|
automaticGainControl = AutomaticGainControl.create(audioRecord!!.audioSessionId) |
||||
|
automaticGainControl?.enabled = true |
||||
|
} |
||||
|
|
||||
|
// 开始录音 |
||||
|
audioRecord?.startRecording() |
||||
|
|
||||
|
// 处理线程 |
||||
|
Thread { |
||||
|
android.os.Process.setThreadPriority(Process.THREAD_PRIORITY_AUDIO) |
||||
|
processAudio() |
||||
|
}.start() |
||||
|
|
||||
|
isProcessing.set(true) |
||||
|
FileLogger.d(TAG, "音频处理已启动") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "创建音频处理器失败: ${e.message}") |
||||
|
releaseAudioResources() |
||||
|
throw e |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 停止音频处理 |
||||
|
fun stopProcessing() { |
||||
|
if (!isProcessing.get()) return |
||||
|
|
||||
|
isProcessing.set(false) |
||||
|
releaseAudioResources() |
||||
|
FileLogger.d(TAG, "音频处理已停止") |
||||
|
} |
||||
|
|
||||
|
// 释放音频资源 |
||||
|
private fun releaseAudioResources() { |
||||
|
try { |
||||
|
audioRecord?.stop() |
||||
|
|
||||
|
echoCanceler?.release() |
||||
|
echoCanceler = null |
||||
|
|
||||
|
noiseSuppressor?.release() |
||||
|
noiseSuppressor = null |
||||
|
|
||||
|
automaticGainControl?.release() |
||||
|
automaticGainControl = null |
||||
|
|
||||
|
audioRecord?.release() |
||||
|
audioRecord = null |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "释放音频资源失败: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 音频处理线程 |
||||
|
private fun processAudio() { |
||||
|
// 设置线程优先级 |
||||
|
try { |
||||
|
Process.setThreadPriority(Process.THREAD_PRIORITY_URGENT_AUDIO) |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置线程优先级失败") |
||||
|
} |
||||
|
|
||||
|
val buffer = ByteArray(bufferSize) |
||||
|
|
||||
|
while (isProcessing.get()) { |
||||
|
try { |
||||
|
val readSize = audioRecord?.read(buffer, 0, buffer.size) ?: -1 |
||||
|
|
||||
|
if (readSize > 0) { |
||||
|
// 修复: 只传入buffer,不传readSize |
||||
|
// 创建新的byte数组,只包含读取到的数据 |
||||
|
val audioData = buffer.copyOfRange(0, readSize) |
||||
|
pushStream.write(audioData) |
||||
|
} |
||||
|
|
||||
|
// 适当休眠,避免占用过多 CPU |
||||
|
Thread.sleep(5) |
||||
|
} catch (e: Exception) { |
||||
|
if (isProcessing.get()) { |
||||
|
FileLogger.e(TAG, "处理音频数据异常: ${e.message}") |
||||
|
} |
||||
|
break |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 一次性识别回调接口 |
||||
|
interface RecognizeCallback { |
||||
|
fun onResult(text: String, detectedLanguage: String) |
||||
|
fun onError(error: String) |
||||
|
} |
||||
|
|
||||
|
// 连续识别回调接口 |
||||
|
interface ContinuousRecognizeCallback { |
||||
|
fun onResult(text: String, detectedLanguage: String) |
||||
|
fun onRecognizing(recognizing: String, detectedLanguage: String) |
||||
|
fun onSessionStarted() |
||||
|
fun onSessionStopped() |
||||
|
fun onCanceled(reason: String, errorDetails: String) |
||||
|
fun onError(error: String) |
||||
|
|
||||
|
// 兼容旧版本的接口 |
||||
|
fun onSuccess(message: String) {} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,297 @@ |
|||||
|
package com.yunqiinnovation.azure_speech |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.app.Activity |
||||
|
import android.os.Handler |
||||
|
import android.os.Looper |
||||
|
import androidx.annotation.NonNull |
||||
|
import com.yunqiinnovation.azure_speech.utils.FileLogger |
||||
|
|
||||
|
import io.flutter.embedding.engine.plugins.FlutterPlugin |
||||
|
import io.flutter.plugin.common.MethodCall |
||||
|
import io.flutter.plugin.common.MethodChannel |
||||
|
import io.flutter.plugin.common.MethodChannel.MethodCallHandler |
||||
|
import io.flutter.plugin.common.MethodChannel.Result |
||||
|
import io.flutter.plugin.common.EventChannel |
||||
|
|
||||
|
/** AzureSpeechPlugin */ |
||||
|
class AzureSpeechPlugin: FlutterPlugin { |
||||
|
private val TAG = "AzureSpeechPlugin" |
||||
|
private lateinit var context: Context |
||||
|
private val mainHandler = Handler(Looper.getMainLooper()) |
||||
|
|
||||
|
// ASR相关 |
||||
|
private lateinit var asrChannel : MethodChannel |
||||
|
private lateinit var asrEventChannel: EventChannel |
||||
|
private var asrEventSink: EventChannel.EventSink? = null |
||||
|
private lateinit var azureAsrHelper: AzureAsrHelper |
||||
|
|
||||
|
// TTS相关 |
||||
|
private lateinit var ttsChannel : MethodChannel |
||||
|
private lateinit var azureTtsHelper: AzureTtsHelper |
||||
|
|
||||
|
// ASR 事件发送方法 |
||||
|
private fun sendAsrEvent(event: Map<String, Any>) { |
||||
|
FileLogger.d(TAG, "发送ASR事件: $event") |
||||
|
if (asrEventSink == null) { |
||||
|
FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
mainHandler.post { |
||||
|
try { |
||||
|
asrEventSink?.success(event) |
||||
|
FileLogger.d(TAG, "ASR事件发送成功") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
context = flutterPluginBinding.applicationContext |
||||
|
|
||||
|
// 初始化ASR通道 |
||||
|
asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr") |
||||
|
asrChannel.setMethodCallHandler(AsrMethodHandler()) |
||||
|
|
||||
|
// 初始化TTS通道 |
||||
|
ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/tts") |
||||
|
ttsChannel.setMethodCallHandler(TtsMethodHandler()) |
||||
|
|
||||
|
// 初始化ASR事件通道 |
||||
|
asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events") |
||||
|
asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { |
||||
|
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { |
||||
|
asrEventSink = events |
||||
|
} |
||||
|
|
||||
|
override fun onCancel(arguments: Any?) { |
||||
|
asrEventSink = null |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
// 初始化Azure语音服务 |
||||
|
azureTtsHelper = AzureTtsHelper(context) |
||||
|
azureAsrHelper = AzureAsrHelper(context) |
||||
|
} |
||||
|
|
||||
|
// ASR方法处理器 |
||||
|
inner class AsrMethodHandler : MethodCallHandler { |
||||
|
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
||||
|
when (call.method) { |
||||
|
"initialize" -> { |
||||
|
val subscriptionKey = call.argument<String>("subscriptionKey") ?: "" |
||||
|
val region = call.argument<String>("region") ?: "" |
||||
|
val supportedLanguages = call.argument<List<String>>("supportedLanguages") ?: listOf("zh-CN") |
||||
|
|
||||
|
try { |
||||
|
val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages.toTypedArray()) |
||||
|
result.success(success) |
||||
|
} catch (e: Exception) { |
||||
|
result.error("INITIALIZATION_ERROR", e.message, null) |
||||
|
} |
||||
|
} |
||||
|
"recognizeOnce" -> { |
||||
|
azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { |
||||
|
override fun onResult(text: String, detectedLanguage: String) { |
||||
|
mainHandler.post { |
||||
|
result.success(mapOf( |
||||
|
"text" to text, |
||||
|
"detectedLanguage" to detectedLanguage |
||||
|
)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
mainHandler.post { |
||||
|
result.error("RECOGNITION_ERROR", error, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"startContinuousRecognition" -> { |
||||
|
// 确保事件通道已准备好 |
||||
|
if (asrEventSink == null) { |
||||
|
result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onResult(text: String, detectedLanguage: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "result", |
||||
|
"text" to text, |
||||
|
"detectedLanguage" to detectedLanguage |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "recognizing", |
||||
|
"text" to recognizing, |
||||
|
"detectedLanguage" to detectedLanguage |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStarted() { |
||||
|
sendAsrEvent(mapOf("type" to "sessionStarted")) |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStopped() { |
||||
|
sendAsrEvent(mapOf("type" to "sessionStopped")) |
||||
|
} |
||||
|
|
||||
|
override fun onCanceled(reason: String, errorDetails: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "canceled", |
||||
|
"reason" to reason, |
||||
|
"errorDetails" to errorDetails |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
sendAsrEvent(mapOf("type" to "error", "message" to error)) |
||||
|
} |
||||
|
|
||||
|
override fun onSuccess(message: String) { |
||||
|
sendAsrEvent(mapOf("type" to "success", "message" to message)) |
||||
|
} |
||||
|
}) |
||||
|
result.success(success) |
||||
|
} |
||||
|
"stopContinuousRecognition" -> { |
||||
|
try { |
||||
|
if (!azureAsrHelper.isContinuousRecognitionActive()) { |
||||
|
result.success(true) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
||||
|
override fun onResult(text: String, detectedLanguage: String) {} |
||||
|
override fun onRecognizing(recognizing: String, detectedLanguage: String) {} |
||||
|
override fun onSessionStarted() {} |
||||
|
override fun onSessionStopped() {} |
||||
|
override fun onCanceled(reason: String, errorDetails: String) {} |
||||
|
override fun onError(error: String) { |
||||
|
mainHandler.post { |
||||
|
result.error("STOP_ERROR", error, null) |
||||
|
} |
||||
|
} |
||||
|
override fun onSuccess(message: String) {} |
||||
|
}) |
||||
|
result.success(success) |
||||
|
} catch (e: Exception) { |
||||
|
result.error("STOP_ERROR", e.message, null) |
||||
|
} |
||||
|
} |
||||
|
"isContinuousRecognitionActive" -> { |
||||
|
result.success(azureAsrHelper.isContinuousRecognitionActive()) |
||||
|
} |
||||
|
"dispose" -> { |
||||
|
azureAsrHelper.dispose() |
||||
|
result.success(true) |
||||
|
} |
||||
|
else -> { |
||||
|
result.notImplemented() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// TTS方法处理器 |
||||
|
inner class TtsMethodHandler : MethodCallHandler { |
||||
|
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
||||
|
when (call.method) { |
||||
|
"initialize" -> { |
||||
|
val subscriptionKey = call.argument<String>("subscriptionKey") ?: "" |
||||
|
val region = call.argument<String>("region") ?: "" |
||||
|
val language = call.argument<String>("language") ?: "zh-CN" |
||||
|
|
||||
|
val success = azureTtsHelper.initialize(subscriptionKey, region, language) |
||||
|
result.success(success) |
||||
|
} |
||||
|
"setVoice" -> { |
||||
|
val voiceName = call.argument<String>("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) |
||||
|
result.success(azureTtsHelper.setVoice(voiceName)) |
||||
|
} |
||||
|
"setSpeechParams" -> { |
||||
|
val rate = call.argument<Int>("rate") ?: 0 |
||||
|
val pitch = call.argument<Int>("pitch") ?: 0 |
||||
|
val volume = call.argument<Int>("volume") ?: 100 |
||||
|
result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) |
||||
|
} |
||||
|
"setAudioOutputType" -> { |
||||
|
val outputTypeStr = call.argument<String>("outputType") ?: "speaker" |
||||
|
val outputType = when (outputTypeStr.lowercase()) { |
||||
|
"speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER |
||||
|
"earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE |
||||
|
"auto" -> AzureTtsHelper.AudioOutputType.AUTO |
||||
|
else -> AzureTtsHelper.AudioOutputType.SPEAKER |
||||
|
} |
||||
|
result.success(azureTtsHelper.setAudioOutputType(outputType)) |
||||
|
} |
||||
|
"speakText" -> { |
||||
|
val text = call.argument<String>("text") ?: return result.error("INVALID_ARGUMENTS", "文本不能为空", null) |
||||
|
|
||||
|
azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { |
||||
|
override fun onSuccess(message: String) { |
||||
|
mainHandler.post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
mainHandler.post { |
||||
|
result.error("SPEAK_ERROR", error, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"speakSsml" -> { |
||||
|
val ssml = call.argument<String>("ssml") ?: return result.error("INVALID_ARGUMENTS", "SSML不能为空", null) |
||||
|
|
||||
|
azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { |
||||
|
override fun onSuccess(message: String) { |
||||
|
mainHandler.post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
mainHandler.post { |
||||
|
result.error("SPEAK_ERROR", error, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"stopSpeaking" -> { |
||||
|
result.success(azureTtsHelper.stopSpeaking()) |
||||
|
} |
||||
|
"isSpeaking" -> { |
||||
|
result.success(azureTtsHelper.isSpeaking()) |
||||
|
} |
||||
|
"dispose" -> { |
||||
|
azureTtsHelper.dispose() |
||||
|
result.success(true) |
||||
|
} |
||||
|
else -> { |
||||
|
result.notImplemented() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
asrChannel.setMethodCallHandler(null) |
||||
|
ttsChannel.setMethodCallHandler(null) |
||||
|
asrEventChannel.setStreamHandler(null) |
||||
|
|
||||
|
try { |
||||
|
azureTtsHelper.dispose() |
||||
|
azureAsrHelper.dispose() |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "Dispose resources error: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,45 @@ |
|||||
|
package com.yunqiinnovation.azure_speech.utils |
||||
|
|
||||
|
import android.util.Log |
||||
|
|
||||
|
/** |
||||
|
* 简单的文件日志工具类 |
||||
|
*/ |
||||
|
object FileLogger { |
||||
|
private const val TAG_PREFIX = "AzureSpeech_" |
||||
|
|
||||
|
/** |
||||
|
* 记录调试信息 |
||||
|
*/ |
||||
|
fun d(tag: String, message: String) { |
||||
|
Log.d("$TAG_PREFIX$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录信息 |
||||
|
*/ |
||||
|
fun i(tag: String, message: String) { |
||||
|
Log.i("$TAG_PREFIX$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录警告信息 |
||||
|
*/ |
||||
|
fun w(tag: String, message: String) { |
||||
|
Log.w("$TAG_PREFIX$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录错误信息 |
||||
|
*/ |
||||
|
fun e(tag: String, message: String) { |
||||
|
Log.e("$TAG_PREFIX$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录异常 |
||||
|
*/ |
||||
|
fun e(tag: String, message: String, throwable: Throwable) { |
||||
|
Log.e("$TAG_PREFIX$tag", message, throwable) |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,461 @@ |
|||||
|
import Foundation |
||||
|
import AVFoundation |
||||
|
import MicrosoftCognitiveServicesSpeech |
||||
|
|
||||
|
/// Azure 语音识别辅助类 |
||||
|
class AzureAsrHelper: NSObject { |
||||
|
private var recognizer: SPXSpeechRecognizer? |
||||
|
private var speechConfig: SPXSpeechConfig? |
||||
|
private var audioConfig: SPXAudioConfig? |
||||
|
private var initialized = false |
||||
|
private var isContinuousRecognitionActive = false |
||||
|
private var currentLanguage = "zh-CN" |
||||
|
private var subscriptionKey = "" |
||||
|
private var serviceRegion = "" |
||||
|
private var isAutoDetectLanguage = false |
||||
|
private var supportedLanguages = ["zh-CN", "en-US"] |
||||
|
|
||||
|
// 音频会话管理 |
||||
|
private let audioSession = AVAudioSession.sharedInstance() |
||||
|
|
||||
|
// 事件回调 |
||||
|
private var eventHandler: (([String: Any]) -> Void)? |
||||
|
|
||||
|
/// 设置事件处理器 |
||||
|
/// |
||||
|
/// - Parameter handler: 事件处理回调 |
||||
|
func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) { |
||||
|
self.eventHandler = handler |
||||
|
} |
||||
|
|
||||
|
/// 初始化语音识别服务 |
||||
|
/// |
||||
|
/// - Parameters: |
||||
|
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
||||
|
/// - serviceRegion: Azure 语音服务区域 |
||||
|
/// - supportedLanguages: 支持的语言列表,默认为 ["zh-CN", "en-US"] |
||||
|
/// - Returns: 是否初始化成功 |
||||
|
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String] = ["zh-CN", "en-US"]) -> Bool { |
||||
|
print("[AzureAsrHelper] 初始化 Azure 语音服务") |
||||
|
|
||||
|
// 检查配置是否为空 |
||||
|
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
||||
|
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 释放之前的资源 |
||||
|
dispose() |
||||
|
|
||||
|
// 保存配置 |
||||
|
self.subscriptionKey = speechSubscriptionKey |
||||
|
self.serviceRegion = serviceRegion |
||||
|
|
||||
|
// 设置语言 |
||||
|
if supportedLanguages.isEmpty { |
||||
|
print("[AzureAsrHelper] 警告: 传入的支持语言列表为空,将使用默认语言") |
||||
|
} else { |
||||
|
self.supportedLanguages = supportedLanguages |
||||
|
} |
||||
|
|
||||
|
// 根据支持的语言数量决定是否启用自动语言检测 |
||||
|
self.isAutoDetectLanguage = supportedLanguages.count >= 2 |
||||
|
|
||||
|
// 如果只有一种语言,设置为当前语言 |
||||
|
if !isAutoDetectLanguage && !supportedLanguages.isEmpty { |
||||
|
self.currentLanguage = supportedLanguages[0] |
||||
|
} |
||||
|
|
||||
|
// 创建语音配置 |
||||
|
do { |
||||
|
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) |
||||
|
|
||||
|
// 设置语言配置 |
||||
|
if isAutoDetectLanguage { |
||||
|
// 设置自动语言检测 |
||||
|
try speechConfig?.setPropertyTo("Continuous", byId: SPXPropertyId.SpeechServiceConnection_LanguageIdMode) |
||||
|
} else { |
||||
|
// 设置指定的识别语言 |
||||
|
speechConfig?.speechRecognitionLanguage = currentLanguage |
||||
|
} |
||||
|
|
||||
|
// 创建音频配置 - 使用默认麦克风 |
||||
|
audioConfig = SPXAudioConfig.default() |
||||
|
|
||||
|
// 创建识别器 |
||||
|
if isAutoDetectLanguage { |
||||
|
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) |
||||
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) |
||||
|
} else { |
||||
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
||||
|
} |
||||
|
|
||||
|
// 配置音频会话 |
||||
|
try configureAudioSession() |
||||
|
|
||||
|
initialized = true |
||||
|
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 配置音频会话 |
||||
|
private func configureAudioSession() throws { |
||||
|
print("[AzureAsrHelper] 开始配置音频会话...") |
||||
|
|
||||
|
do { |
||||
|
// 设置音频会话类别和模式 |
||||
|
try audioSession.setCategory(.record, mode: .measurement, options: [.duckOthers, .allowBluetooth]) |
||||
|
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
||||
|
} catch { |
||||
|
print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败") |
||||
|
throw error |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 一次性语音识别 |
||||
|
/// |
||||
|
/// - Parameter completion: 完成回调,返回是否成功、识别文本、识别语言和可能的错误信息 |
||||
|
func recognizeOnce(completion: @escaping (Bool, String?, String?, String?) -> Void) { |
||||
|
if !initialized { |
||||
|
completion(false, nil, nil, "语音服务未初始化") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 重置 recognizer |
||||
|
if !resetRecognizer() { |
||||
|
completion(false, nil, nil, "重置识别器失败") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
// 激活音频会话 |
||||
|
try audioSession.setActive(true) |
||||
|
|
||||
|
// 添加识别事件处理 |
||||
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
if event.result.reason == .recognizedSpeech { |
||||
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
||||
|
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
||||
|
completion(true, event.result.text, detectedLanguage, nil) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
recognizer?.addRecognizingEventHandler { [weak self] _, event in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
if event.result.reason == .recognizingSpeech { |
||||
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
||||
|
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 添加会话事件处理 |
||||
|
recognizer?.addSessionStartedEventHandler { _, _ in |
||||
|
print("[AzureAsrHelper] 识别会话已开始") |
||||
|
} |
||||
|
|
||||
|
recognizer?.addSessionStoppedEventHandler { _, _ in |
||||
|
print("[AzureAsrHelper] 识别会话已结束") |
||||
|
} |
||||
|
|
||||
|
// 添加取消事件处理 |
||||
|
recognizer?.addCanceledEventHandler { _, event in |
||||
|
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { |
||||
|
let errorDetails = cancellationDetails.errorDetails ?? "未知错误" |
||||
|
print("[AzureAsrHelper] 识别取消: \(errorDetails)") |
||||
|
completion(false, nil, nil, "识别取消: \(errorDetails)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 执行识别 |
||||
|
let result = try recognizer?.recognizeOnceAsync().get() |
||||
|
|
||||
|
if result?.reason != .recognizedSpeech { |
||||
|
completion(false, nil, nil, "未能识别语音") |
||||
|
} |
||||
|
} catch { |
||||
|
completion(false, nil, nil, "识别异常: \(error.localizedDescription)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 重置识别器 |
||||
|
/// |
||||
|
/// - Returns: 是否重置成功 |
||||
|
private func resetRecognizer() -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 检查配置是否为空 |
||||
|
if subscriptionKey.isEmpty || serviceRegion.isEmpty { |
||||
|
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
// 释放之前的 recognizer |
||||
|
recognizer = nil |
||||
|
|
||||
|
// 创建音频配置 - 使用默认麦克风 |
||||
|
audioConfig = SPXAudioConfig.default() |
||||
|
|
||||
|
// 重新创建识别器 |
||||
|
if isAutoDetectLanguage { |
||||
|
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) |
||||
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) |
||||
|
} else { |
||||
|
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 获取检测到的语言 |
||||
|
/// |
||||
|
/// - Parameter result: 识别结果 |
||||
|
/// - Returns: 检测到的语言代码 |
||||
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
||||
|
if isAutoDetectLanguage { |
||||
|
do { |
||||
|
if let autoDetectResult = try SPXAutoDetectSourceLanguageResult(fromRecognitionResult: result) { |
||||
|
return autoDetectResult.language |
||||
|
} |
||||
|
return "" |
||||
|
} catch { |
||||
|
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") |
||||
|
return "" |
||||
|
} |
||||
|
} else { |
||||
|
return currentLanguage |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 开始连续语音识别 |
||||
|
/// |
||||
|
/// - Returns: 是否成功启动连续识别 |
||||
|
func startContinuousRecognition() -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 检查配置是否为空 |
||||
|
if subscriptionKey.isEmpty || serviceRegion.isEmpty { |
||||
|
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 如果已经在进行连续识别,直接返回 |
||||
|
if isContinuousRecognitionActive { |
||||
|
print("[AzureAsrHelper] 已经在进行连续识别中,忽略请求") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
// 重置 recognizer |
||||
|
if !resetRecognizer() { |
||||
|
print("[AzureAsrHelper] 尝试重新创建识别器...") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
// 激活音频会话 |
||||
|
try audioSession.setActive(true) |
||||
|
|
||||
|
// 添加识别事件处理 |
||||
|
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
if event.result.reason == .recognizedSpeech { |
||||
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "finalResult", |
||||
|
"text": event.result.text ?? "", |
||||
|
"language": detectedLanguage |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 识别中事件 |
||||
|
recognizer?.addRecognizingEventHandler { [weak self] _, event in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
if event.result.reason == .recognizingSpeech { |
||||
|
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "recognizing", |
||||
|
"text": event.result.text ?? "", |
||||
|
"language": detectedLanguage |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 会话事件 |
||||
|
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "sessionStarted" |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
|
||||
|
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
self.isContinuousRecognitionActive = false |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "sessionStopped" |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
|
||||
|
// 取消事件 |
||||
|
recognizer?.addCanceledEventHandler { [weak self] _, event in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
self.isContinuousRecognitionActive = false |
||||
|
var errorMessage = "未知错误" |
||||
|
|
||||
|
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { |
||||
|
errorMessage = cancellationDetails.errorDetails ?? "未知错误" |
||||
|
} |
||||
|
|
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "error", |
||||
|
"error": "识别取消: \(errorMessage)" |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
|
||||
|
// 开始连续识别 |
||||
|
try recognizer?.startContinuousRecognition() |
||||
|
isContinuousRecognitionActive = true |
||||
|
print("[AzureAsrHelper] 连续识别已启动") |
||||
|
|
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 停止连续语音识别 |
||||
|
/// |
||||
|
/// - Returns: 是否成功停止连续识别 |
||||
|
func stopContinuousRecognition() -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if !isContinuousRecognitionActive { |
||||
|
print("[AzureAsrHelper] 未进行连续识别,忽略停止请求") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
print("[AzureAsrHelper] 停止连续语音识别") |
||||
|
|
||||
|
if recognizer == nil { |
||||
|
if isContinuousRecognitionActive { |
||||
|
print("[AzureAsrHelper] 警告: 识别器为空,但状态显示活跃") |
||||
|
} |
||||
|
isContinuousRecognitionActive = false |
||||
|
|
||||
|
// 通知停止成功 |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "success", |
||||
|
"message": "连续识别已停止" |
||||
|
] |
||||
|
eventHandler?(eventData) |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
// 停止连续识别 |
||||
|
try recognizer?.stopContinuousRecognition() |
||||
|
|
||||
|
// 延迟一点时间确保处理完成 |
||||
|
DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { [weak self] in |
||||
|
guard let self = self else { return } |
||||
|
|
||||
|
// 重置状态 |
||||
|
self.isContinuousRecognitionActive = false |
||||
|
|
||||
|
// 恢复音频会话 |
||||
|
do { |
||||
|
try self.audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
||||
|
} catch { |
||||
|
// 忽略错误 |
||||
|
} |
||||
|
|
||||
|
print("[AzureAsrHelper] 连续识别已停止") |
||||
|
|
||||
|
// 通知停止成功 |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "success", |
||||
|
"message": "连续识别已停止" |
||||
|
] |
||||
|
self.eventHandler?(eventData) |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch { |
||||
|
// 强制重置状态 |
||||
|
isContinuousRecognitionActive = false |
||||
|
print("[AzureAsrHelper] 警告: 停止连续识别失败: \(error.localizedDescription)") |
||||
|
|
||||
|
// 通知停止失败,但仍然视为处理完成 |
||||
|
let eventData: [String: Any] = [ |
||||
|
"eventType": "success", |
||||
|
"message": "连续识别已停止(但有错误)" |
||||
|
] |
||||
|
eventHandler?(eventData) |
||||
|
|
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 释放资源 |
||||
|
func dispose() { |
||||
|
// 如果正在进行连续识别,先停止 |
||||
|
if isContinuousRecognitionActive { |
||||
|
_ = stopContinuousRecognition() |
||||
|
} |
||||
|
|
||||
|
// 恢复音频会话 |
||||
|
do { |
||||
|
try audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
||||
|
} catch { |
||||
|
// 忽略错误 |
||||
|
} |
||||
|
|
||||
|
// 释放资源 |
||||
|
recognizer = nil |
||||
|
speechConfig = nil |
||||
|
audioConfig = nil |
||||
|
|
||||
|
initialized = false |
||||
|
isContinuousRecognitionActive = false |
||||
|
print("[AzureAsrHelper] 资源已释放") |
||||
|
} |
||||
|
|
||||
|
/// 检查连续识别是否处于活跃状态 |
||||
|
/// |
||||
|
/// - Returns: 是否正在进行连续识别 |
||||
|
func isContinuousRecognitionActive() -> Bool { |
||||
|
return isContinuousRecognitionActive |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,182 @@ |
|||||
|
import Flutter |
||||
|
import UIKit |
||||
|
|
||||
|
public class AzureSpeechPlugin: NSObject, FlutterPlugin { |
||||
|
private var ttsHelper: AzureTtsHelper? |
||||
|
private var asrHelper: AzureAsrHelper? |
||||
|
private var eventSink: FlutterEventSink? |
||||
|
|
||||
|
public static func register(with registrar: FlutterPluginRegistrar) { |
||||
|
let channel = FlutterMethodChannel(name: "azure_speech", binaryMessenger: registrar.messenger()) |
||||
|
let instance = AzureSpeechPlugin() |
||||
|
registrar.addMethodCallDelegate(instance, channel: channel) |
||||
|
|
||||
|
// 初始化事件通道 |
||||
|
let eventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) |
||||
|
eventChannel.setStreamHandler(AsrStreamHandler(instance: instance)) |
||||
|
} |
||||
|
|
||||
|
override init() { |
||||
|
super.init() |
||||
|
ttsHelper = AzureTtsHelper() |
||||
|
asrHelper = AzureAsrHelper() |
||||
|
|
||||
|
// 设置ASR事件处理 |
||||
|
asrHelper?.setEventHandler { [weak self] event in |
||||
|
self?.handleAsrEvent(event) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
||||
|
switch call.method { |
||||
|
// TTS相关方法 |
||||
|
case "initializeTts": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let subscriptionKey = args["subscriptionKey"] as? String, |
||||
|
let serviceRegion = args["serviceRegion"] as? String else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
let language = args["language"] as? String ?? "zh-CN" |
||||
|
let success = ttsHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, language: language) ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "setTtsVoice": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let voiceName = args["voiceName"] as? String else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
let success = ttsHelper?.setVoice(voiceName: voiceName) ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "setTtsSpeechParams": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let rate = args["rate"] as? Int, |
||||
|
let pitch = args["pitch"] as? Int, |
||||
|
let volume = args["volume"] as? Int else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
let success = ttsHelper?.setSpeechParams(rate: rate, pitch: pitch, volume: volume) ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "speakText": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let text = args["text"] as? String else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
ttsHelper?.speakText(text: text) { success, _ in |
||||
|
result(success) |
||||
|
} |
||||
|
|
||||
|
case "stopSpeaking": |
||||
|
let success = ttsHelper?.stopSpeaking() ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "isSpeaking": |
||||
|
let speaking = ttsHelper?.isSpeaking() ?? false |
||||
|
result(speaking) |
||||
|
|
||||
|
case "setTtsAudioOutputType": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let outputType = args["outputType"] as? String else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
var type: AzureTtsHelper.AudioOutputType = .auto |
||||
|
switch outputType.uppercased() { |
||||
|
case "SPEAKER": |
||||
|
type = .speaker |
||||
|
case "EARPIECE": |
||||
|
type = .earpiece |
||||
|
default: |
||||
|
type = .auto |
||||
|
} |
||||
|
|
||||
|
let success = ttsHelper?.setAudioOutputType(outputType: type) ?? false |
||||
|
result(success) |
||||
|
|
||||
|
// ASR相关方法 |
||||
|
case "initializeAsr": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let subscriptionKey = args["subscriptionKey"] as? String, |
||||
|
let serviceRegion = args["serviceRegion"] as? String, |
||||
|
let supportedLanguages = args["supportedLanguages"] as? [String] else { |
||||
|
result(false) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
let success = asrHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, supportedLanguages: supportedLanguages) ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "recognizeOnce": |
||||
|
asrHelper?.recognizeOnce { success, text, language, error in |
||||
|
var resultMap: [String: Any] = ["success": success] |
||||
|
if success { |
||||
|
resultMap["text"] = text |
||||
|
resultMap["language"] = language |
||||
|
} else { |
||||
|
resultMap["error"] = error |
||||
|
} |
||||
|
result(resultMap) |
||||
|
} |
||||
|
|
||||
|
case "startContinuousRecognition": |
||||
|
let success = asrHelper?.startContinuousRecognition() ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "stopContinuousRecognition": |
||||
|
let success = asrHelper?.stopContinuousRecognition() ?? false |
||||
|
result(success) |
||||
|
|
||||
|
case "isContinuousRecognitionActive": |
||||
|
let isActive = asrHelper?.isContinuousRecognitionActive() ?? false |
||||
|
result(isActive) |
||||
|
|
||||
|
case "dispose": |
||||
|
ttsHelper?.dispose() |
||||
|
asrHelper?.dispose() |
||||
|
result(nil) |
||||
|
|
||||
|
default: |
||||
|
result(FlutterMethodNotImplemented) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 设置事件接收器 |
||||
|
func setEventSink(_ sink: FlutterEventSink?) { |
||||
|
self.eventSink = sink |
||||
|
} |
||||
|
|
||||
|
// 处理ASR事件 |
||||
|
private func handleAsrEvent(_ event: [String: Any]) { |
||||
|
self.eventSink?(event) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// ASR事件流处理器 |
||||
|
class AsrStreamHandler: NSObject, FlutterStreamHandler { |
||||
|
private weak var plugin: AzureSpeechPlugin? |
||||
|
|
||||
|
init(instance: AzureSpeechPlugin) { |
||||
|
self.plugin = instance |
||||
|
super.init() |
||||
|
} |
||||
|
|
||||
|
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
||||
|
plugin?.setEventSink(events) |
||||
|
return nil |
||||
|
} |
||||
|
|
||||
|
func onCancel(withArguments arguments: Any?) -> FlutterError? { |
||||
|
plugin?.setEventSink(nil) |
||||
|
return nil |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,330 @@ |
|||||
|
import Foundation |
||||
|
import AVFoundation |
||||
|
import MicrosoftCognitiveServicesSpeech |
||||
|
|
||||
|
/// Azure 语音合成辅助类 |
||||
|
class AzureTtsHelper: NSObject { |
||||
|
private var synthesizer: SPXSpeechSynthesizer? |
||||
|
private var speechConfig: SPXSpeechConfig? |
||||
|
private var audioConfig: SPXAudioConfig? |
||||
|
private var initialized = false |
||||
|
private var speaking = false |
||||
|
|
||||
|
// 音频输出类型 |
||||
|
enum AudioOutputType { |
||||
|
case speaker // 扬声器 |
||||
|
case earpiece // 听筒 |
||||
|
case auto // 自动选择 |
||||
|
} |
||||
|
|
||||
|
// 当前设置 |
||||
|
private var currentVoiceName = "zh-CN-XiaoxiaoNeural" |
||||
|
private var currentSpeechRate = 0 |
||||
|
private var currentPitch = 0 |
||||
|
private var currentVolume = 100 |
||||
|
private var currentAudioOutputType: AudioOutputType = .auto |
||||
|
|
||||
|
// 音频会话管理 |
||||
|
private let audioSession = AVAudioSession.sharedInstance() |
||||
|
|
||||
|
/// 初始化 TTS 引擎 |
||||
|
/// |
||||
|
/// - Parameters: |
||||
|
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
||||
|
/// - serviceRegion: Azure 语音服务区域 |
||||
|
/// - language: 可选,默认语言,默认为 "zh-CN" |
||||
|
/// - Returns: 是否初始化成功 |
||||
|
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { |
||||
|
print("[AzureTtsHelper] 初始化 Azure 语音服务") |
||||
|
|
||||
|
// 检查配置是否为空 |
||||
|
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
||||
|
print("[AzureTtsHelper] 错误: Azure 配置信息不完整") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 释放之前的资源 |
||||
|
dispose() |
||||
|
|
||||
|
do { |
||||
|
// 创建语音配置 |
||||
|
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) |
||||
|
|
||||
|
// 设置语音合成输出格式为高质量音频 |
||||
|
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm) |
||||
|
|
||||
|
// 设置默认语言 |
||||
|
speechConfig?.setSpeechSynthesisLanguage(language) |
||||
|
|
||||
|
// 设置默认语音 |
||||
|
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) |
||||
|
|
||||
|
// 创建音频配置 - 使用默认扬声器 |
||||
|
audioConfig = SPXAudioConfig.default() |
||||
|
|
||||
|
// 创建语音合成器 |
||||
|
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) |
||||
|
|
||||
|
initialized = true |
||||
|
|
||||
|
// 设置默认音频输出类型为自动 |
||||
|
setAudioOutputType(outputType: .auto) |
||||
|
|
||||
|
print("[AzureTtsHelper] TTS 引擎初始化成功") |
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 设置音频输出设备类型 |
||||
|
/// |
||||
|
/// - Parameter outputType: 音频输出设备类型 |
||||
|
/// - Returns: 是否设置成功 |
||||
|
func setAudioOutputType(outputType: AudioOutputType) -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
currentAudioOutputType = outputType |
||||
|
|
||||
|
switch outputType { |
||||
|
case .speaker: |
||||
|
// 使用扬声器 |
||||
|
try audioSession.setCategory(.playback, mode: .default) |
||||
|
try audioSession.overrideOutputAudioPort(.speaker) |
||||
|
print("[AzureTtsHelper] 已设置音频输出设备为扬声器") |
||||
|
|
||||
|
case .earpiece: |
||||
|
// 使用听筒 |
||||
|
try audioSession.setCategory(.playback, mode: .voiceChat) |
||||
|
try audioSession.overrideOutputAudioPort(.none) |
||||
|
print("[AzureTtsHelper] 已设置音频输出设备为听筒") |
||||
|
|
||||
|
case .auto: |
||||
|
// 检查是否有耳机连接 |
||||
|
let outputs = audioSession.currentRoute.outputs |
||||
|
let hasHeadphones = outputs.contains { output in |
||||
|
return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP |
||||
|
} |
||||
|
|
||||
|
if hasHeadphones { |
||||
|
// 有耳机,使用耳机 |
||||
|
try audioSession.setCategory(.playback, mode: .default) |
||||
|
try audioSession.overrideOutputAudioPort(.none) |
||||
|
print("[AzureTtsHelper] 已设置音频输出设备为耳机") |
||||
|
} else { |
||||
|
// 无耳机,使用听筒 |
||||
|
try audioSession.setCategory(.playback, mode: .voiceChat) |
||||
|
try audioSession.overrideOutputAudioPort(.none) |
||||
|
print("[AzureTtsHelper] 已设置音频输出设备为听筒") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
try audioSession.setActive(true) |
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 设置语音 |
||||
|
/// |
||||
|
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural" |
||||
|
/// - Returns: 是否设置成功 |
||||
|
func setVoice(voiceName: String) -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if voiceName == currentVoiceName { |
||||
|
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
currentVoiceName = voiceName |
||||
|
speechConfig?.setSpeechSynthesisVoiceName(voiceName) |
||||
|
|
||||
|
// 重新创建合成器 |
||||
|
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) |
||||
|
|
||||
|
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 设置语音合成参数 |
||||
|
/// |
||||
|
/// - Parameters: |
||||
|
/// - rate: 语速,范围 -100 到 100,默认为 0 |
||||
|
/// - pitch: 音调,范围 -100 到 100,默认为 0 |
||||
|
/// - volume: 音量,范围 0 到 100,默认为 100 |
||||
|
/// - Returns: 是否设置成功 |
||||
|
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
currentSpeechRate = rate |
||||
|
currentPitch = pitch |
||||
|
currentVolume = volume |
||||
|
|
||||
|
print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
/// 合成文本为语音并播放 |
||||
|
/// |
||||
|
/// - Parameters: |
||||
|
/// - text: 要合成的文本 |
||||
|
/// - completion: 完成回调,返回是否成功和可能的错误信息 |
||||
|
func speakText(text: String, completion: @escaping (Bool, String?) -> Void) { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
completion(false, "TTS 引擎尚未初始化") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
print("[AzureTtsHelper] 开始合成文本: \(text)") |
||||
|
|
||||
|
// 生成 SSML |
||||
|
let ssml = generateSsml(text: text) |
||||
|
|
||||
|
// 使用 SSML 合成语音 |
||||
|
speakSsml(ssml: ssml, completion: completion) |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") |
||||
|
completion(false, "语音合成异常: \(error.localizedDescription)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 生成 SSML 文本 |
||||
|
/// |
||||
|
/// - Parameter text: 要转换的文本 |
||||
|
/// - Returns: SSML 格式的文本 |
||||
|
private func generateSsml(text: String) -> String { |
||||
|
// 计算 SSML 参数 |
||||
|
let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%") |
||||
|
let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%" |
||||
|
let volumeParam = "\(min(max(currentVolume, 0), 100))%" |
||||
|
|
||||
|
return """ |
||||
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN"> |
||||
|
<voice name="\(currentVoiceName)"> |
||||
|
<prosody rate="\(rateParam)" pitch="\(pitchParam)" volume="\(volumeParam)"> |
||||
|
\(text) |
||||
|
</prosody> |
||||
|
</voice> |
||||
|
</speak> |
||||
|
""" |
||||
|
} |
||||
|
|
||||
|
/// 合成 SSML 为语音并播放 |
||||
|
/// |
||||
|
/// - Parameters: |
||||
|
/// - ssml: SSML 格式的文本 |
||||
|
/// - completion: 完成回调,返回是否成功和可能的错误信息 |
||||
|
private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
completion(false, "TTS 引擎尚未初始化") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
print("[AzureTtsHelper] 开始合成 SSML") |
||||
|
|
||||
|
// 标记为正在播放 |
||||
|
speaking = true |
||||
|
|
||||
|
// 激活音频会话 |
||||
|
try audioSession.setActive(true) |
||||
|
|
||||
|
// 异步合成语音 |
||||
|
let result = try synthesizer!.speakSsml(ssml) |
||||
|
|
||||
|
switch result.reason { |
||||
|
case .synthesizingAudioCompleted: |
||||
|
print("[AzureTtsHelper] 语音合成完成") |
||||
|
speaking = false |
||||
|
completion(true, "语音合成完成") |
||||
|
case .canceled: |
||||
|
if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) { |
||||
|
print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") |
||||
|
speaking = false |
||||
|
completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") |
||||
|
} else { |
||||
|
print("[AzureTtsHelper] 语音合成取消") |
||||
|
speaking = false |
||||
|
completion(false, "语音合成取消") |
||||
|
} |
||||
|
default: |
||||
|
print("[AzureTtsHelper] 语音合成失败: \(result.reason)") |
||||
|
speaking = false |
||||
|
completion(false, "语音合成失败: \(result.reason)") |
||||
|
} |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") |
||||
|
speaking = false |
||||
|
completion(false, "语音合成异常: \(error.localizedDescription)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 停止当前语音合成 |
||||
|
/// |
||||
|
/// - Returns: 是否停止成功 |
||||
|
func stopSpeaking() -> Bool { |
||||
|
if !initialized { |
||||
|
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
try synthesizer?.stopSpeaking() |
||||
|
speaking = false |
||||
|
print("[AzureTtsHelper] 已停止语音合成") |
||||
|
return true |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)") |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 释放资源 |
||||
|
func dispose() { |
||||
|
do { |
||||
|
stopSpeaking() |
||||
|
|
||||
|
// 恢复音频会话 |
||||
|
try audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
||||
|
|
||||
|
synthesizer = nil |
||||
|
speechConfig = nil |
||||
|
audioConfig = nil |
||||
|
|
||||
|
initialized = false |
||||
|
speaking = false |
||||
|
print("[AzureTtsHelper] TTS 引擎已释放") |
||||
|
} catch { |
||||
|
print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 检查当前是否正在播放语音 |
||||
|
/// |
||||
|
/// - Returns: 是否正在播放语音 |
||||
|
func isSpeaking() -> Bool { |
||||
|
return speaking |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,32 @@ |
|||||
|
name: azure_speech |
||||
|
description: Azure语音服务插件,包含TTS和ASR服务 |
||||
|
version: 0.0.1 |
||||
|
homepage: |
||||
|
|
||||
|
environment: |
||||
|
sdk: ">=2.17.0 <3.0.0" |
||||
|
flutter: ">=2.5.0" |
||||
|
|
||||
|
dependencies: |
||||
|
flutter: |
||||
|
sdk: flutter |
||||
|
|
||||
|
dev_dependencies: |
||||
|
flutter_test: |
||||
|
sdk: flutter |
||||
|
flutter_lints: ^2.0.0 |
||||
|
|
||||
|
# For information on the generic Dart part of this file, see the |
||||
|
# following page: https://dart.dev/tools/pub/pubspec |
||||
|
|
||||
|
# The following section is specific to Flutter packages. |
||||
|
flutter: |
||||
|
# This section identifies this Flutter project as a plugin project. |
||||
|
plugin: |
||||
|
platforms: |
||||
|
android: |
||||
|
package: com.yunqiinnovation.azure_speech |
||||
|
pluginClass: AzureSpeechPlugin |
||||
|
ios: |
||||
|
pluginClass: AzureSpeechPlugin |
||||
|
|
||||
@ -0,0 +1,253 @@ |
|||||
|
# OpenAI Service Plugin |
||||
|
|
||||
|
一个用于Flutter应用的OpenAI服务插件,支持Android和iOS平台。 |
||||
|
|
||||
|
## 功能 |
||||
|
|
||||
|
- 支持文本生成(completions) |
||||
|
- 支持流式输出(streaming) |
||||
|
- 支持函数调用(function calling) |
||||
|
- 支持自定义API基础URL |
||||
|
- 支持自定义模型选择 |
||||
|
|
||||
|
## 安装 |
||||
|
|
||||
|
在你的`pubspec.yaml`文件中添加以下依赖: |
||||
|
|
||||
|
```yaml |
||||
|
dependencies: |
||||
|
open_ai_service: |
||||
|
path: 本地路径/open_ai_service |
||||
|
``` |
||||
|
|
||||
|
## 使用方法 |
||||
|
|
||||
|
### 初始化服务 |
||||
|
|
||||
|
```dart |
||||
|
import 'package:open_ai_service/open_ai_service.dart'; |
||||
|
|
||||
|
final openAIService = OpenAIService(); |
||||
|
|
||||
|
// 初始化服务 |
||||
|
await openAIService.initialize( |
||||
|
apiKey: 'your_openai_api_key', |
||||
|
baseUrl: 'https://api.openai.com/v1/chat/completions', // 可选 |
||||
|
model: 'gpt-4-turbo', // 可选 |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### 发送非流式请求 |
||||
|
|
||||
|
```dart |
||||
|
// 创建消息 |
||||
|
final userMessage = await openAIService.createUserMessage('你好,请介绍一下自己'); |
||||
|
|
||||
|
// 发送请求 |
||||
|
final response = await openAIService.sendMessage( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个有用的AI助手', |
||||
|
); |
||||
|
|
||||
|
print('AI回复: $response'); |
||||
|
``` |
||||
|
|
||||
|
### 发送流式请求(回调方式) |
||||
|
|
||||
|
```dart |
||||
|
// 创建消息 |
||||
|
final userMessage = await openAIService.createUserMessage('写一个短故事'); |
||||
|
|
||||
|
// 发送流式请求 |
||||
|
await openAIService.sendMessageStream( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个善于讲故事的AI助手', |
||||
|
); |
||||
|
|
||||
|
// 处理事件 |
||||
|
final subscription = openAIService.processEvents( |
||||
|
onToken: (token) { |
||||
|
// 处理每个返回的token |
||||
|
print(token); |
||||
|
}, |
||||
|
onComplete: () { |
||||
|
// 处理完成事件 |
||||
|
print('生成完成'); |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
// 处理错误 |
||||
|
print('错误: $error'); |
||||
|
}, |
||||
|
onFunctionCall: (functionCall) { |
||||
|
// 处理函数调用 |
||||
|
print('函数调用: ${functionCall['name']}'); |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
// 在不需要时取消订阅 |
||||
|
subscription.cancel(); |
||||
|
``` |
||||
|
|
||||
|
### 发送流式请求(Stream方式) |
||||
|
|
||||
|
```dart |
||||
|
// 创建消息 |
||||
|
final userMessage = await openAIService.createUserMessage('写一个短故事'); |
||||
|
|
||||
|
// 获取字符串流 |
||||
|
final stream = openAIService.streamMessage( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个善于讲故事的AI助手', |
||||
|
); |
||||
|
|
||||
|
// 使用流 |
||||
|
final StringBuilder responseBuilder = StringBuilder(); |
||||
|
|
||||
|
stream.listen( |
||||
|
(token) { |
||||
|
// 处理每个token |
||||
|
responseBuilder.write(token); |
||||
|
print(token); // 实时输出 |
||||
|
}, |
||||
|
onDone: () { |
||||
|
// 流结束 |
||||
|
print('完整回复: ${responseBuilder.toString()}'); |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
// 错误处理 |
||||
|
print('错误: $error'); |
||||
|
} |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### 注册函数 |
||||
|
|
||||
|
```dart |
||||
|
// 注册一个函数 |
||||
|
await openAIService.registerFunction( |
||||
|
name: 'get_weather', |
||||
|
description: '获取指定城市的天气信息', |
||||
|
parameters: { |
||||
|
'type': 'object', |
||||
|
'properties': { |
||||
|
'city': { |
||||
|
'type': 'string', |
||||
|
'description': '城市名称', |
||||
|
}, |
||||
|
'date': { |
||||
|
'type': 'string', |
||||
|
'description': '日期,格式为YYYY-MM-DD', |
||||
|
}, |
||||
|
}, |
||||
|
'required': ['city'], |
||||
|
}, |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### 处理函数调用 |
||||
|
|
||||
|
```dart |
||||
|
// 创建消息 |
||||
|
final userMessage = await openAIService.createUserMessage('明天北京的天气如何?'); |
||||
|
|
||||
|
// 发送流式请求 |
||||
|
await openAIService.sendMessageStream( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个有用的AI助手', |
||||
|
); |
||||
|
|
||||
|
// 处理事件 |
||||
|
openAIService.processEvents( |
||||
|
onToken: (token) { |
||||
|
print(token); |
||||
|
}, |
||||
|
onComplete: () { |
||||
|
print('生成完成'); |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
print('错误: $error'); |
||||
|
}, |
||||
|
onFunctionCall: (functionCall) { |
||||
|
// 处理函数调用 |
||||
|
final name = functionCall['name']; |
||||
|
final arguments = functionCall['arguments']; |
||||
|
|
||||
|
print('收到函数调用: $name, 参数: $arguments'); |
||||
|
|
||||
|
// 假设处理了函数调用并获得结果 |
||||
|
final result = '{"temperature": 25, "condition": "sunny"}'; |
||||
|
|
||||
|
// 发送函数调用结果 |
||||
|
openAIService.sendFunctionCallResult( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个有用的AI助手', |
||||
|
functionCall: functionCall, |
||||
|
functionResult: result, |
||||
|
); |
||||
|
}, |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### 使用Stream API处理函数调用 |
||||
|
|
||||
|
```dart |
||||
|
// 创建消息和响应处理器 |
||||
|
final userMessage = await openAIService.createUserMessage('明天北京的天气如何?'); |
||||
|
final responseBuilder = StringBuilder(); |
||||
|
|
||||
|
// 处理事件流以捕获函数调用 |
||||
|
final subscription = openAIService.processEvents( |
||||
|
onFunctionCall: (functionCall) async { |
||||
|
// 取消当前事件监听 |
||||
|
subscription.cancel(); |
||||
|
|
||||
|
// 处理函数调用 |
||||
|
final name = functionCall['name']; |
||||
|
final arguments = functionCall['arguments']; |
||||
|
|
||||
|
print('收到函数调用: $name, 参数: $arguments'); |
||||
|
|
||||
|
// 假设处理了函数调用并获得结果 |
||||
|
final result = '{"temperature": 25, "condition": "sunny"}'; |
||||
|
|
||||
|
// 使用Stream API发送函数调用结果 |
||||
|
final resultStream = openAIService.streamFunctionResult( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个有用的AI助手', |
||||
|
functionCall: functionCall, |
||||
|
functionResult: result, |
||||
|
); |
||||
|
|
||||
|
// 处理结果流 |
||||
|
resultStream.listen( |
||||
|
(token) { |
||||
|
responseBuilder.write(token); |
||||
|
print(token); // 实时输出 |
||||
|
}, |
||||
|
onDone: () { |
||||
|
print('完整回复: ${responseBuilder.toString()}'); |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
print('错误: $error'); |
||||
|
} |
||||
|
); |
||||
|
} |
||||
|
); |
||||
|
|
||||
|
// 启动请求 |
||||
|
await openAIService.sendMessageStream( |
||||
|
messages: [userMessage], |
||||
|
systemPrompt: '你是一个有用的AI助手', |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
## 注意事项 |
||||
|
|
||||
|
1. 确保在使用前已正确初始化服务 |
||||
|
2. 对于流式请求,确保在不需要时取消订阅 |
||||
|
3. 处理函数调用时,确保提供有效的结果格式 |
||||
|
4. 网络请求可能会失败,请确保加入适当的错误处理 |
||||
|
|
||||
|
## 许可证 |
||||
|
|
||||
|
[MIT License](LICENSE) |
||||
@ -0,0 +1,37 @@ |
|||||
|
plugins { |
||||
|
// Android Library 插件 |
||||
|
id("com.android.library") |
||||
|
// Kotlin Android 插件 |
||||
|
id("org.jetbrains.kotlin.android") |
||||
|
} |
||||
|
|
||||
|
android { |
||||
|
// 命名空间,对应你插件的包名(需与代码内包名保持一致) |
||||
|
namespace = "com.yunqiinnovation.open_ai_service" |
||||
|
|
||||
|
// 目标 SDK 版本 |
||||
|
compileSdk = 33 |
||||
|
|
||||
|
defaultConfig { |
||||
|
// 最低 SDK 版本 |
||||
|
minSdk = 21 |
||||
|
targetSdk = 33 |
||||
|
} |
||||
|
|
||||
|
// Java 语言级别兼容配置 |
||||
|
compileOptions { |
||||
|
sourceCompatibility = JavaVersion.VERSION_11 |
||||
|
targetCompatibility = JavaVersion.VERSION_11 |
||||
|
} |
||||
|
|
||||
|
// Kotlin 语言级别 |
||||
|
kotlinOptions { |
||||
|
jvmTarget = "11" |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
dependencies { |
||||
|
|
||||
|
implementation("com.squareup.okhttp3:okhttp:4.10.0") |
||||
|
|
||||
|
} |
||||
@ -0,0 +1 @@ |
|||||
|
rootProject.name = "open_ai_service" |
||||
@ -0,0 +1,11 @@ |
|||||
|
<?xml version="1.0" encoding="utf-8"?> |
||||
|
<manifest xmlns:android="http://schemas.android.com/apk/res/android" |
||||
|
package="com.yunqiinnovation.open_ai_service"> |
||||
|
|
||||
|
<uses-permission android:name="android.permission.INTERNET" /> |
||||
|
<uses-permission android:name="android.permission.ACCESS_NETWORK_STATE" /> |
||||
|
|
||||
|
<application> |
||||
|
<!-- 插件不需要额外的组件声明 --> |
||||
|
</application> |
||||
|
</manifest> |
||||
@ -0,0 +1,469 @@ |
|||||
|
package com.yunqiinnovation.open_ai_service |
||||
|
|
||||
|
import android.util.Log |
||||
|
import okhttp3.* |
||||
|
import okhttp3.MediaType.Companion.toMediaTypeOrNull |
||||
|
import okhttp3.RequestBody.Companion.toRequestBody |
||||
|
import org.json.JSONArray |
||||
|
import org.json.JSONObject |
||||
|
import java.io.IOException |
||||
|
import java.util.concurrent.TimeUnit |
||||
|
|
||||
|
/** |
||||
|
* OpenAI服务的原生实现 |
||||
|
*/ |
||||
|
class OpenAIService() { |
||||
|
private val TAG = "OpenAIService" |
||||
|
private var baseUrl = "" |
||||
|
private val client = OkHttpClient.Builder() |
||||
|
.connectTimeout(30, TimeUnit.SECONDS) |
||||
|
.readTimeout(30, TimeUnit.SECONDS) |
||||
|
.writeTimeout(30, TimeUnit.SECONDS) |
||||
|
.build() |
||||
|
|
||||
|
private var apiKey: String = "" |
||||
|
private var isInitialized = false |
||||
|
private var model: String = "" // 默认模型 |
||||
|
|
||||
|
// 用于存储注册的函数 |
||||
|
private val registeredFunctions = mutableListOf<JSONObject>() |
||||
|
|
||||
|
/** |
||||
|
* 构建curl命令用于测试 |
||||
|
*/ |
||||
|
private fun buildCurlCommand(request: Request, body: String): String { |
||||
|
val command = StringBuilder("curl -v -X ${request.method}") |
||||
|
|
||||
|
// 添加请求头 |
||||
|
request.headers.forEach { header -> |
||||
|
// 敏感信息处理:不显示真实的API Key |
||||
|
if (header.first == "Authorization") { |
||||
|
command.append(" -H '${header.first}: Bearer $apiKey'") |
||||
|
} else { |
||||
|
command.append(" -H '${header.first}: ${header.second}'") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 添加请求体 |
||||
|
if (request.method == "POST" || request.method == "PUT") { |
||||
|
// 转义JSON中的单引号,确保curl命令正确 |
||||
|
val escapedBody = body.replace("'", "\\'") |
||||
|
command.append(" -d '${escapedBody}'") |
||||
|
} |
||||
|
|
||||
|
// 添加URL |
||||
|
command.append(" '${request.url}'") |
||||
|
|
||||
|
return command.toString() |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 创建用户消息 |
||||
|
*/ |
||||
|
fun createUserMessage(content: String): JSONObject { |
||||
|
return JSONObject().apply { |
||||
|
put("role", "user") |
||||
|
put("content", content) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 创建助手消息 |
||||
|
*/ |
||||
|
fun createAssistantMessage(content: String): JSONObject { |
||||
|
return JSONObject().apply { |
||||
|
put("role", "assistant") |
||||
|
put("content", content) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 初始化OpenAI服务 |
||||
|
*/ |
||||
|
fun initialize(apiKey: String, baseUrl: String, model: String): Boolean { |
||||
|
this.apiKey = apiKey |
||||
|
if (baseUrl.isNotEmpty()) { |
||||
|
this.baseUrl = baseUrl |
||||
|
} |
||||
|
if (model.isNotEmpty()) { |
||||
|
this.model = model |
||||
|
} |
||||
|
isInitialized = apiKey.isNotEmpty() |
||||
|
return isInitialized |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 注册函数 |
||||
|
*/ |
||||
|
fun registerFunction(name: String, description: String, parameters: JSONObject): Boolean { |
||||
|
try { |
||||
|
val function = JSONObject().apply { |
||||
|
put("name", name) |
||||
|
put("description", description) |
||||
|
put("parameters", parameters) |
||||
|
} |
||||
|
|
||||
|
// 检查是否已存在相同名称的函数 |
||||
|
val existingIndex = registeredFunctions.indexOfFirst { |
||||
|
it.getString("name") == name |
||||
|
} |
||||
|
|
||||
|
if (existingIndex >= 0) { |
||||
|
// 如果已存在,则替换 |
||||
|
registeredFunctions[existingIndex] = function |
||||
|
} else { |
||||
|
// 如果不存在,则添加 |
||||
|
registeredFunctions.add(function) |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 发送消息(非流式输出) |
||||
|
*/ |
||||
|
@Throws(OpenAIException::class) |
||||
|
fun sendMessage(messages: JSONArray): String { |
||||
|
if (!isInitialized || apiKey.isEmpty()) { |
||||
|
throw OpenAIException("OpenAI服务未初始化") |
||||
|
} |
||||
|
|
||||
|
val requestBody = JSONObject().apply { |
||||
|
put("model", model) |
||||
|
put("messages", messages) |
||||
|
put("temperature", 0.7) |
||||
|
put("max_tokens", 2000) |
||||
|
put("stream", false) |
||||
|
|
||||
|
// 如果有注册的函数,则添加到请求中 |
||||
|
if (registeredFunctions.isNotEmpty()) { |
||||
|
val tools = JSONArray() |
||||
|
for (function in registeredFunctions) { |
||||
|
val tool = JSONObject().apply { |
||||
|
put("type", "function") |
||||
|
put("function", function) |
||||
|
} |
||||
|
tools.put(tool) |
||||
|
} |
||||
|
put("tools", tools) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
val mediaType = "application/json".toMediaTypeOrNull() |
||||
|
val request = Request.Builder() |
||||
|
.url(baseUrl) |
||||
|
.addHeader("Content-Type", "application/json") |
||||
|
.addHeader("Authorization", "Bearer $apiKey") |
||||
|
.post(requestBody.toString().toRequestBody(mediaType)) |
||||
|
.build() |
||||
|
|
||||
|
try { |
||||
|
// 输出用于测试的curl命令 |
||||
|
// val curlCommand = buildCurlCommand(request, requestBody.toString()) |
||||
|
// Log.d(TAG, "curl command: \n$curlCommand") |
||||
|
|
||||
|
client.newCall(request).execute().use { response -> |
||||
|
if (!response.isSuccessful) { |
||||
|
throw OpenAIException("API调用失败: ${response.code}") |
||||
|
} |
||||
|
|
||||
|
val responseBody = response.body?.string() ?: throw OpenAIException("Empty response") |
||||
|
val jsonResponse = JSONObject(responseBody) |
||||
|
|
||||
|
// 检查是否有函数调用 |
||||
|
if (jsonResponse.has("choices") && |
||||
|
jsonResponse.getJSONArray("choices").length() > 0) { |
||||
|
|
||||
|
val choice = jsonResponse.getJSONArray("choices").getJSONObject(0) |
||||
|
|
||||
|
// 检查是否是函数调用 |
||||
|
if (choice.has("message")) { |
||||
|
val message = choice.getJSONObject("message") |
||||
|
|
||||
|
// 检查是否有工具调用 |
||||
|
if (message.has("tool_calls")) { |
||||
|
val toolCalls = message.getJSONArray("tool_calls") |
||||
|
if (toolCalls.length() > 0) { |
||||
|
val toolCall = toolCalls.getJSONObject(0) |
||||
|
if (toolCall.has("function")) { |
||||
|
val function = toolCall.getJSONObject("function") |
||||
|
val functionCall = JSONObject().apply { |
||||
|
put("name", function.getString("name")) |
||||
|
put("arguments", function.getString("arguments")) |
||||
|
put("id", toolCall.getString("id")) |
||||
|
} |
||||
|
return functionCall.toString() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 如果没有工具调用,返回消息内容 |
||||
|
if (message.has("content")) { |
||||
|
return message.getString("content") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
throw OpenAIException("Invalid response format") |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
if (e is OpenAIException) throw e |
||||
|
throw OpenAIException("Failed to communicate with AI service: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 发送消息(流式输出) |
||||
|
*/ |
||||
|
fun sendMessageStream(messages: JSONArray, callback: StreamCallback) { |
||||
|
if (!isInitialized || apiKey.isEmpty()) { |
||||
|
callback.onError(OpenAIException("OpenAI服务未初始化")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val requestBody = JSONObject().apply { |
||||
|
put("model", model) |
||||
|
put("messages", messages) |
||||
|
put("temperature", 0.7) |
||||
|
put("max_tokens", 2000) |
||||
|
put("stream", true) |
||||
|
|
||||
|
// 如果有注册的函数,则添加到请求中 |
||||
|
if (registeredFunctions.isNotEmpty()) { |
||||
|
val tools = JSONArray() |
||||
|
for (function in registeredFunctions) { |
||||
|
val tool = JSONObject().apply { |
||||
|
put("type", "function") |
||||
|
put("function", function) |
||||
|
} |
||||
|
tools.put(tool) |
||||
|
} |
||||
|
put("tools", tools) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
val mediaType = "application/json".toMediaTypeOrNull() |
||||
|
val request = Request.Builder() |
||||
|
.url(baseUrl) |
||||
|
.addHeader("Content-Type", "application/json") |
||||
|
.addHeader("Authorization", "Bearer $apiKey") |
||||
|
.addHeader("Accept", "text/event-stream") |
||||
|
.post(requestBody.toString().toRequestBody(mediaType)) |
||||
|
.build() |
||||
|
|
||||
|
// 输出用于测试的curl命令 |
||||
|
// val curlCommand = buildCurlCommand(request, requestBody.toString()) |
||||
|
// Log.d(TAG, "curl command: $curlCommand") |
||||
|
|
||||
|
client.newCall(request).enqueue(object : Callback { |
||||
|
override fun onFailure(call: Call, e: IOException) { |
||||
|
callback.onError(OpenAIException(e.message ?: "请求失败")) |
||||
|
} |
||||
|
|
||||
|
override fun onResponse(call: Call, response: Response) { |
||||
|
|
||||
|
if (!response.isSuccessful) { |
||||
|
callback.onError(OpenAIException("API调用失败: ${response.code}")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val responseBody = response.body ?: return |
||||
|
val source = responseBody.source() |
||||
|
|
||||
|
try { |
||||
|
// 预取数据到缓冲区 |
||||
|
source.request(Long.MAX_VALUE) |
||||
|
val bufferedSource = source.buffer |
||||
|
|
||||
|
// 用于存储函数调用的各个部分 |
||||
|
val finalToolCalls = mutableMapOf<Int, ToolCallInfo>() |
||||
|
|
||||
|
while (!bufferedSource.exhausted()) { |
||||
|
val line = bufferedSource.readUtf8Line()?.trim() ?: continue |
||||
|
if (line.isEmpty()) continue |
||||
|
if (line.startsWith("data:")) { |
||||
|
val data = line.substring(5).trim() |
||||
|
|
||||
|
// 处理[DONE]消息 |
||||
|
if (data == "[DONE]" || data == "[\"DONE\"]") { |
||||
|
processToolCalls(finalToolCalls, callback) |
||||
|
callback.onComplete() |
||||
|
break |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val jsonData = JSONObject(data) |
||||
|
|
||||
|
// 处理消息内容 |
||||
|
if (jsonData.has("choices")) { |
||||
|
val choices = jsonData.getJSONArray("choices") |
||||
|
if (choices.length() > 0) { |
||||
|
val choice = choices.getJSONObject(0) |
||||
|
|
||||
|
if (choice.has("delta")) { |
||||
|
val delta = choice.getJSONObject("delta") |
||||
|
|
||||
|
// 处理普通文本内容 |
||||
|
if (delta.has("content")) { |
||||
|
val content = delta.getString("content") |
||||
|
callback.onToken(content) |
||||
|
} |
||||
|
|
||||
|
// 处理工具调用(函数调用) |
||||
|
if (delta.has("tool_calls")) { |
||||
|
val toolCalls = delta.getJSONArray("tool_calls") |
||||
|
for (i in 0 until toolCalls.length()) { |
||||
|
val toolCall = toolCalls.getJSONObject(i) |
||||
|
val index = toolCall.getInt("index") |
||||
|
|
||||
|
// 创建或获取现有的工具调用信息 |
||||
|
val toolCallInfo = finalToolCalls.getOrPut(index) { ToolCallInfo() } |
||||
|
|
||||
|
// 更新ID |
||||
|
if (toolCall.has("id")) { |
||||
|
toolCallInfo.id = toolCall.getString("id") |
||||
|
} |
||||
|
|
||||
|
// 更新函数信息 |
||||
|
if (toolCall.has("function")) { |
||||
|
val function = toolCall.getJSONObject("function") |
||||
|
|
||||
|
if (function.has("name")) { |
||||
|
toolCallInfo.name = function.getString("name") |
||||
|
} |
||||
|
|
||||
|
if (function.has("arguments")) { |
||||
|
toolCallInfo.arguments += function.getString("arguments") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
// 忽略解析错误 |
||||
|
Log.e(TAG, "解析JSON出错: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
callback.onError(OpenAIException("处理响应流时出错: ${e.message}")) |
||||
|
} finally { |
||||
|
responseBody.close() |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 发送函数调用结果 |
||||
|
*/ |
||||
|
fun sendFunctionCallResult( |
||||
|
messages: JSONArray, |
||||
|
functionCall: JSONObject, |
||||
|
functionResult: String, |
||||
|
callback: StreamCallback |
||||
|
) { |
||||
|
try { |
||||
|
val fullMessages = JSONArray() |
||||
|
|
||||
|
// 添加用户消息 |
||||
|
for (i in 0 until messages.length()) { |
||||
|
fullMessages.put(messages.getJSONObject(i)) |
||||
|
} |
||||
|
|
||||
|
// 添加函数调用消息 |
||||
|
fullMessages.put(JSONObject().apply { |
||||
|
put("role", "assistant") |
||||
|
put("content", "") |
||||
|
|
||||
|
// 添加工具调用 |
||||
|
val toolCalls = JSONArray().apply { |
||||
|
val toolCall = JSONObject().apply { |
||||
|
put("id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) |
||||
|
put("type", "function") |
||||
|
put("function", JSONObject().apply { |
||||
|
put("name", functionCall.getString("name")) |
||||
|
put("arguments", functionCall.getString("arguments")) |
||||
|
}) |
||||
|
} |
||||
|
put(toolCall) |
||||
|
} |
||||
|
put("tool_calls", toolCalls) |
||||
|
}) |
||||
|
|
||||
|
// 添加函数调用结果 |
||||
|
fullMessages.put(JSONObject().apply { |
||||
|
put("role", "tool") |
||||
|
put("content", functionResult) |
||||
|
put("tool_call_id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) |
||||
|
}) |
||||
|
|
||||
|
// 添加一个带有content的assistant消息,确保API请求不会因为缺少content而失败 |
||||
|
fullMessages.put(JSONObject().apply { |
||||
|
put("role", "assistant") |
||||
|
put("content", "") // 空内容,让模型生成新的回复 |
||||
|
}) |
||||
|
|
||||
|
// 发送完整对话 |
||||
|
sendMessageStream(fullMessages, callback) |
||||
|
|
||||
|
} catch (e: Exception) { |
||||
|
callback.onError(OpenAIException("发送函数调用结果失败: ${e.message}")) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 处理工具调用(处理完整的函数调用并回调) |
||||
|
*/ |
||||
|
private fun processToolCalls(toolCalls: Map<Int, ToolCallInfo>, callback: StreamCallback) { |
||||
|
if (toolCalls.isEmpty()) return |
||||
|
|
||||
|
// 只处理第一个工具调用 |
||||
|
val firstToolCall = toolCalls.entries.firstOrNull()?.value ?: return |
||||
|
|
||||
|
if (firstToolCall.isValid()) { |
||||
|
// 创建函数调用JSON对象 |
||||
|
val functionCall = JSONObject().apply { |
||||
|
put("name", firstToolCall.name) |
||||
|
put("arguments", firstToolCall.arguments) |
||||
|
put("id", firstToolCall.id) |
||||
|
} |
||||
|
|
||||
|
// 回调 |
||||
|
callback.onFunctionCall(functionCall) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 工具调用信息类 |
||||
|
*/ |
||||
|
private class ToolCallInfo { |
||||
|
var id: String = "" |
||||
|
var name: String = "" |
||||
|
var arguments: String = "" |
||||
|
|
||||
|
fun isValid(): Boolean { |
||||
|
return id.isNotEmpty() && name.isNotEmpty() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 流式输出回调接口 |
||||
|
*/ |
||||
|
interface StreamCallback { |
||||
|
fun onToken(token: String) |
||||
|
fun onComplete() |
||||
|
fun onError(e: Exception) |
||||
|
fun onFunctionCall(functionCall: JSONObject) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* OpenAI服务异常 |
||||
|
*/ |
||||
|
class OpenAIException(message: String) : Exception(message) |
||||
@ -0,0 +1,317 @@ |
|||||
|
package com.yunqiinnovation.open_ai_service |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.util.Log |
||||
|
import androidx.annotation.NonNull |
||||
|
import io.flutter.embedding.engine.plugins.FlutterPlugin |
||||
|
import io.flutter.plugin.common.MethodCall |
||||
|
import io.flutter.plugin.common.MethodChannel |
||||
|
import io.flutter.plugin.common.MethodChannel.MethodCallHandler |
||||
|
import io.flutter.plugin.common.MethodChannel.Result |
||||
|
import io.flutter.plugin.common.EventChannel |
||||
|
import io.flutter.plugin.common.EventChannel.EventSink |
||||
|
import io.flutter.plugin.common.EventChannel.StreamHandler |
||||
|
import org.json.JSONArray |
||||
|
import org.json.JSONObject |
||||
|
import java.util.concurrent.CountDownLatch |
||||
|
import java.util.concurrent.Executors |
||||
|
|
||||
|
/** OpenAIServicePlugin */ |
||||
|
class OpenAIServicePlugin : FlutterPlugin, MethodCallHandler, StreamHandler { |
||||
|
/// 方法通道名称 |
||||
|
private val methodChannelName = "com.yunqiinnovation.open_ai_service/methods" |
||||
|
|
||||
|
/// 事件通道名称 |
||||
|
private val eventChannelName = "com.yunqiinnovation.open_ai_service/events" |
||||
|
|
||||
|
/// 方法通道 |
||||
|
private lateinit var methodChannel: MethodChannel |
||||
|
|
||||
|
/// 事件通道 |
||||
|
private lateinit var eventChannel: EventChannel |
||||
|
|
||||
|
/// 应用上下文 |
||||
|
private lateinit var context: Context |
||||
|
|
||||
|
/// OpenAI服务实例 |
||||
|
private val openAIService = OpenAIService() |
||||
|
|
||||
|
/// 事件接收器(用于流式输出) |
||||
|
private var eventSink: EventSink? = null |
||||
|
|
||||
|
/// 执行器(用于后台线程) |
||||
|
private val executor = Executors.newSingleThreadExecutor() |
||||
|
|
||||
|
override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
// 保存上下文 |
||||
|
context = flutterPluginBinding.applicationContext |
||||
|
|
||||
|
// 初始化方法通道 |
||||
|
methodChannel = MethodChannel(flutterPluginBinding.binaryMessenger, methodChannelName) |
||||
|
methodChannel.setMethodCallHandler(this) |
||||
|
|
||||
|
// 初始化事件通道 |
||||
|
eventChannel = EventChannel(flutterPluginBinding.binaryMessenger, eventChannelName) |
||||
|
eventChannel.setStreamHandler(this) |
||||
|
} |
||||
|
|
||||
|
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
||||
|
when (call.method) { |
||||
|
"initialize" -> { |
||||
|
val apiKey = call.argument<String>("apiKey") ?: "" |
||||
|
val baseUrl = call.argument<String>("baseUrl") ?: "" |
||||
|
val model = call.argument<String>("model") ?: "" |
||||
|
|
||||
|
val initialized = openAIService.initialize(apiKey, baseUrl, model) |
||||
|
result.success(initialized) |
||||
|
} |
||||
|
|
||||
|
"registerFunction" -> { |
||||
|
val name = call.argument<String>("name") ?: "" |
||||
|
val description = call.argument<String>("description") ?: "" |
||||
|
val parameters = call.argument<Map<String, Any>>("parameters") |
||||
|
|
||||
|
if (name.isEmpty() || parameters == null) { |
||||
|
result.error("INVALID_ARGUMENT", "函数注册参数无效", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val parametersJson = JSONObject(parameters) |
||||
|
val registered = openAIService.registerFunction(name, description, parametersJson) |
||||
|
result.success(registered) |
||||
|
} |
||||
|
|
||||
|
"sendMessage" -> { |
||||
|
val messagesRaw = call.argument<List<Map<String, Any>>>("messages") ?: emptyList() |
||||
|
|
||||
|
// 转换消息格式 |
||||
|
val messages = JSONArray() |
||||
|
for (message in messagesRaw) { |
||||
|
messages.put(JSONObject(message)) |
||||
|
} |
||||
|
|
||||
|
// 在后台线程执行请求 |
||||
|
executor.execute { |
||||
|
try { |
||||
|
val response = openAIService.sendMessage(messages) |
||||
|
// 在主线程返回结果 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.success(response) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
// 在主线程返回错误 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.error("OPENAI_ERROR", e.message, null) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
"sendMessageStream" -> { |
||||
|
val messagesRaw = call.argument<List<Map<String, Any>>>("messages") ?: emptyList() |
||||
|
|
||||
|
// 检查事件接收器 |
||||
|
if (eventSink == null) { |
||||
|
result.error("NO_EVENT_SINK", "没有可用的事件流接收器", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 转换消息格式 |
||||
|
val messages = JSONArray() |
||||
|
for (message in messagesRaw) { |
||||
|
messages.put(JSONObject(message)) |
||||
|
} |
||||
|
|
||||
|
// 在后台线程执行请求 |
||||
|
executor.execute { |
||||
|
try { |
||||
|
openAIService.sendMessageStream( |
||||
|
messages = messages, |
||||
|
callback = object : OpenAIService.StreamCallback { |
||||
|
override fun onToken(token: String) { |
||||
|
// 发送token事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "token", "content" to token)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onComplete() { |
||||
|
// 发送完成事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "complete")) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(e: Exception) { |
||||
|
// 发送错误事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "error", "content" to e.message)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onFunctionCall(functionCall: JSONObject) { |
||||
|
// 发送函数调用事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
val functionCallMap = functionCall.toMap() |
||||
|
eventSink?.success(mapOf("type" to "functionCall", "content" to functionCallMap)) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 请求已开始 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
// 在主线程返回错误 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.error("OPENAI_ERROR", e.message, null) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
"sendFunctionCallResult" -> { |
||||
|
val messagesRaw = call.argument<List<Map<String, Any>>>("messages") ?: emptyList() |
||||
|
val functionCallRaw = call.argument<Map<String, Any>>("functionCall") ?: emptyMap() |
||||
|
val functionResult = call.argument<String>("functionResult") ?: "" |
||||
|
|
||||
|
// 检查事件接收器 |
||||
|
if (eventSink == null) { |
||||
|
result.error("NO_EVENT_SINK", "没有可用的事件流接收器", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 转换消息格式 |
||||
|
val messages = JSONArray() |
||||
|
for (message in messagesRaw) { |
||||
|
messages.put(JSONObject(message)) |
||||
|
} |
||||
|
|
||||
|
// 转换函数调用 |
||||
|
val functionCall = JSONObject(functionCallRaw) |
||||
|
|
||||
|
// 在后台线程执行请求 |
||||
|
executor.execute { |
||||
|
try { |
||||
|
openAIService.sendFunctionCallResult( |
||||
|
messages = messages, |
||||
|
functionCall = functionCall, |
||||
|
functionResult = functionResult, |
||||
|
callback = object : OpenAIService.StreamCallback { |
||||
|
override fun onToken(token: String) { |
||||
|
// 发送token事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "token", "content" to token)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onComplete() { |
||||
|
// 发送完成事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "complete")) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(e: Exception) { |
||||
|
// 发送错误事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
eventSink?.success(mapOf("type" to "error", "content" to e.message)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onFunctionCall(nestedFunctionCall: JSONObject) { |
||||
|
// 发送函数调用事件 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
val functionCallMap = nestedFunctionCall.toMap() |
||||
|
eventSink?.success(mapOf("type" to "functionCall", "content" to functionCallMap)) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
// 请求已开始 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
// 在主线程返回错误 |
||||
|
android.os.Handler(android.os.Looper.getMainLooper()).post { |
||||
|
result.error("OPENAI_ERROR", e.message, null) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
"createUserMessage" -> { |
||||
|
val content = call.argument<String>("content") ?: "" |
||||
|
val message = openAIService.createUserMessage(content) |
||||
|
result.success(message.toMap()) |
||||
|
} |
||||
|
|
||||
|
"createAssistantMessage" -> { |
||||
|
val content = call.argument<String>("content") ?: "" |
||||
|
val message = openAIService.createAssistantMessage(content) |
||||
|
result.success(message.toMap()) |
||||
|
} |
||||
|
|
||||
|
else -> { |
||||
|
result.notImplemented() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
methodChannel.setMethodCallHandler(null) |
||||
|
eventChannel.setStreamHandler(null) |
||||
|
executor.shutdown() |
||||
|
} |
||||
|
|
||||
|
// Stream事件处理 |
||||
|
override fun onListen(arguments: Any?, eventSink: EventSink?) { |
||||
|
this.eventSink = eventSink |
||||
|
} |
||||
|
|
||||
|
override fun onCancel(arguments: Any?) { |
||||
|
this.eventSink = null |
||||
|
} |
||||
|
|
||||
|
// 工具方法:JSONObject转Map |
||||
|
private fun JSONObject.toMap(): Map<String, Any?> { |
||||
|
val map = mutableMapOf<String, Any?>() |
||||
|
val keys = this.keys() |
||||
|
while (keys.hasNext()) { |
||||
|
val key = keys.next() |
||||
|
var value: Any? = this.opt(key) |
||||
|
|
||||
|
value = when (value) { |
||||
|
JSONObject.NULL -> null |
||||
|
is JSONObject -> value.toMap() |
||||
|
is JSONArray -> value.toList() |
||||
|
else -> value |
||||
|
} |
||||
|
|
||||
|
map[key] = value |
||||
|
} |
||||
|
return map |
||||
|
} |
||||
|
|
||||
|
// 工具方法:JSONArray转List |
||||
|
private fun JSONArray.toList(): List<Any?> { |
||||
|
val list = mutableListOf<Any?>() |
||||
|
for (i in 0 until this.length()) { |
||||
|
var value: Any? = this.opt(i) |
||||
|
|
||||
|
value = when (value) { |
||||
|
JSONObject.NULL -> null |
||||
|
is JSONObject -> value.toMap() |
||||
|
is JSONArray -> value.toList() |
||||
|
else -> value |
||||
|
} |
||||
|
|
||||
|
list.add(value) |
||||
|
} |
||||
|
return list |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,473 @@ |
|||||
|
import Foundation |
||||
|
|
||||
|
/// OpenAI服务异常 |
||||
|
public class OpenAIError: Error { |
||||
|
let message: String |
||||
|
|
||||
|
init(_ message: String) { |
||||
|
self.message = message |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 工具调用信息 |
||||
|
private class ToolCallInfo { |
||||
|
var id: String = "" |
||||
|
var name: String = "" |
||||
|
var arguments: String = "" |
||||
|
|
||||
|
var isValid: Bool { |
||||
|
return !id.isEmpty && !name.isEmpty |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// OpenAI服务iOS原生实现 |
||||
|
public class OpenAIService { |
||||
|
private let TAG = "OpenAIService" |
||||
|
private var baseUrl = "https://api.openai.com/v1/chat/completions" |
||||
|
private var apiKey: String = "" |
||||
|
private var isInitialized = false |
||||
|
private var model: String = "doubao-1-5-lite-32k-250115" // 默认模型 |
||||
|
|
||||
|
// 用于存储注册的函数 |
||||
|
private var registeredFunctions: [[String: Any]] = [] |
||||
|
|
||||
|
// URL会话 |
||||
|
private let session: URLSession |
||||
|
|
||||
|
public init() { |
||||
|
// 创建URL会话配置 |
||||
|
let config = URLSessionConfiguration.default |
||||
|
config.timeoutIntervalForRequest = 30.0 |
||||
|
config.timeoutIntervalForResource = 30.0 |
||||
|
session = URLSession(configuration: config) |
||||
|
} |
||||
|
|
||||
|
/// 创建用户消息 |
||||
|
public func createUserMessage(content: String) -> [String: Any] { |
||||
|
return ["role": "user", "content": content] |
||||
|
} |
||||
|
|
||||
|
/// 创建助手消息 |
||||
|
public func createAssistantMessage(content: String) -> [String: Any] { |
||||
|
return ["role": "assistant", "content": content] |
||||
|
} |
||||
|
|
||||
|
/// 初始化OpenAI服务 |
||||
|
public func initialize(apiKey: String, baseUrl: String = "", model: String = "") -> Bool { |
||||
|
self.apiKey = apiKey |
||||
|
if !baseUrl.isEmpty { |
||||
|
self.baseUrl = baseUrl |
||||
|
} |
||||
|
if !model.isEmpty { |
||||
|
self.model = model |
||||
|
} |
||||
|
isInitialized = !apiKey.isEmpty |
||||
|
return isInitialized |
||||
|
} |
||||
|
|
||||
|
/// 注册函数 |
||||
|
public func registerFunction(name: String, description: String, parameters: [String: Any]) -> Bool { |
||||
|
do { |
||||
|
let function: [String: Any] = [ |
||||
|
"name": name, |
||||
|
"description": description, |
||||
|
"parameters": parameters |
||||
|
] |
||||
|
|
||||
|
// 检查是否已存在相同名称的函数 |
||||
|
if let existingIndex = registeredFunctions.firstIndex(where: { ($0["name"] as? String) == name }) { |
||||
|
// 如果已存在,则替换 |
||||
|
registeredFunctions[existingIndex] = function |
||||
|
} else { |
||||
|
// 如果不存在,则添加 |
||||
|
registeredFunctions.append(function) |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch { |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息(非流式输出) |
||||
|
public func sendMessage(messages: [[String: Any]], systemPrompt: String) throws -> String { |
||||
|
guard isInitialized, !apiKey.isEmpty else { |
||||
|
throw OpenAIError("OpenAI服务未初始化") |
||||
|
} |
||||
|
|
||||
|
// 构建完整消息,添加系统提示 |
||||
|
var fullMessages: [[String: Any]] = [ |
||||
|
["role": "system", "content": systemPrompt] |
||||
|
] |
||||
|
fullMessages.append(contentsOf: messages) |
||||
|
|
||||
|
// 构建请求体 |
||||
|
var requestDict: [String: Any] = [ |
||||
|
"model": model, |
||||
|
"messages": fullMessages, |
||||
|
"temperature": 0.7, |
||||
|
"max_tokens": 2000, |
||||
|
"stream": false |
||||
|
] |
||||
|
|
||||
|
// 如果有注册的函数,添加到请求中 |
||||
|
if !registeredFunctions.isEmpty { |
||||
|
var tools: [[String: Any]] = [] |
||||
|
for function in registeredFunctions { |
||||
|
let tool: [String: Any] = [ |
||||
|
"type": "function", |
||||
|
"function": function |
||||
|
] |
||||
|
tools.append(tool) |
||||
|
} |
||||
|
requestDict["tools"] = tools |
||||
|
} |
||||
|
|
||||
|
// 将请求数据转换为JSON数据 |
||||
|
guard let jsonData = try? JSONSerialization.data(withJSONObject: requestDict) else { |
||||
|
throw OpenAIError("无法序列化请求数据") |
||||
|
} |
||||
|
|
||||
|
// 创建URL请求 |
||||
|
guard let url = URL(string: baseUrl) else { |
||||
|
throw OpenAIError("无效的URL") |
||||
|
} |
||||
|
|
||||
|
var request = URLRequest(url: url) |
||||
|
request.httpMethod = "POST" |
||||
|
request.addValue("application/json", forHTTPHeaderField: "Content-Type") |
||||
|
request.addValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization") |
||||
|
request.httpBody = jsonData |
||||
|
|
||||
|
// 创建信号量用于同步请求 |
||||
|
let semaphore = DispatchSemaphore(value: 0) |
||||
|
var responseResult: Result<String, Error> = .failure(OpenAIError("未收到响应")) |
||||
|
|
||||
|
// 执行请求 |
||||
|
let task = session.dataTask(with: request) { data, response, error in |
||||
|
if let error = error { |
||||
|
responseResult = .failure(OpenAIError("请求失败: \(error.localizedDescription)")) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard let httpResponse = response as? HTTPURLResponse else { |
||||
|
responseResult = .failure(OpenAIError("无效的HTTP响应")) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard httpResponse.statusCode == 200 else { |
||||
|
responseResult = .failure(OpenAIError("API调用失败: \(httpResponse.statusCode)")) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard let data = data else { |
||||
|
responseResult = .failure(OpenAIError("响应数据为空")) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
do { |
||||
|
// 解析JSON响应 |
||||
|
guard let jsonResponse = try JSONSerialization.jsonObject(with: data) as? [String: Any] else { |
||||
|
responseResult = .failure(OpenAIError("无法解析JSON响应")) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 检查是否有函数调用 |
||||
|
if let choices = jsonResponse["choices"] as? [[String: Any]], !choices.isEmpty, |
||||
|
let choice = choices.first, |
||||
|
let message = choice["message"] as? [String: Any] { |
||||
|
|
||||
|
// 检查是否有工具调用 |
||||
|
if let toolCalls = message["tool_calls"] as? [[String: Any]], !toolCalls.isEmpty, |
||||
|
let toolCall = toolCalls.first, |
||||
|
let function = toolCall["function"] as? [String: Any], |
||||
|
let name = function["name"] as? String, |
||||
|
let arguments = function["arguments"] as? String, |
||||
|
let id = toolCall["id"] as? String { |
||||
|
|
||||
|
let functionCallDict: [String: Any] = [ |
||||
|
"name": name, |
||||
|
"arguments": arguments, |
||||
|
"id": id |
||||
|
] |
||||
|
|
||||
|
// 将函数调用转为JSON字符串 |
||||
|
if let functionCallData = try? JSONSerialization.data(withJSONObject: functionCallDict), |
||||
|
let functionCallString = String(data: functionCallData, encoding: .utf8) { |
||||
|
responseResult = .success(functionCallString) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 如果没有工具调用,返回消息内容 |
||||
|
if let content = message["content"] as? String { |
||||
|
responseResult = .success(content) |
||||
|
semaphore.signal() |
||||
|
return |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
responseResult = .failure(OpenAIError("无效的响应格式")) |
||||
|
semaphore.signal() |
||||
|
|
||||
|
} catch { |
||||
|
responseResult = .failure(OpenAIError("解析响应时出错: \(error.localizedDescription)")) |
||||
|
semaphore.signal() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
task.resume() |
||||
|
|
||||
|
// 等待响应完成 |
||||
|
_ = semaphore.wait(timeout: .distantFuture) |
||||
|
|
||||
|
// 返回结果或抛出错误 |
||||
|
switch responseResult { |
||||
|
case .success(let result): |
||||
|
return result |
||||
|
case .failure(let error): |
||||
|
throw error |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息(流式输出) |
||||
|
public func sendMessageStream(messages: [[String: Any]], systemPrompt: String, callback: @escaping StreamCallback) { |
||||
|
guard isInitialized, !apiKey.isEmpty else { |
||||
|
callback.onError(OpenAIError("OpenAI服务未初始化")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 构建完整消息,添加系统提示 |
||||
|
var fullMessages: [[String: Any]] = [ |
||||
|
["role": "system", "content": systemPrompt] |
||||
|
] |
||||
|
fullMessages.append(contentsOf: messages) |
||||
|
|
||||
|
// 构建请求体 |
||||
|
var requestDict: [String: Any] = [ |
||||
|
"model": model, |
||||
|
"messages": fullMessages, |
||||
|
"temperature": 0.7, |
||||
|
"max_tokens": 2000, |
||||
|
"stream": true |
||||
|
] |
||||
|
|
||||
|
// 如果有注册的函数,添加到请求中 |
||||
|
if !registeredFunctions.isEmpty { |
||||
|
var tools: [[String: Any]] = [] |
||||
|
for function in registeredFunctions { |
||||
|
let tool: [String: Any] = [ |
||||
|
"type": "function", |
||||
|
"function": function |
||||
|
] |
||||
|
tools.append(tool) |
||||
|
} |
||||
|
requestDict["tools"] = tools |
||||
|
} |
||||
|
|
||||
|
// 将请求数据转换为JSON数据 |
||||
|
guard let jsonData = try? JSONSerialization.data(withJSONObject: requestDict) else { |
||||
|
callback.onError(OpenAIError("无法序列化请求数据")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 创建URL请求 |
||||
|
guard let url = URL(string: baseUrl) else { |
||||
|
callback.onError(OpenAIError("无效的URL")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
var request = URLRequest(url: url) |
||||
|
request.httpMethod = "POST" |
||||
|
request.addValue("application/json", forHTTPHeaderField: "Content-Type") |
||||
|
request.addValue("Bearer \(apiKey)", forHTTPHeaderField: "Authorization") |
||||
|
request.addValue("text/event-stream", forHTTPHeaderField: "Accept") |
||||
|
request.httpBody = jsonData |
||||
|
|
||||
|
// 用于存储函数调用的各个部分 |
||||
|
var finalToolCalls: [Int: ToolCallInfo] = [:] |
||||
|
|
||||
|
// 创建数据任务 |
||||
|
let task = session.dataTask(with: request) { data, response, error in |
||||
|
if let error = error { |
||||
|
callback.onError(OpenAIError("请求失败: \(error.localizedDescription)")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard let httpResponse = response as? HTTPURLResponse else { |
||||
|
callback.onError(OpenAIError("无效的HTTP响应")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard httpResponse.statusCode == 200 else { |
||||
|
callback.onError(OpenAIError("API调用失败: \(httpResponse.statusCode)")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
guard let data = data else { |
||||
|
callback.onError(OpenAIError("响应数据为空")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 处理SSE数据流 |
||||
|
if let text = String(data: data, encoding: .utf8) { |
||||
|
let lines = text.components(separatedBy: "\n") |
||||
|
|
||||
|
for line in lines { |
||||
|
if line.isEmpty { continue } |
||||
|
|
||||
|
if line.hasPrefix("data: ") { |
||||
|
let dataContent = line.dropFirst(6) |
||||
|
|
||||
|
// 处理[DONE]消息 |
||||
|
if dataContent == "[DONE]" { |
||||
|
self.processToolCalls(finalToolCalls, callback: callback) |
||||
|
callback.onComplete() |
||||
|
break |
||||
|
} |
||||
|
|
||||
|
// 解析JSON数据 |
||||
|
do { |
||||
|
if let data = dataContent.data(using: .utf8), |
||||
|
let jsonData = try JSONSerialization.jsonObject(with: data) as? [String: Any] { |
||||
|
|
||||
|
// 处理消息内容 |
||||
|
if let choices = jsonData["choices"] as? [[String: Any]], !choices.isEmpty, |
||||
|
let choice = choices.first { |
||||
|
|
||||
|
if let delta = choice["delta"] as? [String: Any] { |
||||
|
// 处理普通文本内容 |
||||
|
if let content = delta["content"] as? String { |
||||
|
callback.onToken(content) |
||||
|
} |
||||
|
|
||||
|
// 处理工具调用(函数调用) |
||||
|
if let toolCalls = delta["tool_calls"] as? [[String: Any]] { |
||||
|
for toolCall in toolCalls { |
||||
|
if let index = toolCall["index"] as? Int { |
||||
|
// 创建或获取现有的工具调用信息 |
||||
|
let toolCallInfo = finalToolCalls[index] ?? ToolCallInfo() |
||||
|
|
||||
|
// 更新ID |
||||
|
if let id = toolCall["id"] as? String { |
||||
|
toolCallInfo.id = id |
||||
|
} |
||||
|
|
||||
|
// 更新函数信息 |
||||
|
if let function = toolCall["function"] as? [String: Any] { |
||||
|
if let name = function["name"] as? String { |
||||
|
toolCallInfo.name = name |
||||
|
} |
||||
|
|
||||
|
if let arguments = function["arguments"] as? String { |
||||
|
toolCallInfo.arguments += arguments |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
finalToolCalls[index] = toolCallInfo |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} catch { |
||||
|
NSLog("解析JSON出错: \(error.localizedDescription)") |
||||
|
// 忽略解析错误,继续处理其他行 |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
task.resume() |
||||
|
} |
||||
|
|
||||
|
/// 发送函数调用结果 |
||||
|
public func sendFunctionCallResult( |
||||
|
messages: [[String: Any]], |
||||
|
systemPrompt: String, |
||||
|
functionCall: [String: Any], |
||||
|
functionResult: String, |
||||
|
callback: @escaping StreamCallback |
||||
|
) { |
||||
|
do { |
||||
|
// 构建完整消息数组 |
||||
|
var fullMessages: [[String: Any]] = [ |
||||
|
// 添加系统提示 |
||||
|
["role": "system", "content": systemPrompt] |
||||
|
] |
||||
|
|
||||
|
// 添加用户消息 |
||||
|
fullMessages.append(contentsOf: messages) |
||||
|
|
||||
|
// 获取函数相关信息 |
||||
|
guard let name = functionCall["name"] as? String, |
||||
|
let arguments = functionCall["arguments"] as? String else { |
||||
|
callback.onError(OpenAIError("函数调用信息不完整")) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
let id = functionCall["id"] as? String ?? "call_\(Int(Date().timeIntervalSince1970 * 1000))" |
||||
|
|
||||
|
// 添加函数调用消息 |
||||
|
fullMessages.append([ |
||||
|
"role": "assistant", |
||||
|
"content": NSNull(), |
||||
|
"tool_calls": [ |
||||
|
[ |
||||
|
"id": id, |
||||
|
"type": "function", |
||||
|
"function": [ |
||||
|
"name": name, |
||||
|
"arguments": arguments |
||||
|
] |
||||
|
] |
||||
|
] |
||||
|
]) |
||||
|
|
||||
|
// 添加函数调用结果 |
||||
|
fullMessages.append([ |
||||
|
"role": "tool", |
||||
|
"content": functionResult, |
||||
|
"tool_call_id": id |
||||
|
]) |
||||
|
|
||||
|
// 发送完整对话 |
||||
|
sendMessageStream(messages: fullMessages, systemPrompt: systemPrompt, callback: callback) |
||||
|
|
||||
|
} catch { |
||||
|
callback.onError(OpenAIError("发送函数调用结果失败: \(error.localizedDescription)")) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 处理工具调用(函数调用)并回调 |
||||
|
private func processToolCalls(_ toolCalls: [Int: ToolCallInfo], callback: StreamCallback) { |
||||
|
if toolCalls.isEmpty { return } |
||||
|
|
||||
|
// 只处理第一个工具调用 |
||||
|
guard let firstToolCall = toolCalls.values.first, firstToolCall.isValid else { return } |
||||
|
|
||||
|
// 创建函数调用字典 |
||||
|
let functionCall: [String: Any] = [ |
||||
|
"name": firstToolCall.name, |
||||
|
"arguments": firstToolCall.arguments, |
||||
|
"id": firstToolCall.id |
||||
|
] |
||||
|
|
||||
|
// 回调 |
||||
|
callback.onFunctionCall(functionCall) |
||||
|
} |
||||
|
|
||||
|
/// 流式输出回调协议 |
||||
|
public typealias StreamCallback = (onToken: (String) -> Void, |
||||
|
onComplete: () -> Void, |
||||
|
onError: (Error) -> Void, |
||||
|
onFunctionCall: ([String: Any]) -> Void) |
||||
|
} |
||||
@ -0,0 +1,215 @@ |
|||||
|
import Flutter |
||||
|
import UIKit |
||||
|
|
||||
|
public class OpenAIServicePlugin: NSObject, FlutterPlugin, FlutterStreamHandler { |
||||
|
// OpenAI服务实例 |
||||
|
private let openAIService = OpenAIService() |
||||
|
|
||||
|
// 事件接收器 |
||||
|
private var eventSink: FlutterEventSink? |
||||
|
|
||||
|
// 注册插件 |
||||
|
public static func register(with registrar: FlutterPluginRegistrar) { |
||||
|
let methodChannel = FlutterMethodChannel(name: "com.yunqiinnovation.open_ai_service/methods", binaryMessenger: registrar.messenger()) |
||||
|
let eventChannel = FlutterEventChannel(name: "com.yunqiinnovation.open_ai_service/events", binaryMessenger: registrar.messenger()) |
||||
|
|
||||
|
let instance = OpenAIServicePlugin() |
||||
|
registrar.addMethodCallDelegate(instance, channel: methodChannel) |
||||
|
eventChannel.setStreamHandler(instance) |
||||
|
} |
||||
|
|
||||
|
// 处理方法调用 |
||||
|
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
||||
|
switch call.method { |
||||
|
case "initialize": |
||||
|
if let args = call.arguments as? [String: Any], |
||||
|
let apiKey = args["apiKey"] as? String { |
||||
|
let baseUrl = args["baseUrl"] as? String ?? "" |
||||
|
let model = args["model"] as? String ?? "" |
||||
|
let initialized = openAIService.initialize(apiKey: apiKey, baseUrl: baseUrl, model: model) |
||||
|
result(initialized) |
||||
|
} else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "初始化参数无效", details: nil)) |
||||
|
} |
||||
|
|
||||
|
case "registerFunction": |
||||
|
if let args = call.arguments as? [String: Any], |
||||
|
let name = args["name"] as? String, |
||||
|
let description = args["description"] as? String, |
||||
|
let parameters = args["parameters"] as? [String: Any] { |
||||
|
|
||||
|
let registered = openAIService.registerFunction(name: name, description: description, parameters: parameters) |
||||
|
result(registered) |
||||
|
} else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "函数注册参数无效", details: nil)) |
||||
|
} |
||||
|
|
||||
|
case "sendMessage": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let messagesRaw = args["messages"] as? [[String: Any]], |
||||
|
let systemPrompt = args["systemPrompt"] as? String else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "发送消息参数无效", details: nil)) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 在后台线程执行 |
||||
|
DispatchQueue.global(qos: .userInitiated).async { |
||||
|
do { |
||||
|
let response = try self.openAIService.sendMessage(messages: messagesRaw, systemPrompt: systemPrompt) |
||||
|
// 在主线程返回结果 |
||||
|
DispatchQueue.main.async { |
||||
|
result(response) |
||||
|
} |
||||
|
} catch { |
||||
|
// 在主线程返回错误 |
||||
|
DispatchQueue.main.async { |
||||
|
result(FlutterError(code: "OPENAI_ERROR", message: error.localizedDescription, details: nil)) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
case "sendMessageStream": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let messagesRaw = args["messages"] as? [[String: Any]], |
||||
|
let systemPrompt = args["systemPrompt"] as? String else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "发送消息参数无效", details: nil)) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 检查事件接收器 |
||||
|
guard let eventSink = self.eventSink else { |
||||
|
result(FlutterError(code: "NO_EVENT_SINK", message: "没有可用的事件流接收器", details: nil)) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 在后台线程执行 |
||||
|
DispatchQueue.global(qos: .userInitiated).async { |
||||
|
let callback: OpenAIService.StreamCallback = ( |
||||
|
onToken: { token in |
||||
|
// 发送token事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "token", "content": token]) |
||||
|
} |
||||
|
}, |
||||
|
onComplete: { |
||||
|
// 发送完成事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "complete"]) |
||||
|
} |
||||
|
}, |
||||
|
onError: { error in |
||||
|
// 发送错误事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "error", "content": error.localizedDescription]) |
||||
|
} |
||||
|
}, |
||||
|
onFunctionCall: { functionCall in |
||||
|
// 发送函数调用事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "functionCall", "content": functionCall]) |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
self.openAIService.sendMessageStream(messages: messagesRaw, systemPrompt: systemPrompt, callback: callback) |
||||
|
|
||||
|
// 请求已开始 |
||||
|
DispatchQueue.main.async { |
||||
|
result(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
case "sendFunctionCallResult": |
||||
|
guard let args = call.arguments as? [String: Any], |
||||
|
let messagesRaw = args["messages"] as? [[String: Any]], |
||||
|
let systemPrompt = args["systemPrompt"] as? String, |
||||
|
let functionCallRaw = args["functionCall"] as? [String: Any], |
||||
|
let functionResult = args["functionResult"] as? String else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "发送函数调用结果参数无效", details: nil)) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 检查事件接收器 |
||||
|
guard let eventSink = self.eventSink else { |
||||
|
result(FlutterError(code: "NO_EVENT_SINK", message: "没有可用的事件流接收器", details: nil)) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
// 在后台线程执行 |
||||
|
DispatchQueue.global(qos: .userInitiated).async { |
||||
|
let callback: OpenAIService.StreamCallback = ( |
||||
|
onToken: { token in |
||||
|
// 发送token事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "token", "content": token]) |
||||
|
} |
||||
|
}, |
||||
|
onComplete: { |
||||
|
// 发送完成事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "complete"]) |
||||
|
} |
||||
|
}, |
||||
|
onError: { error in |
||||
|
// 发送错误事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "error", "content": error.localizedDescription]) |
||||
|
} |
||||
|
}, |
||||
|
onFunctionCall: { functionCall in |
||||
|
// 发送函数调用事件 |
||||
|
DispatchQueue.main.async { |
||||
|
eventSink(["type": "functionCall", "content": functionCall]) |
||||
|
} |
||||
|
} |
||||
|
) |
||||
|
|
||||
|
self.openAIService.sendFunctionCallResult( |
||||
|
messages: messagesRaw, |
||||
|
systemPrompt: systemPrompt, |
||||
|
functionCall: functionCallRaw, |
||||
|
functionResult: functionResult, |
||||
|
callback: callback |
||||
|
) |
||||
|
|
||||
|
// 请求已开始 |
||||
|
DispatchQueue.main.async { |
||||
|
result(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
case "createUserMessage": |
||||
|
if let args = call.arguments as? [String: Any], |
||||
|
let content = args["content"] as? String { |
||||
|
let message = openAIService.createUserMessage(content: content) |
||||
|
result(message) |
||||
|
} else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "创建用户消息参数无效", details: nil)) |
||||
|
} |
||||
|
|
||||
|
case "createAssistantMessage": |
||||
|
if let args = call.arguments as? [String: Any], |
||||
|
let content = args["content"] as? String { |
||||
|
let message = openAIService.createAssistantMessage(content: content) |
||||
|
result(message) |
||||
|
} else { |
||||
|
result(FlutterError(code: "INVALID_ARGUMENT", message: "创建助手消息参数无效", details: nil)) |
||||
|
} |
||||
|
|
||||
|
default: |
||||
|
result(FlutterMethodNotImplemented) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// MARK: - FlutterStreamHandler |
||||
|
|
||||
|
public func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
||||
|
self.eventSink = events |
||||
|
return nil |
||||
|
} |
||||
|
|
||||
|
public func onCancel(withArguments arguments: Any?) -> FlutterError? { |
||||
|
self.eventSink = nil |
||||
|
return nil |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,267 @@ |
|||||
|
import 'dart:async'; |
||||
|
import 'dart:convert'; |
||||
|
|
||||
|
import 'package:flutter/services.dart'; |
||||
|
|
||||
|
/// OpenAI服务异常 |
||||
|
class OpenAIException implements Exception { |
||||
|
final String message; |
||||
|
|
||||
|
OpenAIException(this.message); |
||||
|
|
||||
|
@override |
||||
|
String toString() => 'OpenAIException: $message'; |
||||
|
} |
||||
|
|
||||
|
/// OpenAI服务事件类型 |
||||
|
enum OpenAIEventType { |
||||
|
token, |
||||
|
complete, |
||||
|
error, |
||||
|
functionCall, |
||||
|
} |
||||
|
|
||||
|
/// OpenAI服务事件 |
||||
|
class OpenAIEvent { |
||||
|
final OpenAIEventType type; |
||||
|
final dynamic content; |
||||
|
|
||||
|
OpenAIEvent({required this.type, this.content}); |
||||
|
|
||||
|
factory OpenAIEvent.fromMap(Map<String, dynamic> map) { |
||||
|
final typeStr = map['type'] as String; |
||||
|
final content = map['content']; |
||||
|
|
||||
|
return OpenAIEvent( |
||||
|
type: _typeFromString(typeStr), |
||||
|
content: content, |
||||
|
); |
||||
|
} |
||||
|
|
||||
|
static OpenAIEventType _typeFromString(String typeStr) { |
||||
|
switch (typeStr) { |
||||
|
case 'token': |
||||
|
return OpenAIEventType.token; |
||||
|
case 'complete': |
||||
|
return OpenAIEventType.complete; |
||||
|
case 'error': |
||||
|
return OpenAIEventType.error; |
||||
|
case 'functionCall': |
||||
|
return OpenAIEventType.functionCall; |
||||
|
default: |
||||
|
throw ArgumentError('未知的事件类型: $typeStr'); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// OpenAI服务插件 |
||||
|
class OpenAIService { |
||||
|
static const MethodChannel _channel = MethodChannel('com.yunqiinnovation.open_ai_service/methods'); |
||||
|
static const EventChannel _eventChannel = EventChannel('com.yunqiinnovation.open_ai_service/events'); |
||||
|
|
||||
|
/// 事件流控制器 |
||||
|
StreamController<OpenAIEvent>? _eventStreamController; |
||||
|
|
||||
|
/// 事件流 |
||||
|
Stream<OpenAIEvent>? _eventStream; |
||||
|
|
||||
|
/// 获取事件流 |
||||
|
Stream<OpenAIEvent> get eventStream { |
||||
|
if (_eventStream == null) { |
||||
|
_eventStreamController = StreamController<OpenAIEvent>.broadcast(); |
||||
|
_eventStream = _eventStreamController!.stream; |
||||
|
|
||||
|
// 监听原生事件 |
||||
|
_eventChannel.receiveBroadcastStream().listen( |
||||
|
(dynamic event) { |
||||
|
if (event is Map<dynamic, dynamic>) { |
||||
|
final eventMap = Map<String, dynamic>.from(event); |
||||
|
final openAIEvent = OpenAIEvent.fromMap(eventMap); |
||||
|
_eventStreamController!.add(openAIEvent); |
||||
|
} |
||||
|
}, |
||||
|
onError: (error) { |
||||
|
_eventStreamController!.addError(OpenAIException('事件流错误: $error')); |
||||
|
}, |
||||
|
); |
||||
|
} |
||||
|
|
||||
|
return _eventStream!; |
||||
|
} |
||||
|
|
||||
|
/// 初始化OpenAI服务 |
||||
|
/// |
||||
|
/// [apiKey] OpenAI API密钥 |
||||
|
/// [baseUrl] 可选,自定义API基础URL |
||||
|
/// [model] 可选,自定义使用的模型 |
||||
|
Future<bool> initialize({ |
||||
|
required String apiKey, |
||||
|
String baseUrl = '', |
||||
|
String model = '', |
||||
|
}) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<bool>( |
||||
|
'initialize', |
||||
|
{ |
||||
|
'apiKey': apiKey, |
||||
|
'baseUrl': baseUrl, |
||||
|
'model': model, |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
return result ?? false; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('初始化失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 注册函数 |
||||
|
/// |
||||
|
/// [name] 函数名称 |
||||
|
/// [description] 函数描述 |
||||
|
/// [parameters] 函数参数 |
||||
|
Future<bool> registerFunction({ |
||||
|
required String name, |
||||
|
required String description, |
||||
|
required Map<String, dynamic> parameters, |
||||
|
}) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<bool>( |
||||
|
'registerFunction', |
||||
|
{ |
||||
|
'name': name, |
||||
|
'description': description, |
||||
|
'parameters': parameters, |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
return result ?? false; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('注册函数失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 创建用户消息 |
||||
|
/// |
||||
|
/// [content] 消息内容 |
||||
|
Future<Map<String, dynamic>> createUserMessage(String content) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<Map<dynamic, dynamic>>( |
||||
|
'createUserMessage', |
||||
|
{'content': content}, |
||||
|
); |
||||
|
|
||||
|
if (result == null) { |
||||
|
throw OpenAIException('创建用户消息失败: 结果为空'); |
||||
|
} |
||||
|
|
||||
|
return Map<String, dynamic>.from(result); |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('创建用户消息失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 创建助手消息 |
||||
|
/// |
||||
|
/// [content] 消息内容 |
||||
|
Future<Map<String, dynamic>> createAssistantMessage(String content) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<Map<dynamic, dynamic>>( |
||||
|
'createAssistantMessage', |
||||
|
{'content': content}, |
||||
|
); |
||||
|
|
||||
|
if (result == null) { |
||||
|
throw OpenAIException('创建助手消息失败: 结果为空'); |
||||
|
} |
||||
|
|
||||
|
return Map<String, dynamic>.from(result); |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('创建助手消息失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息(非流式输出) |
||||
|
/// |
||||
|
/// [messages] 消息列表 |
||||
|
Future<String> sendMessage({ |
||||
|
required List<Map<String, dynamic>> messages, |
||||
|
}) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<String>( |
||||
|
'sendMessage', |
||||
|
{ |
||||
|
'messages': messages, |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
if (result == null) { |
||||
|
throw OpenAIException('发送消息失败: 结果为空'); |
||||
|
} |
||||
|
|
||||
|
return result; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('发送消息失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送消息(流式输出) |
||||
|
/// |
||||
|
/// [messages] 消息列表 |
||||
|
/// |
||||
|
/// 返回一个布尔值,表示请求是否已开始 |
||||
|
Future<bool> sendMessageStream({ |
||||
|
required List<Map<String, dynamic>> messages, |
||||
|
}) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<bool>( |
||||
|
'sendMessageStream', |
||||
|
{ |
||||
|
'messages': messages, |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
return result ?? false; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('发送流式消息失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 发送函数调用结果 |
||||
|
/// |
||||
|
/// [messages] 消息列表 |
||||
|
/// [functionCall] 函数调用信息 |
||||
|
/// [functionResult] 函数调用结果 |
||||
|
/// |
||||
|
/// 返回一个布尔值,表示请求是否已开始 |
||||
|
Future<bool> sendFunctionCallResult({ |
||||
|
required List<Map<String, dynamic>> messages, |
||||
|
required Map<String, dynamic> functionCall, |
||||
|
required String functionResult, |
||||
|
}) async { |
||||
|
try { |
||||
|
final result = await _channel.invokeMethod<bool>( |
||||
|
'sendFunctionCallResult', |
||||
|
{ |
||||
|
'messages': messages, |
||||
|
'functionCall': functionCall, |
||||
|
'functionResult': functionResult, |
||||
|
}, |
||||
|
); |
||||
|
|
||||
|
return result ?? false; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('发送函数调用结果失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
|
||||
|
/// 从JSON字符串解析函数调用 |
||||
|
Map<String, dynamic> parseFunctionCall(String functionCallJson) { |
||||
|
try { |
||||
|
return json.decode(functionCallJson) as Map<String, dynamic>; |
||||
|
} catch (e) { |
||||
|
throw OpenAIException('解析函数调用失败: $e'); |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,26 @@ |
|||||
|
name: open_ai_service |
||||
|
description: 原生OpenAI服务插件,提供与OpenAI API的交互功能,支持流式输出和函数调用。 |
||||
|
version: 0.0.1 |
||||
|
homepage: https://github.com/yunqiinnovation/deep_voice |
||||
|
|
||||
|
environment: |
||||
|
sdk: ">=2.17.0 <4.0.0" |
||||
|
flutter: ">=2.5.0" |
||||
|
|
||||
|
dependencies: |
||||
|
flutter: |
||||
|
sdk: flutter |
||||
|
|
||||
|
dev_dependencies: |
||||
|
flutter_test: |
||||
|
sdk: flutter |
||||
|
flutter_lints: ^2.0.0 |
||||
|
|
||||
|
flutter: |
||||
|
plugin: |
||||
|
platforms: |
||||
|
android: |
||||
|
package: com.yunqiinnovation.open_ai_service |
||||
|
pluginClass: OpenAIServicePlugin |
||||
|
ios: |
||||
|
pluginClass: OpenAIServicePlugin |
||||
@ -0,0 +1,206 @@ |
|||||
|
# 火山引擎语音服务插件 |
||||
|
|
||||
|
基于火山引擎语音SDK封装的Flutter插件,支持Android平台。 |
||||
|
|
||||
|
## 功能 |
||||
|
- 语音合成(TTS) |
||||
|
- 支持在线/离线/混合合成模式 |
||||
|
- 支持SSML格式文本 |
||||
|
- 支持情感合成和情感预测 |
||||
|
- 支持音量、语速、音高等多种参数调节 |
||||
|
- 复刻音色支持 |
||||
|
- 离线资源管理 |
||||
|
- 语音识别(ASR) |
||||
|
- 一次性识别 |
||||
|
- 连续识别 |
||||
|
- 长按识别 |
||||
|
- 热词优化 |
||||
|
- 多语言支持 |
||||
|
|
||||
|
## 使用说明 |
||||
|
|
||||
|
### TTS 语音合成 |
||||
|
|
||||
|
#### 基础用法 |
||||
|
```dart |
||||
|
import 'package:volcano_speech/volcano_speech.dart'; |
||||
|
|
||||
|
// 初始化 |
||||
|
await VolcanoSpeech.initialize(appId: 'YOUR_APP_ID', apiKey: 'YOUR_API_KEY'); |
||||
|
|
||||
|
// 设置音量、语速等参数 |
||||
|
await VolcanoSpeech.setSpeechParams( |
||||
|
rate: 0, // 语速: -500到500,0为正常速度 |
||||
|
volume: 100, // 音量: 0到100,默认100 |
||||
|
pitch: 0, // 音高: -500到500,0为正常音高 |
||||
|
silenceDuration: 500, // 静音段时长: 毫秒 |
||||
|
); |
||||
|
|
||||
|
// 设置音色 |
||||
|
await VolcanoSpeech.setVoice( |
||||
|
voiceName: 'zh_female_qingxin', |
||||
|
voiceType: 'qingxin' |
||||
|
); |
||||
|
|
||||
|
// 开始合成并播放 |
||||
|
await VolcanoSpeech.speakText(text: '这是一段测试文本'); |
||||
|
|
||||
|
// 暂停播放 |
||||
|
await VolcanoSpeech.pausePlayback(); |
||||
|
|
||||
|
// 继续播放 |
||||
|
await VolcanoSpeech.resumePlayback(); |
||||
|
|
||||
|
// 停止播放 |
||||
|
await VolcanoSpeech.stopSpeaking(); |
||||
|
|
||||
|
// 销毁引擎 |
||||
|
await VolcanoSpeech.dispose(); |
||||
|
``` |
||||
|
|
||||
|
#### 进阶用法 |
||||
|
```dart |
||||
|
// 设置工作模式 |
||||
|
await VolcanoSpeech.setWorkMode(mode: TtsWorkMode.alternate); // 先在线,断网时切换离线 |
||||
|
|
||||
|
// 设置离线发音人 |
||||
|
await VolcanoSpeech.setOfflineVoice( |
||||
|
voiceName: 'xifei', |
||||
|
voiceType: 'qingxin' |
||||
|
); |
||||
|
|
||||
|
// 下载离线资源 |
||||
|
await VolcanoSpeech.downloadOfflineResource( |
||||
|
voiceTypes: ['qingxin', 'zhenjiang'], |
||||
|
languages: ['zh-CN'] |
||||
|
); |
||||
|
|
||||
|
// 使用SSML格式文本 |
||||
|
await VolcanoSpeech.setTextType(type: TtsTextType.ssml); |
||||
|
await VolcanoSpeech.speakText( |
||||
|
text: '<speak>这是一段<say-as interpret-as="date">2023-10-01</say-as>的语音合成</speak>' |
||||
|
); |
||||
|
|
||||
|
// 设置情感 |
||||
|
await VolcanoSpeech.setEmotion(emotion: 'happy'); |
||||
|
|
||||
|
// 启用情感预测 |
||||
|
await VolcanoSpeech.setEnableEmotionPredict(enable: true); |
||||
|
|
||||
|
// 启用服务端缓存 |
||||
|
await VolcanoSpeech.setEnableCache(enable: true); |
||||
|
|
||||
|
// 监听TTS进度事件 |
||||
|
VolcanoSpeech.ttsProgressEvents.listen((event) { |
||||
|
print('播放进度: ${(event.progress * 100).toStringAsFixed(1)}%'); |
||||
|
}); |
||||
|
|
||||
|
// 复刻音色支持 |
||||
|
await VolcanoSpeech.setEnableVoiceClone( |
||||
|
enable: true, |
||||
|
backendCluster: 'your_cluster_name' |
||||
|
); |
||||
|
``` |
||||
|
|
||||
|
### ASR 语音识别 |
||||
|
#### 一次性识别 |
||||
|
```dart |
||||
|
import 'package:volcano_speech/volcano_speech.dart'; |
||||
|
|
||||
|
// 初始化 |
||||
|
await VolcanoSpeech.initializeAsr( |
||||
|
appId: 'YOUR_APP_ID', |
||||
|
apiKey: 'YOUR_API_KEY', |
||||
|
supportedLanguages: ['zh-CN'], |
||||
|
); |
||||
|
|
||||
|
// 设置识别语言 |
||||
|
await VolcanoSpeech.setAsrLanguage(language: 'zh-CN'); |
||||
|
|
||||
|
// 设置热词(可选) |
||||
|
await VolcanoSpeech.setAsrHotWords( |
||||
|
hotWords: '{"hotwords":[{"word":"火山引擎","scale":2.0}]}', |
||||
|
); |
||||
|
|
||||
|
// 开始一次性识别 |
||||
|
try { |
||||
|
final result = await VolcanoSpeech.recognizeOnce(); |
||||
|
print('识别结果: ${result['text']}, 语言: ${result['language']}'); |
||||
|
} catch (e) { |
||||
|
print('识别出错: $e'); |
||||
|
} |
||||
|
|
||||
|
// 销毁引擎 |
||||
|
await VolcanoSpeech.disposeAsr(); |
||||
|
``` |
||||
|
|
||||
|
#### 连续识别 |
||||
|
```dart |
||||
|
import 'package:volcano_speech/volcano_speech.dart'; |
||||
|
|
||||
|
// 初始化 |
||||
|
await VolcanoSpeech.initializeAsr( |
||||
|
appId: 'YOUR_APP_ID', |
||||
|
apiKey: 'YOUR_API_KEY', |
||||
|
); |
||||
|
|
||||
|
// 监听识别事件 |
||||
|
VolcanoSpeech.asrEvents.listen((event) { |
||||
|
switch (event.type) { |
||||
|
case AsrEventType.sessionStarted: |
||||
|
print('识别会话开始'); |
||||
|
break; |
||||
|
case AsrEventType.sessionStopped: |
||||
|
print('识别会话结束'); |
||||
|
break; |
||||
|
case AsrEventType.recognizing: |
||||
|
print('正在识别: ${event.text}'); |
||||
|
break; |
||||
|
case AsrEventType.result: |
||||
|
print('识别结果: ${event.text}'); |
||||
|
break; |
||||
|
case AsrEventType.volumeChanged: |
||||
|
print('音量: ${event.volume}'); |
||||
|
break; |
||||
|
case AsrEventType.error: |
||||
|
print('识别错误: ${event.errorMessage}'); |
||||
|
break; |
||||
|
} |
||||
|
}); |
||||
|
|
||||
|
// 开始连续识别 |
||||
|
await VolcanoSpeech.startContinuousRecognition(); |
||||
|
|
||||
|
// 检查是否正在识别 |
||||
|
final isActive = await VolcanoSpeech.isContinuousRecognitionActive(); |
||||
|
print('是否正在识别: $isActive'); |
||||
|
|
||||
|
// 停止连续识别 |
||||
|
await VolcanoSpeech.stopContinuousRecognition(); |
||||
|
``` |
||||
|
|
||||
|
#### 长按识别 |
||||
|
```dart |
||||
|
import 'package:volcano_speech/volcano_speech.dart'; |
||||
|
|
||||
|
// 初始化 |
||||
|
await VolcanoSpeech.initializeAsr( |
||||
|
appId: 'YOUR_APP_ID', |
||||
|
apiKey: 'YOUR_API_KEY', |
||||
|
); |
||||
|
|
||||
|
// 监听识别事件 |
||||
|
VolcanoSpeech.asrEvents.listen((event) { |
||||
|
// 处理事件... |
||||
|
}); |
||||
|
|
||||
|
// 用户按下按钮时开始识别 |
||||
|
onPressed: () async { |
||||
|
await VolcanoSpeech.startListening(); |
||||
|
}, |
||||
|
|
||||
|
// 用户释放按钮时停止识别 |
||||
|
onReleased: () async { |
||||
|
await VolcanoSpeech.stopListening(); |
||||
|
}, |
||||
|
``` |
||||
@ -0,0 +1,64 @@ |
|||||
|
import com.android.build.gradle.LibraryExtension |
||||
|
|
||||
|
buildscript { |
||||
|
repositories { |
||||
|
google() |
||||
|
mavenCentral() |
||||
|
} |
||||
|
dependencies { |
||||
|
classpath("com.android.tools.build:gradle:7.3.0") |
||||
|
classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
allprojects { |
||||
|
repositories { |
||||
|
google() |
||||
|
mavenCentral() |
||||
|
|
||||
|
} |
||||
|
} |
||||
|
|
||||
|
plugins { |
||||
|
id("com.android.library") |
||||
|
kotlin("android") |
||||
|
} |
||||
|
|
||||
|
// 配置android扩展 |
||||
|
configure<LibraryExtension> { |
||||
|
namespace = "com.yunqiinnovation.volcano_speech" |
||||
|
compileSdkVersion(33) |
||||
|
|
||||
|
defaultConfig { |
||||
|
minSdk = 21 |
||||
|
} |
||||
|
|
||||
|
compileOptions { |
||||
|
sourceCompatibility = JavaVersion.VERSION_11 |
||||
|
targetCompatibility = JavaVersion.VERSION_11 |
||||
|
} |
||||
|
|
||||
|
sourceSets { |
||||
|
getByName("main") { |
||||
|
manifest.srcFile("src/main/AndroidManifest.xml") |
||||
|
java.srcDirs("src/main/kotlin") |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 添加lint选项 |
||||
|
lintOptions { |
||||
|
isCheckReleaseBuilds = false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 显式设置Kotlin JVM目标版本 |
||||
|
tasks.withType<org.jetbrains.kotlin.gradle.tasks.KotlinCompile> { |
||||
|
kotlinOptions { |
||||
|
jvmTarget = "11" |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
dependencies { |
||||
|
// 添加火山引擎语音合成SDK |
||||
|
implementation("com.bytedance.speechengine:speechengine_tob:0.0.5") |
||||
|
} |
||||
@ -0,0 +1 @@ |
|||||
|
rootProject.name = "volcano_speech" |
||||
@ -0,0 +1,9 @@ |
|||||
|
<?xml version="1.0" encoding="utf-8"?> |
||||
|
<manifest xmlns:android="http://schemas.android.com/apk/res/android" |
||||
|
package="com.yunqiinnovation.volcano_speech"> |
||||
|
<uses-permission android:name="android.permission.READ_EXTERNAL_STORAGE" /> |
||||
|
<uses-permission android:name="android.permission.WRITE_EXTERNAL_STORAGE" /> |
||||
|
<uses-permission android:name="android.permission.INTERNET" /> |
||||
|
<uses-permission android:name="android.permission.ACCESS_NETWORK_STATE" /> |
||||
|
<uses-permission android:name="android.permission.FOREGROUND_SERVICE"/> |
||||
|
</manifest> |
||||
@ -0,0 +1,529 @@ |
|||||
|
package com.yunqiinnovation.volcano_speech |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.os.Handler |
||||
|
import android.os.Looper |
||||
|
import com.bytedance.speech.speechengine.SpeechEngine |
||||
|
import com.bytedance.speech.speechengine.SpeechEngineDefines |
||||
|
import com.bytedance.speech.speechengine.SpeechEngineGenerator |
||||
|
import com.yunqiinnovation.volcano_speech.utils.FileLogger |
||||
|
import org.json.JSONArray |
||||
|
import org.json.JSONException |
||||
|
import org.json.JSONObject |
||||
|
|
||||
|
/** |
||||
|
* 火山语音识别帮助类 (大模型版本) |
||||
|
*/ |
||||
|
class VolcanoAsrHelper(private val context: Context) { |
||||
|
private val TAG = "VolcanoAsrHelper" |
||||
|
private val mainHandler = Handler(Looper.getMainLooper()) |
||||
|
|
||||
|
// 语音引擎相关 |
||||
|
private var engine: SpeechEngine? = null |
||||
|
private var engineHandler: Long = -1 |
||||
|
private var isInitialized = false |
||||
|
|
||||
|
// 当前回调 |
||||
|
private var currentAsrCallback: ASRCallback? = null |
||||
|
private var currentContinuousCallback: ASRContinuousCallback? = null |
||||
|
|
||||
|
// 识别状态 |
||||
|
private var isContinuousRecognitionActive = false |
||||
|
|
||||
|
// 配置参数 |
||||
|
private var language = "zh-CN" |
||||
|
private var enableVolume = false |
||||
|
private var showUtterances = false |
||||
|
|
||||
|
/** |
||||
|
* ASR一次性识别回调 |
||||
|
*/ |
||||
|
interface ASRCallback { |
||||
|
fun onSuccess(text: String) |
||||
|
fun onError(error: String) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* ASR连续识别回调 |
||||
|
*/ |
||||
|
interface ASRContinuousCallback { |
||||
|
fun onResult(text: String) |
||||
|
fun onRecognizing(text: String) |
||||
|
fun onSessionStarted() |
||||
|
fun onSessionStopped() |
||||
|
fun onVolumeChanged(volume: Int) |
||||
|
fun onError(error: String) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 初始化语音识别引擎 |
||||
|
*/ |
||||
|
fun initialize(appId: String, token: String, resourceId: String): Boolean { |
||||
|
if (isInitialized) { |
||||
|
FileLogger.i(TAG, "引擎已经初始化") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 准备环境 |
||||
|
SpeechEngineGenerator.PrepareEnvironment(context, null) |
||||
|
|
||||
|
// 创建引擎 |
||||
|
engine = SpeechEngineGenerator.getInstance() |
||||
|
engineHandler = engine?.createEngine() ?: -1 |
||||
|
|
||||
|
if (engineHandler == -1L) { |
||||
|
FileLogger.e(TAG, "创建引擎失败") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 设置上下文 |
||||
|
engine?.setContext(context) |
||||
|
|
||||
|
// 设置引擎类型为ASR |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, SpeechEngineDefines.ASR_ENGINE) |
||||
|
|
||||
|
// 设置日志级别 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, SpeechEngineDefines.LOG_LEVEL_WARN) |
||||
|
|
||||
|
// 设置用户ID和设备ID (使用静态值,实际项目中应替换为真实值) |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_UID_STRING, "user_id") |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, "device_id") |
||||
|
|
||||
|
// 设置鉴权信息 - 大模型版本不需要Bearer前缀 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, appId) |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, token) |
||||
|
|
||||
|
// 设置资源ID - 大模型必需 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RESOURCE_ID_STRING, resourceId) |
||||
|
|
||||
|
// 设置协议类型为Seed - 大模型必需 |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_PROTOCOL_TYPE_INT, SpeechEngineDefines.PROTOCOL_TYPE_SEED) |
||||
|
|
||||
|
// 设置网络配置 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_ADDRESS_STRING, "wss://openspeech.bytedance.com") |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_URI_STRING, "/api/v3/sauc/bigmodel") |
||||
|
|
||||
|
// 设置超时时间 |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_CONN_TIMEOUT_INT, 12000) |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_RECV_TIMEOUT_INT, 8000) |
||||
|
|
||||
|
// 设置音频来源为录音机 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RECORDER_TYPE_STRING, SpeechEngineDefines.RECORDER_TYPE_RECORDER) |
||||
|
|
||||
|
// 设置最大录音时长 (默认60秒) |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_VAD_MAX_SPEECH_DURATION_INT, 60000) |
||||
|
|
||||
|
// 设置回声消除 (用于ASR识别时不会收到TTS的声音) |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_RECORDER_PRESET_INT, SpeechEngineDefines.RECORDER_PRESET_VOICE_COMMUNICATION) |
||||
|
|
||||
|
// 初始化引擎 |
||||
|
val result = engine?.initEngine(engineHandler) |
||||
|
isInitialized = result == SpeechEngineDefines.ERR_NO_ERROR |
||||
|
|
||||
|
if (isInitialized) { |
||||
|
FileLogger.i(TAG, "引擎初始化成功") |
||||
|
|
||||
|
// 设置回调监听 |
||||
|
engine?.setListener(object : SpeechEngine.SpeechListener { |
||||
|
override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { |
||||
|
val stdData = String(data) |
||||
|
handleEngineEvent(type, stdData) |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
return true |
||||
|
} else { |
||||
|
FileLogger.e(TAG, "引擎初始化失败: $result") |
||||
|
return false |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "初始化异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置语言 |
||||
|
*/ |
||||
|
fun setLanguage(language: String): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.language = language |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_LANGUAGE_STRING, language) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置语言失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置热词 |
||||
|
*/ |
||||
|
fun setHotWords(hotWordsId: String): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
if (hotWordsId.isNotEmpty()) { |
||||
|
// 大模型ASR需要通过请求参数设置热词 |
||||
|
val reqParams = "{\"corpus\":{\"boosting_table_id\":\"$hotWordsId\"}}" |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) |
||||
|
} |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置热词失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置是否返回音量 |
||||
|
*/ |
||||
|
fun setEnableVolume(enable: Boolean): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.enableVolume = enable |
||||
|
engine?.setOptionBoolean(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENABLE_GET_VOLUME_BOOL, enable) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置音量返回失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置是否显示语音停顿、分句、分词信息 |
||||
|
*/ |
||||
|
fun setShowUtterances(show: Boolean): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.showUtterances = show |
||||
|
engine?.setOptionBoolean(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_SHOW_UTTER_BOOL, show) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置语音信息显示失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置VAD切句参数 |
||||
|
*/ |
||||
|
fun setVadParams(forceToSpeechTime: Int, endWindowSize: Int): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val reqParams = "{\"force_to_speech_time\":$forceToSpeechTime, \"end_window_size\":$endWindowSize}" |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置VAD参数失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置纠错词表 |
||||
|
*/ |
||||
|
fun setCorrectWords(correctWordsJson: String): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val reqParams = "{\"context\": \"{\\\"correct_words\\\": $correctWordsJson}\"}" |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ASR_REQ_PARAMS_STRING, reqParams) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置纠错词表失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 一次性识别 |
||||
|
*/ |
||||
|
fun recognizeOnce(callback: ASRCallback): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (isContinuousRecognitionActive) { |
||||
|
FileLogger.e(TAG, "当前正在连续识别中") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.currentAsrCallback = callback |
||||
|
|
||||
|
// 停止当前引擎 |
||||
|
engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") |
||||
|
|
||||
|
// 启动引擎开始识别 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "启动识别失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "识别异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止一次性识别 |
||||
|
*/ |
||||
|
fun stopRecognize(): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 告知引擎音频输入完成 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_FINISH_TALKING, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "停止识别失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止识别异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 开始连续识别 |
||||
|
*/ |
||||
|
fun startContinuousRecognition(callback: ASRContinuousCallback): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (isContinuousRecognitionActive) { |
||||
|
FileLogger.e(TAG, "当前已经在连续识别中") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.currentContinuousCallback = callback |
||||
|
|
||||
|
// 停止当前引擎 |
||||
|
engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") |
||||
|
|
||||
|
// 启动引擎开始识别 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "启动连续识别失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
isContinuousRecognitionActive = true |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "连续识别异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止连续识别 |
||||
|
*/ |
||||
|
fun stopContinuousRecognition(): Boolean { |
||||
|
if (!isInitialized || !isContinuousRecognitionActive) { |
||||
|
FileLogger.e(TAG, "引擎未初始化或未在连续识别中") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 停止引擎 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "停止连续识别失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
isContinuousRecognitionActive = false |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止连续识别异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 判断是否在连续识别中 |
||||
|
*/ |
||||
|
fun isContinuousRecognitionActive(): Boolean { |
||||
|
return isContinuousRecognitionActive |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 释放资源 |
||||
|
*/ |
||||
|
fun release() { |
||||
|
if (!isInitialized) { |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 停止引擎 |
||||
|
if (isContinuousRecognitionActive) { |
||||
|
stopContinuousRecognition() |
||||
|
} else { |
||||
|
engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") |
||||
|
} |
||||
|
|
||||
|
// 销毁引擎 |
||||
|
engine?.destroyEngine(engineHandler) |
||||
|
engineHandler = -1 |
||||
|
engine = null |
||||
|
isInitialized = false |
||||
|
|
||||
|
FileLogger.i(TAG, "引擎已释放") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "释放引擎异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 处理引擎事件 |
||||
|
*/ |
||||
|
private fun handleEngineEvent(type: Int, data: String) { |
||||
|
when (type) { |
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { |
||||
|
// 引擎启动成功 |
||||
|
mainHandler.post { |
||||
|
currentContinuousCallback?.onSessionStarted() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { |
||||
|
// 引擎停止 |
||||
|
isContinuousRecognitionActive = false |
||||
|
mainHandler.post { |
||||
|
currentContinuousCallback?.onSessionStopped() |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_PARTIAL_RESULT -> { |
||||
|
// 中间识别结果 |
||||
|
try { |
||||
|
val json = JSONObject(data) |
||||
|
if (!json.has("result")) { |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val resultArray = json.getJSONArray("result") |
||||
|
if (resultArray.length() > 0) { |
||||
|
val result = resultArray.getJSONObject(0) |
||||
|
val text = result.optString("text", "") |
||||
|
|
||||
|
if (text.isNotEmpty()) { |
||||
|
mainHandler.post { |
||||
|
currentContinuousCallback?.onRecognizing(text) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} catch (e: JSONException) { |
||||
|
FileLogger.e(TAG, "解析中间识别结果异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_FINAL_RESULT -> { |
||||
|
// 最终识别结果 |
||||
|
try { |
||||
|
val json = JSONObject(data) |
||||
|
if (!json.has("result")) { |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val resultArray = json.getJSONArray("result") |
||||
|
if (resultArray.length() > 0) { |
||||
|
val result = resultArray.getJSONObject(0) |
||||
|
val text = result.optString("text", "") |
||||
|
|
||||
|
if (text.isNotEmpty()) { |
||||
|
mainHandler.post { |
||||
|
if (isContinuousRecognitionActive) { |
||||
|
currentContinuousCallback?.onResult(text) |
||||
|
} else { |
||||
|
currentAsrCallback?.onSuccess(text) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} catch (e: JSONException) { |
||||
|
FileLogger.e(TAG, "解析最终识别结果异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_VOLUME_LEVEL -> { |
||||
|
// 音量回调 |
||||
|
if (enableVolume && isContinuousRecognitionActive) { |
||||
|
try { |
||||
|
val volume = (data.toFloat() * 100).toInt() |
||||
|
mainHandler.post { |
||||
|
currentContinuousCallback?.onVolumeChanged(volume) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "解析音量数据异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { |
||||
|
// 错误信息 |
||||
|
try { |
||||
|
val json = JSONObject(data) |
||||
|
val errCode = json.optInt("err_code", -1) |
||||
|
val errMsg = json.optString("err_msg", "未知错误") |
||||
|
|
||||
|
FileLogger.e(TAG, "引擎错误: $errCode, $errMsg") |
||||
|
|
||||
|
mainHandler.post { |
||||
|
if (isContinuousRecognitionActive) { |
||||
|
currentContinuousCallback?.onError(errMsg) |
||||
|
isContinuousRecognitionActive = false |
||||
|
} else { |
||||
|
currentAsrCallback?.onError(errMsg) |
||||
|
} |
||||
|
} |
||||
|
} catch (e: JSONException) { |
||||
|
FileLogger.e(TAG, "解析错误信息异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,445 @@ |
|||||
|
package com.yunqiinnovation.volcano_speech |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.os.Handler |
||||
|
import android.os.Looper |
||||
|
import androidx.annotation.NonNull |
||||
|
import com.yunqiinnovation.volcano_speech.utils.FileLogger |
||||
|
|
||||
|
import io.flutter.embedding.engine.plugins.FlutterPlugin |
||||
|
import io.flutter.plugin.common.MethodCall |
||||
|
import io.flutter.plugin.common.MethodChannel |
||||
|
import io.flutter.plugin.common.MethodChannel.MethodCallHandler |
||||
|
import io.flutter.plugin.common.MethodChannel.Result |
||||
|
import io.flutter.plugin.common.EventChannel |
||||
|
|
||||
|
/** VolcanoSpeechPlugin */ |
||||
|
class VolcanoSpeechPlugin: FlutterPlugin { |
||||
|
private val TAG = "VolcanoSpeechPlugin" |
||||
|
private lateinit var context: Context |
||||
|
private val mainHandler = Handler(Looper.getMainLooper()) |
||||
|
|
||||
|
// ASR相关 |
||||
|
private lateinit var asrChannel: MethodChannel |
||||
|
private lateinit var asrEventChannel: EventChannel |
||||
|
private var asrEventSink: EventChannel.EventSink? = null |
||||
|
|
||||
|
// TTS相关 |
||||
|
private lateinit var ttsChannel: MethodChannel |
||||
|
private lateinit var ttsEventChannel: EventChannel |
||||
|
private var ttsEventSink: EventChannel.EventSink? = null |
||||
|
|
||||
|
// TTS帮助类 |
||||
|
private lateinit var volcanoTtsHelper: VolcanoTtsHelper |
||||
|
|
||||
|
// ASR帮助类 |
||||
|
private lateinit var volcanoAsrHelper: VolcanoAsrHelper |
||||
|
|
||||
|
override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
context = flutterPluginBinding.applicationContext |
||||
|
|
||||
|
// 初始化ASR通道 |
||||
|
asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/asr") |
||||
|
asrChannel.setMethodCallHandler(AsrMethodHandler()) |
||||
|
|
||||
|
// 初始化TTS通道 |
||||
|
ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/tts") |
||||
|
ttsChannel.setMethodCallHandler(TtsMethodHandler()) |
||||
|
|
||||
|
// 初始化ASR事件通道 |
||||
|
asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/asr_events") |
||||
|
asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { |
||||
|
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { |
||||
|
asrEventSink = events |
||||
|
} |
||||
|
|
||||
|
override fun onCancel(arguments: Any?) { |
||||
|
asrEventSink = null |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
// 初始化TTS事件通道 |
||||
|
ttsEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "volcano_speech/tts_events") |
||||
|
ttsEventChannel.setStreamHandler(object : EventChannel.StreamHandler { |
||||
|
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { |
||||
|
ttsEventSink = events |
||||
|
} |
||||
|
|
||||
|
override fun onCancel(arguments: Any?) { |
||||
|
ttsEventSink = null |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
// 初始化TTS帮助类 |
||||
|
volcanoTtsHelper = VolcanoTtsHelper(context) |
||||
|
|
||||
|
// 初始化ASR帮助类 |
||||
|
volcanoAsrHelper = VolcanoAsrHelper(context) |
||||
|
} |
||||
|
|
||||
|
// 发送ASR事件 |
||||
|
private fun sendAsrEvent(event: Map<String, Any>) { |
||||
|
FileLogger.d(TAG, "发送ASR事件: $event") |
||||
|
if (asrEventSink == null) { |
||||
|
FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
mainHandler.post { |
||||
|
try { |
||||
|
asrEventSink?.success(event) |
||||
|
FileLogger.d(TAG, "ASR事件发送成功") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// 发送TTS事件 |
||||
|
private fun sendTtsEvent(event: Map<String, Any>) { |
||||
|
FileLogger.d(TAG, "发送TTS事件: $event") |
||||
|
if (ttsEventSink == null) { |
||||
|
FileLogger.w(TAG, "无法发送TTS事件:事件通道未准备好") |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
mainHandler.post { |
||||
|
try { |
||||
|
ttsEventSink?.success(event) |
||||
|
FileLogger.d(TAG, "TTS事件发送成功") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "发送TTS事件失败: ${e.message}") |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// ASR方法处理器 |
||||
|
inner class AsrMethodHandler : MethodCallHandler { |
||||
|
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
||||
|
when (call.method) { |
||||
|
"getPlatformVersion" -> { |
||||
|
result.success("Android ${android.os.Build.VERSION.RELEASE}") |
||||
|
} |
||||
|
"initialize" -> { |
||||
|
val appId = call.argument<String>("appId") ?: "" |
||||
|
val apiKey = call.argument<String>("apiKey") ?: "" |
||||
|
val resourceId = call.argument<String>("resourceId") ?: "" |
||||
|
|
||||
|
if (appId.isEmpty() || apiKey.isEmpty() || resourceId.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "appId、apiKey和resourceId不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = volcanoAsrHelper.initialize(appId, apiKey, resourceId) |
||||
|
result.success(success) |
||||
|
} |
||||
|
"setLanguage" -> { |
||||
|
val language = call.argument<String>("language") ?: "zh-CN" |
||||
|
result.success(volcanoAsrHelper.setLanguage(language)) |
||||
|
} |
||||
|
"setHotWords" -> { |
||||
|
val hotWordsId = call.argument<String>("hotWordsId") ?: "" |
||||
|
result.success(volcanoAsrHelper.setHotWords(hotWordsId)) |
||||
|
} |
||||
|
"setVadParams" -> { |
||||
|
val forceToSpeechTime = call.argument<Int>("forceToSpeechTime") ?: 0 |
||||
|
val endWindowSize = call.argument<Int>("endWindowSize") ?: 800 |
||||
|
result.success(volcanoAsrHelper.setVadParams(forceToSpeechTime, endWindowSize)) |
||||
|
} |
||||
|
"setCorrectWords" -> { |
||||
|
val correctWordsJson = call.argument<String>("correctWordsJson") ?: "{}" |
||||
|
result.success(volcanoAsrHelper.setCorrectWords(correctWordsJson)) |
||||
|
} |
||||
|
"setEnableVolume" -> { |
||||
|
val enable = call.argument<Boolean>("enable") ?: false |
||||
|
result.success(volcanoAsrHelper.setEnableVolume(enable)) |
||||
|
} |
||||
|
"setShowUtterances" -> { |
||||
|
val enable = call.argument<Boolean>("enable") ?: false |
||||
|
result.success(volcanoAsrHelper.setShowUtterances(enable)) |
||||
|
} |
||||
|
"recognizeOnce" -> { |
||||
|
// 确保当前不在连续识别中 |
||||
|
if (volcanoAsrHelper.isContinuousRecognitionActive()) { |
||||
|
result.error("ASR_BUSY", "当前正在连续识别中", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
volcanoAsrHelper.recognizeOnce(object : VolcanoAsrHelper.ASRCallback { |
||||
|
override fun onSuccess(text: String) { |
||||
|
mainHandler.post { |
||||
|
result.success(mapOf( |
||||
|
"text" to text |
||||
|
)) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
mainHandler.post { |
||||
|
result.error("ASR_ERROR", error, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"startContinuousRecognition" -> { |
||||
|
// 确保事件通道已准备好 |
||||
|
if (asrEventSink == null) { |
||||
|
result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = volcanoAsrHelper.startContinuousRecognition(object : VolcanoAsrHelper.ASRContinuousCallback { |
||||
|
override fun onResult(text: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "result", |
||||
|
"text" to text |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onRecognizing(text: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "recognizing", |
||||
|
"text" to text |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStarted() { |
||||
|
sendAsrEvent(mapOf("type" to "sessionStarted")) |
||||
|
} |
||||
|
|
||||
|
override fun onSessionStopped() { |
||||
|
sendAsrEvent(mapOf("type" to "sessionStopped")) |
||||
|
} |
||||
|
|
||||
|
override fun onVolumeChanged(volume: Int) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "volumeChanged", |
||||
|
"volume" to volume |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onError(error: String) { |
||||
|
sendAsrEvent(mapOf( |
||||
|
"type" to "error", |
||||
|
"message" to error |
||||
|
)) |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
result.success(success) |
||||
|
} |
||||
|
"stopContinuousRecognition" -> { |
||||
|
val success = volcanoAsrHelper.stopContinuousRecognition() |
||||
|
result.success(success) |
||||
|
} |
||||
|
"stopRecognize" -> { |
||||
|
val success = volcanoAsrHelper.stopRecognize() |
||||
|
result.success(success) |
||||
|
} |
||||
|
"isContinuousRecognitionActive" -> { |
||||
|
result.success(volcanoAsrHelper.isContinuousRecognitionActive()) |
||||
|
} |
||||
|
"release" -> { |
||||
|
volcanoAsrHelper.release() |
||||
|
result.success(true) |
||||
|
} |
||||
|
else -> { |
||||
|
result.notImplemented() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
// TTS方法处理器 |
||||
|
inner class TtsMethodHandler : MethodCallHandler { |
||||
|
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
||||
|
when (call.method) { |
||||
|
"getPlatformVersion" -> { |
||||
|
result.success("Android ${android.os.Build.VERSION.RELEASE}") |
||||
|
} |
||||
|
"initialize" -> { |
||||
|
val appId = call.argument<String>("appId") ?: "" |
||||
|
val token = call.argument<String>("token") ?: "" |
||||
|
val resourceId = call.argument<String>("resourceId") ?: "" |
||||
|
|
||||
|
if (appId.isEmpty() || token.isEmpty() || resourceId.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "appId、token和resourceId不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = volcanoTtsHelper.initialize(appId, token, resourceId) |
||||
|
result.success(success) |
||||
|
} |
||||
|
"setVoice" -> { |
||||
|
val voice = call.argument<String>("voice") ?: "" |
||||
|
|
||||
|
if (voice.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "voice不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
result.success(volcanoTtsHelper.setVoice(voice)) |
||||
|
} |
||||
|
"setContinuousMode" -> { |
||||
|
val isContinuous = call.argument<Boolean>("isContinuous") ?: false |
||||
|
result.success(volcanoTtsHelper.setContinuousMode(isContinuous)) |
||||
|
} |
||||
|
"speak" -> { |
||||
|
val text = call.argument<String>("text") ?: "" |
||||
|
|
||||
|
if (text.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
volcanoTtsHelper.speak(text, object : VolcanoTtsHelper.TTSCallback { |
||||
|
override fun onStart(reqId: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "start", |
||||
|
"reqId" to reqId |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onProgress(reqId: String, progress: Double) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "progress", |
||||
|
"reqId" to reqId, |
||||
|
"progress" to progress |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onComplete(reqId: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "complete", |
||||
|
"reqId" to reqId |
||||
|
)) |
||||
|
|
||||
|
mainHandler.post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(reqId: String, errorCode: Int, errorMsg: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "error", |
||||
|
"reqId" to reqId, |
||||
|
"errorCode" to errorCode, |
||||
|
"errorMsg" to errorMsg |
||||
|
)) |
||||
|
|
||||
|
mainHandler.post { |
||||
|
result.error("TTS_ERROR", errorMsg, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"synthesisNext" -> { |
||||
|
val text = call.argument<String>("text") ?: "" |
||||
|
|
||||
|
if (text.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
val success = volcanoTtsHelper.synthesisNext(text) |
||||
|
result.success(success) |
||||
|
} |
||||
|
"pause" -> { |
||||
|
result.success(volcanoTtsHelper.pause()) |
||||
|
} |
||||
|
"resume" -> { |
||||
|
result.success(volcanoTtsHelper.resume()) |
||||
|
} |
||||
|
"stop" -> { |
||||
|
result.success(volcanoTtsHelper.stop()) |
||||
|
} |
||||
|
"release" -> { |
||||
|
volcanoTtsHelper.release() |
||||
|
result.success(true) |
||||
|
} |
||||
|
// 以下方法用于与Azure Speech版本兼容 |
||||
|
"setSpeechSynthesisVoice" -> { |
||||
|
val voiceName = call.argument<String>("voiceName") ?: "" |
||||
|
|
||||
|
if (voiceName.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
result.success(volcanoTtsHelper.setVoice(voiceName)) |
||||
|
} |
||||
|
"speakText" -> { |
||||
|
val text = call.argument<String>("text") ?: "" |
||||
|
|
||||
|
if (text.isEmpty()) { |
||||
|
result.error("INVALID_ARGUMENTS", "合成文本不能为空", null) |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
volcanoTtsHelper.speak(text, object : VolcanoTtsHelper.TTSCallback { |
||||
|
override fun onStart(reqId: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "start", |
||||
|
"reqId" to reqId |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onProgress(reqId: String, progress: Double) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "progress", |
||||
|
"reqId" to reqId, |
||||
|
"progress" to progress |
||||
|
)) |
||||
|
} |
||||
|
|
||||
|
override fun onComplete(reqId: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "complete", |
||||
|
"reqId" to reqId |
||||
|
)) |
||||
|
|
||||
|
mainHandler.post { |
||||
|
result.success(true) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onError(reqId: String, errorCode: Int, errorMsg: String) { |
||||
|
sendTtsEvent(mapOf( |
||||
|
"type" to "error", |
||||
|
"reqId" to reqId, |
||||
|
"errorCode" to errorCode, |
||||
|
"errorMsg" to errorMsg |
||||
|
)) |
||||
|
|
||||
|
mainHandler.post { |
||||
|
result.error("TTS_ERROR", errorMsg, null) |
||||
|
} |
||||
|
} |
||||
|
}) |
||||
|
} |
||||
|
"stopSpeaking" -> { |
||||
|
result.success(volcanoTtsHelper.stop()) |
||||
|
} |
||||
|
"pauseSpeaking" -> { |
||||
|
result.success(volcanoTtsHelper.pause()) |
||||
|
} |
||||
|
"resumeSpeaking" -> { |
||||
|
result.success(volcanoTtsHelper.resume()) |
||||
|
} |
||||
|
else -> { |
||||
|
result.notImplemented() |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { |
||||
|
asrChannel.setMethodCallHandler(null) |
||||
|
ttsChannel.setMethodCallHandler(null) |
||||
|
asrEventChannel.setStreamHandler(null) |
||||
|
ttsEventChannel.setStreamHandler(null) |
||||
|
|
||||
|
volcanoTtsHelper.release() |
||||
|
volcanoAsrHelper.release() |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,399 @@ |
|||||
|
package com.yunqiinnovation.volcano_speech |
||||
|
|
||||
|
import android.content.Context |
||||
|
import android.os.Handler |
||||
|
import android.os.Looper |
||||
|
import com.bytedance.speech.speechengine.SpeechEngine |
||||
|
import com.bytedance.speech.speechengine.SpeechEngineDefines |
||||
|
import com.bytedance.speech.speechengine.SpeechEngineGenerator |
||||
|
import com.yunqiinnovation.volcano_speech.utils.FileLogger |
||||
|
import org.json.JSONObject |
||||
|
|
||||
|
/** |
||||
|
* 火山语音合成帮助类 (大模型版本) |
||||
|
*/ |
||||
|
class VolcanoTtsHelper(private val context: Context) { |
||||
|
private val TAG = "VolcanoTtsHelper" |
||||
|
private val mainHandler = Handler(Looper.getMainLooper()) |
||||
|
|
||||
|
// 语音引擎相关 |
||||
|
private var engine: SpeechEngine? = null |
||||
|
private var engineHandler: Long = -1 |
||||
|
private var isInitialized = false |
||||
|
|
||||
|
// 当前回调 |
||||
|
private var currentTtsCallback: TTSCallback? = null |
||||
|
|
||||
|
// 合成状态 |
||||
|
private var isPlaying = false |
||||
|
private var currentReqId = "" |
||||
|
|
||||
|
// 配置参数 |
||||
|
private var voice = "zh_female_yuxi" |
||||
|
private var ttsText = "" |
||||
|
private var isContinuous = false |
||||
|
|
||||
|
/** |
||||
|
* TTS合成回调接口 |
||||
|
*/ |
||||
|
interface TTSCallback { |
||||
|
fun onStart(reqId: String) |
||||
|
fun onProgress(reqId: String, progress: Double) |
||||
|
fun onComplete(reqId: String) |
||||
|
fun onError(reqId: String, errorCode: Int, errorMsg: String) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 初始化语音合成引擎 |
||||
|
*/ |
||||
|
fun initialize(appId: String, token: String, resourceId: String): Boolean { |
||||
|
if (isInitialized) { |
||||
|
FileLogger.i(TAG, "引擎已经初始化") |
||||
|
return true |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 准备环境 |
||||
|
SpeechEngineGenerator.PrepareEnvironment(context, null) |
||||
|
|
||||
|
// 创建引擎 |
||||
|
engine = SpeechEngineGenerator.getInstance() |
||||
|
engineHandler = engine?.createEngine() ?: -1 |
||||
|
|
||||
|
if (engineHandler == -1L) { |
||||
|
FileLogger.e(TAG, "创建引擎失败") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
// 设置上下文 |
||||
|
engine?.setContext(context) |
||||
|
|
||||
|
// 设置引擎类型为TTS |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_ENGINE_NAME_STRING, SpeechEngineDefines.TTS_ENGINE) |
||||
|
|
||||
|
// 设置日志级别 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_LOG_LEVEL_STRING, SpeechEngineDefines.LOG_LEVEL_WARN) |
||||
|
|
||||
|
// 设置用户ID和设备ID (使用静态值,实际项目中应替换为真实值) |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_UID_STRING, "user_id") |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_DEVICE_ID_STRING, "device_id") |
||||
|
|
||||
|
// 设置授权信息 - 大模型版本不需要Bearer前缀 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_ID_STRING, appId) |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_APP_TOKEN_STRING, token) |
||||
|
|
||||
|
// 设置资源ID - 大模型必需 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_RESOURCE_ID_STRING, resourceId) |
||||
|
|
||||
|
// 设置协议类型为Seed - 大模型必需 |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_PROTOCOL_TYPE_INT, SpeechEngineDefines.PROTOCOL_TYPE_SEED) |
||||
|
|
||||
|
// 设置合成策略为在线合成 |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_WORK_MODE_INT, SpeechEngineDefines.TTS_WORK_MODE_ONLINE) |
||||
|
|
||||
|
// 设置在线请求资源配置 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_ADDRESS_STRING, "wss://openspeech.bytedance.com") |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_URI_STRING, "/api/v3/sauc/bigmodel") |
||||
|
|
||||
|
// 设置发音人 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, voice) |
||||
|
|
||||
|
// 设置播放进度回调 |
||||
|
engine?.setOptionInt(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_WITH_FRONTEND_INT, 1) |
||||
|
|
||||
|
// 初始化引擎 |
||||
|
val result = engine?.initEngine(engineHandler) |
||||
|
isInitialized = result == SpeechEngineDefines.ERR_NO_ERROR |
||||
|
|
||||
|
if (isInitialized) { |
||||
|
FileLogger.i(TAG, "引擎初始化成功") |
||||
|
|
||||
|
// 设置回调监听 |
||||
|
engine?.setListener(object : SpeechEngine.SpeechListener { |
||||
|
override fun onSpeechMessage(type: Int, data: ByteArray, len: Int) { |
||||
|
val stdData = String(data) |
||||
|
handleEngineEvent(type, stdData) |
||||
|
} |
||||
|
}) |
||||
|
|
||||
|
return true |
||||
|
} else { |
||||
|
FileLogger.e(TAG, "引擎初始化失败: $result") |
||||
|
return false |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "初始化异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置发音人 |
||||
|
*/ |
||||
|
fun setVoice(voice: String): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.voice = voice |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_VOICE_ONLINE_STRING, voice) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置发音人失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 设置合成场景(单次或连续) |
||||
|
*/ |
||||
|
fun setContinuousMode(isContinuous: Boolean): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.isContinuous = isContinuous |
||||
|
val scenarioType = if (isContinuous) { |
||||
|
SpeechEngineDefines.TTS_SCENARIO_TYPE_NOVEL |
||||
|
} else { |
||||
|
SpeechEngineDefines.TTS_SCENARIO_TYPE_NORMAL |
||||
|
} |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_SCENARIO_STRING, scenarioType) |
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "设置合成场景失败: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 开始合成并播放 |
||||
|
*/ |
||||
|
fun speak(text: String, callback: TTSCallback? = null): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (text.isEmpty()) { |
||||
|
FileLogger.e(TAG, "合成文本为空") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
this.ttsText = text |
||||
|
this.currentTtsCallback = callback |
||||
|
|
||||
|
// 设置要合成的文本 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, text) |
||||
|
|
||||
|
// 停止当前引擎 |
||||
|
engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNC_STOP_ENGINE, "") |
||||
|
|
||||
|
// 启动引擎开始合成 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_START_ENGINE, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "启动合成失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "合成异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 仅用于连续合成场景:在引擎启动后添加新的文本进行合成 |
||||
|
*/ |
||||
|
fun synthesisNext(text: String): Boolean { |
||||
|
if (!isInitialized || !isContinuous) { |
||||
|
FileLogger.e(TAG, "引擎未初始化或非连续合成模式") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
if (text.isEmpty()) { |
||||
|
FileLogger.e(TAG, "合成文本为空") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 设置要合成的文本 |
||||
|
engine?.setOptionString(engineHandler, SpeechEngineDefines.PARAMS_KEY_TTS_TEXT_STRING, text) |
||||
|
|
||||
|
// 发送合成指令 |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_SYNTHESIS, "") |
||||
|
|
||||
|
if (ret != SpeechEngineDefines.ERR_NO_ERROR) { |
||||
|
FileLogger.e(TAG, "添加合成文本失败: $ret") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
return true |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "添加合成文本异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 暂停播放 |
||||
|
*/ |
||||
|
fun pause(): Boolean { |
||||
|
if (!isInitialized || !isPlaying) { |
||||
|
FileLogger.e(TAG, "引擎未初始化或未在播放") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_PAUSE_PLAYER, "") |
||||
|
return ret == SpeechEngineDefines.ERR_NO_ERROR |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "暂停播放异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 恢复播放 |
||||
|
*/ |
||||
|
fun resume(): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_RESUME_PLAYER, "") |
||||
|
return ret == SpeechEngineDefines.ERR_NO_ERROR |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "恢复播放异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 停止播放 |
||||
|
*/ |
||||
|
fun stop(): Boolean { |
||||
|
if (!isInitialized) { |
||||
|
FileLogger.e(TAG, "引擎未初始化") |
||||
|
return false |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
val ret = engine?.sendDirective(engineHandler, SpeechEngineDefines.DIRECTIVE_STOP_ENGINE, "") |
||||
|
isPlaying = false |
||||
|
return ret == SpeechEngineDefines.ERR_NO_ERROR |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "停止播放异常: ${e.message}", e) |
||||
|
return false |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 释放资源 |
||||
|
*/ |
||||
|
fun release() { |
||||
|
if (!isInitialized) { |
||||
|
return |
||||
|
} |
||||
|
|
||||
|
try { |
||||
|
// 停止引擎 |
||||
|
stop() |
||||
|
|
||||
|
// 销毁引擎 |
||||
|
engine?.destroyEngine(engineHandler) |
||||
|
engineHandler = -1 |
||||
|
engine = null |
||||
|
isInitialized = false |
||||
|
|
||||
|
FileLogger.i(TAG, "引擎已释放") |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "释放引擎异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 处理引擎事件 |
||||
|
*/ |
||||
|
private fun handleEngineEvent(type: Int, data: String) { |
||||
|
when (type) { |
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_START -> { |
||||
|
// 引擎启动成功,获取请求ID |
||||
|
currentReqId = data |
||||
|
isPlaying = true |
||||
|
|
||||
|
mainHandler.post { |
||||
|
currentTtsCallback?.onStart(currentReqId) |
||||
|
} |
||||
|
|
||||
|
if (isContinuous) { |
||||
|
// 在连续合成模式下,需要单独发送合成指令 |
||||
|
synthesisNext(ttsText) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_STOP -> { |
||||
|
// 引擎停止 |
||||
|
isPlaying = false |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_TTS_START_PLAYING -> { |
||||
|
// 开始播放 |
||||
|
FileLogger.d(TAG, "开始播放: $data") |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_TTS_FINISH_PLAYING -> { |
||||
|
// 播放结束 |
||||
|
isPlaying = false |
||||
|
|
||||
|
mainHandler.post { |
||||
|
currentTtsCallback?.onComplete(currentReqId) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_TTS_PLAYBACK_PROGRESS -> { |
||||
|
// 播放进度 |
||||
|
try { |
||||
|
val json = JSONObject(data) |
||||
|
val progress = json.optDouble("progress", 0.0) |
||||
|
val reqId = json.optString("reqid", "") |
||||
|
|
||||
|
mainHandler.post { |
||||
|
currentTtsCallback?.onProgress(reqId, progress) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "解析进度信息异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
SpeechEngineDefines.MESSAGE_TYPE_ENGINE_ERROR -> { |
||||
|
// 错误信息 |
||||
|
try { |
||||
|
val json = JSONObject(data) |
||||
|
val reqId = json.optString("reqid", "") |
||||
|
val errCode = json.optInt("err_code", -1) |
||||
|
val errMsg = json.optString("err_msg", "未知错误") |
||||
|
|
||||
|
FileLogger.e(TAG, "引擎错误: $errCode, $errMsg") |
||||
|
|
||||
|
isPlaying = false |
||||
|
|
||||
|
mainHandler.post { |
||||
|
currentTtsCallback?.onError(reqId, errCode, errMsg) |
||||
|
} |
||||
|
} catch (e: Exception) { |
||||
|
FileLogger.e(TAG, "解析错误信息异常: ${e.message}", e) |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,55 @@ |
|||||
|
package com.yunqiinnovation.volcano_speech.utils |
||||
|
|
||||
|
import android.util.Log |
||||
|
|
||||
|
/** |
||||
|
* 文件日志记录工具 |
||||
|
*/ |
||||
|
object FileLogger { |
||||
|
private const val TAG = "VolcanoSpeech" |
||||
|
private var isDebugEnabled = true |
||||
|
|
||||
|
/** |
||||
|
* 设置是否启用调试日志 |
||||
|
*/ |
||||
|
fun setDebugEnabled(enabled: Boolean) { |
||||
|
isDebugEnabled = enabled |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录调试日志 |
||||
|
*/ |
||||
|
fun d(tag: String, message: String) { |
||||
|
if (isDebugEnabled) { |
||||
|
Log.d("$TAG-$tag", message) |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录信息日志 |
||||
|
*/ |
||||
|
fun i(tag: String, message: String) { |
||||
|
Log.i("$TAG-$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录警告日志 |
||||
|
*/ |
||||
|
fun w(tag: String, message: String) { |
||||
|
Log.w("$TAG-$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录错误日志 |
||||
|
*/ |
||||
|
fun e(tag: String, message: String) { |
||||
|
Log.e("$TAG-$tag", message) |
||||
|
} |
||||
|
|
||||
|
/** |
||||
|
* 记录错误日志,带异常 |
||||
|
*/ |
||||
|
fun e(tag: String, message: String, throwable: Throwable) { |
||||
|
Log.e("$TAG-$tag", message, throwable) |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,603 @@ |
|||||
|
import 'dart:async'; |
||||
|
import 'package:flutter/services.dart'; |
||||
|
|
||||
|
/// ASR 事件类型 |
||||
|
enum AsrEventType { |
||||
|
/// 会话开始 |
||||
|
sessionStarted, |
||||
|
|
||||
|
/// 会话结束 |
||||
|
sessionStopped, |
||||
|
|
||||
|
/// 正在识别(中间结果) |
||||
|
recognizing, |
||||
|
|
||||
|
/// 识别结果(最终结果) |
||||
|
result, |
||||
|
|
||||
|
/// 音量变化 |
||||
|
volumeChanged, |
||||
|
|
||||
|
/// 错误 |
||||
|
error |
||||
|
} |
||||
|
|
||||
|
/// TTS 工作模式 |
||||
|
enum TtsWorkMode { |
||||
|
/// 在线合成 |
||||
|
online, |
||||
|
|
||||
|
/// 离线合成 |
||||
|
offline, |
||||
|
|
||||
|
/// 同时在线离线 |
||||
|
both, |
||||
|
|
||||
|
/// 先在线再离线(网络不好时自动切换) |
||||
|
alternate, |
||||
|
|
||||
|
/// 文件模式 |
||||
|
file |
||||
|
} |
||||
|
|
||||
|
/// TTS 文本类型 |
||||
|
enum TtsTextType { |
||||
|
/// 纯文本 |
||||
|
plain, |
||||
|
|
||||
|
/// SSML格式 |
||||
|
ssml |
||||
|
} |
||||
|
|
||||
|
/// 协议类型 |
||||
|
enum ProtocolType { |
||||
|
/// 默认协议 |
||||
|
defaultProtocol, |
||||
|
|
||||
|
/// Seed协议(用于大模型) |
||||
|
seed |
||||
|
} |
||||
|
|
||||
|
/// TTS 播放进度事件 |
||||
|
class TtsProgressEvent { |
||||
|
/// 播放进度 0.0-1.0 |
||||
|
final double progress; |
||||
|
|
||||
|
/// 请求ID |
||||
|
final String reqId; |
||||
|
|
||||
|
const TtsProgressEvent({ |
||||
|
required this.progress, |
||||
|
required this.reqId, |
||||
|
}); |
||||
|
|
||||
|
@override |
||||
|
String toString() { |
||||
|
return 'TtsProgressEvent{progress: $progress, reqId: $reqId}'; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// ASR 事件 |
||||
|
class AsrEvent { |
||||
|
/// 事件类型 |
||||
|
final AsrEventType type; |
||||
|
|
||||
|
/// 识别文本(仅在recognizing和result类型时有效) |
||||
|
final String? text; |
||||
|
|
||||
|
/// 识别语言(仅在recognizing和result类型时有效) |
||||
|
final String? language; |
||||
|
|
||||
|
/// 音量值(仅在volumeChanged类型时有效) |
||||
|
final int? volume; |
||||
|
|
||||
|
/// 错误信息(仅在error类型时有效) |
||||
|
final String? errorMessage; |
||||
|
|
||||
|
const AsrEvent({ |
||||
|
required this.type, |
||||
|
this.text, |
||||
|
this.language, |
||||
|
this.volume, |
||||
|
this.errorMessage, |
||||
|
}); |
||||
|
|
||||
|
factory AsrEvent.fromMap(Map<dynamic, dynamic> map) { |
||||
|
final typeStr = map['type'] as String; |
||||
|
|
||||
|
AsrEventType type; |
||||
|
switch (typeStr) { |
||||
|
case 'sessionStarted': |
||||
|
type = AsrEventType.sessionStarted; |
||||
|
break; |
||||
|
case 'sessionStopped': |
||||
|
type = AsrEventType.sessionStopped; |
||||
|
break; |
||||
|
case 'recognizing': |
||||
|
type = AsrEventType.recognizing; |
||||
|
break; |
||||
|
case 'result': |
||||
|
type = AsrEventType.result; |
||||
|
break; |
||||
|
case 'volumeChanged': |
||||
|
type = AsrEventType.volumeChanged; |
||||
|
break; |
||||
|
case 'error': |
||||
|
type = AsrEventType.error; |
||||
|
break; |
||||
|
default: |
||||
|
throw ArgumentError('未知的事件类型: $typeStr'); |
||||
|
} |
||||
|
|
||||
|
return AsrEvent( |
||||
|
type: type, |
||||
|
text: map['text'] as String?, |
||||
|
language: map['language'] as String?, |
||||
|
volume: map['volume'] as int?, |
||||
|
errorMessage: map['message'] as String?, |
||||
|
); |
||||
|
} |
||||
|
|
||||
|
@override |
||||
|
String toString() { |
||||
|
return 'AsrEvent{type: $type, text: $text, language: $language, volume: $volume, errorMessage: $errorMessage}'; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 火山引擎语音服务插件 |
||||
|
class VolcanoSpeech { |
||||
|
static final VolcanoSpeechAsr asr = VolcanoSpeechAsr._(); |
||||
|
static final VolcanoSpeechTts tts = VolcanoSpeechTts._(); |
||||
|
|
||||
|
/// 释放资源 |
||||
|
static Future<void> dispose() async { |
||||
|
await asr.dispose(); |
||||
|
await tts.dispose(); |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 火山引擎语音识别服务 |
||||
|
class VolcanoSpeechAsr { |
||||
|
static const MethodChannel _channel = MethodChannel('volcano_speech/asr'); |
||||
|
static const EventChannel _eventChannel = EventChannel('volcano_speech/asr_events'); |
||||
|
|
||||
|
/// ASR事件流控制器 |
||||
|
static final StreamController<AsrEvent> _eventStreamController = StreamController<AsrEvent>.broadcast(); |
||||
|
|
||||
|
/// ASR事件流 |
||||
|
Stream<AsrEvent> get events => _eventStreamController.stream; |
||||
|
|
||||
|
/// 是否已初始化事件监听 |
||||
|
bool _eventListenerInitialized = false; |
||||
|
|
||||
|
VolcanoSpeechAsr._() { |
||||
|
_initEventListener(); |
||||
|
} |
||||
|
|
||||
|
/// 初始化ASR事件监听 |
||||
|
void _initEventListener() { |
||||
|
if (_eventListenerInitialized) return; |
||||
|
|
||||
|
_eventChannel.receiveBroadcastStream().listen((dynamic event) { |
||||
|
if (event is Map) { |
||||
|
_eventStreamController.add(AsrEvent.fromMap(event)); |
||||
|
} |
||||
|
}); |
||||
|
|
||||
|
_eventListenerInitialized = true; |
||||
|
} |
||||
|
|
||||
|
/// 获取平台版本信息 |
||||
|
Future<String?> getPlatformVersion() async { |
||||
|
return await _channel.invokeMethod('getPlatformVersion'); |
||||
|
} |
||||
|
|
||||
|
/// 初始化语音识别引擎 |
||||
|
/// |
||||
|
/// [appId] 火山引擎AppID |
||||
|
/// [apiKey] 火山引擎ApiKey |
||||
|
/// [supportedLanguages] 支持的语言列表 |
||||
|
/// [useBigModel] 是否使用大模型识别 |
||||
|
/// [resourceId] 大模型资源ID(仅当useBigModel为true时有效) |
||||
|
Future<bool> initialize({ |
||||
|
required String appId, |
||||
|
required String apiKey, |
||||
|
List<String> supportedLanguages = const ['zh-CN'], |
||||
|
bool useBigModel = false, |
||||
|
String resourceId = '', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('initialize', { |
||||
|
'appId': appId, |
||||
|
'apiKey': apiKey, |
||||
|
'supportedLanguages': supportedLanguages, |
||||
|
'useBigModel': useBigModel, |
||||
|
'resourceId': resourceId, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置识别语言 |
||||
|
/// |
||||
|
/// [language] 语言代码,例如 zh-CN、en-US |
||||
|
Future<bool> setLanguage(String language) async { |
||||
|
return await _channel.invokeMethod('setLanguage', { |
||||
|
'language': language, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置热词 |
||||
|
/// |
||||
|
/// [hotWords] 热词JSON字符串,例如 {"hotwords":[{"word":"快速入门","scale":2.0}]} |
||||
|
Future<bool> setHotWords(String hotWords) async { |
||||
|
return await _channel.invokeMethod('setHotWords', { |
||||
|
'hotWords': hotWords, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置ASR请求参数 |
||||
|
/// |
||||
|
/// [params] 请求参数JSON字符串 |
||||
|
Future<bool> setRequestParams(String params) async { |
||||
|
return await _channel.invokeMethod('setRequestParams', { |
||||
|
'params': params, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 启用语音停顿、分句、分词信息输出 |
||||
|
/// |
||||
|
/// [enable] 是否启用 |
||||
|
Future<bool> setShowUtterances(bool enable) async { |
||||
|
return await _channel.invokeMethod('setShowUtterances', { |
||||
|
'enable': enable, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 一次性识别(直到说话结束) |
||||
|
/// |
||||
|
/// 返回识别结果文本和语言 |
||||
|
Future<Map<String, String>> recognizeOnce() async { |
||||
|
final result = await _channel.invokeMethod('recognizeOnce'); |
||||
|
return { |
||||
|
'text': result['text'] ?? '', |
||||
|
'language': result['language'] ?? 'zh-CN', |
||||
|
}; |
||||
|
} |
||||
|
|
||||
|
/// 开始连续识别 |
||||
|
/// |
||||
|
/// 通过[events]流监听识别结果 |
||||
|
Future<bool> startContinuousRecognition() async { |
||||
|
return await _channel.invokeMethod('startContinuousRecognition') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 停止连续识别 |
||||
|
Future<bool> stopContinuousRecognition() async { |
||||
|
return await _channel.invokeMethod('stopContinuousRecognition') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 检查是否正在连续识别 |
||||
|
Future<bool> isContinuousRecognitionActive() async { |
||||
|
return await _channel.invokeMethod('isContinuousRecognitionActive') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 开始长按识别(按下开始,抬起结束) |
||||
|
/// |
||||
|
/// 通过[events]流监听识别结果 |
||||
|
Future<bool> startListening() async { |
||||
|
return await _channel.invokeMethod('startListening') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 停止长按识别 |
||||
|
Future<bool> stopListening() async { |
||||
|
return await _channel.invokeMethod('stopListening') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 释放ASR资源 |
||||
|
Future<bool> dispose() async { |
||||
|
return await _channel.invokeMethod('dispose') ?? false; |
||||
|
} |
||||
|
} |
||||
|
|
||||
|
/// 火山引擎语音合成服务 |
||||
|
class VolcanoSpeechTts { |
||||
|
static const MethodChannel _channel = MethodChannel('volcano_speech/tts'); |
||||
|
static const EventChannel _eventChannel = EventChannel('volcano_speech/tts_events'); |
||||
|
|
||||
|
/// TTS进度事件流控制器 |
||||
|
static final StreamController<TtsProgressEvent> _progressStreamController = StreamController<TtsProgressEvent>.broadcast(); |
||||
|
|
||||
|
/// TTS进度事件流 |
||||
|
Stream<TtsProgressEvent> get progressEvents => _progressStreamController.stream; |
||||
|
|
||||
|
/// 是否已初始化事件监听 |
||||
|
bool _eventListenerInitialized = false; |
||||
|
|
||||
|
VolcanoSpeechTts._() { |
||||
|
_initEventListener(); |
||||
|
} |
||||
|
|
||||
|
/// 初始化TTS事件监听 |
||||
|
void _initEventListener() { |
||||
|
if (_eventListenerInitialized) return; |
||||
|
|
||||
|
_eventChannel.receiveBroadcastStream().listen((dynamic event) { |
||||
|
if (event is Map) { |
||||
|
final progress = event['progress'] as double?; |
||||
|
final reqId = event['reqId'] as String?; |
||||
|
|
||||
|
if (progress != null && reqId != null) { |
||||
|
_progressStreamController.add(TtsProgressEvent( |
||||
|
progress: progress, |
||||
|
reqId: reqId, |
||||
|
)); |
||||
|
} |
||||
|
} |
||||
|
}); |
||||
|
|
||||
|
_eventListenerInitialized = true; |
||||
|
} |
||||
|
|
||||
|
/// 获取平台版本信息 |
||||
|
Future<String?> getPlatformVersion() async { |
||||
|
return await _channel.invokeMethod('getPlatformVersion'); |
||||
|
} |
||||
|
|
||||
|
/// 启用大模型TTS |
||||
|
/// |
||||
|
/// [enable] 是否启用大模型TTS |
||||
|
/// [resourceId] 资源ID |
||||
|
Future<bool> enableBigModelTts({ |
||||
|
required bool enable, |
||||
|
String resourceId = '', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('enableBigModelTts', { |
||||
|
'enable': enable, |
||||
|
'resourceId': resourceId, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 初始化语音合成引擎 |
||||
|
/// |
||||
|
/// [appId] 火山引擎AppID |
||||
|
/// [apiKey] 火山引擎ApiKey |
||||
|
Future<bool> initialize({ |
||||
|
required String appId, |
||||
|
required String apiKey, |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('initialize', { |
||||
|
'appId': appId, |
||||
|
'apiKey': apiKey, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置音色 |
||||
|
/// |
||||
|
/// [voiceName] 音色名称 |
||||
|
/// [voiceType] 音色类型,默认 qingxin |
||||
|
Future<bool> setVoice({ |
||||
|
required String voiceName, |
||||
|
String voiceType = 'qingxin', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('setVoice', { |
||||
|
'voiceName': voiceName, |
||||
|
'voiceType': voiceType, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置离线音色 |
||||
|
/// |
||||
|
/// [voiceName] 音色名称 |
||||
|
/// [voiceType] 音色类型,默认 qingxin |
||||
|
Future<bool> setOfflineVoice({ |
||||
|
required String voiceName, |
||||
|
String voiceType = 'qingxin', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('setOfflineVoice', { |
||||
|
'voiceName': voiceName, |
||||
|
'voiceType': voiceType, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置大模型声音ID |
||||
|
/// |
||||
|
/// [voiceId] 声音ID |
||||
|
Future<bool> setBigModelVoiceId(String voiceId) async { |
||||
|
return await _channel.invokeMethod('setBigModelVoiceId', { |
||||
|
'voiceId': voiceId, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置大模型请求参数 |
||||
|
/// |
||||
|
/// [params] 请求参数,JSON字符串 |
||||
|
Future<bool> setBigModelRequestParams(String params) async { |
||||
|
return await _channel.invokeMethod('setBigModelRequestParams', { |
||||
|
'params': params, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置工作模式 |
||||
|
/// |
||||
|
/// [mode] 工作模式 |
||||
|
Future<bool> setWorkMode(TtsWorkMode mode) async { |
||||
|
String modeStr; |
||||
|
switch (mode) { |
||||
|
case TtsWorkMode.online: |
||||
|
modeStr = 'online'; |
||||
|
break; |
||||
|
case TtsWorkMode.offline: |
||||
|
modeStr = 'offline'; |
||||
|
break; |
||||
|
case TtsWorkMode.both: |
||||
|
modeStr = 'both'; |
||||
|
break; |
||||
|
case TtsWorkMode.alternate: |
||||
|
modeStr = 'alternate'; |
||||
|
break; |
||||
|
case TtsWorkMode.file: |
||||
|
modeStr = 'file'; |
||||
|
break; |
||||
|
} |
||||
|
|
||||
|
return await _channel.invokeMethod('setWorkMode', { |
||||
|
'mode': modeStr, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置文本类型 |
||||
|
/// |
||||
|
/// [type] 文本类型,plain或ssml |
||||
|
Future<bool> setTextType(TtsTextType type) async { |
||||
|
String typeStr; |
||||
|
switch (type) { |
||||
|
case TtsTextType.plain: |
||||
|
typeStr = 'plain'; |
||||
|
break; |
||||
|
case TtsTextType.ssml: |
||||
|
typeStr = 'ssml'; |
||||
|
break; |
||||
|
} |
||||
|
|
||||
|
return await _channel.invokeMethod('setTextType', { |
||||
|
'type': typeStr, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置是否启用缓存 |
||||
|
/// |
||||
|
/// [enable] 是否启用 |
||||
|
Future<bool> setEnableCache(bool enable) async { |
||||
|
return await _channel.invokeMethod('setEnableCache', { |
||||
|
'enable': enable, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置情感 |
||||
|
/// |
||||
|
/// [emotion] 情感,例如 neutral、happy、angry、sad等 |
||||
|
Future<bool> setEmotion(String emotion) async { |
||||
|
return await _channel.invokeMethod('setEmotion', { |
||||
|
'emotion': emotion, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置是否启用情感预测 |
||||
|
/// |
||||
|
/// [enable] 是否启用 |
||||
|
Future<bool> setEnableEmotionPredict(bool enable) async { |
||||
|
return await _channel.invokeMethod('setEnableEmotionPredict', { |
||||
|
'enable': enable, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置是否启用声音克隆 |
||||
|
/// |
||||
|
/// [enable] 是否启用 |
||||
|
/// [backendCluster] 后端集群 |
||||
|
Future<bool> setEnableVoiceClone({ |
||||
|
required bool enable, |
||||
|
String backendCluster = '', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('setEnableVoiceClone', { |
||||
|
'enable': enable, |
||||
|
'backendCluster': backendCluster, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置是否启用回声消除 |
||||
|
/// |
||||
|
/// [enable] 是否启用 |
||||
|
Future<bool> setEnableAEC(bool enable) async { |
||||
|
return await _channel.invokeMethod('setEnableAEC', { |
||||
|
'enable': enable, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 下载离线资源 |
||||
|
/// |
||||
|
/// [voiceTypes] 音色类型列表 |
||||
|
/// [languages] 语言列表 |
||||
|
Future<bool> downloadOfflineResource({ |
||||
|
List<String> voiceTypes = const ['qingxin'], |
||||
|
List<String> languages = const ['zh-CN'], |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('downloadOfflineResource', { |
||||
|
'voiceTypes': voiceTypes, |
||||
|
'languages': languages, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置语音参数 |
||||
|
/// |
||||
|
/// [rate] 语速 -500~500 |
||||
|
/// [volume] 音量 0~100 |
||||
|
/// [pitch] 音调 -500~500 |
||||
|
/// [silenceDuration] 静音时长,毫秒 |
||||
|
Future<bool> setSpeechParams({ |
||||
|
int rate = 0, |
||||
|
int volume = 100, |
||||
|
int pitch = 0, |
||||
|
int silenceDuration = 0, |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('setSpeechParams', { |
||||
|
'rate': rate, |
||||
|
'volume': volume, |
||||
|
'pitch': pitch, |
||||
|
'silenceDuration': silenceDuration, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 设置音频输出类型 |
||||
|
/// |
||||
|
/// [outputType] 输出类型,speaker(扬声器),earpiece(听筒),auto(自动) |
||||
|
Future<bool> setAudioOutputType(String outputType) async { |
||||
|
return await _channel.invokeMethod('setAudioOutputType', { |
||||
|
'outputType': outputType, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 合成并播放文本 |
||||
|
/// |
||||
|
/// [text] 待合成的文本 |
||||
|
Future<bool> speakText(String text) async { |
||||
|
return await _channel.invokeMethod('speakText', { |
||||
|
'text': text, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 使用大模型合成并播放文本 |
||||
|
/// |
||||
|
/// [text] 待合成的文本 |
||||
|
/// [voiceId] 声音ID |
||||
|
/// [params] 额外参数,JSON字符串 |
||||
|
Future<bool> speakWithBigModel({ |
||||
|
required String text, |
||||
|
required String voiceId, |
||||
|
String params = '', |
||||
|
}) async { |
||||
|
return await _channel.invokeMethod('speakWithBigModel', { |
||||
|
'text': text, |
||||
|
'voiceId': voiceId, |
||||
|
'params': params, |
||||
|
}) ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 暂停播放 |
||||
|
Future<bool> pausePlayback() async { |
||||
|
return await _channel.invokeMethod('pausePlayback') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 恢复播放 |
||||
|
Future<bool> resumePlayback() async { |
||||
|
return await _channel.invokeMethod('resumePlayback') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 停止播放 |
||||
|
Future<bool> stopSpeaking() async { |
||||
|
return await _channel.invokeMethod('stopSpeaking') ?? false; |
||||
|
} |
||||
|
|
||||
|
/// 释放TTS资源 |
||||
|
Future<bool> dispose() async { |
||||
|
return await _channel.invokeMethod('dispose') ?? false; |
||||
|
} |
||||
|
} |
||||
@ -0,0 +1,28 @@ |
|||||
|
name: volcano_speech |
||||
|
description: 火山引擎语音合成服务插件 |
||||
|
version: 0.0.1 |
||||
|
homepage: |
||||
|
|
||||
|
environment: |
||||
|
sdk: ">=2.17.0 <3.0.0" |
||||
|
flutter: ">=2.5.0" |
||||
|
|
||||
|
dependencies: |
||||
|
flutter: |
||||
|
sdk: flutter |
||||
|
|
||||
|
dev_dependencies: |
||||
|
flutter_test: |
||||
|
sdk: flutter |
||||
|
flutter_lints: ^2.0.0 |
||||
|
|
||||
|
# The following section is specific to Flutter packages. |
||||
|
flutter: |
||||
|
# This section identifies this Flutter project as a plugin project. |
||||
|
plugin: |
||||
|
platforms: |
||||
|
android: |
||||
|
package: com.yunqiinnovation.volcano_speech |
||||
|
pluginClass: VolcanoSpeechPlugin |
||||
|
ios: |
||||
|
pluginClass: VolcanoSpeechPlugin |
||||
@ -0,0 +1 @@ |
|||||
|
curl -v -X POST -H 'Content-Type: application/json' -H 'Authorization: Bearer 168deb3d-fd0c-4912-b9f1-aaee5c6743e6' -H 'Accept: text/event-stream' -d '{"model":"bot-20250405211523-l7c9r","messages":[{"role":"system","content":" 你是一个智能语音助手,能够简洁明了地回答用户的问题。\n时刻关心用户的情绪和需求,主动提供鼓励和温暖。\n\n语言风格活泼、亲切,能够幽默地互动,陪伴用户,缓解压力,增添生活乐趣。\n\n请始终以用户为中心,保持回应的高效性、准确性和温暖体贴,成为用户真正的灵魂伴侣。\n \n 当用户说\"退出\"、\"再见\"、\"结束对话\"等类似意图时,你应该使用exit_interaction函数来结束对话,\n 并在结束前说一句友好的告别语,例如\"再见,有需要随时找我\"。"},{"role":"system","content":" 你是一个智能语音助手,能够简洁明了地回答用户的问题。\n时刻关心用户的情绪和需求,主动提供鼓励和温暖。\n\n语言风格活泼、亲切,能够幽默地互动,陪伴用户,缓解压力,增添生活乐趣。\n\n请始终以用户为中心,保持回应的高效性、准确性和温暖体贴,成为用户真正的灵魂伴侣。\n \n 当用户说\"退出\"、\"再见\"、\"结束对话\"等类似意图时,你应该使用exit_interaction函数来结束对话,\n 并在结束前说一句友好的告别语,例如\"再见,有需要随时找我\"。"},{"role":"user","content":"退下吧。"},{"role":"assistant","content":"","tool_calls":[{"id":"call_8k680azmfc4thqrrnwpwqxah","type":"function","function":{"name":"exit_interaction","arguments":" {}"}}]},{"role":"tool","content":"{\"result\": \"已退出语音交互\"}","tool_call_id":"call_8k680azmfc4thqrrnwpwqxah"}],"temperature":0.7,"max_tokens":2000,"stream":true,"tools":[{"type":"function","function":{"name":"exit_interaction","description":"退出当前语音交互","parameters":{"type":"object","properties":{},"required":[]}}}]}' 'https://ark.cn-beijing.volces.com/api/v3/bots/chat/completions' |
||||
Loading…
Reference in new issue