35 changed files with 3148 additions and 2896 deletions
@ -1,654 +0,0 @@ |
|||
package com.yunqiinnovation.deepsound |
|||
|
|||
import android.content.Context |
|||
import android.media.AudioAttributes |
|||
import android.media.AudioFormat |
|||
import android.media.AudioRecord |
|||
import android.media.MediaRecorder |
|||
import android.media.audiofx.AcousticEchoCanceler |
|||
import android.media.audiofx.NoiseSuppressor |
|||
import android.media.audiofx.AutomaticGainControl |
|||
import android.os.Process |
|||
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
|||
import com.microsoft.cognitiveservices.speech.* |
|||
import com.microsoft.cognitiveservices.speech.audio.* |
|||
import com.microsoft.cognitiveservices.speech.util.EventHandler |
|||
import java.util.concurrent.ExecutionException |
|||
import java.util.concurrent.atomic.AtomicBoolean |
|||
// 移除WebRTC相关导入 |
|||
|
|||
class AzureAsrHelper(private val context: Context) { |
|||
private var recognizer: SpeechRecognizer? = null |
|||
private var speechConfig: SpeechConfig? = null |
|||
private val TAG = "AzureAsrHelper" |
|||
private var isContinuousRecognitionActive = false |
|||
private var currentLanguage = "zh-CN" |
|||
private var subscriptionKey = "" |
|||
private var serviceRegion = "" |
|||
private var isAutoDetectLanguage = false |
|||
private var supportedLanguages = arrayOf("zh-CN", "en-US") |
|||
|
|||
// 是否使用回音消除 - 内部控制常量 |
|||
private val useEchoCancellation = true |
|||
|
|||
// 自定义音频处理相关 |
|||
private var customAudioProcessor: CustomAudioProcessor? = null |
|||
private var pushStream: PushAudioInputStream? = null |
|||
private var audioConfig: AudioConfig? = null |
|||
|
|||
// 初始化SDK并创建recognizer |
|||
fun initialize(subscriptionKey: String, serviceRegion: String, |
|||
supportedLanguages: Array<String> = arrayOf("zh-CN", "en-US")): Boolean { |
|||
try { |
|||
FileLogger.d(TAG, "初始化 Azure 语音服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if (subscriptionKey.isEmpty() || serviceRegion.isEmpty()) { |
|||
FileLogger.e(TAG, "Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
this.subscriptionKey = subscriptionKey |
|||
this.serviceRegion = serviceRegion |
|||
|
|||
// 设置语言 |
|||
if (supportedLanguages.isNotEmpty()) { |
|||
this.supportedLanguages = supportedLanguages |
|||
} |
|||
|
|||
// 根据支持的语言数量决定是否启用自动语言检测 |
|||
this.isAutoDetectLanguage = supportedLanguages.size >= 2 |
|||
|
|||
// 如果只有一种语言,设置为当前语言 |
|||
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { |
|||
this.currentLanguage = supportedLanguages[0] |
|||
} |
|||
|
|||
// 创建语音配置 |
|||
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) |
|||
|
|||
// 设置语言配置 |
|||
if (isAutoDetectLanguage) { |
|||
// 设置自动语言检测 |
|||
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") |
|||
} else { |
|||
// 设置指定的识别语言 |
|||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|||
} |
|||
|
|||
// 创建识别器 |
|||
try { |
|||
if (useEchoCancellation) { |
|||
// 如果使用回音消除,创建自定义音频输入流 |
|||
setupCustomAudioProcessing() |
|||
|
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
} |
|||
} else { |
|||
// 使用默认麦克风输入 |
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig) |
|||
} |
|||
} |
|||
|
|||
FileLogger.d(TAG, "Azure 语音服务初始化成功") |
|||
return true |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建识别器失败: ${e.message}") |
|||
stopCustomAudioProcessing() |
|||
return false |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "初始化失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
private fun resetRecognizer(): Boolean { |
|||
try { |
|||
// 释放之前的 recognizer |
|||
recognizer?.close() |
|||
recognizer = null |
|||
|
|||
// 停止当前的音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
// 使用现有配置重新创建 recognizer |
|||
if (speechConfig != null) { |
|||
if (useEchoCancellation) { |
|||
// 如果使用回音消除,创建自定义音频输入流 |
|||
setupCustomAudioProcessing() |
|||
|
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
} |
|||
} else { |
|||
// 使用默认麦克风输入 |
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig) |
|||
} |
|||
} |
|||
return true |
|||
} else { |
|||
FileLogger.e(TAG, "语音配置未初始化") |
|||
return false |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "重置识别器失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 开始一次性语音识别 |
|||
fun recognizeOnce(callback: RecognizeCallback) { |
|||
if (speechConfig == null) { |
|||
callback.onError("语音服务未初始化") |
|||
return |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if (!resetRecognizer()) { |
|||
callback.onError("重置识别器失败") |
|||
return |
|||
} |
|||
|
|||
try { |
|||
// 启动音频处理 |
|||
startCustomAudioProcessing() |
|||
|
|||
// 执行识别 |
|||
val result = recognizer?.recognizeOnceAsync()?.get() |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
|||
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language |
|||
callback.onResult(result.text, detectedLanguage ?: "") |
|||
} else { |
|||
callback.onError("未能识别语音") |
|||
} |
|||
} catch (e: Exception) { |
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
callback.onError("识别异常: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 开始连续语音识别 |
|||
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|||
if (speechConfig == null) { |
|||
callback.onError("语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
// 如果已经在进行连续识别,先停止 |
|||
if (isContinuousRecognitionActive) { |
|||
stopContinuousRecognition(callback) |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if (!resetRecognizer()) { |
|||
callback.onError("重置识别器失败") |
|||
return false |
|||
} |
|||
|
|||
try { |
|||
// 启动音频处理 |
|||
startCustomAudioProcessing() |
|||
|
|||
// 设置识别事件处理 |
|||
// 最终识别结果 |
|||
recognizer?.recognized?.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
if (event.result.reason == ResultReason.RecognizedSpeech) { |
|||
val detectedLanguage = if (isAutoDetectLanguage) { |
|||
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|||
} else { |
|||
currentLanguage |
|||
} |
|||
// FileLogger.d(TAG, "最终识别结果: ${event.result.text}") |
|||
callback.onResult(event.result.text, detectedLanguage) |
|||
} |
|||
} |
|||
) |
|||
|
|||
// 识别中事件 |
|||
recognizer?.recognizing?.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
if (event.result.reason == ResultReason.RecognizingSpeech) { |
|||
val detectedLanguage = if (isAutoDetectLanguage) { |
|||
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|||
} else { |
|||
currentLanguage |
|||
} |
|||
// FileLogger.d(TAG, "识别中结果: ${event.result.text}") |
|||
callback.onRecognizing(event.result.text, detectedLanguage) |
|||
} |
|||
} |
|||
) |
|||
|
|||
// 会话事件 |
|||
recognizer?.sessionStarted?.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
isContinuousRecognitionActive = true |
|||
callback.onSessionStarted() |
|||
} |
|||
) |
|||
|
|||
recognizer?.sessionStopped?.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
isContinuousRecognitionActive = false |
|||
callback.onSessionStopped() |
|||
} |
|||
) |
|||
|
|||
// 取消事件 |
|||
recognizer?.canceled?.addEventListener( |
|||
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
|||
val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else "" |
|||
callback.onCanceled(event.reason.toString(), errorDetails) |
|||
isContinuousRecognitionActive = false |
|||
} |
|||
) |
|||
|
|||
// 开始连续识别 |
|||
recognizer?.startContinuousRecognitionAsync()?.get() |
|||
isContinuousRecognitionActive = true |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
callback.onError("开始连续识别失败: ${e.message}") |
|||
isContinuousRecognitionActive = false |
|||
stopCustomAudioProcessing() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 停止连续语音识别 |
|||
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|||
if (!isContinuousRecognitionActive || recognizer == null) { |
|||
return true |
|||
} |
|||
|
|||
try { |
|||
recognizer?.stopContinuousRecognitionAsync() |
|||
isContinuousRecognitionActive = false |
|||
callback.onSessionStopped() |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
callback.onError("停止连续识别失败: ${e.message}") |
|||
isContinuousRecognitionActive = false |
|||
stopCustomAudioProcessing() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 检查连续识别是否活跃 |
|||
fun isContinuousRecognitionActive(): Boolean { |
|||
return isContinuousRecognitionActive |
|||
} |
|||
|
|||
// 设置自定义音频处理 |
|||
private fun setupCustomAudioProcessing() { |
|||
if (!useEchoCancellation) { |
|||
return |
|||
} |
|||
|
|||
try { |
|||
// 1. 创建PushAudioInputStream |
|||
pushStream = PushAudioInputStream.create() |
|||
|
|||
// 2. 创建AudioConfig |
|||
audioConfig = AudioConfig.fromStreamInput(pushStream) |
|||
|
|||
// 3. 创建自定义音频处理器 |
|||
customAudioProcessor = CustomAudioProcessor(pushStream) |
|||
|
|||
FileLogger.d(TAG, "自定义音频处理设置完成") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") |
|||
releaseCustomAudioProcessing() |
|||
|
|||
// 降级处理:如果自定义处理设置失败,尝试使用默认麦克风 |
|||
try { |
|||
FileLogger.d(TAG, "尝试降级到默认麦克风输入") |
|||
audioConfig = AudioConfig.fromDefaultMicrophoneInput() |
|||
} catch (e2: Exception) { |
|||
FileLogger.e(TAG, "默认麦克风输入设置也失败: ${e2.message}") |
|||
audioConfig = null |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 启动自定义音频处理 |
|||
private fun startCustomAudioProcessing() { |
|||
if (!useEchoCancellation || customAudioProcessor == null) { |
|||
return |
|||
} |
|||
|
|||
try { |
|||
customAudioProcessor?.startRecording() |
|||
FileLogger.d(TAG, "自定义音频处理已启动") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "启动自定义音频处理失败: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 停止自定义音频处理 |
|||
private fun stopCustomAudioProcessing() { |
|||
if (!useEchoCancellation || customAudioProcessor == null) { |
|||
return |
|||
} |
|||
|
|||
try { |
|||
customAudioProcessor?.stopRecording() |
|||
FileLogger.d(TAG, "自定义音频处理已停止") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "停止自定义音频处理失败: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 释放自定义音频处理资源 |
|||
private fun releaseCustomAudioProcessing() { |
|||
stopCustomAudioProcessing() |
|||
|
|||
try { |
|||
customAudioProcessor = null |
|||
pushStream?.close() |
|||
pushStream = null |
|||
audioConfig?.close() |
|||
audioConfig = null |
|||
|
|||
FileLogger.d(TAG, "自定义音频处理资源已释放") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "释放自定义音频处理资源时出错: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 释放所有资源 |
|||
fun dispose() { |
|||
try { |
|||
// 停止和释放音频处理 |
|||
releaseCustomAudioProcessing() |
|||
|
|||
recognizer?.close() |
|||
recognizer = null |
|||
|
|||
speechConfig?.close() |
|||
speechConfig = null |
|||
|
|||
isContinuousRecognitionActive = false |
|||
|
|||
FileLogger.d(TAG, "语音识别资源已释放") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "释放资源出错: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 自定义音频处理器 - 使用Android原生回音消除 |
|||
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream?) { |
|||
private val SAMPLE_RATE = 16000 |
|||
private val CHANNEL_CONFIG = AudioFormat.CHANNEL_IN_MONO |
|||
private val AUDIO_FORMAT = AudioFormat.ENCODING_PCM_16BIT |
|||
private val BUFFER_SIZE = SAMPLE_RATE * 2 // 简化缓冲区大小计算,更稳定 |
|||
|
|||
private var audioRecord: AudioRecord? = null |
|||
private var echoCanceler: AcousticEchoCanceler? = null |
|||
private val isRecording = AtomicBoolean(false) |
|||
private var recordingThread: Thread? = null |
|||
|
|||
// 启动录音并处理音频数据 |
|||
fun startRecording() { |
|||
if (isRecording.get() || pushStream == null) { |
|||
return |
|||
} |
|||
|
|||
try { |
|||
// 使用Builder模式构建AudioFormat |
|||
val audioFormat = AudioFormat.Builder() |
|||
.setSampleRate(SAMPLE_RATE) |
|||
.setEncoding(AUDIO_FORMAT) |
|||
.setChannelMask(CHANNEL_CONFIG) |
|||
.build() |
|||
|
|||
// 使用Builder模式创建AudioRecord实例 |
|||
audioRecord = AudioRecord.Builder() |
|||
.setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION) |
|||
.setAudioFormat(audioFormat) |
|||
.setBufferSizeInBytes(BUFFER_SIZE) |
|||
.build() |
|||
|
|||
// 检查AudioRecord初始化状态 |
|||
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { |
|||
FileLogger.e(TAG, "AudioRecord初始化失败,状态: ${audioRecord?.state}") |
|||
// 尝试使用DEFAULT音频源重试一次 |
|||
audioRecord?.release() |
|||
audioRecord = AudioRecord.Builder() |
|||
.setAudioSource(MediaRecorder.AudioSource.DEFAULT) |
|||
.setAudioFormat(audioFormat) |
|||
.setBufferSizeInBytes(BUFFER_SIZE) |
|||
.build() |
|||
|
|||
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { |
|||
FileLogger.e(TAG, "AudioRecord初始化第二次尝试也失败,放弃") |
|||
releaseAudioResources() |
|||
return |
|||
} else { |
|||
FileLogger.d(TAG, "使用默认音频源成功初始化AudioRecord") |
|||
} |
|||
} |
|||
|
|||
// 启用音频效果(回音消除、噪声抑制等) |
|||
enableAudioEffects() |
|||
|
|||
// 启动录音 |
|||
audioRecord?.startRecording() |
|||
isRecording.set(true) |
|||
|
|||
// 创建录音线程 |
|||
recordingThread = Thread({ |
|||
val buffer = ByteArray(BUFFER_SIZE) |
|||
|
|||
while (isRecording.get()) { |
|||
try { |
|||
val readSize = audioRecord?.read(buffer, 0, BUFFER_SIZE) ?: 0 |
|||
|
|||
if (readSize > 0) { |
|||
try { |
|||
// 将处理后的音频数据推送到流 |
|||
if (readSize == buffer.size) { |
|||
// 如果读取的大小等于buffer的大小,直接写入整个buffer |
|||
pushStream.write(buffer) |
|||
} else { |
|||
// 如果只读取了部分数据,创建新的数组只包含有效数据 |
|||
val validData = buffer.copyOfRange(0, readSize) |
|||
pushStream.write(validData) |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "写入音频数据失败: ${e.message}") |
|||
break |
|||
} |
|||
} else if (readSize == 0) { |
|||
// 读取为0,可能是临时的,等待一下继续尝试 |
|||
Thread.sleep(10) |
|||
} else { |
|||
// 负值表示错误 |
|||
FileLogger.e(TAG, "读取音频数据失败,错误码: $readSize") |
|||
break |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "录音线程异常: ${e.message}") |
|||
break |
|||
} |
|||
} |
|||
}, "AudioRecordingThread") |
|||
|
|||
// 设置线程优先级并启动 |
|||
recordingThread?.priority = Thread.MAX_PRIORITY |
|||
recordingThread?.start() |
|||
|
|||
FileLogger.d(TAG, "音频录制已启动" + (if(echoCanceler?.enabled == true) ",回音消除已启用" else "")) |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "启动音频录制失败: ${e.message}") |
|||
releaseAudioResources() |
|||
} |
|||
} |
|||
|
|||
// 启用音频效果(回音消除、噪声抑制等) |
|||
private fun enableAudioEffects() { |
|||
try { |
|||
val audioSessionId = audioRecord?.audioSessionId ?: -1 |
|||
|
|||
if (audioSessionId != -1) { |
|||
// 启用回音消除 |
|||
if (AcousticEchoCanceler.isAvailable()) { |
|||
try { |
|||
echoCanceler = AcousticEchoCanceler.create(audioSessionId) |
|||
if (echoCanceler != null) { |
|||
echoCanceler?.enabled = true |
|||
FileLogger.d(TAG, "回音消除已启用,会话ID: $audioSessionId") |
|||
} else { |
|||
FileLogger.w(TAG, "回音消除器创建返回null") |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建回音消除器时出错: ${e.message}") |
|||
} |
|||
} else { |
|||
FileLogger.d(TAG, "设备不支持回音消除") |
|||
} |
|||
|
|||
// 以下功能暂时不启用,可根据需要取消注释 |
|||
/* |
|||
// 启用噪声抑制 |
|||
if (NoiseSuppressor.isAvailable()) { |
|||
try { |
|||
val ns = NoiseSuppressor.create(audioSessionId) |
|||
ns?.enabled = true |
|||
FileLogger.d(TAG, "噪声抑制已启用") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建噪声抑制器时出错: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 启用自动增益控制 |
|||
if (AutomaticGainControl.isAvailable()) { |
|||
try { |
|||
val agc = AutomaticGainControl.create(audioSessionId) |
|||
agc?.enabled = true |
|||
FileLogger.d(TAG, "自动增益控制已启用") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建自动增益控制时出错: ${e.message}") |
|||
} |
|||
} |
|||
*/ |
|||
} else { |
|||
FileLogger.w(TAG, "无效的音频会话ID,无法启用音频效果") |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "启用音频效果时出错: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 停止录音 |
|||
fun stopRecording() { |
|||
if (!isRecording.get()) { |
|||
return |
|||
} |
|||
|
|||
isRecording.set(false) |
|||
|
|||
try { |
|||
// 等待录音线程结束 |
|||
recordingThread?.join(1000) |
|||
|
|||
// 释放资源 |
|||
releaseAudioResources() |
|||
|
|||
FileLogger.d(TAG, "音频录制已停止") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "停止音频录制失败: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 释放音频资源 |
|||
private fun releaseAudioResources() { |
|||
try { |
|||
// 停止录音 |
|||
try { |
|||
if (audioRecord?.state == AudioRecord.STATE_INITIALIZED) { |
|||
audioRecord?.stop() |
|||
} |
|||
} catch (e: Exception) { |
|||
// 忽略可能的IllegalStateException |
|||
FileLogger.w(TAG, "停止AudioRecord时出错: ${e.message}") |
|||
} |
|||
|
|||
// 释放回音消除器 |
|||
try { |
|||
if (echoCanceler != null) { |
|||
echoCanceler?.enabled = false |
|||
echoCanceler?.release() |
|||
echoCanceler = null |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.w(TAG, "释放回音消除器时出错: ${e.message}") |
|||
} finally { |
|||
echoCanceler = null |
|||
} |
|||
|
|||
// 释放音频记录器 |
|||
try { |
|||
audioRecord?.release() |
|||
} catch (e: Exception) { |
|||
FileLogger.w(TAG, "释放AudioRecord时出错: ${e.message}") |
|||
} finally { |
|||
audioRecord = null |
|||
} |
|||
|
|||
// 重置线程 |
|||
recordingThread = null |
|||
|
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "释放音频资源失败: ${e.message}") |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 一次性识别回调接口 |
|||
interface RecognizeCallback { |
|||
fun onResult(result: String, detectedLanguage: String = "") |
|||
fun onError(error: String) |
|||
} |
|||
|
|||
// 连续识别回调接口 |
|||
interface ContinuousRecognizeCallback { |
|||
fun onResult(result: String, detectedLanguage: String = "") |
|||
fun onRecognizing(recognizing: String, detectedLanguage: String = "") |
|||
fun onSessionStarted() |
|||
fun onSessionStopped() |
|||
fun onCanceled(reason: String, errorDetails: String) |
|||
fun onError(error: String) |
|||
} |
|||
} |
|||
@ -1,344 +0,0 @@ |
|||
package com.yunqiinnovation.deepsound |
|||
|
|||
import android.util.Log |
|||
import okhttp3.* |
|||
import okhttp3.MediaType.Companion.toMediaTypeOrNull |
|||
import okhttp3.RequestBody.Companion.toRequestBody |
|||
import org.json.JSONArray |
|||
import org.json.JSONObject |
|||
import java.io.IOException |
|||
import java.util.concurrent.CountDownLatch |
|||
import java.util.concurrent.TimeUnit |
|||
|
|||
/** |
|||
* 火山AI服务的原生实现 |
|||
* |
|||
* 参考Flutter端的VolcanoAIService实现,提供同步和异步的API调用方式 |
|||
*/ |
|||
class VolcanoAIService() { |
|||
private val TAG = "VolcanoAIService" |
|||
private val baseUrl = "https://ark.cn-beijing.volces.com/api/v3" |
|||
private val chatEndpoint = "/chat/completions" |
|||
private val client = OkHttpClient.Builder() |
|||
.connectTimeout(30, TimeUnit.SECONDS) |
|||
.readTimeout(30, TimeUnit.SECONDS) |
|||
.writeTimeout(30, TimeUnit.SECONDS) |
|||
.build() |
|||
|
|||
private var apiKey: String = "" |
|||
private var isInitialized = false |
|||
|
|||
/** |
|||
* 初始化火山AI服务 |
|||
* |
|||
* @param apiKey 火山AI API密钥 |
|||
* @return 初始化是否成功 |
|||
*/ |
|||
fun initialize(apiKey: String): Boolean { |
|||
this.apiKey = apiKey |
|||
isInitialized = apiKey.isNotEmpty() |
|||
|
|||
if (!isInitialized) { |
|||
Log.e(TAG, "初始化失败:API key 不能为空") |
|||
} else { |
|||
Log.d(TAG, "火山AI服务初始化成功") |
|||
} |
|||
|
|||
return isInitialized |
|||
} |
|||
|
|||
/** |
|||
* 生成个性化问候语 |
|||
* |
|||
* @param agentName 代理名称 |
|||
* @param systemPrompt 系统提示词 |
|||
* @param callback 回调函数,返回生成的问候语 |
|||
*/ |
|||
fun generateGreeting(agentName: String, systemPrompt: String, callback: (String?, Exception?) -> Unit) { |
|||
val messages = JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
put(JSONObject().apply { |
|||
put("role", "user") |
|||
put("content", "请用一句简短的话向我打个招呼,要符合你的身份和性格特点,不要超过18个字。") |
|||
}) |
|||
} |
|||
|
|||
sendMessageStream(messages, systemPrompt, object : StreamCallback { |
|||
val stringBuilder = StringBuilder() |
|||
|
|||
override fun onToken(token: String) { |
|||
stringBuilder.append(token) |
|||
} |
|||
|
|||
override fun onComplete() { |
|||
callback(stringBuilder.toString(), null) |
|||
} |
|||
|
|||
override fun onError(e: Exception) { |
|||
callback(null, e) |
|||
} |
|||
}) |
|||
} |
|||
|
|||
/** |
|||
* 发送消息(非流式输出) |
|||
* |
|||
* @param messages 消息列表 |
|||
* @param systemPrompt 系统提示词 |
|||
* @return 返回AI的回复 |
|||
* @throws VolcanoAIException 如果API调用失败 |
|||
*/ |
|||
@Throws(VolcanoAIException::class) |
|||
fun sendMessage(messages: JSONArray, systemPrompt: String): String { |
|||
// 检查是否已初始化 |
|||
if (!isInitialized || apiKey.isEmpty()) { |
|||
throw VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法") |
|||
} |
|||
|
|||
val fullMessages = JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
for (i in 0 until messages.length()) { |
|||
put(messages.getJSONObject(i)) |
|||
} |
|||
} |
|||
|
|||
val requestBody = JSONObject().apply { |
|||
put("model", "doubao-1-5-lite-32k-250115") |
|||
put("messages", fullMessages) |
|||
put("temperature", 0.7) |
|||
put("max_tokens", 2000) |
|||
put("stream", false) |
|||
} |
|||
|
|||
val mediaType = "application/json".toMediaTypeOrNull() |
|||
val request = Request.Builder() |
|||
.url("$baseUrl$chatEndpoint") |
|||
.addHeader("Content-Type", "application/json") |
|||
.addHeader("Authorization", "Bearer $apiKey") |
|||
.post(requestBody.toString().toRequestBody(mediaType)) |
|||
.build() |
|||
|
|||
try { |
|||
client.newCall(request).execute().use { response -> |
|||
if (!response.isSuccessful) { |
|||
val errorBody = response.body?.string() ?: "" |
|||
val errorMessage = try { |
|||
JSONObject(errorBody).getJSONObject("error").getString("message") |
|||
} catch (e: Exception) { |
|||
"Unknown error occurred" |
|||
} |
|||
throw VolcanoAIException(errorMessage) |
|||
} |
|||
|
|||
val responseBody = response.body?.string() ?: throw VolcanoAIException("Empty response") |
|||
val jsonResponse = JSONObject(responseBody) |
|||
|
|||
if (jsonResponse.has("choices") && |
|||
jsonResponse.getJSONArray("choices").length() > 0 && |
|||
jsonResponse.getJSONArray("choices").getJSONObject(0).has("message")) { |
|||
return jsonResponse.getJSONArray("choices") |
|||
.getJSONObject(0) |
|||
.getJSONObject("message") |
|||
.getString("content") |
|||
} |
|||
|
|||
throw VolcanoAIException("Invalid response format") |
|||
} |
|||
} catch (e: Exception) { |
|||
if (e is VolcanoAIException) throw e |
|||
throw VolcanoAIException("Failed to communicate with AI service: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 发送消息(流式输出) |
|||
* |
|||
* @param messages 消息列表 |
|||
* @param systemPrompt 系统提示词 |
|||
* @param callback 回调函数,用于接收流式输出的结果 |
|||
*/ |
|||
fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { |
|||
// 检查是否已初始化 |
|||
if (!isInitialized || apiKey.isEmpty()) { |
|||
callback.onError(VolcanoAIException("火山AI服务未初始化或API key为空,请先调用initialize方法")) |
|||
return |
|||
} |
|||
|
|||
val fullMessages = JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
for (i in 0 until messages.length()) { |
|||
put(messages.getJSONObject(i)) |
|||
} |
|||
} |
|||
|
|||
val requestBody = JSONObject().apply { |
|||
put("model", "doubao-1-5-lite-32k-250115") |
|||
put("messages", fullMessages) |
|||
put("temperature", 0.7) |
|||
put("max_tokens", 2000) |
|||
put("stream", true) |
|||
} |
|||
|
|||
val mediaType = "application/json".toMediaTypeOrNull() |
|||
val request = Request.Builder() |
|||
.url("$baseUrl$chatEndpoint") |
|||
.addHeader("Content-Type", "application/json") |
|||
.addHeader("Authorization", "Bearer $apiKey") |
|||
.addHeader("Accept", "text/event-stream") |
|||
.post(requestBody.toString().toRequestBody(mediaType)) |
|||
.build() |
|||
|
|||
client.newCall(request).enqueue(object : Callback { |
|||
override fun onFailure(call: Call, e: IOException) { |
|||
callback.onError(VolcanoAIException("Failed to communicate with AI service: ${e.message}")) |
|||
} |
|||
|
|||
override fun onResponse(call: Call, response: Response) { |
|||
if (!response.isSuccessful) { |
|||
val errorBody = response.body?.string() ?: "" |
|||
val errorMessage = try { |
|||
JSONObject(errorBody).getJSONObject("error").getString("message") |
|||
} catch (e: Exception) { |
|||
"Unknown error occurred" |
|||
} |
|||
callback.onError(VolcanoAIException(errorMessage)) |
|||
return |
|||
} |
|||
|
|||
val responseBody = response.body ?: return |
|||
val source = responseBody.source() |
|||
val bufferedSource = source.buffer |
|||
|
|||
try { |
|||
while (!bufferedSource.exhausted()) { |
|||
val line = bufferedSource.readUtf8Line() ?: continue |
|||
|
|||
if (line.isEmpty()) continue |
|||
if (line.startsWith("data: ")) { |
|||
val data = line.substring(6) |
|||
if (data == "[DONE]") { |
|||
callback.onComplete() |
|||
break |
|||
} |
|||
|
|||
try { |
|||
val jsonData = JSONObject(data) |
|||
if (jsonData.has("choices") && |
|||
jsonData.getJSONArray("choices").length() > 0 && |
|||
jsonData.getJSONArray("choices").getJSONObject(0).has("delta") && |
|||
jsonData.getJSONArray("choices").getJSONObject(0).getJSONObject("delta").has("content")) { |
|||
val content = jsonData.getJSONArray("choices") |
|||
.getJSONObject(0) |
|||
.getJSONObject("delta") |
|||
.getString("content") |
|||
callback.onToken(content) |
|||
} |
|||
} catch (e: Exception) { |
|||
// 忽略无效的JSON数据 |
|||
continue |
|||
} |
|||
} |
|||
} |
|||
} catch (e: Exception) { |
|||
callback.onError(VolcanoAIException("Error processing stream: ${e.message}")) |
|||
} finally { |
|||
response.close() |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
|
|||
/** |
|||
* 同步方式发送消息(流式输出) |
|||
* |
|||
* 注意:此方法会阻塞当前线程,请在后台线程中调用 |
|||
* |
|||
* @param messages 消息列表 |
|||
* @param systemPrompt 系统提示词 |
|||
* @return 返回完整的AI回复 |
|||
* @throws VolcanoAIException 如果API调用失败 |
|||
*/ |
|||
@Throws(VolcanoAIException::class) |
|||
fun sendMessageStreamSync(messages: JSONArray, systemPrompt: String): String { |
|||
val result = StringBuilder() |
|||
val latch = CountDownLatch(1) |
|||
var exception: Exception? = null |
|||
|
|||
sendMessageStream(messages, systemPrompt, object : StreamCallback { |
|||
override fun onToken(token: String) { |
|||
result.append(token) |
|||
} |
|||
|
|||
override fun onComplete() { |
|||
latch.countDown() |
|||
} |
|||
|
|||
override fun onError(e: Exception) { |
|||
exception = e |
|||
latch.countDown() |
|||
} |
|||
}) |
|||
|
|||
// 等待流式输出完成或出错 |
|||
latch.await(60, TimeUnit.SECONDS) |
|||
|
|||
if (exception != null) { |
|||
throw exception as VolcanoAIException |
|||
} |
|||
|
|||
return result.toString() |
|||
} |
|||
|
|||
/** |
|||
* 创建用户消息 |
|||
*/ |
|||
fun createUserMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "user") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 创建系统消息 |
|||
*/ |
|||
fun createSystemMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 创建助手消息 |
|||
*/ |
|||
fun createAssistantMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "assistant") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 流式输出回调接口 |
|||
*/ |
|||
interface StreamCallback { |
|||
fun onToken(token: String) |
|||
fun onComplete() |
|||
fun onError(e: Exception) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 火山AI异常 |
|||
*/ |
|||
class VolcanoAIException(message: String) : Exception(message) |
|||
@ -0,0 +1,606 @@ |
|||
package com.yunqiinnovation.deepsound |
|||
|
|||
import android.util.Log |
|||
import okhttp3.* |
|||
import okhttp3.MediaType.Companion.toMediaTypeOrNull |
|||
import okhttp3.RequestBody.Companion.toRequestBody |
|||
import org.json.JSONArray |
|||
import org.json.JSONObject |
|||
import java.io.IOException |
|||
import java.util.concurrent.TimeUnit |
|||
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
|||
|
|||
/** |
|||
* OpenAI服务的原生实现 |
|||
*/ |
|||
class OpenAIService() { |
|||
private val TAG = "OpenAIService" |
|||
private var baseUrl = "https://api.openai.com/v1/chat/completions" |
|||
private val client = OkHttpClient.Builder() |
|||
.connectTimeout(30, TimeUnit.SECONDS) |
|||
.readTimeout(30, TimeUnit.SECONDS) |
|||
.writeTimeout(30, TimeUnit.SECONDS) |
|||
.build() |
|||
|
|||
private var apiKey: String = "" |
|||
private var isInitialized = false |
|||
private var model: String = "doubao-1-5-lite-32k-250115" // 默认模型 |
|||
|
|||
// 用于存储注册的函数 |
|||
private val registeredFunctions = mutableListOf<JSONObject>() |
|||
|
|||
|
|||
/** |
|||
* 初始化OpenAI服务 |
|||
*/ |
|||
fun initialize(apiKey: String, baseUrl: String = ""): Boolean { |
|||
this.apiKey = apiKey |
|||
if (baseUrl.isNotEmpty()) { |
|||
this.baseUrl = baseUrl |
|||
} |
|||
isInitialized = apiKey.isNotEmpty() |
|||
return isInitialized |
|||
} |
|||
|
|||
/** |
|||
* 注册函数 |
|||
*/ |
|||
fun registerFunction(name: String, description: String, parameters: JSONObject): Boolean { |
|||
try { |
|||
val function = JSONObject().apply { |
|||
put("name", name) |
|||
put("description", description) |
|||
put("parameters", parameters) |
|||
} |
|||
|
|||
// 检查是否已存在相同名称的函数 |
|||
val existingIndex = registeredFunctions.indexOfFirst { |
|||
it.getString("name") == name |
|||
} |
|||
|
|||
if (existingIndex >= 0) { |
|||
// 如果已存在,则替换 |
|||
registeredFunctions[existingIndex] = function |
|||
} else { |
|||
// 如果不存在,则添加 |
|||
registeredFunctions.add(function) |
|||
} |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 发送消息(非流式输出) |
|||
*/ |
|||
@Throws(OpenAIException::class) |
|||
fun sendMessage(messages: JSONArray, systemPrompt: String): String { |
|||
if (!isInitialized || apiKey.isEmpty()) { |
|||
throw OpenAIException("OpenAI服务未初始化") |
|||
} |
|||
|
|||
val fullMessages = JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
for (i in 0 until messages.length()) { |
|||
put(messages.getJSONObject(i)) |
|||
} |
|||
} |
|||
|
|||
val requestBody = JSONObject().apply { |
|||
put("model", model) |
|||
put("messages", fullMessages) |
|||
put("temperature", 0.7) |
|||
put("max_tokens", 2000) |
|||
put("stream", false) |
|||
|
|||
// 如果有注册的函数,则添加到请求中 |
|||
if (registeredFunctions.isNotEmpty()) { |
|||
val tools = JSONArray() |
|||
for (function in registeredFunctions) { |
|||
val tool = JSONObject().apply { |
|||
put("type", "function") |
|||
put("function", function) |
|||
} |
|||
tools.put(tool) |
|||
} |
|||
put("tools", tools) |
|||
} |
|||
} |
|||
|
|||
val mediaType = "application/json".toMediaTypeOrNull() |
|||
val request = Request.Builder() |
|||
.url(baseUrl) |
|||
.addHeader("Content-Type", "application/json") |
|||
.addHeader("Authorization", "Bearer $apiKey") |
|||
.post(requestBody.toString().toRequestBody(mediaType)) |
|||
.build() |
|||
|
|||
try { |
|||
client.newCall(request).execute().use { response -> |
|||
if (!response.isSuccessful) { |
|||
throw OpenAIException("API调用失败: ${response.code}") |
|||
} |
|||
|
|||
val responseBody = response.body?.string() ?: throw OpenAIException("Empty response") |
|||
val jsonResponse = JSONObject(responseBody) |
|||
|
|||
// 检查是否有函数调用 |
|||
if (jsonResponse.has("choices") && |
|||
jsonResponse.getJSONArray("choices").length() > 0) { |
|||
|
|||
val choice = jsonResponse.getJSONArray("choices").getJSONObject(0) |
|||
|
|||
// 检查是否是函数调用 |
|||
if (choice.has("message")) { |
|||
val message = choice.getJSONObject("message") |
|||
|
|||
// 检查是否有工具调用 |
|||
if (message.has("tool_calls")) { |
|||
val toolCalls = message.getJSONArray("tool_calls") |
|||
if (toolCalls.length() > 0) { |
|||
val toolCall = toolCalls.getJSONObject(0) |
|||
if (toolCall.has("function")) { |
|||
val function = toolCall.getJSONObject("function") |
|||
val functionCall = JSONObject().apply { |
|||
put("name", function.getString("name")) |
|||
put("arguments", function.getString("arguments")) |
|||
put("id", toolCall.getString("id")) |
|||
} |
|||
return functionCall.toString() |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 如果没有工具调用,返回消息内容 |
|||
if (message.has("content")) { |
|||
return message.getString("content") |
|||
} |
|||
} |
|||
} |
|||
|
|||
throw OpenAIException("Invalid response format") |
|||
} |
|||
} catch (e: Exception) { |
|||
if (e is OpenAIException) throw e |
|||
throw OpenAIException("Failed to communicate with AI service: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 发送消息(流式输出) |
|||
*/ |
|||
fun sendMessageStream(messages: JSONArray, systemPrompt: String, callback: StreamCallback) { |
|||
if (!isInitialized || apiKey.isEmpty()) { |
|||
callback.onError(OpenAIException("OpenAI服务未初始化")) |
|||
return |
|||
} |
|||
|
|||
val fullMessages = JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
for (i in 0 until messages.length()) { |
|||
put(messages.getJSONObject(i)) |
|||
} |
|||
} |
|||
|
|||
val requestBody = JSONObject().apply { |
|||
put("model", model) |
|||
put("messages", fullMessages) |
|||
put("temperature", 0.7) |
|||
put("max_tokens", 2000) |
|||
put("stream", true) |
|||
|
|||
// 如果有注册的函数,则添加到请求中 |
|||
if (registeredFunctions.isNotEmpty()) { |
|||
val tools = JSONArray() |
|||
for (function in registeredFunctions) { |
|||
val tool = JSONObject().apply { |
|||
put("type", "function") |
|||
put("function", function) |
|||
} |
|||
tools.put(tool) |
|||
} |
|||
put("tools", tools) |
|||
} |
|||
} |
|||
|
|||
val mediaType = "application/json".toMediaTypeOrNull() |
|||
val request = Request.Builder() |
|||
.url(baseUrl) |
|||
.addHeader("Content-Type", "application/json") |
|||
.addHeader("Authorization", "Bearer $apiKey") |
|||
.addHeader("Accept", "text/event-stream") |
|||
.post(requestBody.toString().toRequestBody(mediaType)) |
|||
.build() |
|||
|
|||
client.newCall(request).enqueue(object : Callback { |
|||
override fun onFailure(call: Call, e: IOException) { |
|||
callback.onError(OpenAIException(e.message ?: "请求失败")) |
|||
} |
|||
|
|||
override fun onResponse(call: Call, response: Response) { |
|||
if (!response.isSuccessful) { |
|||
callback.onError(OpenAIException("API调用失败: ${response.code}")) |
|||
return |
|||
} |
|||
|
|||
val responseBody = response.body ?: return |
|||
val source = responseBody.source() |
|||
|
|||
try { |
|||
// 预取数据到缓冲区 |
|||
source.request(Long.MAX_VALUE) |
|||
val bufferedSource = source.buffer |
|||
|
|||
// 用于存储函数调用的各个部分 |
|||
val finalToolCalls = mutableMapOf<Int, ToolCallInfo>() |
|||
|
|||
while (!bufferedSource.exhausted()) { |
|||
val line = bufferedSource.readUtf8Line() ?: continue |
|||
|
|||
if (line.isEmpty()) continue |
|||
if (line.startsWith("data: ")) { |
|||
val data = line.substring(6) |
|||
if (data == "[DONE]") { |
|||
callback.onComplete() |
|||
break |
|||
} |
|||
|
|||
try { |
|||
val jsonData = JSONObject(data) |
|||
if (jsonData.has("choices") && |
|||
jsonData.getJSONArray("choices").length() > 0) { |
|||
|
|||
val choice = jsonData.getJSONArray("choices").getJSONObject(0) |
|||
|
|||
// 检查是否有delta |
|||
if (choice.has("delta")) { |
|||
val delta = choice.getJSONObject("delta") |
|||
|
|||
// 检查是否有工具调用 |
|||
if (delta.has("tool_calls")) { |
|||
val toolCalls = delta.getJSONArray("tool_calls") |
|||
for (i in 0 until toolCalls.length()) { |
|||
val toolCall = toolCalls.getJSONObject(i) |
|||
val index = toolCall.optInt("index", i) |
|||
|
|||
// 如果是新的工具调用,初始化 |
|||
if (!finalToolCalls.containsKey(index)) { |
|||
finalToolCalls[index] = ToolCallInfo() |
|||
} |
|||
|
|||
// 获取ID |
|||
if (toolCall.has("id")) { |
|||
finalToolCalls[index]?.id = toolCall.getString("id") |
|||
} |
|||
|
|||
// 处理函数信息 |
|||
if (toolCall.has("function")) { |
|||
val function = toolCall.getJSONObject("function") |
|||
|
|||
if (function.has("name")) { |
|||
finalToolCalls[index]?.name = function.getString("name") |
|||
} |
|||
|
|||
if (function.has("arguments")) { |
|||
finalToolCalls[index]?.arguments += function.getString("arguments") |
|||
} |
|||
} |
|||
} |
|||
continue |
|||
} |
|||
|
|||
// 如果有内容,发送给回调 |
|||
if (delta.has("content") && !delta.isNull("content")) { |
|||
val content = delta.getString("content") |
|||
callback.onToken(content) |
|||
} |
|||
} |
|||
} |
|||
} catch (e: Exception) { |
|||
// 忽略无效的JSON |
|||
continue |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 处理完整的函数调用 |
|||
for ((_, toolCallInfo) in finalToolCalls) { |
|||
if (toolCallInfo.name.isNotEmpty()) { |
|||
try { |
|||
// 创建函数调用对象 |
|||
val functionCall = JSONObject().apply { |
|||
put("id", toolCallInfo.id) |
|||
put("name", toolCallInfo.name) |
|||
put("arguments", toolCallInfo.arguments.trim()) |
|||
} |
|||
|
|||
callback.onFunctionCall(functionCall) |
|||
} catch (e: Exception) { |
|||
// 出错时使用空参数 |
|||
val functionCall = JSONObject().apply { |
|||
put("id", toolCallInfo.id) |
|||
put("name", toolCallInfo.name) |
|||
put("arguments", "{}") |
|||
} |
|||
callback.onFunctionCall(functionCall) |
|||
} |
|||
} |
|||
} |
|||
} catch (e: Exception) { |
|||
callback.onError(OpenAIException("处理流式响应出错: ${e.message}")) |
|||
} finally { |
|||
response.close() |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
|
|||
/** |
|||
* 发送函数调用结果 |
|||
*/ |
|||
fun sendFunctionCallResult( |
|||
messages: JSONArray, |
|||
systemPrompt: String, |
|||
functionCall: JSONObject, |
|||
functionResult: String, |
|||
callback: StreamCallback |
|||
) { |
|||
if (!isInitialized || apiKey.isEmpty()) { |
|||
callback.onError(OpenAIException("OpenAI服务未初始化")) |
|||
return |
|||
} |
|||
|
|||
// 构建完整的消息历史 |
|||
val fullMessages = JSONArray().apply { |
|||
// 添加系统提示 |
|||
put(JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", systemPrompt) |
|||
}) |
|||
|
|||
// 添加历史消息 |
|||
for (i in 0 until messages.length()) { |
|||
put(messages.getJSONObject(i)) |
|||
} |
|||
|
|||
// 添加函数调用信息 |
|||
put(JSONObject().apply { |
|||
put("role", "assistant") |
|||
put("content", null) |
|||
put("tool_calls", JSONArray().apply { |
|||
put(JSONObject().apply { |
|||
put("id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) |
|||
put("type", "function") |
|||
put("function", JSONObject().apply { |
|||
put("name", functionCall.getString("name")) |
|||
put("arguments", functionCall.getString("arguments")) |
|||
}) |
|||
}) |
|||
}) |
|||
}) |
|||
|
|||
// 添加函数返回结果 |
|||
put(JSONObject().apply { |
|||
put("role", "tool") |
|||
put("tool_call_id", functionCall.optString("id", "call_${System.currentTimeMillis()}")) |
|||
put("content", functionResult) |
|||
}) |
|||
} |
|||
|
|||
// 构建请求 |
|||
val requestBody = JSONObject().apply { |
|||
put("model", model) |
|||
put("messages", fullMessages) |
|||
put("temperature", 0.7) |
|||
put("max_tokens", 2000) |
|||
put("stream", true) |
|||
|
|||
// 如果有注册的函数,则添加到请求中 |
|||
if (registeredFunctions.isNotEmpty()) { |
|||
val tools = JSONArray() |
|||
for (function in registeredFunctions) { |
|||
val tool = JSONObject().apply { |
|||
put("type", "function") |
|||
put("function", function) |
|||
} |
|||
tools.put(tool) |
|||
} |
|||
put("tools", tools) |
|||
} |
|||
} |
|||
|
|||
val mediaType = "application/json".toMediaTypeOrNull() |
|||
val request = Request.Builder() |
|||
.url(baseUrl) |
|||
.addHeader("Content-Type", "application/json") |
|||
.addHeader("Authorization", "Bearer $apiKey") |
|||
.addHeader("Accept", "text/event-stream") |
|||
.post(requestBody.toString().toRequestBody(mediaType)) |
|||
.build() |
|||
|
|||
// 发送请求 |
|||
client.newCall(request).enqueue(object : Callback { |
|||
override fun onFailure(call: Call, e: IOException) { |
|||
callback.onError(OpenAIException(e.message ?: "请求失败")) |
|||
} |
|||
|
|||
override fun onResponse(call: Call, response: Response) { |
|||
if (!response.isSuccessful) { |
|||
callback.onError(OpenAIException("API调用失败: ${response.code}")) |
|||
return |
|||
} |
|||
|
|||
val responseBody = response.body ?: return |
|||
val source = responseBody.source() |
|||
|
|||
try { |
|||
// 预取数据到缓冲区 |
|||
source.request(Long.MAX_VALUE) |
|||
val bufferedSource = source.buffer |
|||
|
|||
// 用于存储函数调用的各个部分 |
|||
val finalToolCalls = mutableMapOf<Int, ToolCallInfo>() |
|||
|
|||
while (!bufferedSource.exhausted()) { |
|||
val line = bufferedSource.readUtf8Line() ?: continue |
|||
|
|||
if (line.isEmpty()) continue |
|||
if (line.startsWith("data: ")) { |
|||
val data = line.substring(6) |
|||
if (data == "[DONE]") { |
|||
callback.onComplete() |
|||
break |
|||
} |
|||
|
|||
try { |
|||
val jsonData = JSONObject(data) |
|||
if (jsonData.has("choices") && |
|||
jsonData.getJSONArray("choices").length() > 0) { |
|||
|
|||
val choice = jsonData.getJSONArray("choices").getJSONObject(0) |
|||
|
|||
// 检查是否有delta |
|||
if (choice.has("delta")) { |
|||
val delta = choice.getJSONObject("delta") |
|||
|
|||
// 检查是否有工具调用 |
|||
if (delta.has("tool_calls")) { |
|||
val toolCalls = delta.getJSONArray("tool_calls") |
|||
for (i in 0 until toolCalls.length()) { |
|||
val toolCall = toolCalls.getJSONObject(i) |
|||
val index = toolCall.optInt("index", i) |
|||
|
|||
// 如果是新的工具调用,初始化 |
|||
if (!finalToolCalls.containsKey(index)) { |
|||
finalToolCalls[index] = ToolCallInfo() |
|||
} |
|||
|
|||
// 获取ID |
|||
if (toolCall.has("id")) { |
|||
finalToolCalls[index]?.id = toolCall.getString("id") |
|||
} |
|||
|
|||
// 处理函数信息 |
|||
if (toolCall.has("function")) { |
|||
val function = toolCall.getJSONObject("function") |
|||
|
|||
if (function.has("name")) { |
|||
finalToolCalls[index]?.name = function.getString("name") |
|||
} |
|||
|
|||
if (function.has("arguments")) { |
|||
finalToolCalls[index]?.arguments += function.getString("arguments") |
|||
} |
|||
} |
|||
} |
|||
continue |
|||
} |
|||
|
|||
// 如果有内容,发送给回调 |
|||
if (delta.has("content") && !delta.isNull("content")) { |
|||
val content = delta.getString("content") |
|||
callback.onToken(content) |
|||
} |
|||
} |
|||
} |
|||
} catch (e: Exception) { |
|||
// 忽略无效的JSON |
|||
continue |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 处理完整的函数调用 |
|||
for ((_, toolCallInfo) in finalToolCalls) { |
|||
if (toolCallInfo.name.isNotEmpty()) { |
|||
try { |
|||
// 创建函数调用对象 |
|||
val functionCall = JSONObject().apply { |
|||
put("id", toolCallInfo.id) |
|||
put("name", toolCallInfo.name) |
|||
put("arguments", toolCallInfo.arguments.trim()) |
|||
} |
|||
|
|||
callback.onFunctionCall(functionCall) |
|||
} catch (e: Exception) { |
|||
// 出错时使用空参数 |
|||
val functionCall = JSONObject().apply { |
|||
put("id", toolCallInfo.id) |
|||
put("name", toolCallInfo.name) |
|||
put("arguments", "{}") |
|||
} |
|||
callback.onFunctionCall(functionCall) |
|||
} |
|||
} |
|||
} |
|||
} catch (e: Exception) { |
|||
callback.onError(OpenAIException("处理流式响应出错: ${e.message}")) |
|||
} finally { |
|||
response.close() |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
|
|||
/** |
|||
* 创建用户消息 |
|||
*/ |
|||
fun createUserMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "user") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 创建系统消息 |
|||
*/ |
|||
fun createSystemMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "system") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 创建助手消息 |
|||
*/ |
|||
fun createAssistantMessage(content: String): JSONObject { |
|||
return JSONObject().apply { |
|||
put("role", "assistant") |
|||
put("content", content) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 流式输出回调接口 |
|||
*/ |
|||
interface StreamCallback { |
|||
fun onToken(token: String) |
|||
fun onComplete() |
|||
fun onError(e: Exception) |
|||
fun onFunctionCall(functionCall: JSONObject) {} |
|||
} |
|||
|
|||
/** |
|||
* 用于存储工具调用信息的辅助类 |
|||
*/ |
|||
private class ToolCallInfo { |
|||
var id: String = "" |
|||
var name: String = "" |
|||
var arguments: String = "" |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* OpenAI异常 |
|||
*/ |
|||
class OpenAIException(message: String) : Exception(message) |
|||
@ -0,0 +1,180 @@ |
|||
package com.yunqiinnovation.deepsound |
|||
|
|||
import org.json.JSONArray |
|||
import org.json.JSONObject |
|||
import com.yunqiinnovation.deepsound.OpenAIService |
|||
import com.yunqiinnovation.deepsound.core.utils.FileLogger |
|||
|
|||
/** |
|||
* 语音功能处理器 - 处理AI函数调用 |
|||
*/ |
|||
class VoiceFunctionHandler( |
|||
private val openAIService: OpenAIService, |
|||
private val systemPrompt: String |
|||
) { |
|||
companion object { |
|||
private const val TAG = "VoiceFunctionHandler" |
|||
} |
|||
|
|||
/** |
|||
* 初始化并注册所有可用的函数 |
|||
*/ |
|||
fun initialize() { |
|||
try { |
|||
// 注册退出交互函数 |
|||
registerExitInteractionFunction() |
|||
|
|||
// 在这里可以注册更多函数 |
|||
|
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "初始化函数处理器失败: ${e.message}", e) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 注册退出交互函数 |
|||
*/ |
|||
private fun registerExitInteractionFunction() { |
|||
try { |
|||
openAIService.registerFunction( |
|||
"exit_interaction", |
|||
"退出当前语音交互", |
|||
JSONObject(""" |
|||
{ |
|||
"type": "object", |
|||
"properties": {}, |
|||
"required": [] |
|||
} |
|||
""") |
|||
) |
|||
FileLogger.d(TAG, "退出交互功能已注册") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "注册退出交互函数失败: ${e.message}", e) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 处理函数调用 |
|||
* |
|||
* @param functionCall 函数调用信息 |
|||
* @param messages 消息历史 |
|||
* @param callback 回调,处理退出等操作 |
|||
* @return 是否已处理函数调用 |
|||
*/ |
|||
fun handleFunctionCall( |
|||
functionCall: JSONObject, |
|||
messages: JSONArray, |
|||
callback: FunctionCallCallback |
|||
): Boolean { |
|||
val functionName = functionCall.getString("name") |
|||
FileLogger.d(TAG, "处理函数调用: $functionName") |
|||
|
|||
return when (functionName) { |
|||
"exit_interaction" -> { |
|||
handleExitInteraction(functionCall, messages, callback) |
|||
true |
|||
} |
|||
else -> { |
|||
// 未知函数,返回默认结果 |
|||
handleUnknownFunction(functionCall, messages, callback) |
|||
false |
|||
} |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 处理退出交互函数 |
|||
*/ |
|||
private fun handleExitInteraction( |
|||
functionCall: JSONObject, |
|||
messages: JSONArray, |
|||
callback: FunctionCallCallback |
|||
) { |
|||
FileLogger.d(TAG, "处理退出交互函数") |
|||
|
|||
val responseBuilder = StringBuilder() |
|||
|
|||
openAIService.sendFunctionCallResult( |
|||
messages = messages, |
|||
systemPrompt = systemPrompt, |
|||
functionCall = functionCall, |
|||
functionResult = "{\"result\": \"已退出语音交互\"}", |
|||
callback = object : OpenAIService.StreamCallback { |
|||
override fun onToken(token: String) { |
|||
responseBuilder.append(token) |
|||
} |
|||
|
|||
override fun onComplete() { |
|||
FileLogger.d(TAG, "handleExitInteraction onComplete: ${responseBuilder.toString()}") |
|||
val response = responseBuilder.toString() |
|||
if (response.isNotEmpty()) { |
|||
callback.onExitWithMessage(response) |
|||
} else { |
|||
callback.onExitWithMessage("已退出语音交互") |
|||
} |
|||
} |
|||
|
|||
override fun onError(e: Exception) { |
|||
FileLogger.e(TAG, "处理退出交互函数调用出错: ${e.message}") |
|||
callback.onError("退出交互时出错") |
|||
} |
|||
|
|||
override fun onFunctionCall(nestedCall: JSONObject) { |
|||
FileLogger.e(TAG, "意外收到嵌套函数调用: ${nestedCall.getString("name")}") |
|||
} |
|||
} |
|||
) |
|||
} |
|||
|
|||
/** |
|||
* 处理未知函数调用 |
|||
*/ |
|||
private fun handleUnknownFunction( |
|||
functionCall: JSONObject, |
|||
messages: JSONArray, |
|||
callback: FunctionCallCallback |
|||
) { |
|||
FileLogger.d(TAG, "处理未知函数: ${functionCall.getString("name")}") |
|||
|
|||
try { |
|||
openAIService.sendFunctionCallResult( |
|||
messages = messages, |
|||
systemPrompt = systemPrompt, |
|||
functionCall = functionCall, |
|||
functionResult = "{\"result\": \"处理函数调用中\"}", |
|||
callback = object : OpenAIService.StreamCallback { |
|||
override fun onToken(token: String) { |
|||
callback.onTokenReceived(token) |
|||
} |
|||
|
|||
override fun onComplete() { |
|||
callback.onComplete() |
|||
} |
|||
|
|||
override fun onError(e: Exception) { |
|||
FileLogger.e(TAG, "处理函数调用失败: ${e.message}") |
|||
callback.onError("处理函数调用失败: ${e.message}") |
|||
} |
|||
|
|||
override fun onFunctionCall(nestedCall: JSONObject) { |
|||
callback.onFunctionCall(nestedCall) |
|||
} |
|||
} |
|||
) |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "处理函数调用失败: ${e.message}") |
|||
callback.onError("处理函数调用失败: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 函数调用回调接口 |
|||
*/ |
|||
interface FunctionCallCallback { |
|||
fun onTokenReceived(token: String) |
|||
fun onComplete() |
|||
fun onError(message: String) |
|||
fun onFunctionCall(functionCall: JSONObject) |
|||
fun onExitWithMessage(farewell: String) |
|||
} |
|||
} |
|||
@ -1,21 +0,0 @@ |
|||
MIT License |
|||
|
|||
Copyright (c) 2024 Your Company |
|||
|
|||
Permission is hereby granted, free of charge, to any person obtaining a copy |
|||
of this software and associated documentation files (the "Software"), to deal |
|||
in the Software without restriction, including without limitation the rights |
|||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell |
|||
copies of the Software, and to permit persons to whom the Software is |
|||
furnished to do so, subject to the following conditions: |
|||
|
|||
The above copyright notice and this permission notice shall be included in all |
|||
copies or substantial portions of the Software. |
|||
|
|||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR |
|||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, |
|||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE |
|||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER |
|||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, |
|||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE |
|||
SOFTWARE. |
|||
@ -1,66 +0,0 @@ |
|||
# Azure Speech Recognition |
|||
|
|||
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. |
|||
|
|||
## Features |
|||
|
|||
- Speech-to-text (Azure Speech Recognition) |
|||
- Text-to-speech (Azure Speech Synthesis) |
|||
- Support for multiple languages |
|||
- Language detection |
|||
- Continuous recognition |
|||
- Streaming synthesis |
|||
|
|||
## Getting Started |
|||
|
|||
### Prerequisites |
|||
|
|||
- Azure Speech service subscription key |
|||
- Azure Speech service region |
|||
|
|||
### Installation |
|||
|
|||
Add this to your package's `pubspec.yaml` file: |
|||
|
|||
```yaml |
|||
dependencies: |
|||
azure_speech_recognition: |
|||
path: ./azure |
|||
``` |
|||
|
|||
### Usage |
|||
|
|||
```dart |
|||
import 'package:azure_speech_recognition/azure_speech_recognition.dart'; |
|||
|
|||
// Initialize the service |
|||
await AzureSpeechRecognition.initialize( |
|||
subscriptionKey: 'your_subscription_key', |
|||
region: 'your_region', |
|||
supportedLanguages: ['zh-CN', 'en-US'], |
|||
); |
|||
|
|||
// Start continuous recognition |
|||
await AzureSpeechRecognition.startContinuousRecognition(); |
|||
|
|||
// Listen for recognition events |
|||
AzureSpeechRecognition.onRecognitionEvent.listen((event) { |
|||
if (event['type'] == 'result') { |
|||
print('Recognized: ${event['text']}'); |
|||
print('Detected language: ${event['detectedLanguage']}'); |
|||
} |
|||
}); |
|||
|
|||
// Stop recognition when done |
|||
await AzureSpeechRecognition.stopContinuousRecognition(); |
|||
|
|||
// Speak text |
|||
await AzureSpeechRecognition.speakText('Hello, world!'); |
|||
|
|||
// Clean up |
|||
await AzureSpeechRecognition.dispose(); |
|||
``` |
|||
|
|||
## License |
|||
|
|||
This project is licensed under the MIT License - see the LICENSE file for details. |
|||
@ -1,757 +0,0 @@ |
|||
import Foundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
import AVFoundation |
|||
import AudioToolbox |
|||
|
|||
/// Azure ASR工具类,负责实现语音识别服务接口 |
|||
@available(iOS 13.0, *) |
|||
class AzureAsrHelper: NSObject { |
|||
// MARK: - 属性 |
|||
|
|||
/// 事件处理回调 |
|||
private var eventHandler: (String, [String: Any]) -> Void |
|||
|
|||
/// 语音配置信息 |
|||
private var speechSubscriptionKey: String = "" |
|||
private var serviceRegion: String = "" |
|||
|
|||
/// 语音识别相关 |
|||
private var speechConfig: SPXSpeechConfiguration? |
|||
private var recognizer: SPXSpeechRecognizer? |
|||
private var audioConfig: SPXAudioConfiguration? |
|||
private var pushStream: SPXPushAudioInputStream? |
|||
|
|||
/// 音频处理相关 |
|||
private var audioProcessor: CustomAudioProcessor? |
|||
private var isProcessingAudio = false |
|||
private var audioProcessingTimer: Timer? |
|||
|
|||
/// 状态标志 |
|||
private var isInitialized = false |
|||
private var _isContinuousRecognitionActive = false |
|||
|
|||
/// 当前语言和支持的语言 |
|||
private var currentLanguage = "zh-CN" |
|||
private var supportedLanguages: [String] = ["zh-CN", "en-US"] |
|||
private var isAutoDetectLanguage = false |
|||
|
|||
// MARK: - 初始化 |
|||
|
|||
init(eventHandler: @escaping (String, [String: Any]) -> Void) { |
|||
self.eventHandler = eventHandler |
|||
super.init() |
|||
} |
|||
|
|||
deinit { |
|||
dispose() |
|||
} |
|||
|
|||
// MARK: - ASR Service 接口实现 |
|||
|
|||
/// 初始化语音识别服务 |
|||
/// - Parameters: |
|||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|||
/// - serviceRegion: Azure 服务区域 (如 eastasia) |
|||
/// - supportedLanguages: 支持的语言代码数组 (可选) |
|||
/// - Returns: 初始化是否成功 |
|||
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String]? = nil) -> Bool { |
|||
print("[AzureAsrHelper] 初始化 Azure 语音服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
|||
eventHandler("error", ["message": "Azure 配置信息不完整"]) |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
// 记录配置信息 |
|||
self.speechSubscriptionKey = speechSubscriptionKey |
|||
self.serviceRegion = serviceRegion |
|||
|
|||
// 设置语言 |
|||
if let languages = supportedLanguages, !languages.isEmpty { |
|||
self.supportedLanguages = languages |
|||
} |
|||
|
|||
// 根据支持的语言数量决定是否启用自动语言检测 |
|||
isAutoDetectLanguage = self.supportedLanguages.count >= 2 |
|||
|
|||
// 如果只有一种语言,设置为当前语言 |
|||
if !isAutoDetectLanguage && !self.supportedLanguages.isEmpty { |
|||
currentLanguage = self.supportedLanguages[0] |
|||
} |
|||
|
|||
// 创建识别器和设置回调 |
|||
if !createRecognizerAndSetupCallbacks() { |
|||
return false |
|||
} |
|||
|
|||
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
|||
isInitialized = true |
|||
return true |
|||
} |
|||
|
|||
/// 创建识别器并设置回调 |
|||
private func createRecognizerAndSetupCallbacks() -> Bool { |
|||
// 释放之前的 recognizer |
|||
recognizer = nil |
|||
audioConfig = nil |
|||
|
|||
do { |
|||
// 创建语音配置 |
|||
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|||
|
|||
// 设置音频输入参数 |
|||
try setupAudioSession() |
|||
|
|||
// 创建自定义推送流,替代默认的麦克风输入 |
|||
pushStream = try SPXPushAudioInputStream() |
|||
audioConfig = try SPXAudioConfiguration(streamInput: pushStream!) |
|||
|
|||
// 初始化自定义音频处理器 |
|||
audioProcessor = CustomAudioProcessor() |
|||
|
|||
// 设置语言配置 |
|||
if isAutoDetectLanguage { |
|||
// 设置自动语言检测 |
|||
speechConfig?.setPropertyTo("Continuous", by: SPXPropertyId.speechServiceConnectionLanguageIdMode) |
|||
|
|||
// 创建自动语言检测配置 |
|||
let autoDetectSourceLanguageConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) |
|||
|
|||
// 创建识别器 |
|||
recognizer = try SPXSpeechRecognizer( |
|||
speechConfiguration: speechConfig!, |
|||
autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig, |
|||
audioConfiguration: audioConfig! |
|||
) |
|||
} else { |
|||
// 设置指定的识别语言 |
|||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|||
|
|||
// 创建识别器 |
|||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|||
} |
|||
|
|||
// 设置所有回调 |
|||
setupAllCallbacks() |
|||
|
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") |
|||
eventHandler("error", ["message": "创建识别器失败: \(error.localizedDescription)"]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 设置音频会话 |
|||
private func setupAudioSession() throws { |
|||
let audioSession = AVAudioSession.sharedInstance() |
|||
|
|||
// 使用playAndRecord类别允许同时录音和播放 |
|||
try audioSession.setCategory(.playAndRecord, |
|||
mode: .voiceChat, // 使用voiceChat模式能够更好地支持回音消除 |
|||
options: [.allowBluetooth, .defaultToSpeaker, .allowAirPlay, .mixWithOthers]) |
|||
|
|||
// 设置首选的输入和输出 |
|||
let currentRoute = audioSession.currentRoute |
|||
|
|||
// 获取当前是否连接了耳机或外部麦克风 |
|||
let hasHeadphones = currentRoute.outputs.contains { |
|||
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP |
|||
} |
|||
|
|||
// 如果没有耳机,明确启用内置麦克风和扬声器的回音消除 |
|||
if !hasHeadphones { |
|||
try audioSession.setMode(.voiceChat) // 语音聊天模式有更强的回音消除 |
|||
|
|||
// 启用回音消除和噪声抑制 |
|||
try audioSession.setInputGain(0.8) // 适当降低输入增益以减少扬声器音频被麦克风捕获的可能性 |
|||
} else { |
|||
// 耳机模式,可以使用不同的设置 |
|||
try audioSession.setMode(.voiceChat) |
|||
try audioSession.setInputGain(1.0) |
|||
} |
|||
|
|||
// 设置合适的采样率 |
|||
try audioSession.setPreferredSampleRate(16000.0) // Azure语音识别推荐的采样率 |
|||
try audioSession.setPreferredIOBufferDuration(0.01) // 较小的缓冲区大小以减少延迟 |
|||
|
|||
// 激活音频会话 |
|||
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|||
|
|||
print("[AzureAsrHelper] 音频会话配置成功,已启用回音消除") |
|||
} |
|||
|
|||
/// 设置所有回调 |
|||
private func setupAllCallbacks() { |
|||
guard let recognizer = recognizer else { return } |
|||
|
|||
// 最终识别结果 |
|||
recognizer.addRecognizedEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == SPXResultReason.recognizedSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|||
self.eventHandler("result", [ |
|||
"text": event.result.text ?? "", |
|||
"detectedLanguage": detectedLanguage |
|||
]) |
|||
} |
|||
} |
|||
|
|||
// 识别中事件 |
|||
recognizer.addRecognizingEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == SPXResultReason.recognizingSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|||
self.eventHandler("recognizing", [ |
|||
"text": event.result.text ?? "", |
|||
"detectedLanguage": detectedLanguage |
|||
]) |
|||
} |
|||
} |
|||
|
|||
// 会话事件 |
|||
recognizer.addSessionStartedEventHandler { [weak self] _, _ in |
|||
guard let self = self else { return } |
|||
|
|||
print("[AzureAsrHelper] 识别会话已开始") |
|||
self._isContinuousRecognitionActive = true |
|||
self.eventHandler("sessionStarted", [:]) |
|||
} |
|||
|
|||
recognizer.addSessionStoppedEventHandler { [weak self] _, _ in |
|||
guard let self = self else { return } |
|||
|
|||
print("[AzureAsrHelper] 识别会话已结束") |
|||
self._isContinuousRecognitionActive = false |
|||
self.eventHandler("sessionStopped", [:]) |
|||
} |
|||
|
|||
// 取消事件 |
|||
recognizer.addCanceledEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
let reason = event.reason.rawValue |
|||
let errorDetails = event.errorDetails ?? "未知错误" |
|||
|
|||
print("[AzureAsrHelper] 识别取消: \(errorDetails)") |
|||
|
|||
self.eventHandler("canceled", [ |
|||
"reason": reason, |
|||
"errorDetails": errorDetails |
|||
]) |
|||
|
|||
self._isContinuousRecognitionActive = false |
|||
} |
|||
} |
|||
|
|||
/// 执行一次性语音识别 |
|||
/// - Returns: 是否成功启动识别 |
|||
func recognizeOnce() -> Bool { |
|||
if !isInitialized { |
|||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|||
eventHandler("error", ["message": "语音服务未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
// 如果正在连续识别,先停止 |
|||
if _isContinuousRecognitionActive { |
|||
stopContinuousRecognition() |
|||
} |
|||
|
|||
// 确保识别器已创建 |
|||
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 启动音频处理 |
|||
startAudioProcessing() |
|||
|
|||
// 通知会话开始 |
|||
eventHandler("sessionStarted", [:]) |
|||
|
|||
// 执行识别 |
|||
try recognizer?.recognizeOnceAsync { [weak self] result in |
|||
guard let self = self else { return } |
|||
|
|||
// 停止音频处理 |
|||
self.stopAudioProcessing() |
|||
|
|||
if result.reason == SPXResultReason.recognizedSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|||
self.eventHandler("result", [ |
|||
"text": result.text ?? "", |
|||
"detectedLanguage": detectedLanguage |
|||
]) |
|||
} else if result.reason == SPXResultReason.noMatch { |
|||
print("[AzureAsrHelper] 无匹配结果") |
|||
self.eventHandler("noMatch", [:]) |
|||
} else if result.reason == SPXResultReason.canceled { |
|||
do { |
|||
let details = try SPXCancellationDetails(fromCanceledRecognitionResult: result) |
|||
let errorDetails = details.errorDetails ?? "未知错误" |
|||
self.eventHandler("error", ["message": "识别取消: \(errorDetails)"]) |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 获取取消详情失败: \(error.localizedDescription)") |
|||
self.eventHandler("error", ["message": "识别取消,无法获取详细原因"]) |
|||
} |
|||
} |
|||
} |
|||
|
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 识别异常: \(error.localizedDescription)") |
|||
eventHandler("error", ["message": "识别异常: \(error.localizedDescription)"]) |
|||
stopAudioProcessing() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 开始连续语音识别 |
|||
/// - Returns: 是否成功启动识别 |
|||
func startContinuousRecognition() -> Bool { |
|||
if !isInitialized { |
|||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|||
eventHandler("error", ["message": "语音服务未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
// 如果已经在进行连续识别,先停止 |
|||
if _isContinuousRecognitionActive { |
|||
stopContinuousRecognition() |
|||
} |
|||
|
|||
// 确保识别器已创建 |
|||
if recognizer == nil && !createRecognizerAndSetupCallbacks() { |
|||
return false |
|||
} |
|||
|
|||
// 重新确保音频设置正确 |
|||
do { |
|||
try setupAudioSession() |
|||
} catch { |
|||
print("[AzureAsrHelper] 警告: 设置音频会话失败: \(error.localizedDescription)") |
|||
} |
|||
|
|||
do { |
|||
// 启动音频处理 |
|||
startAudioProcessing() |
|||
|
|||
// 启动连续识别 |
|||
try recognizer?.startContinuousRecognition() |
|||
_isContinuousRecognitionActive = true |
|||
|
|||
print("[AzureAsrHelper] 连续识别开始") |
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
|||
eventHandler("error", ["message": "开始连续识别失败: \(error.localizedDescription)"]) |
|||
_isContinuousRecognitionActive = false |
|||
stopAudioProcessing() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 停止连续语音识别 |
|||
/// - Returns: 是否成功停止识别 |
|||
func stopContinuousRecognition() -> Bool { |
|||
// 停止音频处理 |
|||
stopAudioProcessing() |
|||
|
|||
if !_isContinuousRecognitionActive || recognizer == nil { |
|||
return true |
|||
} |
|||
|
|||
do { |
|||
try recognizer?.stopContinuousRecognition() |
|||
_isContinuousRecognitionActive = false |
|||
print("[AzureAsrHelper] 连续识别已停止") |
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 停止连续识别失败: \(error.localizedDescription)") |
|||
eventHandler("error", ["message": "停止连续识别失败: \(error.localizedDescription)"]) |
|||
_isContinuousRecognitionActive = false |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 检查连续识别是否活跃 |
|||
/// - Returns: 连续识别是否处于活跃状态 |
|||
func isContinuousRecognitionActive() -> Bool { |
|||
return _isContinuousRecognitionActive |
|||
} |
|||
|
|||
/// 释放资源 |
|||
func dispose() { |
|||
print("[AzureAsrHelper] 释放资源") |
|||
|
|||
// 停止音频处理 |
|||
stopAudioProcessing() |
|||
|
|||
// 停止连续识别 |
|||
if _isContinuousRecognitionActive { |
|||
stopContinuousRecognition() |
|||
} |
|||
|
|||
// 释放音频会话 |
|||
do { |
|||
try AVAudioSession.sharedInstance().setActive(false, options: .notifyOthersOnDeactivation) |
|||
} catch { |
|||
print("[AzureAsrHelper] 警告: 释放音频会话失败: \(error.localizedDescription)") |
|||
} |
|||
|
|||
// 释放资源 |
|||
recognizer = nil |
|||
speechConfig = nil |
|||
audioConfig = nil |
|||
pushStream = nil |
|||
audioProcessor = nil |
|||
|
|||
// 重置状态 |
|||
_isContinuousRecognitionActive = false |
|||
isInitialized = false |
|||
} |
|||
|
|||
/// 从结果中获取检测到的语言 |
|||
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|||
if isAutoDetectLanguage { |
|||
do { |
|||
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|||
return langResult.language ?? currentLanguage |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") |
|||
return currentLanguage |
|||
} |
|||
} else { |
|||
return currentLanguage |
|||
} |
|||
} |
|||
|
|||
// MARK: - 音频处理 |
|||
|
|||
/// 开始音频处理 |
|||
private func startAudioProcessing() { |
|||
guard !isProcessingAudio, let audioProcessor = audioProcessor else { return } |
|||
|
|||
isProcessingAudio = true |
|||
|
|||
// 启动音频处理器 |
|||
if !audioProcessor.startRecord() { |
|||
print("[AzureAsrHelper] 错误: 启动音频处理器失败") |
|||
eventHandler("error", ["message": "启动音频处理器失败"]) |
|||
return |
|||
} |
|||
|
|||
// 启动音频处理定时器 |
|||
audioProcessingTimer = Timer.scheduledTimer(withTimeInterval: 0.08, repeats: true) { [weak self] _ in |
|||
guard let self = self, self.isProcessingAudio, let processor = self.audioProcessor, let stream = self.pushStream else { |
|||
return |
|||
} |
|||
|
|||
// 读取处理后的音频数据 |
|||
var bytes = [UInt8](repeating: 0, count: 2560) |
|||
let bytesRead = processor.read(bytes: &bytes) |
|||
|
|||
if bytesRead > 0 { |
|||
// 推送数据到Azure语音服务 |
|||
let data = Data(bytes: bytes, count: bytesRead) |
|||
stream.write(data) |
|||
|
|||
// 通知音频数据可用 |
|||
self.eventHandler("audioData", ["data": bytes]) |
|||
} |
|||
} |
|||
|
|||
print("[AzureAsrHelper] 音频处理已启动") |
|||
} |
|||
|
|||
/// 停止音频处理 |
|||
private func stopAudioProcessing() { |
|||
// 停止定时器 |
|||
audioProcessingTimer?.invalidate() |
|||
audioProcessingTimer = nil |
|||
|
|||
// 停止音频处理器 |
|||
audioProcessor?.stopRecord() |
|||
|
|||
isProcessingAudio = false |
|||
print("[AzureAsrHelper] 音频处理已停止") |
|||
} |
|||
} |
|||
|
|||
// MARK: - 自定义音频处理器 |
|||
|
|||
@available(iOS 13.0, *) |
|||
class CustomAudioProcessor: NSObject { |
|||
// 音频单元 |
|||
private var ioUnit: AudioUnit? |
|||
|
|||
// 音频格式 |
|||
private var audioFormat: AudioStreamBasicDescription |
|||
|
|||
// 音频缓冲 |
|||
private var audioBufferList: AudioBufferList |
|||
private var audioList: [Float] = [] |
|||
private let audioListQueue = DispatchQueue(label: "audioListQueue") |
|||
|
|||
// 回音消除状态 |
|||
private var isEchoCancellationEnabled = true |
|||
|
|||
override init() { |
|||
// 设置音频格式 - 16kHz, 16位, 单声道 |
|||
audioFormat = AudioStreamBasicDescription( |
|||
mSampleRate: 16000.0, |
|||
mFormatID: kAudioFormatLinearPCM, |
|||
mFormatFlags: kAudioFormatFlagIsSignedInteger | kAudioFormatFlagIsPacked, |
|||
mBytesPerPacket: 2, |
|||
mFramesPerPacket: 1, |
|||
mBytesPerFrame: 2, |
|||
mChannelsPerFrame: 1, |
|||
mBitsPerChannel: 16, |
|||
mReserved: 0 |
|||
) |
|||
|
|||
// 初始化音频缓冲 |
|||
audioBufferList = AudioBufferList( |
|||
mNumberBuffers: 1, |
|||
mBuffers: AudioBuffer( |
|||
mNumberChannels: 1, |
|||
mDataByteSize: 4096, |
|||
mData: malloc(4096) |
|||
) |
|||
) |
|||
|
|||
super.init() |
|||
} |
|||
|
|||
deinit { |
|||
stopRecord() |
|||
free(audioBufferList.mBuffers.mData) |
|||
} |
|||
|
|||
/// 启动音频处理 |
|||
/// - Returns: 是否成功启动 |
|||
func startRecord() -> Bool { |
|||
print("[CustomAudioProcessor] 配置音频单元") |
|||
|
|||
// 创建音频组件描述 - 使用VoiceProcessingIO类型获取回音消除 |
|||
var ioUnitDescription = AudioComponentDescription( |
|||
componentType: kAudioUnitType_Output, |
|||
componentSubType: kAudioUnitSubType_VoiceProcessingIO, |
|||
componentManufacturer: kAudioUnitManufacturer_Apple, |
|||
componentFlags: 0, |
|||
componentFlagsMask: 0 |
|||
) |
|||
|
|||
// 查找音频组件 |
|||
guard let ioUnitRef = AudioComponentFindNext(nil, &ioUnitDescription) else { |
|||
print("[CustomAudioProcessor] 错误: 未找到音频组件") |
|||
return false |
|||
} |
|||
|
|||
// 创建音频单元实例 |
|||
if checkError(AudioComponentInstanceNew(ioUnitRef, &ioUnit), "创建音频单元") { |
|||
ioUnit = nil |
|||
return false |
|||
} |
|||
|
|||
// 启用输入端口 |
|||
var enableInput: UInt32 = 1 |
|||
let kInputBus: AudioUnitElement = 1 |
|||
let kOutputBus: AudioUnitElement = 0 |
|||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|||
kAudioUnitScope_Input, kInputBus, &enableInput, |
|||
UInt32(MemoryLayout<UInt32>.size)), "启用输入端口") { |
|||
return false |
|||
} |
|||
|
|||
// 禁用输出端口 (我们只需要输入) |
|||
var enableOutput: UInt32 = 0 |
|||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioOutputUnitProperty_EnableIO, |
|||
kAudioUnitScope_Output, kOutputBus, |
|||
&enableOutput, UInt32(MemoryLayout<UInt32>.size)), "禁用输出端口") { |
|||
return false |
|||
} |
|||
|
|||
// 设置缓冲区分配标志 |
|||
var flag: UInt32 = 0 |
|||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_ShouldAllocateBuffer, |
|||
kAudioUnitScope_Output, kInputBus, &flag, UInt32(MemoryLayout<UInt32>.size)), "设置缓冲区分配标志") { |
|||
return false |
|||
} |
|||
|
|||
// 设置音频格式 |
|||
let size = UInt32(MemoryLayout<AudioStreamBasicDescription>.size) |
|||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|||
kAudioUnitScope_Output, kInputBus, &audioFormat, size), "设置输入总线输出范围的流格式") { |
|||
return false |
|||
} |
|||
|
|||
if checkError(AudioUnitSetProperty(ioUnit!, kAudioUnitProperty_StreamFormat, |
|||
kAudioUnitScope_Input, kOutputBus, &audioFormat, size), "设置输出总线输入范围的流格式") { |
|||
return false |
|||
} |
|||
|
|||
// 启用回音消除 |
|||
if isEchoCancellationEnabled { |
|||
var echoCancellation: UInt32 = 1 |
|||
AudioUnitSetProperty(ioUnit!, kAUVoiceIOProperty_BypassVoiceProcessing, |
|||
kAudioUnitScope_Global, 0, &echoCancellation, UInt32(MemoryLayout<UInt32>.size)) |
|||
} |
|||
|
|||
// 设置输入回调 - 当有新音频数据时调用 |
|||
var inputCallback = AURenderCallbackStruct( |
|||
inputProc: CustomAudioProcessor.onAudioDataAvailable, |
|||
inputProcRefCon: UnsafeMutableRawPointer(Unmanaged.passUnretained(self).toOpaque()) |
|||
) |
|||
|
|||
if checkError(AudioUnitSetProperty(ioUnit!, |
|||
kAudioOutputUnitProperty_SetInputCallback, |
|||
kAudioUnitScope_Global, kInputBus, |
|||
&inputCallback, UInt32(MemoryLayout<AURenderCallbackStruct>.size)), "设置输入回调") { |
|||
return false |
|||
} |
|||
|
|||
// 初始化音频单元 |
|||
var hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") |
|||
while hasError { |
|||
Thread.sleep(forTimeInterval: 0.1) |
|||
hasError = checkError(AudioUnitInitialize(ioUnit!), "初始化音频单元") |
|||
} |
|||
|
|||
// 启动音频单元 |
|||
hasError = checkError(AudioOutputUnitStart(ioUnit!), "启动音频单元") |
|||
|
|||
print("[CustomAudioProcessor] 音频处理器已启动,回音消除\(isEchoCancellationEnabled ? "已启用" : "已禁用")") |
|||
return !hasError |
|||
} |
|||
|
|||
/// 停止音频处理 |
|||
func stopRecord() { |
|||
print("[CustomAudioProcessor] 停止音频处理器") |
|||
|
|||
if let ioUnit = ioUnit { |
|||
// 停止音频单元 |
|||
_ = checkError(AudioOutputUnitStop(ioUnit), "停止音频单元") |
|||
|
|||
// 关闭音频单元 |
|||
_ = checkError(AudioUnitUninitialize(ioUnit), "反初始化音频单元") |
|||
_ = checkError(AudioComponentInstanceDispose(ioUnit), "释放音频单元") |
|||
|
|||
self.ioUnit = nil |
|||
} |
|||
|
|||
// 清空音频数据缓冲 |
|||
audioListQueue.sync { |
|||
audioList.removeAll() |
|||
} |
|||
} |
|||
|
|||
/// 音频数据回调 - 当有新的音频数据可用时调用 |
|||
private static let onAudioDataAvailable: AURenderCallback = { inRefCon, ioActionFlags, inTimeStamp, inBusNumber, inNumberFrames, ioData in |
|||
// 获取实例 |
|||
let processor = Unmanaged<CustomAudioProcessor>.fromOpaque(inRefCon).takeUnretainedValue() |
|||
|
|||
// 计算预期数据大小 |
|||
let expectedDataByteSize = inNumberFrames * processor.audioFormat.mBytesPerFrame |
|||
|
|||
// 确保缓冲区足够大 |
|||
if processor.audioBufferList.mBuffers.mDataByteSize < expectedDataByteSize { |
|||
processor.audioBufferList.mBuffers.mData = realloc(processor.audioBufferList.mBuffers.mData, Int(expectedDataByteSize)) |
|||
processor.audioBufferList.mBuffers.mDataByteSize = expectedDataByteSize |
|||
} |
|||
|
|||
// 渲染音频数据 |
|||
let status = processor.checkOSStatus(AudioUnitRender(processor.ioUnit!, ioActionFlags, inTimeStamp, |
|||
inBusNumber, inNumberFrames, &processor.audioBufferList), |
|||
"渲染音频数据") |
|||
|
|||
// 将Int16数据转换为浮点数据进行处理 |
|||
var audioDataFloat = [Float](repeating: 0.0, count: Int(inNumberFrames)) |
|||
let buffer = processor.audioBufferList.mBuffers |
|||
let bufferData = buffer.mData!.assumingMemoryBound(to: Int16.self) |
|||
|
|||
for j in 0..<Int(buffer.mDataByteSize / UInt32(MemoryLayout<Int16>.size)) { |
|||
// 归一化到[-1.0, 1.0]范围 |
|||
audioDataFloat[j] = Float(bufferData[j]) / 32768.0 |
|||
} |
|||
|
|||
// 应用附加处理 (如有需要) |
|||
// processor.applyAdditionalProcessing(&audioDataFloat) |
|||
|
|||
// 保存处理后的数据 |
|||
if status == noErr { |
|||
processor.audioListQueue.async { |
|||
processor.audioList.append(contentsOf: audioDataFloat) |
|||
} |
|||
} |
|||
|
|||
return status |
|||
} |
|||
|
|||
/// 读取处理后的音频数据 |
|||
/// - Parameter bytes: 输出字节数组 |
|||
/// - Returns: 读取的字节数 |
|||
func read(bytes: inout [UInt8]) -> Int { |
|||
return audioListQueue.sync { |
|||
// 如果没有数据,返回0 |
|||
if audioList.isEmpty { |
|||
return 0 |
|||
} |
|||
|
|||
// 确保有足够的数据 (至少1280个样本) |
|||
if audioList.count < 1280 { |
|||
return 0 |
|||
} |
|||
|
|||
// 读取一帧数据 (1280个样本) |
|||
let frameLength = 1280 |
|||
let buffer = Array(audioList.prefix(frameLength)) |
|||
audioList.removeFirst(frameLength) |
|||
|
|||
// 将浮点数据转回Int16格式 |
|||
var int16Data = buffer.map { Int16($0 * 32767) } |
|||
|
|||
// 转换为字节数组 |
|||
let data = Data(buffer: UnsafeBufferPointer(start: &int16Data, count: int16Data.count)) |
|||
bytes = [UInt8](data) |
|||
|
|||
// 每个样本2字节 (16位PCM) |
|||
return frameLength * 2 |
|||
} |
|||
} |
|||
|
|||
/// 检查错误并打印日志 |
|||
/// - Parameters: |
|||
/// - status: 操作状态 |
|||
/// - operation: 操作描述 |
|||
/// - Returns: 是否发生错误 |
|||
private func checkError(_ status: OSStatus, _ operation: String) -> Bool { |
|||
if status != noErr { |
|||
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") |
|||
return true |
|||
} |
|||
return false |
|||
} |
|||
|
|||
/// 检查OSStatus并返回状态 |
|||
/// - Parameters: |
|||
/// - status: 操作状态 |
|||
/// - operation: 操作描述 |
|||
/// - Returns: 原始状态 |
|||
private func checkOSStatus(_ status: OSStatus, _ operation: String) -> OSStatus { |
|||
if status != noErr { |
|||
print("[CustomAudioProcessor] 错误: \(operation)失败: \(status)") |
|||
} |
|||
return status |
|||
} |
|||
} |
|||
@ -1,18 +0,0 @@ |
|||
import Flutter |
|||
import UIKit |
|||
|
|||
public class AzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { |
|||
public static func register(with registrar: FlutterPluginRegistrar) { |
|||
if #available(iOS 13.0, *) { |
|||
SwiftAzureSpeechRecognitionPlugin.register(with: registrar) |
|||
} else { |
|||
// 如果低于iOS 13.0,返回不支持的错误 |
|||
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) |
|||
channel.setMethodCallHandler { (call, result) in |
|||
result(FlutterError(code: "UNSUPPORTED", |
|||
message: "需要iOS 13.0及以上系统", |
|||
details: nil)) |
|||
} |
|||
} |
|||
} |
|||
} |
|||
@ -1,427 +0,0 @@ |
|||
import Foundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
import AVFoundation |
|||
|
|||
/// Azure TTS工具类,负责实现TTS服务接口 |
|||
@available(iOS 13.0, *) |
|||
class AzureTtsHelper: NSObject { |
|||
// MARK: - 属性 |
|||
|
|||
/// 事件处理回调 |
|||
private var eventHandler: (String, [String: Any]) -> Void |
|||
|
|||
/// 语音配置信息 |
|||
private var speechSubscriptionKey: String = "" |
|||
private var serviceRegion: String = "" |
|||
|
|||
/// 语音合成配置 |
|||
private var speechConfig: SPXSpeechConfiguration? |
|||
|
|||
/// 语音合成器 |
|||
private var synthesizer: SPXSpeechSynthesizer? |
|||
|
|||
/// 是否初始化成功 |
|||
private var isInitialized = false |
|||
|
|||
/// 当前是否正在播放 |
|||
private var _isSpeaking = false |
|||
|
|||
/// 音频会话配置 |
|||
private var isAudioSessionConfigured = false |
|||
|
|||
// MARK: - 语音设置 |
|||
|
|||
/// 当前语音 |
|||
private var currentVoice = "zh-CN-XiaoxiaoNeural" |
|||
|
|||
/// 支持的语音映射 |
|||
private var voiceMap: [String: String] = [ |
|||
"zh-CN": "zh-CN-XiaoxiaoNeural", |
|||
"en-US": "en-US-JennyNeural", |
|||
"ja-JP": "ja-JP-NanamiNeural", |
|||
"ko-KR": "ko-KR-SunHiNeural", |
|||
"zh-TW": "zh-TW-HsiaoChenNeural", |
|||
"zh-HK": "zh-HK-HiuMaanNeural" |
|||
] |
|||
|
|||
/// 当前语音合成参数 |
|||
private var currentSpeechRate = "0%" |
|||
private var currentPitch = "0%" |
|||
private var currentVolume = "100%" |
|||
|
|||
// MARK: - 初始化 |
|||
|
|||
init(eventHandler: @escaping (String, [String: Any]) -> Void) { |
|||
self.eventHandler = eventHandler |
|||
super.init() |
|||
} |
|||
|
|||
deinit { |
|||
dispose() |
|||
} |
|||
|
|||
// MARK: - TTS 接口实现 |
|||
|
|||
/// 初始化语音合成服务 |
|||
/// - Parameters: |
|||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|||
/// - serviceRegion: Azure 服务区域 (如 eastasia) |
|||
/// - language: 语言代码 (默认 zh-CN) |
|||
/// - Returns: 初始化是否成功 |
|||
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { |
|||
print("[AzureTtsHelper] 初始化语音合成服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureTtsHelper] 错误: Azure 配置信息不完整") |
|||
eventHandler("error", ["error": "Azure 配置信息不完整"]) |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
// 记录配置信息 |
|||
self.speechSubscriptionKey = speechSubscriptionKey |
|||
self.serviceRegion = serviceRegion |
|||
|
|||
// 配置音频会话 |
|||
if !configureAudioSession() { |
|||
print("[AzureTtsHelper] 警告: 音频会话配置失败,将尝试继续初始化") |
|||
} |
|||
|
|||
do { |
|||
// 创建语音配置 |
|||
speechConfig = try SPXSpeechConfiguration(subscription: speechSubscriptionKey, region: serviceRegion) |
|||
|
|||
// 设置默认语音 |
|||
let defaultVoice = getDefaultVoiceForLanguage(language) |
|||
currentVoice = defaultVoice |
|||
speechConfig?.speechSynthesisVoiceName = defaultVoice |
|||
|
|||
// 创建语音合成器 |
|||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|||
|
|||
// 设置事件处理器 |
|||
setupSynthesizerEvents() |
|||
|
|||
isInitialized = true |
|||
print("[AzureTtsHelper] TTS 引擎初始化成功") |
|||
|
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 错误: 初始化语音合成服务失败: \(error.localizedDescription)") |
|||
eventHandler("error", ["error": "初始化语音合成服务失败: \(error.localizedDescription)"]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 配置音频会话 |
|||
private func configureAudioSession() -> Bool { |
|||
let audioSession = AVAudioSession.sharedInstance() |
|||
do { |
|||
// 使用playback类别,但支持混合和空中播放 |
|||
try audioSession.setCategory(.playback, |
|||
mode: .spokenAudio, |
|||
options: [.mixWithOthers, .allowAirPlay, .duckOthers]) |
|||
|
|||
// 根据设备类型选择最佳配置 |
|||
let currentRoute = audioSession.currentRoute |
|||
let hasHeadphones = currentRoute.outputs.contains { |
|||
$0.portType == .headphones || $0.portType == .bluetoothA2DP || $0.portType == .bluetoothHFP |
|||
} |
|||
|
|||
// 优化音频路由 |
|||
if hasHeadphones { |
|||
// 耳机模式,使用默认设置 |
|||
try audioSession.setPreferredIOBufferDuration(0.005) // 较小的缓冲区大小以减少延迟 |
|||
} else { |
|||
// 扬声器模式 |
|||
try audioSession.setPreferredIOBufferDuration(0.005) |
|||
} |
|||
|
|||
// 避免完全激活音频会话,因为ASR可能已经激活 |
|||
// 这里使用setActive(false)是为了不与ASR冲突 |
|||
if !audioSession.isOtherAudioPlaying { |
|||
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|||
} |
|||
|
|||
isAudioSessionConfigured = true |
|||
print("[AzureTtsHelper] 音频会话配置成功") |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 警告: 音频会话配置失败: \(error.localizedDescription)") |
|||
isAudioSessionConfigured = false |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 设置语音 |
|||
/// - Parameter voiceName: 语音名称 (如 "zh-CN-XiaoxiaoNeural") |
|||
/// - Returns: 设置是否成功 |
|||
func setVoice(voiceName: String) -> Bool { |
|||
if !isInitialized { |
|||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
if voiceName.isEmpty { |
|||
print("[AzureTtsHelper] 错误: 声音名称为空") |
|||
eventHandler("error", ["error": "声音名称不能为空"]) |
|||
return false |
|||
} |
|||
|
|||
if voiceName == currentVoice { |
|||
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
|||
return true |
|||
} |
|||
|
|||
print("[AzureTtsHelper] 设置声音: \(voiceName)") |
|||
currentVoice = voiceName |
|||
|
|||
// 更新语音配置 |
|||
if let speechConfig = speechConfig { |
|||
speechConfig.speechSynthesisVoiceName = voiceName |
|||
return true |
|||
} |
|||
|
|||
return false |
|||
} |
|||
|
|||
/// 设置语音合成参数 |
|||
/// - Parameters: |
|||
/// - rate: 语速,范围 -100 到 100,默认为 0 |
|||
/// - pitch: 音调,范围 -100 到 100,默认为 0 |
|||
/// - volume: 音量,范围 0 到 100,默认为 100 |
|||
/// - Returns: 是否设置成功 |
|||
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|||
if !isInitialized { |
|||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
// 转换参数格式 |
|||
currentSpeechRate = formatRateParam(rate) |
|||
currentPitch = formatPitchParam(pitch) |
|||
currentVolume = formatVolumeParam(volume) |
|||
|
|||
print("[AzureTtsHelper] 已设置语音参数: 语速=\(currentSpeechRate), 音调=\(currentPitch), 音量=\(currentVolume)") |
|||
return true |
|||
} |
|||
|
|||
/// 合成文本为语音并播放 |
|||
/// - Parameter text: 要合成的文本 |
|||
/// - Returns: 操作是否成功启动 |
|||
func speakText(text: String) -> Bool { |
|||
if !isInitialized { |
|||
print("[AzureTtsHelper] 错误: TTS 引擎尚未初始化") |
|||
eventHandler("error", ["error": "TTS 引擎尚未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
if text.isEmpty { |
|||
print("[AzureTtsHelper] 警告: 要播放的文本为空") |
|||
return true |
|||
} |
|||
|
|||
// 确保音频会话已配置 |
|||
if !isAudioSessionConfigured { |
|||
_ = configureAudioSession() |
|||
} |
|||
|
|||
print("[AzureTtsHelper] 开始语音合成: \(text.prefix(50))...") |
|||
|
|||
// 生成SSML |
|||
let ssml = generateSsml(text: text) |
|||
|
|||
// 直接进行SSML合成 |
|||
return speakSsmlInternal(text: ssml) |
|||
} |
|||
|
|||
/// 内部SSML合成和播放 |
|||
private func speakSsmlInternal(text: String) -> Bool { |
|||
guard let synthesizer = synthesizer else { |
|||
print("[AzureTtsHelper] 错误: 合成器未初始化") |
|||
eventHandler("error", ["error": "合成器未初始化"]) |
|||
return false |
|||
} |
|||
|
|||
_isSpeaking = true |
|||
eventHandler("started", [:]) |
|||
|
|||
Task { |
|||
do { |
|||
// 使用异步方法进行合成并直接播放 |
|||
_ = try await synthesizer.startSpeakingSsml(text) |
|||
|
|||
} catch { |
|||
print("[AzureTtsHelper] 错误: 语音合成失败: \(error.localizedDescription)") |
|||
DispatchQueue.main.async { |
|||
self._isSpeaking = false |
|||
self.eventHandler("error", ["error": "语音合成失败: \(error.localizedDescription)"]) |
|||
} |
|||
} |
|||
} |
|||
|
|||
return true |
|||
} |
|||
|
|||
/// 停止当前语音合成 |
|||
/// - Returns: 操作是否成功 |
|||
func stopSpeaking() -> Bool { |
|||
if !isInitialized || !_isSpeaking { |
|||
return true |
|||
} |
|||
|
|||
// 停止合成 |
|||
do { |
|||
try synthesizer?.stopSpeaking() |
|||
_isSpeaking = false |
|||
eventHandler("canceled", [:]) |
|||
print("[AzureTtsHelper] 已停止语音合成") |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 错误: 停止语音合成失败: \(error.localizedDescription)") |
|||
eventHandler("error", ["error": "停止语音合成失败: \(error.localizedDescription)"]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 检查是否正在播放 |
|||
/// - Returns: 当前是否正在播放语音 |
|||
func isSpeaking() -> Bool { |
|||
return _isSpeaking |
|||
} |
|||
|
|||
/// 释放资源 |
|||
func dispose() { |
|||
try? stopSpeaking() |
|||
|
|||
// 释放合成器和配置 |
|||
synthesizer = nil |
|||
speechConfig = nil |
|||
|
|||
isInitialized = false |
|||
_isSpeaking = false |
|||
isAudioSessionConfigured = false |
|||
print("[AzureTtsHelper] TTS 引擎已释放") |
|||
} |
|||
|
|||
// MARK: - 私有辅助方法 |
|||
|
|||
/// 设置合成器事件处理 |
|||
private func setupSynthesizerEvents() { |
|||
guard let synthesizer = synthesizer else { return } |
|||
|
|||
// 添加书签到达事件处理 |
|||
synthesizer.addBookmarkReachedEventHandler { _, e in |
|||
print("[AzureTtsHelper] 书签事件: 音频偏移: \((e.audioOffset + 5000) / 10000)ms, 文本: \"\(e.text)\"") |
|||
} |
|||
|
|||
// 合成完成事件 |
|||
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in |
|||
guard let self = self else { return } |
|||
print("[AzureTtsHelper] 语音合成完成: 音频持续时间: \(e.result.audioDuration)") |
|||
DispatchQueue.main.async { |
|||
self._isSpeaking = false |
|||
self.eventHandler("completed", [:]) |
|||
} |
|||
} |
|||
|
|||
// 合成取消事件 |
|||
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in |
|||
guard let self = self else { return } |
|||
|
|||
let result = e.result |
|||
do { |
|||
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: result) |
|||
print("[AzureTtsHelper] 语音合成取消: 原因: \(cancellationDetails.reason)") |
|||
|
|||
if cancellationDetails.reason == SPXCancellationReason.error { |
|||
print("[AzureTtsHelper] 错误代码: \(cancellationDetails.errorCode)") |
|||
print("[AzureTtsHelper] 错误详情: \(cancellationDetails.errorDetails ?? "未知")") |
|||
} |
|||
|
|||
DispatchQueue.main.async { |
|||
self._isSpeaking = false |
|||
self.eventHandler("error", ["error": "语音合成取消: \(cancellationDetails.errorDetails ?? "未知错误")"]) |
|||
} |
|||
} catch { |
|||
print("[AzureTtsHelper] 获取取消详情时出错: \(error)") |
|||
|
|||
DispatchQueue.main.async { |
|||
self._isSpeaking = false |
|||
self.eventHandler("error", ["error": "语音合成被取消"]) |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 合成开始事件 |
|||
synthesizer.addSynthesisStartedEventHandler { _, _ in |
|||
// print("[AzureTtsHelper] 语音合成开始") |
|||
} |
|||
|
|||
// 合成中事件 |
|||
synthesizer.addSynthesizingEventHandler { _, _ in |
|||
// print("[AzureTtsHelper] 语音合成中") |
|||
} |
|||
} |
|||
|
|||
/// 生成 SSML 文本 |
|||
private func generateSsml(text: String) -> String { |
|||
return """ |
|||
<speak version='1.0' xmlns='http://www.w3.org/2001/10/synthesis' xml:lang='zh-CN'> |
|||
<voice name='\(currentVoice)'> |
|||
<prosody rate='\(currentSpeechRate)' pitch='\(currentPitch)' volume='\(currentVolume)'> |
|||
\(text) |
|||
</prosody> |
|||
</voice> |
|||
</speak> |
|||
""" |
|||
} |
|||
|
|||
/// 格式化语速参数 |
|||
private func formatRateParam(_ rate: Int) -> String { |
|||
let clampedRate = rate.clamp(min: -100, max: 100) |
|||
if clampedRate == 0 { |
|||
return "0%" |
|||
} else if clampedRate < 0 { |
|||
return "\(Int(Double(clampedRate) * 0.9))%" |
|||
} else { |
|||
return "+\(clampedRate)%" |
|||
} |
|||
} |
|||
|
|||
/// 格式化音调参数 |
|||
private func formatPitchParam(_ pitch: Int) -> String { |
|||
let clampedPitch = pitch.clamp(min: -100, max: 100) |
|||
if clampedPitch == 0 { |
|||
return "0%" |
|||
} else { |
|||
return "\(Int(Double(clampedPitch) * 0.5))%" |
|||
} |
|||
} |
|||
|
|||
/// 格式化音量参数 |
|||
private func formatVolumeParam(_ volume: Int) -> String { |
|||
let clampedVolume = volume.clamp(min: 0, max: 100) |
|||
return "\(clampedVolume)%" |
|||
} |
|||
|
|||
/// 获取指定语言的默认语音 |
|||
private func getDefaultVoiceForLanguage(_ language: String) -> String { |
|||
return voiceMap[language] ?? "zh-CN-XiaoxiaoNeural" |
|||
} |
|||
} |
|||
|
|||
// MARK: - 扩展 |
|||
|
|||
extension Int { |
|||
func clamp(min: Int, max: Int) -> Int { |
|||
if self < min { return min } |
|||
if self > max { return max } |
|||
return self |
|||
} |
|||
} |
|||
@ -1,259 +0,0 @@ |
|||
import Flutter |
|||
import UIKit |
|||
import MicrosoftCognitiveServicesSpeech |
|||
import AVFoundation |
|||
|
|||
@available(iOS 13.0, *) |
|||
public class SwiftAzureSpeechRecognitionPlugin: NSObject, FlutterPlugin { |
|||
private var azureChannel: FlutterMethodChannel |
|||
private var ttsChannel: FlutterMethodChannel |
|||
private var asrHelper: AzureAsrHelper |
|||
private var ttsHelper: AzureTtsHelper |
|||
private static var eventStreamHandler: AzureEventStreamHandler? |
|||
|
|||
// 创建方法到通道的映射 |
|||
private static var ttsMethodHandlers = [String: FlutterMethodCallHandler]() |
|||
private static var asrMethodHandlers = [String: FlutterMethodCallHandler]() |
|||
|
|||
public static func register(with registrar: FlutterPluginRegistrar) { |
|||
// ASR通道 |
|||
let channel = FlutterMethodChannel(name: "com.deep_voice.azure_asr", binaryMessenger: registrar.messenger()) |
|||
|
|||
// TTS通道 |
|||
let ttsChannel = FlutterMethodChannel(name: "com.deep_voice.azure_tts", binaryMessenger: registrar.messenger()) |
|||
|
|||
// 设置ASR事件通道 |
|||
let eventChannel = FlutterEventChannel(name: "com.deep_voice.azure_asr_events", binaryMessenger: registrar.messenger()) |
|||
eventStreamHandler = AzureEventStreamHandler() |
|||
eventChannel.setStreamHandler(eventStreamHandler) |
|||
|
|||
let instance = SwiftAzureSpeechRecognitionPlugin( |
|||
azureChannel: channel, |
|||
ttsChannel: ttsChannel, |
|||
eventStreamHandler: eventStreamHandler! |
|||
) |
|||
|
|||
// 直接设置各自通道的处理器 |
|||
channel.setMethodCallHandler(instance.handleAsrMethodCalls) |
|||
ttsChannel.setMethodCallHandler(instance.handleTtsMethodCalls) |
|||
} |
|||
|
|||
|
|||
|
|||
// 新增直接处理方法调用的函数 |
|||
private func handleTtsMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|||
handleTtsMethod(call, result) |
|||
} |
|||
|
|||
private func handleAsrMethodCalls(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|||
handleAsrMethod(call, result) |
|||
} |
|||
|
|||
|
|||
init(azureChannel: FlutterMethodChannel, ttsChannel: FlutterMethodChannel, eventStreamHandler: AzureEventStreamHandler) { |
|||
self.azureChannel = azureChannel |
|||
self.ttsChannel = ttsChannel |
|||
|
|||
// 创建辅助类实例,使用自定义事件回调处理器 |
|||
let eventHandler: (String, [String: Any]) -> Void = { eventName, arguments in |
|||
DispatchQueue.main.async { |
|||
if let eventSink = SwiftAzureSpeechRecognitionPlugin.eventStreamHandler?.eventSink { |
|||
var eventData = arguments |
|||
eventData["type"] = eventName |
|||
eventSink(eventData) |
|||
} |
|||
} |
|||
} |
|||
|
|||
asrHelper = AzureAsrHelper(eventHandler: eventHandler) |
|||
ttsHelper = AzureTtsHelper(eventHandler: eventHandler) |
|||
|
|||
super.init() |
|||
} |
|||
|
|||
private func handleAsrMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|||
|
|||
let args = call.arguments as? Dictionary<String, Any> |
|||
|
|||
switch call.method { |
|||
case "initialize": |
|||
// 仅在初始化时读取必要参数 |
|||
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { |
|||
let errorMsg = "语音订阅密钥不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { |
|||
let errorMsg = "服务区域不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
let supportedLanguages = args?["supportedLanguages"] as? [String] ?? [] |
|||
|
|||
let success = asrHelper.initialize( |
|||
speechSubscriptionKey: speechSubscriptionKey, |
|||
serviceRegion: serviceRegion, |
|||
supportedLanguages: supportedLanguages.isEmpty ? nil : supportedLanguages |
|||
) |
|||
result(success) |
|||
|
|||
case "startContinuousRecognition": |
|||
// 只有使用参数时才验证 |
|||
let success = asrHelper.startContinuousRecognition() |
|||
result(success) |
|||
|
|||
case "stopContinuousRecognition": |
|||
// 不需要额外参数 |
|||
let success = asrHelper.stopContinuousRecognition() |
|||
result(success) |
|||
|
|||
case "recognizeOnce": |
|||
// 只有使用参数时才验证 |
|||
let success = asrHelper.recognizeOnce() |
|||
result(success) |
|||
|
|||
case "isContinuousRecognitionActive": |
|||
// 不需要额外参数 |
|||
result(asrHelper.isContinuousRecognitionActive()) |
|||
|
|||
case "dispose": |
|||
// 不需要额外参数 |
|||
print("[AzurePlugin] 释放ASR资源") |
|||
asrHelper.dispose() |
|||
result(true) |
|||
|
|||
default: |
|||
print("[AzurePlugin] 错误: 未知ASR方法: \(call.method)") |
|||
result(FlutterMethodNotImplemented) |
|||
} |
|||
} |
|||
|
|||
private func handleTtsMethod(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|||
|
|||
let args = call.arguments as? Dictionary<String, Any> |
|||
|
|||
switch call.method { |
|||
case "initialize": |
|||
// 仅在初始化时验证参数 |
|||
guard let speechSubscriptionKey = args?["subscriptionKey"] as? String, !speechSubscriptionKey.isEmpty else { |
|||
let errorMsg = "语音订阅密钥不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_SUBSCRIPTION_KEY", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
guard let serviceRegion = args?["region"] as? String, !serviceRegion.isEmpty else { |
|||
let errorMsg = "服务区域不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_REGION", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
let language = args?["language"] as? String ?? "zh-CN" |
|||
|
|||
print("[AzurePlugin] 初始化TTS,语言: \(language)") |
|||
|
|||
let success = ttsHelper.initialize(speechSubscriptionKey: speechSubscriptionKey, serviceRegion: serviceRegion, language: language) |
|||
result(success) |
|||
|
|||
case "setVoice": |
|||
// 仅获取voice参数 |
|||
guard let voiceName = args?["voiceName"] as? String, !voiceName.isEmpty else { |
|||
let errorMsg = "声音名称不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_VOICE", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
print("[AzurePlugin] 设置声音: \(voiceName)") |
|||
|
|||
let success = ttsHelper.setVoice(voiceName: voiceName) |
|||
result(success) |
|||
|
|||
case "speakText": |
|||
// 仅获取text参数 |
|||
let text = args?["text"] as? String ?? "" |
|||
|
|||
if text.isEmpty { |
|||
print("[AzurePlugin] 警告: 要播放的文本为空") |
|||
result("OK") |
|||
return |
|||
} |
|||
|
|||
print("[AzurePlugin] 播放文本: \(text.prefix(50))...") |
|||
|
|||
let success = ttsHelper.speakText(text: text) |
|||
result(success ? "OK" : "ERROR") |
|||
|
|||
case "speakSsml": |
|||
// 仅获取ssml参数 |
|||
guard let ssml = args?["ssml"] as? String, !ssml.isEmpty else { |
|||
let errorMsg = "SSML内容不能为空" |
|||
print("[AzurePlugin] 错误: \(errorMsg)") |
|||
result(FlutterError(code: "INVALID_SSML", message: errorMsg, details: nil)) |
|||
return |
|||
} |
|||
|
|||
print("[AzurePlugin] 播放SSML: \(ssml.prefix(100))...") |
|||
|
|||
// 由于我们移除了speakSsml方法,这里改用speakText方法 |
|||
// Azure SDK内部会自动检测是普通文本还是SSML |
|||
let success = ttsHelper.speakText(text: ssml) |
|||
result(success) |
|||
|
|||
case "stopSpeaking": |
|||
// 不需要参数 |
|||
print("[AzurePlugin] 停止播放") |
|||
let success = ttsHelper.stopSpeaking() |
|||
result(success) |
|||
|
|||
case "isSpeaking": |
|||
// 不需要参数 |
|||
result(ttsHelper.isSpeaking()) |
|||
|
|||
case "setSpeechParams": |
|||
// 仅获取语音参数 |
|||
let rate = args?["rate"] as? Int ?? 0 |
|||
let pitch = args?["pitch"] as? Int ?? 0 |
|||
let volume = args?["volume"] as? Int ?? 100 |
|||
|
|||
print("[AzurePlugin] 设置语音参数: rate=\(rate), pitch=\(pitch), volume=\(volume)") |
|||
let success = ttsHelper.setSpeechParams(rate: rate, pitch: pitch, volume: volume) |
|||
result(success) |
|||
|
|||
case "dispose": |
|||
// 释放TTS资源 |
|||
print("[AzurePlugin] 释放TTS资源") |
|||
ttsHelper.dispose() |
|||
result(true) |
|||
|
|||
default: |
|||
print("[AzurePlugin] 错误: 未知TTS方法: \(call.method)") |
|||
result(FlutterMethodNotImplemented) |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 用于处理事件流的辅助类 |
|||
@available(iOS 13.0, *) |
|||
class AzureEventStreamHandler: NSObject, FlutterStreamHandler { |
|||
var eventSink: FlutterEventSink? |
|||
|
|||
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
|||
self.eventSink = events |
|||
// 通知Flutter端事件通道已准备好 |
|||
DispatchQueue.main.async { |
|||
events(["type": "channelReady"]) |
|||
} |
|||
return nil |
|||
} |
|||
|
|||
func onCancel(withArguments arguments: Any?) -> FlutterError? { |
|||
self.eventSink = nil |
|||
return nil |
|||
} |
|||
} |
|||
@ -1,24 +0,0 @@ |
|||
# |
|||
# To learn more about a Podspec see http://guides.cocoapods.org/syntax/podspec.html. |
|||
# Run `pod lib lint azure_speech_recognition.podspec` to validate before publishing. |
|||
# |
|||
Pod::Spec.new do |s| |
|||
s.name = 'azure_speech_recognition' |
|||
s.version = '0.1.0' |
|||
s.summary = 'Azure Speech Recognition plugin for Flutter' |
|||
s.description = <<-DESC |
|||
A Flutter plugin for Microsoft Azure Speech services, providing both speech recognition (ASR) and text-to-speech (TTS) capabilities. |
|||
DESC |
|||
s.homepage = 'https://github.com/yourusername/azure_speech_recognition' |
|||
s.license = { :type => 'MIT', :file => '../LICENSE' } |
|||
s.author = { 'Your Company' => 'your-email@example.com' } |
|||
s.source = { :path => '.' } |
|||
s.source_files = 'Classes/**/*' |
|||
s.dependency 'Flutter' |
|||
s.dependency 'MicrosoftCognitiveServicesSpeech-iOS', '~> 1.34.0' |
|||
s.platform = :ios, '12.0' |
|||
|
|||
# Flutter.framework does not contain a i386 slice. |
|||
s.pod_target_xcconfig = { 'DEFINES_MODULE' => 'YES', 'EXCLUDED_ARCHS[sdk=iphonesimulator*]' => 'i386' } |
|||
s.swift_version = '5.0' |
|||
end |
|||
@ -1,6 +0,0 @@ |
|||
// This is a placeholder file that exports nothing. |
|||
// The actual implementation is in the app's services folder. |
|||
// This file exists just to satisfy the Flutter plugin structure requirements. |
|||
|
|||
// Empty library to satisfy plugin structure |
|||
library azure_speech_recognition; |
|||
@ -1,23 +0,0 @@ |
|||
name: azure_speech_recognition |
|||
description: Azure Speech Recognition and Text-to-Speech services Flutter plugin |
|||
version: 0.1.0 |
|||
homepage: https://github.com/yourusername/azure_speech_recognition |
|||
|
|||
environment: |
|||
sdk: '>=2.12.0 <3.0.0' |
|||
flutter: ">=2.0.0" |
|||
|
|||
dependencies: |
|||
flutter: |
|||
sdk: flutter |
|||
|
|||
dev_dependencies: |
|||
flutter_test: |
|||
sdk: flutter |
|||
flutter_lints: ^1.0.0 |
|||
|
|||
flutter: |
|||
plugin: |
|||
platforms: |
|||
ios: |
|||
pluginClass: AzureSpeechRecognitionPlugin |
|||
@ -0,0 +1,136 @@ |
|||
# Azure Speech 插件 |
|||
|
|||
本插件为Flutter提供了Azure语音服务的集成,包括: |
|||
|
|||
- 语音合成(TTS) |
|||
- 语音识别(ASR) |
|||
|
|||
## 功能 |
|||
|
|||
### 语音合成(TTS) |
|||
|
|||
- 支持多种语音(如中文、英文等) |
|||
- 语音参数调整(语速、音调、音量) |
|||
- 音频输出设备选择(扬声器、听筒、自动) |
|||
- SSML支持 |
|||
|
|||
### 语音识别(ASR) |
|||
|
|||
- 一次性语音识别 |
|||
- 连续语音识别 |
|||
- 自动语言检测 |
|||
- 音频处理优化(回音消除、噪声抑制等) |
|||
|
|||
## 平台支持 |
|||
|
|||
- Android |
|||
- iOS |
|||
|
|||
## 如何使用 |
|||
|
|||
### 初始化 |
|||
|
|||
```dart |
|||
import 'package:azure_speech/azure_speech.dart'; |
|||
|
|||
// 初始化TTS |
|||
await AzureSpeech.initializeTts( |
|||
'your_subscription_key', |
|||
'your_service_region', |
|||
language: 'zh-CN', |
|||
); |
|||
|
|||
// 初始化ASR |
|||
await AzureSpeech.initializeAsr( |
|||
'your_subscription_key', |
|||
'your_service_region', |
|||
['zh-CN', 'en-US'], |
|||
); |
|||
``` |
|||
|
|||
### 语音合成 |
|||
|
|||
```dart |
|||
// 设置语音 |
|||
await AzureSpeech.setTtsVoice('zh-CN-XiaoxiaoNeural'); |
|||
|
|||
// 设置语音参数 |
|||
await AzureSpeech.setTtsSpeechParams( |
|||
rate: 0, // 语速 -100~100 |
|||
pitch: 0, // 音调 -100~100 |
|||
volume: 100, // 音量 0~100 |
|||
); |
|||
|
|||
// 设置音频输出设备 |
|||
await AzureSpeech.setTtsAudioOutputType('AUTO'); // 'SPEAKER', 'EARPIECE', 'AUTO' |
|||
|
|||
// 播放文本 |
|||
await AzureSpeech.speakText('你好,世界!'); |
|||
|
|||
// 停止播放 |
|||
await AzureSpeech.stopSpeaking(); |
|||
|
|||
// 检查是否正在播放 |
|||
bool isSpeaking = await AzureSpeech.isSpeaking(); |
|||
``` |
|||
|
|||
### 语音识别 |
|||
|
|||
```dart |
|||
// 一次性识别 |
|||
final result = await AzureSpeech.recognizeOnce(); |
|||
if (result['success']) { |
|||
print('识别文本: ${result['text']}'); |
|||
print('识别语言: ${result['language']}'); |
|||
} else { |
|||
print('识别失败: ${result['error']}'); |
|||
} |
|||
|
|||
// 连续识别 |
|||
// 监听识别结果 |
|||
AzureSpeech.asrResultStream.listen((event) { |
|||
switch (event['eventType']) { |
|||
case 'recognizing': |
|||
// 实时识别中的结果 |
|||
print('识别中: ${event['text']}'); |
|||
break; |
|||
case 'finalResult': |
|||
// 最终识别结果 |
|||
print('最终结果: ${event['text']}'); |
|||
break; |
|||
case 'error': |
|||
// 错误 |
|||
print('错误: ${event['error']}'); |
|||
break; |
|||
} |
|||
}); |
|||
|
|||
// 开始连续识别 |
|||
await AzureSpeech.startContinuousRecognition(); |
|||
|
|||
// 停止连续识别 |
|||
await AzureSpeech.stopContinuousRecognition(); |
|||
|
|||
// 检查连续识别是否活跃 |
|||
bool isActive = await AzureSpeech.isContinuousRecognitionActive(); |
|||
``` |
|||
|
|||
### 释放资源 |
|||
|
|||
```dart |
|||
// 释放所有资源 |
|||
await AzureSpeech.dispose(); |
|||
``` |
|||
|
|||
## 依赖项 |
|||
|
|||
本插件依赖于: |
|||
|
|||
- Microsoft Cognitive Services Speech SDK |
|||
- Flutter |
|||
|
|||
## 注意事项 |
|||
|
|||
- 使用前需要在Azure门户中创建语音服务资源,并获取订阅密钥和区域 |
|||
- Android需要相关权限:RECORD_AUDIO, INTERNET等 |
|||
- iOS需要在Info.plist中添加麦克风使用权限描述 |
|||
@ -0,0 +1,65 @@ |
|||
import com.android.build.gradle.LibraryExtension |
|||
|
|||
buildscript { |
|||
repositories { |
|||
google() |
|||
mavenCentral() |
|||
} |
|||
dependencies { |
|||
classpath("com.android.tools.build:gradle:7.3.0") |
|||
classpath("org.jetbrains.kotlin:kotlin-gradle-plugin:1.7.10") |
|||
} |
|||
} |
|||
|
|||
allprojects { |
|||
repositories { |
|||
google() |
|||
mavenCentral() |
|||
} |
|||
} |
|||
|
|||
plugins { |
|||
id("com.android.library") |
|||
kotlin("android") |
|||
} |
|||
|
|||
// 配置android扩展 |
|||
configure<LibraryExtension> { |
|||
namespace = "com.yunqiinnovation.azure_speech" |
|||
compileSdkVersion(33) |
|||
|
|||
defaultConfig { |
|||
minSdk = 21 |
|||
} |
|||
|
|||
compileOptions { |
|||
sourceCompatibility = JavaVersion.VERSION_1_8 |
|||
targetCompatibility = JavaVersion.VERSION_1_8 |
|||
} |
|||
|
|||
sourceSets { |
|||
getByName("main") { |
|||
manifest.srcFile("src/main/AndroidManifest.xml") |
|||
java.srcDirs("src/main/kotlin") |
|||
} |
|||
} |
|||
|
|||
// 添加lint选项 |
|||
lintOptions { |
|||
isCheckReleaseBuilds = false |
|||
} |
|||
} |
|||
|
|||
// 显式设置Kotlin JVM目标版本 |
|||
tasks.withType<org.jetbrains.kotlin.gradle.tasks.KotlinCompile> { |
|||
kotlinOptions { |
|||
jvmTarget = "1.8" |
|||
} |
|||
} |
|||
|
|||
dependencies { |
|||
// 直接通过本地依赖方式添加Flutter |
|||
implementation(fileTree(mapOf("dir" to "libs", "include" to listOf("*.jar")))) |
|||
// 添加Microsoft语音SDK |
|||
implementation("com.microsoft.cognitiveservices.speech:client-sdk:1.30.0") |
|||
} |
|||
@ -0,0 +1 @@ |
|||
rootProject.name = "azure_speech" |
|||
@ -0,0 +1,6 @@ |
|||
<?xml version="1.0" encoding="utf-8"?> |
|||
<manifest xmlns:android="http://schemas.android.com/apk/res/android" |
|||
package="com.yunqiinnovation.azure_speech"> |
|||
<uses-permission android:name="android.permission.INTERNET" /> |
|||
<uses-permission android:name="android.permission.RECORD_AUDIO" /> |
|||
</manifest> |
|||
@ -0,0 +1,593 @@ |
|||
package com.yunqiinnovation.azure_speech |
|||
|
|||
import android.content.Context |
|||
import android.media.AudioAttributes |
|||
import android.media.AudioFormat |
|||
import android.media.AudioRecord |
|||
import android.media.MediaRecorder |
|||
import android.media.audiofx.AcousticEchoCanceler |
|||
import android.media.audiofx.NoiseSuppressor |
|||
import android.media.audiofx.AutomaticGainControl |
|||
import android.os.Process |
|||
import com.yunqiinnovation.azure_speech.utils.FileLogger |
|||
import com.microsoft.cognitiveservices.speech.* |
|||
import com.microsoft.cognitiveservices.speech.audio.* |
|||
import com.microsoft.cognitiveservices.speech.util.EventHandler |
|||
import java.util.concurrent.ExecutionException |
|||
import java.util.concurrent.atomic.AtomicBoolean |
|||
|
|||
class AzureAsrHelper(private val context: Context) { |
|||
private var recognizer: SpeechRecognizer? = null |
|||
private var speechConfig: SpeechConfig? = null |
|||
private val TAG = "AzureAsrHelper" |
|||
private var isContinuousRecognitionActive = false |
|||
private var currentLanguage = "zh-CN" |
|||
private var subscriptionKey = "" |
|||
private var region = "" |
|||
private var isAutoDetectLanguage = false |
|||
private var supportedLanguages = arrayOf("zh-CN", "en-US") |
|||
|
|||
// 是否使用回音消除 - 内部控制常量 |
|||
private val useEchoCancellation = false |
|||
|
|||
// 自定义音频处理相关 |
|||
private var customAudioProcessor: CustomAudioProcessor? = null |
|||
private var pushStream: PushAudioInputStream? = null |
|||
private var audioConfig: AudioConfig? = null |
|||
|
|||
// 初始化SDK并创建recognizer |
|||
fun initialize(subscriptionKey: String, region: String, |
|||
supportedLanguages: Array<String> = arrayOf("zh-CN", "en-US")): Boolean { |
|||
try { |
|||
FileLogger.d(TAG, "初始化 Azure 语音服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if (subscriptionKey.isEmpty() || region.isEmpty()) { |
|||
FileLogger.e(TAG, "Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
this.subscriptionKey = subscriptionKey |
|||
this.region = region |
|||
|
|||
// 设置语言 |
|||
if (supportedLanguages.isNotEmpty()) { |
|||
this.supportedLanguages = supportedLanguages |
|||
} |
|||
|
|||
// 根据支持的语言数量决定是否启用自动语言检测 |
|||
this.isAutoDetectLanguage = supportedLanguages.size >= 2 |
|||
|
|||
// 如果只有一种语言,设置为当前语言 |
|||
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { |
|||
this.currentLanguage = supportedLanguages[0] |
|||
} |
|||
|
|||
// 创建语音配置 |
|||
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region) |
|||
|
|||
// 设置语言配置 |
|||
if (isAutoDetectLanguage) { |
|||
// 设置自动语言检测 |
|||
speechConfig?.setProperty(PropertyId.SpeechServiceConnection_LanguageIdMode, "Continuous") |
|||
} else { |
|||
// 设置指定的识别语言 |
|||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|||
} |
|||
|
|||
// 创建识别器 |
|||
try { |
|||
if (useEchoCancellation) { |
|||
// 如果使用回音消除,创建自定义音频输入流 |
|||
setupCustomAudioProcessing() |
|||
|
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
} |
|||
} else { |
|||
// 使用默认麦克风输入 |
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig) |
|||
} |
|||
} |
|||
|
|||
FileLogger.d(TAG, "Azure 语音服务初始化成功") |
|||
return true |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建识别器失败: ${e.message}") |
|||
stopCustomAudioProcessing() |
|||
return false |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "初始化失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
private fun resetRecognizer(): Boolean { |
|||
try { |
|||
// 释放之前的 recognizer |
|||
recognizer?.close() |
|||
recognizer = null |
|||
|
|||
// 停止当前的音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
// 使用现有配置重新创建 recognizer |
|||
if (speechConfig != null) { |
|||
if (useEchoCancellation) { |
|||
// 如果使用回音消除,创建自定义音频输入流 |
|||
setupCustomAudioProcessing() |
|||
|
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
} |
|||
} else { |
|||
// 使用默认麦克风输入 |
|||
if (isAutoDetectLanguage) { |
|||
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|||
recognizer = SpeechRecognizer(speechConfig, autoDetectConfig) |
|||
} else { |
|||
recognizer = SpeechRecognizer(speechConfig) |
|||
} |
|||
} |
|||
return true |
|||
} else { |
|||
FileLogger.e(TAG, "语音配置未初始化") |
|||
return false |
|||
} |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "重置识别器失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 开始一次性语音识别 |
|||
fun recognizeOnce(callback: RecognizeCallback) { |
|||
if (speechConfig == null) { |
|||
callback.onError("语音服务未初始化") |
|||
return |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if (!resetRecognizer()) { |
|||
callback.onError("重置识别器失败") |
|||
return |
|||
} |
|||
|
|||
try { |
|||
// 启动音频处理 |
|||
startCustomAudioProcessing() |
|||
|
|||
// 执行识别 |
|||
val result = recognizer?.recognizeOnceAsync()?.get() |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
|||
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language |
|||
callback.onResult(result.text, detectedLanguage ?: "") |
|||
} else { |
|||
callback.onError("未能识别语音") |
|||
} |
|||
} catch (e: Exception) { |
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
callback.onError("识别异常: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 开始连续语音识别 |
|||
fun startContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|||
if (speechConfig == null) { |
|||
callback.onError("语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
if (isContinuousRecognitionActive) { |
|||
FileLogger.w(TAG, "已在进行连续识别,忽略请求") |
|||
return true |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if (!resetRecognizer()) { |
|||
callback.onError("重置识别器失败") |
|||
return false |
|||
} |
|||
|
|||
try { |
|||
// 启动音频处理 |
|||
startCustomAudioProcessing() |
|||
|
|||
// 识别中事件 |
|||
recognizer?.recognizing?.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language |
|||
FileLogger.d(TAG, "识别中: ${event.result.text}, 语言: $detectedLanguage") |
|||
callback.onRecognizing(event.result.text, detectedLanguage ?: "") |
|||
} |
|||
) |
|||
|
|||
// 识别完成事件 |
|||
recognizer?.recognized?.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
if (event.result.reason == ResultReason.RecognizedSpeech) { |
|||
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language |
|||
FileLogger.d(TAG, "识别完成: ${event.result.text}, 语言: $detectedLanguage") |
|||
callback.onResult(event.result.text, detectedLanguage ?: "") |
|||
} |
|||
} |
|||
) |
|||
|
|||
// 会话开始事件 |
|||
recognizer?.sessionStarted?.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
FileLogger.d(TAG, "识别会话已开始") |
|||
callback.onSessionStarted() |
|||
callback.onSuccess("开始识别") // 兼容旧接口 |
|||
} |
|||
) |
|||
|
|||
// 会话结束事件 |
|||
recognizer?.sessionStopped?.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
FileLogger.d(TAG, "识别会话已结束") |
|||
isContinuousRecognitionActive = false |
|||
stopCustomAudioProcessing() |
|||
callback.onSessionStopped() |
|||
} |
|||
) |
|||
|
|||
// 取消事件 |
|||
recognizer?.canceled?.addEventListener( |
|||
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
|||
val errorDetails = try { |
|||
event.errorDetails ?: "未知错误" |
|||
} catch (e: Exception) { |
|||
"未知错误" |
|||
} |
|||
val reason = event.reason.toString() |
|||
FileLogger.e(TAG, "识别取消: $errorDetails") |
|||
isContinuousRecognitionActive = false |
|||
stopCustomAudioProcessing() |
|||
callback.onCanceled(reason, errorDetails) |
|||
callback.onError("识别取消: $errorDetails") // 兼容旧接口 |
|||
} |
|||
) |
|||
|
|||
// 开始连续识别 |
|||
recognizer?.startContinuousRecognitionAsync() |
|||
isContinuousRecognitionActive = true |
|||
FileLogger.d(TAG, "连续识别已启动") |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
FileLogger.e(TAG, "开始连续识别失败: ${e.message}") |
|||
e.printStackTrace() |
|||
callback.onError("开始连续识别失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 停止连续语音识别 |
|||
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|||
if (speechConfig == null) { |
|||
callback.onError("语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
if (!isContinuousRecognitionActive) { |
|||
FileLogger.d(TAG, "未进行连续识别,忽略停止请求") |
|||
return true |
|||
} |
|||
|
|||
try { |
|||
FileLogger.d(TAG, "停止连续语音识别") |
|||
|
|||
if (recognizer == null) { |
|||
if (isContinuousRecognitionActive) { |
|||
FileLogger.w(TAG, "识别器为空,但状态显示活跃") |
|||
} |
|||
isContinuousRecognitionActive = false |
|||
return true |
|||
} |
|||
|
|||
// 停止连续识别 |
|||
val future = recognizer?.stopContinuousRecognitionAsync() |
|||
future?.get() |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
isContinuousRecognitionActive = false |
|||
FileLogger.d(TAG, "连续识别已停止") |
|||
callback.onSuccess("连续识别已停止") |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
// 强制重置状态 |
|||
isContinuousRecognitionActive = false |
|||
FileLogger.e(TAG, "停止连续识别失败: ${e.message}") |
|||
e.printStackTrace() |
|||
callback.onError("停止连续识别失败: ${e.message}") |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
// 尝试强制关闭识别器 |
|||
try { |
|||
recognizer?.close() |
|||
recognizer = null |
|||
} catch (ex: Exception) { |
|||
FileLogger.e(TAG, "关闭识别器失败: ${ex.message}") |
|||
} |
|||
|
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 释放资源 |
|||
fun dispose() { |
|||
try { |
|||
// 如果正在进行连续识别,先停止 |
|||
if (isContinuousRecognitionActive) { |
|||
recognizer?.stopContinuousRecognitionAsync()?.get() |
|||
isContinuousRecognitionActive = false |
|||
} |
|||
|
|||
// 停止音频处理 |
|||
stopCustomAudioProcessing() |
|||
|
|||
// 释放recognizer |
|||
recognizer?.close() |
|||
recognizer = null |
|||
|
|||
// 释放speechConfig |
|||
speechConfig?.close() |
|||
speechConfig = null |
|||
|
|||
FileLogger.d(TAG, "资源已释放") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "释放资源失败: ${e.message}") |
|||
|
|||
// 确保状态被重置 |
|||
isContinuousRecognitionActive = false |
|||
customAudioProcessor = null |
|||
pushStream = null |
|||
audioConfig = null |
|||
recognizer = null |
|||
speechConfig = null |
|||
} |
|||
} |
|||
|
|||
// 检查连续识别是否处于活跃状态 |
|||
fun isContinuousRecognitionActive(): Boolean { |
|||
return isContinuousRecognitionActive |
|||
} |
|||
|
|||
// 设置自定义音频处理 |
|||
private fun setupCustomAudioProcessing() { |
|||
try { |
|||
// 创建音频推送流 |
|||
pushStream = PushAudioInputStream.create() |
|||
|
|||
// 创建音频配置 |
|||
audioConfig = AudioConfig.fromStreamInput(pushStream) |
|||
|
|||
// 创建自定义音频处理器 |
|||
customAudioProcessor = CustomAudioProcessor(pushStream!!) |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "设置自定义音频处理失败: ${e.message}") |
|||
e.printStackTrace() |
|||
} |
|||
} |
|||
|
|||
// 启动自定义音频处理 |
|||
private fun startCustomAudioProcessing() { |
|||
if (useEchoCancellation && customAudioProcessor != null) { |
|||
try { |
|||
customAudioProcessor?.startProcessing() |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "启动音频处理器失败") |
|||
e.printStackTrace() |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 停止自定义音频处理 |
|||
private fun stopCustomAudioProcessing() { |
|||
if (customAudioProcessor != null) { |
|||
try { |
|||
customAudioProcessor?.stopProcessing() |
|||
customAudioProcessor = null |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "停止音频处理器失败: ${e.message}") |
|||
e.printStackTrace() |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 修改认证取消事件处理代码 |
|||
private fun setupCancelledEventHandler(callback: ContinuousRecognizeCallback) { |
|||
recognizer?.canceled?.addEventListener( |
|||
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
|||
val errorDetails = try { |
|||
event.errorDetails ?: "未知错误" |
|||
} catch (e: Exception) { |
|||
"未知错误" |
|||
} |
|||
val reason = event.reason.toString() |
|||
FileLogger.e(TAG, "识别取消: $errorDetails") |
|||
isContinuousRecognitionActive = false |
|||
stopCustomAudioProcessing() |
|||
callback.onCanceled(reason, errorDetails) |
|||
callback.onError("识别取消: $errorDetails") // 兼容旧接口 |
|||
} |
|||
) |
|||
} |
|||
|
|||
// 自定义音频处理器 |
|||
private inner class CustomAudioProcessor(private val pushStream: PushAudioInputStream) { |
|||
private val isProcessing = AtomicBoolean(false) |
|||
private var audioRecord: AudioRecord? = null |
|||
private var echoCanceler: AcousticEchoCanceler? = null |
|||
private var noiseSuppressor: NoiseSuppressor? = null |
|||
private var automaticGainControl: AutomaticGainControl? = null |
|||
|
|||
// 音频配置 |
|||
private val sampleRate = 16000 // 16kHz,适合语音识别 |
|||
private val channelConfig = AudioFormat.CHANNEL_IN_MONO |
|||
private val audioFormat = AudioFormat.ENCODING_PCM_16BIT |
|||
|
|||
// 计算最小 buffer 大小 |
|||
private val bufferSize = AudioRecord.getMinBufferSize( |
|||
sampleRate, channelConfig, audioFormat |
|||
) |
|||
|
|||
// 启动音频处理 |
|||
fun startProcessing() { |
|||
if (isProcessing.get()) return |
|||
|
|||
// 创建录音对象 |
|||
try { |
|||
audioRecord = AudioRecord( |
|||
MediaRecorder.AudioSource.VOICE_RECOGNITION, |
|||
sampleRate, |
|||
channelConfig, |
|||
audioFormat, |
|||
bufferSize * 2 // 使用更大的缓冲区以确保不会丢失数据 |
|||
) |
|||
|
|||
// 创建音频处理效果 |
|||
if (AcousticEchoCanceler.isAvailable()) { |
|||
echoCanceler = AcousticEchoCanceler.create(audioRecord!!.audioSessionId) |
|||
echoCanceler?.enabled = true |
|||
} |
|||
|
|||
if (NoiseSuppressor.isAvailable()) { |
|||
noiseSuppressor = NoiseSuppressor.create(audioRecord!!.audioSessionId) |
|||
noiseSuppressor?.enabled = true |
|||
} |
|||
|
|||
if (AutomaticGainControl.isAvailable()) { |
|||
automaticGainControl = AutomaticGainControl.create(audioRecord!!.audioSessionId) |
|||
automaticGainControl?.enabled = true |
|||
} |
|||
|
|||
// 开始录音 |
|||
audioRecord?.startRecording() |
|||
|
|||
// 处理线程 |
|||
Thread { |
|||
android.os.Process.setThreadPriority(Process.THREAD_PRIORITY_AUDIO) |
|||
processAudio() |
|||
}.start() |
|||
|
|||
isProcessing.set(true) |
|||
FileLogger.d(TAG, "音频处理已启动") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "创建音频处理器失败: ${e.message}") |
|||
releaseAudioResources() |
|||
throw e |
|||
} |
|||
} |
|||
|
|||
// 停止音频处理 |
|||
fun stopProcessing() { |
|||
if (!isProcessing.get()) return |
|||
|
|||
isProcessing.set(false) |
|||
releaseAudioResources() |
|||
FileLogger.d(TAG, "音频处理已停止") |
|||
} |
|||
|
|||
// 释放音频资源 |
|||
private fun releaseAudioResources() { |
|||
try { |
|||
audioRecord?.stop() |
|||
|
|||
echoCanceler?.release() |
|||
echoCanceler = null |
|||
|
|||
noiseSuppressor?.release() |
|||
noiseSuppressor = null |
|||
|
|||
automaticGainControl?.release() |
|||
automaticGainControl = null |
|||
|
|||
audioRecord?.release() |
|||
audioRecord = null |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "释放音频资源失败: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 音频处理线程 |
|||
private fun processAudio() { |
|||
// 设置线程优先级 |
|||
try { |
|||
Process.setThreadPriority(Process.THREAD_PRIORITY_URGENT_AUDIO) |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "设置线程优先级失败") |
|||
} |
|||
|
|||
val buffer = ByteArray(bufferSize) |
|||
|
|||
while (isProcessing.get()) { |
|||
try { |
|||
val readSize = audioRecord?.read(buffer, 0, buffer.size) ?: -1 |
|||
|
|||
if (readSize > 0) { |
|||
// 修复: 只传入buffer,不传readSize |
|||
// 创建新的byte数组,只包含读取到的数据 |
|||
val audioData = buffer.copyOfRange(0, readSize) |
|||
pushStream.write(audioData) |
|||
} |
|||
|
|||
// 适当休眠,避免占用过多 CPU |
|||
Thread.sleep(5) |
|||
} catch (e: Exception) { |
|||
if (isProcessing.get()) { |
|||
FileLogger.e(TAG, "处理音频数据异常: ${e.message}") |
|||
} |
|||
break |
|||
} |
|||
} |
|||
} |
|||
} |
|||
|
|||
// 一次性识别回调接口 |
|||
interface RecognizeCallback { |
|||
fun onResult(text: String, detectedLanguage: String) |
|||
fun onError(error: String) |
|||
} |
|||
|
|||
// 连续识别回调接口 |
|||
interface ContinuousRecognizeCallback { |
|||
fun onResult(text: String, detectedLanguage: String) |
|||
fun onRecognizing(recognizing: String, detectedLanguage: String) |
|||
fun onSessionStarted() |
|||
fun onSessionStopped() |
|||
fun onCanceled(reason: String, errorDetails: String) |
|||
fun onError(error: String) |
|||
|
|||
// 兼容旧版本的接口 |
|||
fun onSuccess(message: String) {} |
|||
} |
|||
} |
|||
@ -0,0 +1,297 @@ |
|||
package com.yunqiinnovation.azure_speech |
|||
|
|||
import android.content.Context |
|||
import android.app.Activity |
|||
import android.os.Handler |
|||
import android.os.Looper |
|||
import androidx.annotation.NonNull |
|||
import com.yunqiinnovation.azure_speech.utils.FileLogger |
|||
|
|||
import io.flutter.embedding.engine.plugins.FlutterPlugin |
|||
import io.flutter.plugin.common.MethodCall |
|||
import io.flutter.plugin.common.MethodChannel |
|||
import io.flutter.plugin.common.MethodChannel.MethodCallHandler |
|||
import io.flutter.plugin.common.MethodChannel.Result |
|||
import io.flutter.plugin.common.EventChannel |
|||
|
|||
/** AzureSpeechPlugin */ |
|||
class AzureSpeechPlugin: FlutterPlugin { |
|||
private val TAG = "AzureSpeechPlugin" |
|||
private lateinit var context: Context |
|||
private val mainHandler = Handler(Looper.getMainLooper()) |
|||
|
|||
// ASR相关 |
|||
private lateinit var asrChannel : MethodChannel |
|||
private lateinit var asrEventChannel: EventChannel |
|||
private var asrEventSink: EventChannel.EventSink? = null |
|||
private lateinit var azureAsrHelper: AzureAsrHelper |
|||
|
|||
// TTS相关 |
|||
private lateinit var ttsChannel : MethodChannel |
|||
private lateinit var azureTtsHelper: AzureTtsHelper |
|||
|
|||
// ASR 事件发送方法 |
|||
private fun sendAsrEvent(event: Map<String, Any>) { |
|||
FileLogger.d(TAG, "发送ASR事件: $event") |
|||
if (asrEventSink == null) { |
|||
FileLogger.w(TAG, "无法发送ASR事件:事件通道未准备好") |
|||
return |
|||
} |
|||
|
|||
mainHandler.post { |
|||
try { |
|||
asrEventSink?.success(event) |
|||
FileLogger.d(TAG, "ASR事件发送成功") |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "发送ASR事件失败: ${e.message}") |
|||
} |
|||
} |
|||
} |
|||
|
|||
override fun onAttachedToEngine(@NonNull flutterPluginBinding: FlutterPlugin.FlutterPluginBinding) { |
|||
context = flutterPluginBinding.applicationContext |
|||
|
|||
// 初始化ASR通道 |
|||
asrChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr") |
|||
asrChannel.setMethodCallHandler(AsrMethodHandler()) |
|||
|
|||
// 初始化TTS通道 |
|||
ttsChannel = MethodChannel(flutterPluginBinding.binaryMessenger, "azure_speech/tts") |
|||
ttsChannel.setMethodCallHandler(TtsMethodHandler()) |
|||
|
|||
// 初始化ASR事件通道 |
|||
asrEventChannel = EventChannel(flutterPluginBinding.binaryMessenger, "azure_speech/asr_events") |
|||
asrEventChannel.setStreamHandler(object : EventChannel.StreamHandler { |
|||
override fun onListen(arguments: Any?, events: EventChannel.EventSink?) { |
|||
asrEventSink = events |
|||
} |
|||
|
|||
override fun onCancel(arguments: Any?) { |
|||
asrEventSink = null |
|||
} |
|||
}) |
|||
|
|||
// 初始化Azure语音服务 |
|||
azureTtsHelper = AzureTtsHelper(context) |
|||
azureAsrHelper = AzureAsrHelper(context) |
|||
} |
|||
|
|||
// ASR方法处理器 |
|||
inner class AsrMethodHandler : MethodCallHandler { |
|||
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
|||
when (call.method) { |
|||
"initialize" -> { |
|||
val subscriptionKey = call.argument<String>("subscriptionKey") ?: "" |
|||
val region = call.argument<String>("region") ?: "" |
|||
val supportedLanguages = call.argument<List<String>>("supportedLanguages") ?: listOf("zh-CN") |
|||
|
|||
try { |
|||
val success = azureAsrHelper.initialize(subscriptionKey, region, supportedLanguages.toTypedArray()) |
|||
result.success(success) |
|||
} catch (e: Exception) { |
|||
result.error("INITIALIZATION_ERROR", e.message, null) |
|||
} |
|||
} |
|||
"recognizeOnce" -> { |
|||
azureAsrHelper.recognizeOnce(object : AzureAsrHelper.RecognizeCallback { |
|||
override fun onResult(text: String, detectedLanguage: String) { |
|||
mainHandler.post { |
|||
result.success(mapOf( |
|||
"text" to text, |
|||
"detectedLanguage" to detectedLanguage |
|||
)) |
|||
} |
|||
} |
|||
|
|||
override fun onError(error: String) { |
|||
mainHandler.post { |
|||
result.error("RECOGNITION_ERROR", error, null) |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
"startContinuousRecognition" -> { |
|||
// 确保事件通道已准备好 |
|||
if (asrEventSink == null) { |
|||
result.error("EVENT_CHANNEL_NOT_READY", "事件通道未准备好,无法开始连续识别", null) |
|||
return |
|||
} |
|||
|
|||
val success = azureAsrHelper.startContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
|||
override fun onResult(text: String, detectedLanguage: String) { |
|||
sendAsrEvent(mapOf( |
|||
"type" to "result", |
|||
"text" to text, |
|||
"detectedLanguage" to detectedLanguage |
|||
)) |
|||
} |
|||
|
|||
override fun onRecognizing(recognizing: String, detectedLanguage: String) { |
|||
sendAsrEvent(mapOf( |
|||
"type" to "recognizing", |
|||
"text" to recognizing, |
|||
"detectedLanguage" to detectedLanguage |
|||
)) |
|||
} |
|||
|
|||
override fun onSessionStarted() { |
|||
sendAsrEvent(mapOf("type" to "sessionStarted")) |
|||
} |
|||
|
|||
override fun onSessionStopped() { |
|||
sendAsrEvent(mapOf("type" to "sessionStopped")) |
|||
} |
|||
|
|||
override fun onCanceled(reason: String, errorDetails: String) { |
|||
sendAsrEvent(mapOf( |
|||
"type" to "canceled", |
|||
"reason" to reason, |
|||
"errorDetails" to errorDetails |
|||
)) |
|||
} |
|||
|
|||
override fun onError(error: String) { |
|||
sendAsrEvent(mapOf("type" to "error", "message" to error)) |
|||
} |
|||
|
|||
override fun onSuccess(message: String) { |
|||
sendAsrEvent(mapOf("type" to "success", "message" to message)) |
|||
} |
|||
}) |
|||
result.success(success) |
|||
} |
|||
"stopContinuousRecognition" -> { |
|||
try { |
|||
if (!azureAsrHelper.isContinuousRecognitionActive()) { |
|||
result.success(true) |
|||
return |
|||
} |
|||
|
|||
val success = azureAsrHelper.stopContinuousRecognition(object : AzureAsrHelper.ContinuousRecognizeCallback { |
|||
override fun onResult(text: String, detectedLanguage: String) {} |
|||
override fun onRecognizing(recognizing: String, detectedLanguage: String) {} |
|||
override fun onSessionStarted() {} |
|||
override fun onSessionStopped() {} |
|||
override fun onCanceled(reason: String, errorDetails: String) {} |
|||
override fun onError(error: String) { |
|||
mainHandler.post { |
|||
result.error("STOP_ERROR", error, null) |
|||
} |
|||
} |
|||
override fun onSuccess(message: String) {} |
|||
}) |
|||
result.success(success) |
|||
} catch (e: Exception) { |
|||
result.error("STOP_ERROR", e.message, null) |
|||
} |
|||
} |
|||
"isContinuousRecognitionActive" -> { |
|||
result.success(azureAsrHelper.isContinuousRecognitionActive()) |
|||
} |
|||
"dispose" -> { |
|||
azureAsrHelper.dispose() |
|||
result.success(true) |
|||
} |
|||
else -> { |
|||
result.notImplemented() |
|||
} |
|||
} |
|||
} |
|||
} |
|||
|
|||
// TTS方法处理器 |
|||
inner class TtsMethodHandler : MethodCallHandler { |
|||
override fun onMethodCall(@NonNull call: MethodCall, @NonNull result: Result) { |
|||
when (call.method) { |
|||
"initialize" -> { |
|||
val subscriptionKey = call.argument<String>("subscriptionKey") ?: "" |
|||
val region = call.argument<String>("region") ?: "" |
|||
val language = call.argument<String>("language") ?: "zh-CN" |
|||
|
|||
val success = azureTtsHelper.initialize(subscriptionKey, region, language) |
|||
result.success(success) |
|||
} |
|||
"setVoice" -> { |
|||
val voiceName = call.argument<String>("voiceName") ?: return result.error("INVALID_ARGUMENTS", "语音名称不能为空", null) |
|||
result.success(azureTtsHelper.setVoice(voiceName)) |
|||
} |
|||
"setSpeechParams" -> { |
|||
val rate = call.argument<Int>("rate") ?: 0 |
|||
val pitch = call.argument<Int>("pitch") ?: 0 |
|||
val volume = call.argument<Int>("volume") ?: 100 |
|||
result.success(azureTtsHelper.setSpeechParams(rate, pitch, volume)) |
|||
} |
|||
"setAudioOutputType" -> { |
|||
val outputTypeStr = call.argument<String>("outputType") ?: "speaker" |
|||
val outputType = when (outputTypeStr.lowercase()) { |
|||
"speaker" -> AzureTtsHelper.AudioOutputType.SPEAKER |
|||
"earpiece" -> AzureTtsHelper.AudioOutputType.EARPIECE |
|||
"auto" -> AzureTtsHelper.AudioOutputType.AUTO |
|||
else -> AzureTtsHelper.AudioOutputType.SPEAKER |
|||
} |
|||
result.success(azureTtsHelper.setAudioOutputType(outputType)) |
|||
} |
|||
"speakText" -> { |
|||
val text = call.argument<String>("text") ?: return result.error("INVALID_ARGUMENTS", "文本不能为空", null) |
|||
|
|||
azureTtsHelper.speakText(text, object : AzureTtsHelper.TTSCallback { |
|||
override fun onSuccess(message: String) { |
|||
mainHandler.post { |
|||
result.success(true) |
|||
} |
|||
} |
|||
|
|||
override fun onError(error: String) { |
|||
mainHandler.post { |
|||
result.error("SPEAK_ERROR", error, null) |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
"speakSsml" -> { |
|||
val ssml = call.argument<String>("ssml") ?: return result.error("INVALID_ARGUMENTS", "SSML不能为空", null) |
|||
|
|||
azureTtsHelper.speakSsml(ssml, object : AzureTtsHelper.TTSCallback { |
|||
override fun onSuccess(message: String) { |
|||
mainHandler.post { |
|||
result.success(true) |
|||
} |
|||
} |
|||
|
|||
override fun onError(error: String) { |
|||
mainHandler.post { |
|||
result.error("SPEAK_ERROR", error, null) |
|||
} |
|||
} |
|||
}) |
|||
} |
|||
"stopSpeaking" -> { |
|||
result.success(azureTtsHelper.stopSpeaking()) |
|||
} |
|||
"isSpeaking" -> { |
|||
result.success(azureTtsHelper.isSpeaking()) |
|||
} |
|||
"dispose" -> { |
|||
azureTtsHelper.dispose() |
|||
result.success(true) |
|||
} |
|||
else -> { |
|||
result.notImplemented() |
|||
} |
|||
} |
|||
} |
|||
} |
|||
|
|||
override fun onDetachedFromEngine(@NonNull binding: FlutterPlugin.FlutterPluginBinding) { |
|||
asrChannel.setMethodCallHandler(null) |
|||
ttsChannel.setMethodCallHandler(null) |
|||
asrEventChannel.setStreamHandler(null) |
|||
|
|||
try { |
|||
azureTtsHelper.dispose() |
|||
azureAsrHelper.dispose() |
|||
} catch (e: Exception) { |
|||
FileLogger.e(TAG, "Dispose resources error: ${e.message}") |
|||
} |
|||
} |
|||
} |
|||
@ -0,0 +1,45 @@ |
|||
package com.yunqiinnovation.azure_speech.utils |
|||
|
|||
import android.util.Log |
|||
|
|||
/** |
|||
* 简单的文件日志工具类 |
|||
*/ |
|||
object FileLogger { |
|||
private const val TAG_PREFIX = "AzureSpeech_" |
|||
|
|||
/** |
|||
* 记录调试信息 |
|||
*/ |
|||
fun d(tag: String, message: String) { |
|||
Log.d("$TAG_PREFIX$tag", message) |
|||
} |
|||
|
|||
/** |
|||
* 记录信息 |
|||
*/ |
|||
fun i(tag: String, message: String) { |
|||
Log.i("$TAG_PREFIX$tag", message) |
|||
} |
|||
|
|||
/** |
|||
* 记录警告信息 |
|||
*/ |
|||
fun w(tag: String, message: String) { |
|||
Log.w("$TAG_PREFIX$tag", message) |
|||
} |
|||
|
|||
/** |
|||
* 记录错误信息 |
|||
*/ |
|||
fun e(tag: String, message: String) { |
|||
Log.e("$TAG_PREFIX$tag", message) |
|||
} |
|||
|
|||
/** |
|||
* 记录异常 |
|||
*/ |
|||
fun e(tag: String, message: String, throwable: Throwable) { |
|||
Log.e("$TAG_PREFIX$tag", message, throwable) |
|||
} |
|||
} |
|||
@ -0,0 +1,461 @@ |
|||
import Foundation |
|||
import AVFoundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
|
|||
/// Azure 语音识别辅助类 |
|||
class AzureAsrHelper: NSObject { |
|||
private var recognizer: SPXSpeechRecognizer? |
|||
private var speechConfig: SPXSpeechConfig? |
|||
private var audioConfig: SPXAudioConfig? |
|||
private var initialized = false |
|||
private var isContinuousRecognitionActive = false |
|||
private var currentLanguage = "zh-CN" |
|||
private var subscriptionKey = "" |
|||
private var serviceRegion = "" |
|||
private var isAutoDetectLanguage = false |
|||
private var supportedLanguages = ["zh-CN", "en-US"] |
|||
|
|||
// 音频会话管理 |
|||
private let audioSession = AVAudioSession.sharedInstance() |
|||
|
|||
// 事件回调 |
|||
private var eventHandler: (([String: Any]) -> Void)? |
|||
|
|||
/// 设置事件处理器 |
|||
/// |
|||
/// - Parameter handler: 事件处理回调 |
|||
func setEventHandler(_ handler: @escaping ([String: Any]) -> Void) { |
|||
self.eventHandler = handler |
|||
} |
|||
|
|||
/// 初始化语音识别服务 |
|||
/// |
|||
/// - Parameters: |
|||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|||
/// - serviceRegion: Azure 语音服务区域 |
|||
/// - supportedLanguages: 支持的语言列表,默认为 ["zh-CN", "en-US"] |
|||
/// - Returns: 是否初始化成功 |
|||
func initialize(speechSubscriptionKey: String, serviceRegion: String, supportedLanguages: [String] = ["zh-CN", "en-US"]) -> Bool { |
|||
print("[AzureAsrHelper] 初始化 Azure 语音服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
// 保存配置 |
|||
self.subscriptionKey = speechSubscriptionKey |
|||
self.serviceRegion = serviceRegion |
|||
|
|||
// 设置语言 |
|||
if supportedLanguages.isEmpty { |
|||
print("[AzureAsrHelper] 警告: 传入的支持语言列表为空,将使用默认语言") |
|||
} else { |
|||
self.supportedLanguages = supportedLanguages |
|||
} |
|||
|
|||
// 根据支持的语言数量决定是否启用自动语言检测 |
|||
self.isAutoDetectLanguage = supportedLanguages.count >= 2 |
|||
|
|||
// 如果只有一种语言,设置为当前语言 |
|||
if !isAutoDetectLanguage && !supportedLanguages.isEmpty { |
|||
self.currentLanguage = supportedLanguages[0] |
|||
} |
|||
|
|||
// 创建语音配置 |
|||
do { |
|||
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) |
|||
|
|||
// 设置语言配置 |
|||
if isAutoDetectLanguage { |
|||
// 设置自动语言检测 |
|||
try speechConfig?.setPropertyTo("Continuous", byId: SPXPropertyId.SpeechServiceConnection_LanguageIdMode) |
|||
} else { |
|||
// 设置指定的识别语言 |
|||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|||
} |
|||
|
|||
// 创建音频配置 - 使用默认麦克风 |
|||
audioConfig = SPXAudioConfig.default() |
|||
|
|||
// 创建识别器 |
|||
if isAutoDetectLanguage { |
|||
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) |
|||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) |
|||
} else { |
|||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|||
} |
|||
|
|||
// 配置音频会话 |
|||
try configureAudioSession() |
|||
|
|||
initialized = true |
|||
print("[AzureAsrHelper] Azure 语音服务初始化成功") |
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 创建识别器失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 配置音频会话 |
|||
private func configureAudioSession() throws { |
|||
print("[AzureAsrHelper] 开始配置音频会话...") |
|||
|
|||
do { |
|||
// 设置音频会话类别和模式 |
|||
try audioSession.setCategory(.record, mode: .measurement, options: [.duckOthers, .allowBluetooth]) |
|||
try audioSession.setActive(true, options: .notifyOthersOnDeactivation) |
|||
} catch { |
|||
print("[AzureAsrHelper] 警告: 通过AudioSessionManager配置音频会话失败") |
|||
throw error |
|||
} |
|||
} |
|||
|
|||
/// 一次性语音识别 |
|||
/// |
|||
/// - Parameter completion: 完成回调,返回是否成功、识别文本、识别语言和可能的错误信息 |
|||
func recognizeOnce(completion: @escaping (Bool, String?, String?, String?) -> Void) { |
|||
if !initialized { |
|||
completion(false, nil, nil, "语音服务未初始化") |
|||
return |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if !resetRecognizer() { |
|||
completion(false, nil, nil, "重置识别器失败") |
|||
return |
|||
} |
|||
|
|||
do { |
|||
// 激活音频会话 |
|||
try audioSession.setActive(true) |
|||
|
|||
// 添加识别事件处理 |
|||
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == .recognizedSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
print("[AzureAsrHelper] 识别结果: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|||
completion(true, event.result.text, detectedLanguage, nil) |
|||
} |
|||
} |
|||
|
|||
recognizer?.addRecognizingEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == .recognizingSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
// print("[AzureAsrHelper] 识别中: \(event.result.text ?? ""), 语言: \(detectedLanguage)") |
|||
} |
|||
} |
|||
|
|||
// 添加会话事件处理 |
|||
recognizer?.addSessionStartedEventHandler { _, _ in |
|||
print("[AzureAsrHelper] 识别会话已开始") |
|||
} |
|||
|
|||
recognizer?.addSessionStoppedEventHandler { _, _ in |
|||
print("[AzureAsrHelper] 识别会话已结束") |
|||
} |
|||
|
|||
// 添加取消事件处理 |
|||
recognizer?.addCanceledEventHandler { _, event in |
|||
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { |
|||
let errorDetails = cancellationDetails.errorDetails ?? "未知错误" |
|||
print("[AzureAsrHelper] 识别取消: \(errorDetails)") |
|||
completion(false, nil, nil, "识别取消: \(errorDetails)") |
|||
} |
|||
} |
|||
|
|||
// 执行识别 |
|||
let result = try recognizer?.recognizeOnceAsync().get() |
|||
|
|||
if result?.reason != .recognizedSpeech { |
|||
completion(false, nil, nil, "未能识别语音") |
|||
} |
|||
} catch { |
|||
completion(false, nil, nil, "识别异常: \(error.localizedDescription)") |
|||
} |
|||
} |
|||
|
|||
/// 重置识别器 |
|||
/// |
|||
/// - Returns: 是否重置成功 |
|||
private func resetRecognizer() -> Bool { |
|||
if !initialized { |
|||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
// 检查配置是否为空 |
|||
if subscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 释放之前的 recognizer |
|||
recognizer = nil |
|||
|
|||
// 创建音频配置 - 使用默认麦克风 |
|||
audioConfig = SPXAudioConfig.default() |
|||
|
|||
// 重新创建识别器 |
|||
if isAutoDetectLanguage { |
|||
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguages: supportedLanguages) |
|||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, autoDetectSourceLanguageConfiguration: autoDetectConfig, audioConfiguration: audioConfig!) |
|||
} else { |
|||
recognizer = try SPXSpeechRecognizer(speechConfiguration: speechConfig!, audioConfiguration: audioConfig!) |
|||
} |
|||
|
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 重置识别器失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 获取检测到的语言 |
|||
/// |
|||
/// - Parameter result: 识别结果 |
|||
/// - Returns: 检测到的语言代码 |
|||
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|||
if isAutoDetectLanguage { |
|||
do { |
|||
if let autoDetectResult = try SPXAutoDetectSourceLanguageResult(fromRecognitionResult: result) { |
|||
return autoDetectResult.language |
|||
} |
|||
return "" |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 获取检测到的语言失败: \(error.localizedDescription)") |
|||
return "" |
|||
} |
|||
} else { |
|||
return currentLanguage |
|||
} |
|||
} |
|||
|
|||
/// 开始连续语音识别 |
|||
/// |
|||
/// - Returns: 是否成功启动连续识别 |
|||
func startContinuousRecognition() -> Bool { |
|||
if !initialized { |
|||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
// 检查配置是否为空 |
|||
if subscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureAsrHelper] 错误: Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
// 如果已经在进行连续识别,直接返回 |
|||
if isContinuousRecognitionActive { |
|||
print("[AzureAsrHelper] 已经在进行连续识别中,忽略请求") |
|||
return true |
|||
} |
|||
|
|||
// 重置 recognizer |
|||
if !resetRecognizer() { |
|||
print("[AzureAsrHelper] 尝试重新创建识别器...") |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 激活音频会话 |
|||
try audioSession.setActive(true) |
|||
|
|||
// 添加识别事件处理 |
|||
recognizer?.addRecognizedEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == .recognizedSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "finalResult", |
|||
"text": event.result.text ?? "", |
|||
"language": detectedLanguage |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
} |
|||
|
|||
// 识别中事件 |
|||
recognizer?.addRecognizingEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
if event.result.reason == .recognizingSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: event.result) |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "recognizing", |
|||
"text": event.result.text ?? "", |
|||
"language": detectedLanguage |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
} |
|||
|
|||
// 会话事件 |
|||
recognizer?.addSessionStartedEventHandler { [weak self] _, _ in |
|||
guard let self = self else { return } |
|||
|
|||
let eventData: [String: Any] = [ |
|||
"eventType": "sessionStarted" |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
|
|||
recognizer?.addSessionStoppedEventHandler { [weak self] _, _ in |
|||
guard let self = self else { return } |
|||
|
|||
self.isContinuousRecognitionActive = false |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "sessionStopped" |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
|
|||
// 取消事件 |
|||
recognizer?.addCanceledEventHandler { [weak self] _, event in |
|||
guard let self = self else { return } |
|||
|
|||
self.isContinuousRecognitionActive = false |
|||
var errorMessage = "未知错误" |
|||
|
|||
if let cancellationDetails = try? SPXCancellationDetails(fromCanceledRecognitionResult: event.result) { |
|||
errorMessage = cancellationDetails.errorDetails ?? "未知错误" |
|||
} |
|||
|
|||
let eventData: [String: Any] = [ |
|||
"eventType": "error", |
|||
"error": "识别取消: \(errorMessage)" |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
|
|||
// 开始连续识别 |
|||
try recognizer?.startContinuousRecognition() |
|||
isContinuousRecognitionActive = true |
|||
print("[AzureAsrHelper] 连续识别已启动") |
|||
|
|||
return true |
|||
} catch { |
|||
print("[AzureAsrHelper] 错误: 开始连续识别失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 停止连续语音识别 |
|||
/// |
|||
/// - Returns: 是否成功停止连续识别 |
|||
func stopContinuousRecognition() -> Bool { |
|||
if !initialized { |
|||
print("[AzureAsrHelper] 错误: 语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
if !isContinuousRecognitionActive { |
|||
print("[AzureAsrHelper] 未进行连续识别,忽略停止请求") |
|||
return true |
|||
} |
|||
|
|||
do { |
|||
print("[AzureAsrHelper] 停止连续语音识别") |
|||
|
|||
if recognizer == nil { |
|||
if isContinuousRecognitionActive { |
|||
print("[AzureAsrHelper] 警告: 识别器为空,但状态显示活跃") |
|||
} |
|||
isContinuousRecognitionActive = false |
|||
|
|||
// 通知停止成功 |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "success", |
|||
"message": "连续识别已停止" |
|||
] |
|||
eventHandler?(eventData) |
|||
return true |
|||
} |
|||
|
|||
// 停止连续识别 |
|||
try recognizer?.stopContinuousRecognition() |
|||
|
|||
// 延迟一点时间确保处理完成 |
|||
DispatchQueue.main.asyncAfter(deadline: .now() + 0.5) { [weak self] in |
|||
guard let self = self else { return } |
|||
|
|||
// 重置状态 |
|||
self.isContinuousRecognitionActive = false |
|||
|
|||
// 恢复音频会话 |
|||
do { |
|||
try self.audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
|||
} catch { |
|||
// 忽略错误 |
|||
} |
|||
|
|||
print("[AzureAsrHelper] 连续识别已停止") |
|||
|
|||
// 通知停止成功 |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "success", |
|||
"message": "连续识别已停止" |
|||
] |
|||
self.eventHandler?(eventData) |
|||
} |
|||
|
|||
return true |
|||
} catch { |
|||
// 强制重置状态 |
|||
isContinuousRecognitionActive = false |
|||
print("[AzureAsrHelper] 警告: 停止连续识别失败: \(error.localizedDescription)") |
|||
|
|||
// 通知停止失败,但仍然视为处理完成 |
|||
let eventData: [String: Any] = [ |
|||
"eventType": "success", |
|||
"message": "连续识别已停止(但有错误)" |
|||
] |
|||
eventHandler?(eventData) |
|||
|
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 释放资源 |
|||
func dispose() { |
|||
// 如果正在进行连续识别,先停止 |
|||
if isContinuousRecognitionActive { |
|||
_ = stopContinuousRecognition() |
|||
} |
|||
|
|||
// 恢复音频会话 |
|||
do { |
|||
try audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
|||
} catch { |
|||
// 忽略错误 |
|||
} |
|||
|
|||
// 释放资源 |
|||
recognizer = nil |
|||
speechConfig = nil |
|||
audioConfig = nil |
|||
|
|||
initialized = false |
|||
isContinuousRecognitionActive = false |
|||
print("[AzureAsrHelper] 资源已释放") |
|||
} |
|||
|
|||
/// 检查连续识别是否处于活跃状态 |
|||
/// |
|||
/// - Returns: 是否正在进行连续识别 |
|||
func isContinuousRecognitionActive() -> Bool { |
|||
return isContinuousRecognitionActive |
|||
} |
|||
} |
|||
@ -0,0 +1,182 @@ |
|||
import Flutter |
|||
import UIKit |
|||
|
|||
public class AzureSpeechPlugin: NSObject, FlutterPlugin { |
|||
private var ttsHelper: AzureTtsHelper? |
|||
private var asrHelper: AzureAsrHelper? |
|||
private var eventSink: FlutterEventSink? |
|||
|
|||
public static func register(with registrar: FlutterPluginRegistrar) { |
|||
let channel = FlutterMethodChannel(name: "azure_speech", binaryMessenger: registrar.messenger()) |
|||
let instance = AzureSpeechPlugin() |
|||
registrar.addMethodCallDelegate(instance, channel: channel) |
|||
|
|||
// 初始化事件通道 |
|||
let eventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) |
|||
eventChannel.setStreamHandler(AsrStreamHandler(instance: instance)) |
|||
} |
|||
|
|||
override init() { |
|||
super.init() |
|||
ttsHelper = AzureTtsHelper() |
|||
asrHelper = AzureAsrHelper() |
|||
|
|||
// 设置ASR事件处理 |
|||
asrHelper?.setEventHandler { [weak self] event in |
|||
self?.handleAsrEvent(event) |
|||
} |
|||
} |
|||
|
|||
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|||
switch call.method { |
|||
// TTS相关方法 |
|||
case "initializeTts": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let subscriptionKey = args["subscriptionKey"] as? String, |
|||
let serviceRegion = args["serviceRegion"] as? String else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
let language = args["language"] as? String ?? "zh-CN" |
|||
let success = ttsHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, language: language) ?? false |
|||
result(success) |
|||
|
|||
case "setTtsVoice": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let voiceName = args["voiceName"] as? String else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
let success = ttsHelper?.setVoice(voiceName: voiceName) ?? false |
|||
result(success) |
|||
|
|||
case "setTtsSpeechParams": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let rate = args["rate"] as? Int, |
|||
let pitch = args["pitch"] as? Int, |
|||
let volume = args["volume"] as? Int else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
let success = ttsHelper?.setSpeechParams(rate: rate, pitch: pitch, volume: volume) ?? false |
|||
result(success) |
|||
|
|||
case "speakText": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let text = args["text"] as? String else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
ttsHelper?.speakText(text: text) { success, _ in |
|||
result(success) |
|||
} |
|||
|
|||
case "stopSpeaking": |
|||
let success = ttsHelper?.stopSpeaking() ?? false |
|||
result(success) |
|||
|
|||
case "isSpeaking": |
|||
let speaking = ttsHelper?.isSpeaking() ?? false |
|||
result(speaking) |
|||
|
|||
case "setTtsAudioOutputType": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let outputType = args["outputType"] as? String else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
var type: AzureTtsHelper.AudioOutputType = .auto |
|||
switch outputType.uppercased() { |
|||
case "SPEAKER": |
|||
type = .speaker |
|||
case "EARPIECE": |
|||
type = .earpiece |
|||
default: |
|||
type = .auto |
|||
} |
|||
|
|||
let success = ttsHelper?.setAudioOutputType(outputType: type) ?? false |
|||
result(success) |
|||
|
|||
// ASR相关方法 |
|||
case "initializeAsr": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let subscriptionKey = args["subscriptionKey"] as? String, |
|||
let serviceRegion = args["serviceRegion"] as? String, |
|||
let supportedLanguages = args["supportedLanguages"] as? [String] else { |
|||
result(false) |
|||
return |
|||
} |
|||
|
|||
let success = asrHelper?.initialize(speechSubscriptionKey: subscriptionKey, serviceRegion: serviceRegion, supportedLanguages: supportedLanguages) ?? false |
|||
result(success) |
|||
|
|||
case "recognizeOnce": |
|||
asrHelper?.recognizeOnce { success, text, language, error in |
|||
var resultMap: [String: Any] = ["success": success] |
|||
if success { |
|||
resultMap["text"] = text |
|||
resultMap["language"] = language |
|||
} else { |
|||
resultMap["error"] = error |
|||
} |
|||
result(resultMap) |
|||
} |
|||
|
|||
case "startContinuousRecognition": |
|||
let success = asrHelper?.startContinuousRecognition() ?? false |
|||
result(success) |
|||
|
|||
case "stopContinuousRecognition": |
|||
let success = asrHelper?.stopContinuousRecognition() ?? false |
|||
result(success) |
|||
|
|||
case "isContinuousRecognitionActive": |
|||
let isActive = asrHelper?.isContinuousRecognitionActive() ?? false |
|||
result(isActive) |
|||
|
|||
case "dispose": |
|||
ttsHelper?.dispose() |
|||
asrHelper?.dispose() |
|||
result(nil) |
|||
|
|||
default: |
|||
result(FlutterMethodNotImplemented) |
|||
} |
|||
} |
|||
|
|||
// 设置事件接收器 |
|||
func setEventSink(_ sink: FlutterEventSink?) { |
|||
self.eventSink = sink |
|||
} |
|||
|
|||
// 处理ASR事件 |
|||
private func handleAsrEvent(_ event: [String: Any]) { |
|||
self.eventSink?(event) |
|||
} |
|||
} |
|||
|
|||
// ASR事件流处理器 |
|||
class AsrStreamHandler: NSObject, FlutterStreamHandler { |
|||
private weak var plugin: AzureSpeechPlugin? |
|||
|
|||
init(instance: AzureSpeechPlugin) { |
|||
self.plugin = instance |
|||
super.init() |
|||
} |
|||
|
|||
func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
|||
plugin?.setEventSink(events) |
|||
return nil |
|||
} |
|||
|
|||
func onCancel(withArguments arguments: Any?) -> FlutterError? { |
|||
plugin?.setEventSink(nil) |
|||
return nil |
|||
} |
|||
} |
|||
@ -0,0 +1,330 @@ |
|||
import Foundation |
|||
import AVFoundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
|
|||
/// Azure 语音合成辅助类 |
|||
class AzureTtsHelper: NSObject { |
|||
private var synthesizer: SPXSpeechSynthesizer? |
|||
private var speechConfig: SPXSpeechConfig? |
|||
private var audioConfig: SPXAudioConfig? |
|||
private var initialized = false |
|||
private var speaking = false |
|||
|
|||
// 音频输出类型 |
|||
enum AudioOutputType { |
|||
case speaker // 扬声器 |
|||
case earpiece // 听筒 |
|||
case auto // 自动选择 |
|||
} |
|||
|
|||
// 当前设置 |
|||
private var currentVoiceName = "zh-CN-XiaoxiaoNeural" |
|||
private var currentSpeechRate = 0 |
|||
private var currentPitch = 0 |
|||
private var currentVolume = 100 |
|||
private var currentAudioOutputType: AudioOutputType = .auto |
|||
|
|||
// 音频会话管理 |
|||
private let audioSession = AVAudioSession.sharedInstance() |
|||
|
|||
/// 初始化 TTS 引擎 |
|||
/// |
|||
/// - Parameters: |
|||
/// - speechSubscriptionKey: Azure 语音服务订阅密钥 |
|||
/// - serviceRegion: Azure 语音服务区域 |
|||
/// - language: 可选,默认语言,默认为 "zh-CN" |
|||
/// - Returns: 是否初始化成功 |
|||
func initialize(speechSubscriptionKey: String, serviceRegion: String, language: String = "zh-CN") -> Bool { |
|||
print("[AzureTtsHelper] 初始化 Azure 语音服务") |
|||
|
|||
// 检查配置是否为空 |
|||
if speechSubscriptionKey.isEmpty || serviceRegion.isEmpty { |
|||
print("[AzureTtsHelper] 错误: Azure 配置信息不完整") |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
do { |
|||
// 创建语音配置 |
|||
speechConfig = try SPXSpeechConfig(subscription: speechSubscriptionKey, region: serviceRegion) |
|||
|
|||
// 设置语音合成输出格式为高质量音频 |
|||
speechConfig?.setSpeechSynthesisOutputFormat(.riff24Khz16BitMonoPcm) |
|||
|
|||
// 设置默认语言 |
|||
speechConfig?.setSpeechSynthesisLanguage(language) |
|||
|
|||
// 设置默认语音 |
|||
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) |
|||
|
|||
// 创建音频配置 - 使用默认扬声器 |
|||
audioConfig = SPXAudioConfig.default() |
|||
|
|||
// 创建语音合成器 |
|||
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) |
|||
|
|||
initialized = true |
|||
|
|||
// 设置默认音频输出类型为自动 |
|||
setAudioOutputType(outputType: .auto) |
|||
|
|||
print("[AzureTtsHelper] TTS 引擎初始化成功") |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] TTS 引擎初始化失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 设置音频输出设备类型 |
|||
/// |
|||
/// - Parameter outputType: 音频输出设备类型 |
|||
/// - Returns: 是否设置成功 |
|||
func setAudioOutputType(outputType: AudioOutputType) -> Bool { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
currentAudioOutputType = outputType |
|||
|
|||
switch outputType { |
|||
case .speaker: |
|||
// 使用扬声器 |
|||
try audioSession.setCategory(.playback, mode: .default) |
|||
try audioSession.overrideOutputAudioPort(.speaker) |
|||
print("[AzureTtsHelper] 已设置音频输出设备为扬声器") |
|||
|
|||
case .earpiece: |
|||
// 使用听筒 |
|||
try audioSession.setCategory(.playback, mode: .voiceChat) |
|||
try audioSession.overrideOutputAudioPort(.none) |
|||
print("[AzureTtsHelper] 已设置音频输出设备为听筒") |
|||
|
|||
case .auto: |
|||
// 检查是否有耳机连接 |
|||
let outputs = audioSession.currentRoute.outputs |
|||
let hasHeadphones = outputs.contains { output in |
|||
return output.portType == .headphones || output.portType == .bluetoothA2DP || output.portType == .bluetoothHFP |
|||
} |
|||
|
|||
if hasHeadphones { |
|||
// 有耳机,使用耳机 |
|||
try audioSession.setCategory(.playback, mode: .default) |
|||
try audioSession.overrideOutputAudioPort(.none) |
|||
print("[AzureTtsHelper] 已设置音频输出设备为耳机") |
|||
} else { |
|||
// 无耳机,使用听筒 |
|||
try audioSession.setCategory(.playback, mode: .voiceChat) |
|||
try audioSession.overrideOutputAudioPort(.none) |
|||
print("[AzureTtsHelper] 已设置音频输出设备为听筒") |
|||
} |
|||
} |
|||
|
|||
try audioSession.setActive(true) |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 设置音频输出设备失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 设置语音 |
|||
/// |
|||
/// - Parameter voiceName: 语音名称,例如 "zh-CN-XiaoxiaoNeural" |
|||
/// - Returns: 是否设置成功 |
|||
func setVoice(voiceName: String) -> Bool { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
if voiceName == currentVoiceName { |
|||
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
|||
return true |
|||
} |
|||
|
|||
do { |
|||
currentVoiceName = voiceName |
|||
speechConfig?.setSpeechSynthesisVoiceName(voiceName) |
|||
|
|||
// 重新创建合成器 |
|||
synthesizer = try SPXSpeechSynthesizer(speechConfig: speechConfig!, audioConfig: audioConfig!) |
|||
|
|||
print("[AzureTtsHelper] 已设置语音: \(voiceName)") |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 设置语音失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 设置语音合成参数 |
|||
/// |
|||
/// - Parameters: |
|||
/// - rate: 语速,范围 -100 到 100,默认为 0 |
|||
/// - pitch: 音调,范围 -100 到 100,默认为 0 |
|||
/// - volume: 音量,范围 0 到 100,默认为 100 |
|||
/// - Returns: 是否设置成功 |
|||
func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
currentSpeechRate = rate |
|||
currentPitch = pitch |
|||
currentVolume = volume |
|||
|
|||
print("[AzureTtsHelper] 已设置语音参数: 语速=\(rate), 音调=\(pitch), 音量=\(volume)") |
|||
return true |
|||
} |
|||
|
|||
/// 合成文本为语音并播放 |
|||
/// |
|||
/// - Parameters: |
|||
/// - text: 要合成的文本 |
|||
/// - completion: 完成回调,返回是否成功和可能的错误信息 |
|||
func speakText(text: String, completion: @escaping (Bool, String?) -> Void) { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
completion(false, "TTS 引擎尚未初始化") |
|||
return |
|||
} |
|||
|
|||
do { |
|||
print("[AzureTtsHelper] 开始合成文本: \(text)") |
|||
|
|||
// 生成 SSML |
|||
let ssml = generateSsml(text: text) |
|||
|
|||
// 使用 SSML 合成语音 |
|||
speakSsml(ssml: ssml, completion: completion) |
|||
} catch { |
|||
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") |
|||
completion(false, "语音合成异常: \(error.localizedDescription)") |
|||
} |
|||
} |
|||
|
|||
/// 生成 SSML 文本 |
|||
/// |
|||
/// - Parameter text: 要转换的文本 |
|||
/// - Returns: SSML 格式的文本 |
|||
private func generateSsml(text: String) -> String { |
|||
// 计算 SSML 参数 |
|||
let rateParam = currentSpeechRate == 0 ? "0%" : (currentSpeechRate < 0 ? "\(Int(Double(currentSpeechRate) * 0.9))%" : "\(currentSpeechRate)%") |
|||
let pitchParam = currentPitch == 0 ? "0%" : "\(Int(Double(currentPitch) * 0.5))%" |
|||
let volumeParam = "\(min(max(currentVolume, 0), 100))%" |
|||
|
|||
return """ |
|||
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN"> |
|||
<voice name="\(currentVoiceName)"> |
|||
<prosody rate="\(rateParam)" pitch="\(pitchParam)" volume="\(volumeParam)"> |
|||
\(text) |
|||
</prosody> |
|||
</voice> |
|||
</speak> |
|||
""" |
|||
} |
|||
|
|||
/// 合成 SSML 为语音并播放 |
|||
/// |
|||
/// - Parameters: |
|||
/// - ssml: SSML 格式的文本 |
|||
/// - completion: 完成回调,返回是否成功和可能的错误信息 |
|||
private func speakSsml(ssml: String, completion: @escaping (Bool, String?) -> Void) { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
completion(false, "TTS 引擎尚未初始化") |
|||
return |
|||
} |
|||
|
|||
do { |
|||
print("[AzureTtsHelper] 开始合成 SSML") |
|||
|
|||
// 标记为正在播放 |
|||
speaking = true |
|||
|
|||
// 激活音频会话 |
|||
try audioSession.setActive(true) |
|||
|
|||
// 异步合成语音 |
|||
let result = try synthesizer!.speakSsml(ssml) |
|||
|
|||
switch result.reason { |
|||
case .synthesizingAudioCompleted: |
|||
print("[AzureTtsHelper] 语音合成完成") |
|||
speaking = false |
|||
completion(true, "语音合成完成") |
|||
case .canceled: |
|||
if let cancelDetails = try? SPXSpeechSynthesisCancellationDetails(fromResult: result) { |
|||
print("[AzureTtsHelper] 语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") |
|||
speaking = false |
|||
completion(false, "语音合成取消: \(cancelDetails.errorDetails ?? "未知错误")") |
|||
} else { |
|||
print("[AzureTtsHelper] 语音合成取消") |
|||
speaking = false |
|||
completion(false, "语音合成取消") |
|||
} |
|||
default: |
|||
print("[AzureTtsHelper] 语音合成失败: \(result.reason)") |
|||
speaking = false |
|||
completion(false, "语音合成失败: \(result.reason)") |
|||
} |
|||
} catch { |
|||
print("[AzureTtsHelper] 语音合成异常: \(error.localizedDescription)") |
|||
speaking = false |
|||
completion(false, "语音合成异常: \(error.localizedDescription)") |
|||
} |
|||
} |
|||
|
|||
/// 停止当前语音合成 |
|||
/// |
|||
/// - Returns: 是否停止成功 |
|||
func stopSpeaking() -> Bool { |
|||
if !initialized { |
|||
print("[AzureTtsHelper] TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
try synthesizer?.stopSpeaking() |
|||
speaking = false |
|||
print("[AzureTtsHelper] 已停止语音合成") |
|||
return true |
|||
} catch { |
|||
print("[AzureTtsHelper] 停止语音合成失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/// 释放资源 |
|||
func dispose() { |
|||
do { |
|||
stopSpeaking() |
|||
|
|||
// 恢复音频会话 |
|||
try audioSession.setActive(false, options: .notifyOthersOnDeactivation) |
|||
|
|||
synthesizer = nil |
|||
speechConfig = nil |
|||
audioConfig = nil |
|||
|
|||
initialized = false |
|||
speaking = false |
|||
print("[AzureTtsHelper] TTS 引擎已释放") |
|||
} catch { |
|||
print("[AzureTtsHelper] 释放 TTS 引擎失败: \(error.localizedDescription)") |
|||
} |
|||
} |
|||
|
|||
/// 检查当前是否正在播放语音 |
|||
/// |
|||
/// - Returns: 是否正在播放语音 |
|||
func isSpeaking() -> Bool { |
|||
return speaking |
|||
} |
|||
} |
|||
@ -0,0 +1,32 @@ |
|||
name: azure_speech |
|||
description: Azure语音服务插件,包含TTS和ASR服务 |
|||
version: 0.0.1 |
|||
homepage: |
|||
|
|||
environment: |
|||
sdk: ">=2.17.0 <3.0.0" |
|||
flutter: ">=2.5.0" |
|||
|
|||
dependencies: |
|||
flutter: |
|||
sdk: flutter |
|||
|
|||
dev_dependencies: |
|||
flutter_test: |
|||
sdk: flutter |
|||
flutter_lints: ^2.0.0 |
|||
|
|||
# For information on the generic Dart part of this file, see the |
|||
# following page: https://dart.dev/tools/pub/pubspec |
|||
|
|||
# The following section is specific to Flutter packages. |
|||
flutter: |
|||
# This section identifies this Flutter project as a plugin project. |
|||
plugin: |
|||
platforms: |
|||
android: |
|||
package: com.yunqiinnovation.azure_speech |
|||
pluginClass: AzureSpeechPlugin |
|||
ios: |
|||
pluginClass: AzureSpeechPlugin |
|||
|
|||
Loading…
Reference in new issue