4 changed files with 697 additions and 87 deletions
@ -0,0 +1,218 @@ |
|||
package com.example.deep_voice |
|||
|
|||
import android.util.Log |
|||
import com.microsoft.cognitiveservices.speech.* |
|||
import com.microsoft.cognitiveservices.speech.audio.* |
|||
import com.microsoft.cognitiveservices.speech.util.EventHandler |
|||
import java.util.concurrent.ExecutionException |
|||
|
|||
class AzureAsrHelper { |
|||
private var recognizer: SpeechRecognizer? = null |
|||
private var speechConfig: SpeechConfig? = null |
|||
private var audioConfig: AudioConfig? = null |
|||
private val TAG = "AzureAsrHelper" |
|||
private var isContinuousRecognitionActive = false |
|||
private var currentLanguage = "zh-CN" |
|||
private var subscriptionKey = "" |
|||
private var serviceRegion = "" |
|||
|
|||
// 初始化 SDK |
|||
fun initialize(subscriptionKey: String, serviceRegion: String, language: String = "zh-CN"): Boolean { |
|||
try { |
|||
this.subscriptionKey = subscriptionKey |
|||
this.serviceRegion = serviceRegion |
|||
this.currentLanguage = language |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
// 创建语音配置 |
|||
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) |
|||
speechConfig?.speechRecognitionLanguage = language |
|||
|
|||
// 创建音频配置 |
|||
audioConfig = AudioConfig.fromDefaultMicrophoneInput() |
|||
|
|||
// 创建识别器 |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
return true |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "初始化失败: ${e.message}") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 开始一次性语音识别 |
|||
fun recognizeOnce(language: String? = null, callback: RecognizeCallback) { |
|||
if (recognizer == null) { |
|||
callback.onError("SpeechRecognizer 未初始化") |
|||
return |
|||
} |
|||
|
|||
// 如果指定了新的语言,需要重新初始化 |
|||
if (language != null && language != currentLanguage) { |
|||
initialize(subscriptionKey, serviceRegion, language) |
|||
} |
|||
|
|||
try { |
|||
val result = recognizer?.recognizeOnceAsync()?.get() |
|||
|
|||
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
|||
callback.onResult(result.text) |
|||
} else { |
|||
callback.onError("未能识别语音") |
|||
} |
|||
} catch (e: Exception) { |
|||
callback.onError("识别异常: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
// 开始连续语音识别 |
|||
fun startContinuousRecognition(language: String? = null, callback: ContinuousRecognizeCallback): Boolean { |
|||
if (recognizer == null) { |
|||
callback.onError("SpeechRecognizer 未初始化") |
|||
return false |
|||
} |
|||
|
|||
// 如果已经在进行连续识别,先停止 |
|||
if (isContinuousRecognitionActive) { |
|||
stopContinuousRecognition(callback) |
|||
} |
|||
|
|||
// 如果指定了新的语言,需要重新初始化 |
|||
if (language != null && language != currentLanguage) { |
|||
initialize(subscriptionKey, serviceRegion, language) |
|||
} |
|||
|
|||
try { |
|||
// 先发送会话开始事件 |
|||
callback.onSessionStarted() |
|||
|
|||
// 重新创建识别器 |
|||
recognizer?.close() |
|||
recognizer = SpeechRecognizer(speechConfig, audioConfig) |
|||
|
|||
// 设置识别事件处理 |
|||
recognizer?.let { recognizer -> |
|||
// 最终识别结果 |
|||
recognizer.recognized.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
if (event.result.reason == ResultReason.RecognizedSpeech) { |
|||
callback.onResult(event.result.text) |
|||
} |
|||
} |
|||
) |
|||
|
|||
// 识别中事件 |
|||
recognizer.recognizing.addEventListener( |
|||
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|||
if (event.result.reason == ResultReason.RecognizingSpeech) { |
|||
callback.onRecognizing(event.result.text) |
|||
} |
|||
} |
|||
) |
|||
|
|||
// 会话开始事件 |
|||
recognizer.sessionStarted.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
isContinuousRecognitionActive = true |
|||
callback.onSessionStarted() |
|||
} |
|||
) |
|||
|
|||
// 会话结束事件 |
|||
recognizer.sessionStopped.addEventListener( |
|||
EventHandler<SessionEventArgs> { _, _ -> |
|||
isContinuousRecognitionActive = false |
|||
callback.onSessionStopped() |
|||
} |
|||
) |
|||
|
|||
// 取消事件 |
|||
recognizer.canceled.addEventListener( |
|||
EventHandler<SpeechRecognitionCanceledEventArgs> { _, event -> |
|||
val errorDetails = if (event.reason == CancellationReason.Error) event.errorDetails else "" |
|||
callback.onCanceled(event.reason.toString(), errorDetails) |
|||
isContinuousRecognitionActive = false |
|||
} |
|||
) |
|||
|
|||
// 开始连续识别 |
|||
recognizer.startContinuousRecognitionAsync().get() |
|||
isContinuousRecognitionActive = true |
|||
} |
|||
|
|||
return true |
|||
} catch (e: Exception) { |
|||
callback.onError("开始连续识别失败: ${e.message}") |
|||
isContinuousRecognitionActive = false |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 停止连续语音识别 |
|||
fun stopContinuousRecognition(callback: ContinuousRecognizeCallback): Boolean { |
|||
if (!isContinuousRecognitionActive || recognizer == null) { |
|||
return true |
|||
} |
|||
|
|||
try { |
|||
recognizer?.stopContinuousRecognitionAsync()?.get() |
|||
isContinuousRecognitionActive = false |
|||
callback.onSessionStopped() |
|||
return true |
|||
} catch (e: Exception) { |
|||
callback.onError("停止连续识别失败: ${e.message}") |
|||
isContinuousRecognitionActive = false |
|||
return false |
|||
} |
|||
} |
|||
|
|||
// 检查连续识别是否活跃 |
|||
fun isContinuousRecognitionActive(): Boolean { |
|||
return isContinuousRecognitionActive |
|||
} |
|||
|
|||
// 获取当前语言 |
|||
fun getCurrentLanguage(): String { |
|||
return currentLanguage |
|||
} |
|||
|
|||
// 释放资源 |
|||
fun dispose() { |
|||
try { |
|||
if (isContinuousRecognitionActive && recognizer != null) { |
|||
recognizer?.stopContinuousRecognitionAsync()?.get() |
|||
} |
|||
|
|||
recognizer?.close() |
|||
recognizer = null |
|||
|
|||
audioConfig?.close() |
|||
audioConfig = null |
|||
|
|||
speechConfig?.close() |
|||
speechConfig = null |
|||
|
|||
isContinuousRecognitionActive = false |
|||
} catch (e: Exception) { |
|||
// 忽略异常 |
|||
} |
|||
} |
|||
|
|||
// 一次性识别回调接口 |
|||
interface RecognizeCallback { |
|||
fun onResult(result: String) |
|||
fun onError(error: String) |
|||
} |
|||
|
|||
// 连续识别回调接口 |
|||
interface ContinuousRecognizeCallback { |
|||
fun onResult(result: String) |
|||
fun onRecognizing(recognizing: String) |
|||
fun onSessionStarted() |
|||
fun onSessionStopped() |
|||
fun onCanceled(reason: String, errorDetails: String) |
|||
fun onError(error: String) |
|||
} |
|||
} |
|||
@ -0,0 +1,279 @@ |
|||
package com.example.deep_voice |
|||
|
|||
import android.util.Log |
|||
import com.microsoft.cognitiveservices.speech.* |
|||
import com.microsoft.cognitiveservices.speech.audio.* |
|||
import java.util.concurrent.Future |
|||
import java.util.concurrent.Semaphore |
|||
|
|||
/** |
|||
* Microsoft Text-to-Speech Helper |
|||
* |
|||
* 该类封装了微软语音 SDK 的 TTS 功能,提供简单的接口供 Flutter 调用 |
|||
* 基于微软官方文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis |
|||
*/ |
|||
class AzureTtsHelper { |
|||
private val TAG = "AzureTtsHelper" |
|||
private var speechConfig: SpeechConfig? = null |
|||
private var synthesizer: SpeechSynthesizer? = null |
|||
private var isInitialized = false |
|||
|
|||
// 用于异步操作的信号量 |
|||
private val semaphore = Semaphore(0) |
|||
|
|||
// 当前语音设置 |
|||
private var currentVoiceName = "zh-CN-XiaoxiaoNeural" |
|||
private var currentSpeechRate = "0%" |
|||
private var currentPitch = "0%" |
|||
private var currentVolume = "100%" |
|||
|
|||
/** |
|||
* 初始化 TTS 引擎 |
|||
* |
|||
* @param subscriptionKey Azure 语音服务订阅密钥 |
|||
* @param serviceRegion Azure 语音服务区域 |
|||
* @param language 可选,默认语言,默认为 "zh-CN" |
|||
*/ |
|||
fun initialize(subscriptionKey: String, serviceRegion: String, language: String = "zh-CN"): Boolean { |
|||
try { |
|||
// 创建语音配置 |
|||
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, serviceRegion) |
|||
|
|||
// 设置语音合成输出格式为高质量音频 |
|||
speechConfig?.setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff24Khz16BitMonoPcm) |
|||
|
|||
// 设置默认语言 |
|||
speechConfig?.setSpeechSynthesisLanguage(language) |
|||
|
|||
// 设置默认语音 |
|||
speechConfig?.setSpeechSynthesisVoiceName(currentVoiceName) |
|||
|
|||
// 创建音频配置 - 使用默认扬声器 |
|||
val audioConfig = AudioConfig.fromDefaultSpeakerOutput() |
|||
|
|||
// 创建语音合成器 |
|||
synthesizer = SpeechSynthesizer(speechConfig, audioConfig) |
|||
|
|||
// 设置事件监听 |
|||
synthesizer?.SynthesisCompleted?.addEventListener { _, _ -> semaphore.release() } |
|||
synthesizer?.SynthesisCanceled?.addEventListener { _, _ -> semaphore.release() } |
|||
|
|||
isInitialized = true |
|||
Log.d(TAG, "TTS 引擎初始化成功") |
|||
return true |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "TTS 引擎初始化失败: ${e.message}") |
|||
e.printStackTrace() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置语音 |
|||
* |
|||
* @param voiceName 语音名称,例如 "zh-CN-XiaoxiaoNeural" |
|||
*/ |
|||
fun setVoice(voiceName: String): Boolean { |
|||
if (!isInitialized) { |
|||
Log.e(TAG, "TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
try { |
|||
currentVoiceName = voiceName |
|||
speechConfig?.setSpeechSynthesisVoiceName(voiceName) |
|||
|
|||
// 重新创建合成器以应用新的语音设置 |
|||
val audioConfig = AudioConfig.fromDefaultSpeakerOutput() |
|||
synthesizer?.close() |
|||
synthesizer = SpeechSynthesizer(speechConfig, audioConfig) |
|||
|
|||
// 重新设置事件监听 |
|||
synthesizer?.SynthesisCompleted?.addEventListener { _, _ -> semaphore.release() } |
|||
synthesizer?.SynthesisCanceled?.addEventListener { _, _ -> semaphore.release() } |
|||
|
|||
Log.d(TAG, "已设置语音: $voiceName") |
|||
return true |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "设置语音失败: ${e.message}") |
|||
e.printStackTrace() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置语音合成参数 |
|||
* |
|||
* @param rate 语速,范围 -100 到 100,默认为 0 |
|||
* @param pitch 音调,范围 -100 到 100,默认为 0 |
|||
* @param volume 音量,范围 0 到 100,默认为 100 |
|||
*/ |
|||
fun setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100): Boolean { |
|||
if (!isInitialized) { |
|||
Log.e(TAG, "TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
try { |
|||
currentSpeechRate = if (rate == 0) "0%" else if (rate < 0) "${(rate * 0.9).toInt()}%" else "$rate%" |
|||
currentPitch = if (pitch == 0) "0%" else "${(pitch * 0.5).toInt()}%" |
|||
currentVolume = "${volume.coerceIn(0, 100)}%" |
|||
|
|||
Log.d(TAG, "已设置语音参数: 语速=$currentSpeechRate, 音调=$currentPitch, 音量=$currentVolume") |
|||
return true |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "设置语音参数失败: ${e.message}") |
|||
e.printStackTrace() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 合成文本为语音并播放 |
|||
* |
|||
* @param text 要合成的文本 |
|||
* @param callback 回调接口,用于返回结果或错误 |
|||
*/ |
|||
fun speakText(text: String, callback: TTSCallback) { |
|||
if (!isInitialized) { |
|||
callback.onError("TTS 引擎尚未初始化") |
|||
return |
|||
} |
|||
|
|||
try { |
|||
Log.d(TAG, "开始合成文本: $text") |
|||
|
|||
// 生成 SSML |
|||
val ssml = generateSsml(text) |
|||
|
|||
// 使用 SSML 合成语音 |
|||
speakSsml(ssml, callback) |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "语音合成异常: ${e.message}") |
|||
e.printStackTrace() |
|||
callback.onError("语音合成异常: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 生成 SSML 文本 |
|||
* |
|||
* @param text 要转换的文本 |
|||
* @return SSML 格式的文本 |
|||
*/ |
|||
private fun generateSsml(text: String): String { |
|||
return """ |
|||
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN"> |
|||
<voice name="$currentVoiceName"> |
|||
<prosody rate="$currentSpeechRate" pitch="$currentPitch" volume="$currentVolume"> |
|||
$text |
|||
</prosody> |
|||
</voice> |
|||
</speak> |
|||
""".trimIndent() |
|||
} |
|||
|
|||
/** |
|||
* 合成 SSML 为语音并播放 |
|||
* |
|||
* @param ssml SSML 格式的文本 |
|||
* @param callback 回调接口,用于返回结果或错误 |
|||
*/ |
|||
fun speakSsml(ssml: String, callback: TTSCallback) { |
|||
if (!isInitialized) { |
|||
callback.onError("TTS 引擎尚未初始化") |
|||
return |
|||
} |
|||
|
|||
try { |
|||
Log.d(TAG, "开始合成 SSML") |
|||
|
|||
// 重置信号量 |
|||
semaphore.drainPermits() |
|||
|
|||
// 异步合成语音 |
|||
val task: Future<SpeechSynthesisResult> = synthesizer!!.SpeakSsmlAsync(ssml) |
|||
|
|||
// 等待合成完成 |
|||
Thread { |
|||
try { |
|||
// 等待合成完成信号 |
|||
semaphore.acquire() |
|||
|
|||
// 获取结果 |
|||
val result = task.get() |
|||
|
|||
when (result.reason) { |
|||
ResultReason.SynthesizingAudioCompleted -> { |
|||
Log.d(TAG, "语音合成完成,音频长度: ${result.audioLength / 10000}ms,音频数据大小: ${result.audioData?.size ?: 0} 字节") |
|||
callback.onSuccess("语音合成完成") |
|||
} |
|||
ResultReason.Canceled -> { |
|||
val cancellation = SpeechSynthesisCancellationDetails.fromResult(result) |
|||
Log.e(TAG, "语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") |
|||
callback.onError("语音合成取消: ${cancellation.reason}, ${cancellation.errorDetails}") |
|||
} |
|||
else -> { |
|||
Log.e(TAG, "语音合成失败: ${result.reason}") |
|||
callback.onError("语音合成失败: ${result.reason}") |
|||
} |
|||
} |
|||
|
|||
result.close() |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "处理语音合成结果异常: ${e.message}") |
|||
e.printStackTrace() |
|||
callback.onError("处理语音合成结果异常: ${e.message}") |
|||
} |
|||
}.start() |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "语音合成异常: ${e.message}") |
|||
e.printStackTrace() |
|||
callback.onError("语音合成异常: ${e.message}") |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 停止当前语音合成 |
|||
*/ |
|||
fun stopSpeaking(): Boolean { |
|||
if (!isInitialized) { |
|||
Log.e(TAG, "TTS 引擎尚未初始化") |
|||
return false |
|||
} |
|||
|
|||
try { |
|||
synthesizer?.StopSpeakingAsync() |
|||
Log.d(TAG, "已停止语音合成") |
|||
return true |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "停止语音合成失败: ${e.message}") |
|||
e.printStackTrace() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 释放资源 |
|||
*/ |
|||
fun dispose() { |
|||
try { |
|||
stopSpeaking() |
|||
synthesizer?.close() |
|||
speechConfig?.close() |
|||
isInitialized = false |
|||
Log.d(TAG, "TTS 引擎已释放") |
|||
} catch (e: Exception) { |
|||
Log.e(TAG, "释放 TTS 引擎失败: ${e.message}") |
|||
e.printStackTrace() |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* TTS 回调接口 |
|||
*/ |
|||
interface TTSCallback { |
|||
fun onSuccess(message: String) |
|||
fun onError(error: String) |
|||
} |
|||
} |
|||
Loading…
Reference in new issue