|
|
@ -64,7 +64,7 @@ class IntegratedSpeechTranslationService( |
|
|
private var audioProcessor: AudioProcessor? = null |
|
|
private var audioProcessor: AudioProcessor? = null |
|
|
private var audioConfig: AudioConfig? = null |
|
|
private var audioConfig: AudioConfig? = null |
|
|
// 录音文件处理 |
|
|
// 录音文件处理 |
|
|
var recordfile: RecordFile? = null |
|
|
var recordfile1: RecordFile? = null |
|
|
var filePath: String? = null |
|
|
var filePath: String? = null |
|
|
// 配置管理 |
|
|
// 配置管理 |
|
|
private var serviceConfig = ServiceConfiguration() |
|
|
private var serviceConfig = ServiceConfiguration() |
|
|
@ -91,7 +91,9 @@ class IntegratedSpeechTranslationService( |
|
|
var enableContinuousRecognition: Boolean = true, |
|
|
var enableContinuousRecognition: Boolean = true, |
|
|
var enableAutoLanguageDetection: Boolean = false, |
|
|
var enableAutoLanguageDetection: Boolean = false, |
|
|
var maxRetryAttempts: Int = 3, |
|
|
var maxRetryAttempts: Int = 3, |
|
|
var translationTimeout: Long = 10000L |
|
|
var translationTimeout: Long = 10000L, |
|
|
|
|
|
// 新增:控制是否播放合成的音频 |
|
|
|
|
|
var enableAudioPlayback: Boolean = false |
|
|
) |
|
|
) |
|
|
|
|
|
|
|
|
/** |
|
|
/** |
|
|
@ -122,6 +124,8 @@ class IntegratedSpeechTranslationService( |
|
|
fun onRecognitionStopped() |
|
|
fun onRecognitionStopped() |
|
|
fun onStateChanged(component: String, isActive: Boolean) |
|
|
fun onStateChanged(component: String, isActive: Boolean) |
|
|
fun onError(component: String, error: String) |
|
|
fun onError(component: String, error: String) |
|
|
|
|
|
// 新增:返回合成的音频数据 |
|
|
|
|
|
fun onSynthesisAudioGenerated(text: String, audioData: ByteArray) |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
/** |
|
|
/** |
|
|
@ -190,8 +194,8 @@ class IntegratedSpeechTranslationService( |
|
|
withContext(Dispatchers.Main) { |
|
|
withContext(Dispatchers.Main) { |
|
|
callback.onServiceInitialized() |
|
|
callback.onServiceInitialized() |
|
|
} |
|
|
} |
|
|
startContinuousTranslation() |
|
|
//startContinuousTranslation() |
|
|
recordfile = RecordFile; |
|
|
recordfile1 = RecordFile(); |
|
|
return@withContext true |
|
|
return@withContext true |
|
|
} catch (e: Exception) { |
|
|
} catch (e: Exception) { |
|
|
Log.e(TAG, "初始化失败", e) |
|
|
Log.e(TAG, "初始化失败", e) |
|
|
@ -281,7 +285,7 @@ startContinuousTranslation() |
|
|
recognizing.addEventListener { _, event -> |
|
|
recognizing.addEventListener { _, event -> |
|
|
if (event.result.text.isNotEmpty()) { |
|
|
if (event.result.text.isNotEmpty()) { |
|
|
val confidence = extractConfidence(event.result) |
|
|
val confidence = extractConfidence(event.result) |
|
|
Log.d(TAG, "识别中事件: ${event.result.text}") |
|
|
Log.d(TAG, "识别中事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") |
|
|
eventCallback?.onRecognizing( |
|
|
eventCallback?.onRecognizing( |
|
|
event.result.text, |
|
|
event.result.text, |
|
|
serviceConfig.sourceLanguage, |
|
|
serviceConfig.sourceLanguage, |
|
|
@ -301,7 +305,7 @@ startContinuousTranslation() |
|
|
serviceConfig.sourceLanguage, |
|
|
serviceConfig.sourceLanguage, |
|
|
confidence |
|
|
confidence |
|
|
) |
|
|
) |
|
|
Log.d(TAG, "识别完成事件: ${event.result.text}") |
|
|
Log.d(TAG, "识别完成事件: ${event.result.text},serviceConfig.sourceLanguage:${serviceConfig.sourceLanguage}") |
|
|
// 触发翻译流程 |
|
|
// 触发翻译流程 |
|
|
launch { |
|
|
launch { |
|
|
processTranslationAndSynthesis(event.result.text) |
|
|
processTranslationAndSynthesis(event.result.text) |
|
|
@ -353,8 +357,16 @@ startContinuousTranslation() |
|
|
*/ |
|
|
*/ |
|
|
private fun setupSpeechSynthesizer() { |
|
|
private fun setupSpeechSynthesizer() { |
|
|
synthesizer?.close() |
|
|
synthesizer?.close() |
|
|
synthesizer = |
|
|
|
|
|
SpeechSynthesizer(speechConfig, AudioConfig.fromDefaultSpeakerOutput()).apply { |
|
|
// 根据配置决定音频输出方式 |
|
|
|
|
|
val audioConfig = if (serviceConfig.enableAudioPlayback) { |
|
|
|
|
|
AudioConfig.fromDefaultSpeakerOutput() |
|
|
|
|
|
} else { |
|
|
|
|
|
// 不输出到扬声器,只生成音频数据 |
|
|
|
|
|
null |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
synthesizer = SpeechSynthesizer(speechConfig, audioConfig).apply { |
|
|
|
|
|
|
|
|
// 合成开始事件 |
|
|
// 合成开始事件 |
|
|
SynthesisStarted.addEventListener { _, _ -> |
|
|
SynthesisStarted.addEventListener { _, _ -> |
|
|
@ -374,8 +386,7 @@ startContinuousTranslation() |
|
|
// 合成完成事件 |
|
|
// 合成完成事件 |
|
|
SynthesisCompleted.addEventListener { _, event -> |
|
|
SynthesisCompleted.addEventListener { _, event -> |
|
|
Log.d(TAG, "合成完成事件") |
|
|
Log.d(TAG, "合成完成事件") |
|
|
serviceState.isSynthesizing.set(false) |
|
|
// eventCallback?.onSynthesisCompleted("") |
|
|
eventCallback?.onSynthesisCompleted("") |
|
|
|
|
|
eventCallback?.onStateChanged("Synthesis", false) |
|
|
eventCallback?.onStateChanged("Synthesis", false) |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
@ -389,13 +400,13 @@ startContinuousTranslation() |
|
|
} |
|
|
} |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
Log.d(TAG, "语音合成器设置完成") |
|
|
Log.d(TAG, "语音合成器设置完成,播放模式: ${serviceConfig.enableAudioPlayback}") |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
/** |
|
|
/** |
|
|
* 开始连续语音翻译 |
|
|
* 开始连续语音翻译 |
|
|
*/ |
|
|
*/ |
|
|
suspend fun startContinuousTranslation(): Boolean { |
|
|
fun startContinuousTranslation(): Boolean { |
|
|
if (!serviceState.isInitialized.get()) { |
|
|
if (!serviceState.isInitialized.get()) { |
|
|
eventCallback?.onError("Service", "服务未初始化") |
|
|
eventCallback?.onError("Service", "服务未初始化") |
|
|
return false |
|
|
return false |
|
|
@ -426,10 +437,10 @@ startContinuousTranslation() |
|
|
fun enableRecord(filePath: String) { |
|
|
fun enableRecord(filePath: String) { |
|
|
Log.i(TAG, "开启录音:") |
|
|
Log.i(TAG, "开启录音:") |
|
|
|
|
|
|
|
|
recordfile!!.closeFile(true) |
|
|
recordfile1!!.closeFile(true) |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
recordfile!!.creatingFiles(filePath) |
|
|
recordfile1!!.creatingFiles(filePath) |
|
|
|
|
|
|
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
@ -440,7 +451,7 @@ startContinuousTranslation() |
|
|
try { |
|
|
try { |
|
|
recognizer?.stopContinuousRecognitionAsync() |
|
|
recognizer?.stopContinuousRecognitionAsync() |
|
|
audioProcessor?.stopRecording() |
|
|
audioProcessor?.stopRecording() |
|
|
recordfile!!.closeFile(true) |
|
|
recordfile1!!.closeFile(true) |
|
|
// 等待当前处理完成 |
|
|
// 等待当前处理完成 |
|
|
while (serviceState.isTranslating.get() || serviceState.isSynthesizing.get()) { |
|
|
while (serviceState.isTranslating.get() || serviceState.isSynthesizing.get()) { |
|
|
Thread.sleep(100) |
|
|
Thread.sleep(100) |
|
|
@ -510,11 +521,13 @@ startContinuousTranslation() |
|
|
|
|
|
|
|
|
/** |
|
|
/** |
|
|
* 语音合成 |
|
|
* 语音合成 |
|
|
|
|
|
* @param text 要合成的文本 |
|
|
|
|
|
* @return 返回合成的音频数据,如果合成失败则返回null |
|
|
*/ |
|
|
*/ |
|
|
private suspend fun synthesizeText(text: String) { |
|
|
private suspend fun synthesizeText(text: String): ByteArray? { |
|
|
if (serviceState.isSynthesizing.get()) { |
|
|
if (serviceState.isSynthesizing.get()) { |
|
|
Log.w(TAG, "语音合成正在进行中") |
|
|
Log.w(TAG, "语音合成正在进行中") |
|
|
return |
|
|
return null |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
try { |
|
|
try { |
|
|
@ -524,20 +537,32 @@ startContinuousTranslation() |
|
|
|
|
|
|
|
|
val ssml = generateOptimizedSsml(text) |
|
|
val ssml = generateOptimizedSsml(text) |
|
|
|
|
|
|
|
|
// 使用同步方法确保合成完成后再播放 |
|
|
// 始终使用异步方法获取音频数据 |
|
|
val result = synthesizer?.SpeakSsml(ssml) |
|
|
val result = synthesizer?.SpeakSsmlAsync(ssml)?.get() |
|
|
|
|
|
|
|
|
if (result?.reason == ResultReason.SynthesizingAudioCompleted) { |
|
|
if (result?.reason == ResultReason.SynthesizingAudioCompleted) { |
|
|
Log.d(TAG, "语音合成成功,音频已播放") |
|
|
val playbackStatus = if (serviceConfig.enableAudioPlayback) "音频已播放" else "音频已合成但未播放" |
|
|
eventCallback?.onSynthesisCompleted(text) |
|
|
Log.d(TAG, "语音合成成功,$playbackStatus") |
|
|
|
|
|
|
|
|
|
|
|
// 获取音频数据 |
|
|
|
|
|
val audioData = result.audioData |
|
|
|
|
|
if (audioData != null && audioData.isNotEmpty()) { |
|
|
|
|
|
// 通过回调返回音频数据 |
|
|
|
|
|
eventCallback?.onSynthesisAudioGenerated(text, audioData) |
|
|
|
|
|
Log.d(TAG, "音频数据大小: ${audioData.size} 字节") |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
return audioData |
|
|
} else { |
|
|
} else { |
|
|
Log.e(TAG, "语音合成失败: ${result?.reason}") |
|
|
Log.e(TAG, "语音合成失败: ${result?.reason}") |
|
|
eventCallback?.onSynthesisFailed(text, "合成失败: ${result?.reason}") |
|
|
eventCallback?.onSynthesisFailed(text, "合成失败: ${result?.reason}") |
|
|
|
|
|
return null |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
} catch (e: Exception) { |
|
|
} catch (e: Exception) { |
|
|
Log.e(TAG, "语音合成失败", e) |
|
|
Log.e(TAG, "语音合成失败", e) |
|
|
eventCallback?.onSynthesisFailed(text, "合成失败: ${e.message}") |
|
|
eventCallback?.onSynthesisFailed(text, "合成失败: ${e.message}") |
|
|
|
|
|
return null |
|
|
} finally { |
|
|
} finally { |
|
|
serviceState.isSynthesizing.set(false) |
|
|
serviceState.isSynthesizing.set(false) |
|
|
} |
|
|
} |
|
|
@ -693,6 +718,68 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { |
|
|
Log.d(TAG, "语言对已切换: ${newConfig.sourceLanguage} <-> ${newConfig.targetLanguage}") |
|
|
Log.d(TAG, "语言对已切换: ${newConfig.sourceLanguage} <-> ${newConfig.targetLanguage}") |
|
|
} |
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
|
|
* 设置是否启用音频播放 |
|
|
|
|
|
* @param enabled true表示启用音频播放,false表示只合成不播放 |
|
|
|
|
|
*/ |
|
|
|
|
|
fun setAudioPlaybackEnabled(enabled: Boolean) { |
|
|
|
|
|
serviceConfig.enableAudioPlayback = enabled |
|
|
|
|
|
Log.d(TAG, "音频播放设置更新: $enabled") |
|
|
|
|
|
|
|
|
|
|
|
// 重新设置语音合成器以应用新配置 |
|
|
|
|
|
if (serviceState.isInitialized.get()) { |
|
|
|
|
|
setupSpeechSynthesizer() |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
eventCallback?.onStateChanged("AudioPlayback", enabled) |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
|
|
* 获取当前音频播放状态 |
|
|
|
|
|
* @return true表示启用音频播放,false表示禁用 |
|
|
|
|
|
*/ |
|
|
|
|
|
fun isAudioPlaybackEnabled(): Boolean { |
|
|
|
|
|
return serviceConfig.enableAudioPlayback |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
|
|
* 单独的语音合成方法,只合成不播放,返回音频数据 |
|
|
|
|
|
* @param text 要合成的文本 |
|
|
|
|
|
* @return 合成的音频数据,失败时返回null |
|
|
|
|
|
*/ |
|
|
|
|
|
suspend fun synthesizeTextToAudio(text: String): ByteArray? = withContext(Dispatchers.IO) { |
|
|
|
|
|
try { |
|
|
|
|
|
Log.d(TAG, "开始纯音频合成: $text") |
|
|
|
|
|
|
|
|
|
|
|
val ssml = generateOptimizedSsml(text) |
|
|
|
|
|
|
|
|
|
|
|
// 创建一个临时的合成器,输出到内存而不是扬声器 |
|
|
|
|
|
val audioConfig = AudioConfig.fromStreamOutput(AudioOutputStream.createPullStream()) |
|
|
|
|
|
val tempSynthesizer = SpeechSynthesizer(speechConfig, audioConfig) |
|
|
|
|
|
|
|
|
|
|
|
try { |
|
|
|
|
|
val result = tempSynthesizer.SpeakSsmlAsync(ssml).get() |
|
|
|
|
|
|
|
|
|
|
|
if (result?.reason == ResultReason.SynthesizingAudioCompleted) { |
|
|
|
|
|
val audioData = result.audioData |
|
|
|
|
|
if (audioData != null && audioData.isNotEmpty()) { |
|
|
|
|
|
Log.d(TAG, "纯音频合成成功,大小: ${audioData.size} 字节") |
|
|
|
|
|
return@withContext audioData |
|
|
|
|
|
} |
|
|
|
|
|
} else { |
|
|
|
|
|
Log.e(TAG, "纯音频合成失败: ${result?.reason}") |
|
|
|
|
|
} |
|
|
|
|
|
} finally { |
|
|
|
|
|
tempSynthesizer.close() |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
return@withContext null |
|
|
|
|
|
} catch (e: Exception) { |
|
|
|
|
|
Log.e(TAG, "纯音频合成异常", e) |
|
|
|
|
|
return@withContext null |
|
|
|
|
|
} |
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
/** |
|
|
/** |
|
|
* 推送外部音频数据 |
|
|
* 推送外部音频数据 |
|
|
*/ |
|
|
*/ |
|
|
@ -774,9 +861,9 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) { |
|
|
val audioData = |
|
|
val audioData = |
|
|
audioQueue.poll(100, java.util.concurrent.TimeUnit.MILLISECONDS) |
|
|
audioQueue.poll(100, java.util.concurrent.TimeUnit.MILLISECONDS) |
|
|
audioData?.let { |
|
|
audioData?.let { |
|
|
Log.d(TAG, "音频处理: ${it}") |
|
|
Log.d(TAG, "音频处理: ${it.size}") |
|
|
pushAudioStream?.write(it) |
|
|
pushAudioStream?.write(it) |
|
|
recordfile?.saveAudioDataToWav(it) |
|
|
recordfile1?.saveAudioDataToWav(it) |
|
|
} |
|
|
} |
|
|
} catch (e: InterruptedException) { |
|
|
} catch (e: InterruptedException) { |
|
|
Thread.currentThread().interrupt() |
|
|
Thread.currentThread().interrupt() |
|
|
|