You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
570 lines
15 KiB
570 lines
15 KiB
import 'dart:async';
|
|
import 'package:flutter/services.dart';
|
|
import 'package:get/get.dart';
|
|
import 'package:flutter_dotenv/flutter_dotenv.dart';
|
|
import '../../../core/utils/logger.dart';
|
|
import '../tts_service.dart';
|
|
|
|
/// 音频输出设备类型
|
|
enum AudioOutputType {
|
|
/// 扬声器
|
|
speaker,
|
|
|
|
/// 听筒
|
|
earpiece,
|
|
|
|
/// 自动选择(有耳机用耳机,没有则用听筒)
|
|
auto
|
|
}
|
|
|
|
/// 微软 Text-to-Speech 服务
|
|
///
|
|
/// 该服务通过平台通道与 Android 上的 Microsoft Speech SDK 交互,
|
|
/// 提供文本转语音功能。
|
|
class AzureTtsService extends GetxService implements TtsService {
|
|
static final AzureTtsService to = Get.put(AzureTtsService());
|
|
static const MethodChannel _channel = MethodChannel('com.deep_voice.azure_tts');
|
|
|
|
bool _isInitialized = false;
|
|
late final String _subscriptionKey;
|
|
late final String _serviceRegion;
|
|
|
|
// 当前设置
|
|
String _currentVoice = 'zh-CN-XiaoxiaoNeural';
|
|
int _currentRate = 0;
|
|
int _currentPitch = 0;
|
|
int _currentVolume = 100;
|
|
AudioOutputType _currentAudioOutputType = AudioOutputType.auto;
|
|
|
|
// 支持的语音列表
|
|
final List<String> _supportedVoices = [
|
|
'zh-CN-XiaoxiaoNeural',
|
|
'zh-CN-YunxiNeural',
|
|
'zh-CN-YunjianNeural',
|
|
'en-US-JennyNeural',
|
|
'en-US-GuyNeural'
|
|
];
|
|
|
|
// 语音合成队列
|
|
final List<_SpeechItem> _textQueue = [];
|
|
bool _isProcessingQueue = false;
|
|
|
|
// 事件流控制器
|
|
final StreamController<TtsEvent> _eventController = StreamController<TtsEvent>.broadcast();
|
|
@override
|
|
Stream<TtsEvent> get onEvent => _eventController.stream;
|
|
|
|
// 可观察状态
|
|
final isEnabled = true.obs;
|
|
final _isSpeaking = false.obs;
|
|
final audioOutputType = AudioOutputType.auto.obs;
|
|
|
|
// 流式文本缓冲区
|
|
String _streamBuffer = '';
|
|
|
|
@override
|
|
String get currentVoice => _currentVoice;
|
|
|
|
@override
|
|
bool get isSpeaking => _isSpeaking.value;
|
|
|
|
@override
|
|
List<String> get supportedVoices => _supportedVoices;
|
|
|
|
AudioOutputType get currentAudioOutputType => _currentAudioOutputType;
|
|
|
|
AzureTtsService() {
|
|
_loadConfig();
|
|
}
|
|
|
|
/// 从环境变量加载配置
|
|
void _loadConfig() {
|
|
_subscriptionKey = dotenv.env['AZURE_SPEECH_KEY'] ?? '';
|
|
_serviceRegion = dotenv.env['AZURE_SPEECH_REGION'] ?? '';
|
|
}
|
|
|
|
@override
|
|
Future<bool> initialize({
|
|
List<String>? supportedLanguages,
|
|
}) async {
|
|
if (_isInitialized) return true;
|
|
|
|
try {
|
|
if (_subscriptionKey.isEmpty || _serviceRegion.isEmpty) {
|
|
Logger.error('未找到Azure语音服务配置');
|
|
return false;
|
|
}
|
|
|
|
// 始终优先使用中文作为默认语言
|
|
final String defaultLanguage = 'zh-CN';
|
|
|
|
// 更新支持的语音列表(如果提供)
|
|
if (supportedLanguages != null && supportedLanguages.isNotEmpty) {
|
|
// 如果支持中文,优先使用中文,否则使用提供的第一种语言
|
|
String language = supportedLanguages.contains(defaultLanguage)
|
|
? defaultLanguage
|
|
: supportedLanguages.first;
|
|
|
|
final result = await _channel.invokeMethod('initialize', {
|
|
'subscriptionKey': _subscriptionKey,
|
|
'region': _serviceRegion,
|
|
'language': language,
|
|
});
|
|
|
|
_isInitialized = result;
|
|
|
|
// 确保设置默认的音频输出设备类型
|
|
if (result) {
|
|
// 不需要显式调用 setAudioOutputType,因为原生层已经设置了默认值
|
|
// 只需要确保 Dart 层的状态与原生层一致
|
|
_currentAudioOutputType = AudioOutputType.auto;
|
|
audioOutputType.value = AudioOutputType.auto;
|
|
}
|
|
|
|
return result;
|
|
} else {
|
|
// 使用默认语言初始化
|
|
final result = await _channel.invokeMethod('initialize', {
|
|
'subscriptionKey': _subscriptionKey,
|
|
'region': _serviceRegion,
|
|
'language': 'zh-CN',
|
|
});
|
|
|
|
_isInitialized = result;
|
|
|
|
// 确保设置默认的音频输出设备类型
|
|
if (result) {
|
|
_currentAudioOutputType = AudioOutputType.auto;
|
|
audioOutputType.value = AudioOutputType.auto;
|
|
}
|
|
|
|
return result;
|
|
}
|
|
} catch (e) {
|
|
Logger.error('初始化失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('TTS初始化失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<bool> setVoice(String voiceName) async {
|
|
if (!_isInitialized) await initialize();
|
|
if (voiceName == _currentVoice) return true;
|
|
|
|
try {
|
|
final result = await _channel.invokeMethod('setVoice', {
|
|
'voiceName': voiceName,
|
|
});
|
|
|
|
if (result) _currentVoice = voiceName;
|
|
return result;
|
|
} catch (e) {
|
|
Logger.error('设置语音失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('设置语音失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 设置音频输出设备类型
|
|
Future<bool> setAudioOutputType(AudioOutputType outputType) async {
|
|
if (!_isInitialized) await initialize();
|
|
|
|
try {
|
|
final String outputTypeStr = outputType.toString().split('.').last;
|
|
final result = await _channel.invokeMethod('setAudioOutputType', {
|
|
'outputType': outputTypeStr,
|
|
});
|
|
|
|
if (result) {
|
|
_currentAudioOutputType = outputType;
|
|
audioOutputType.value = outputType;
|
|
}
|
|
|
|
return result;
|
|
} catch (e) {
|
|
Logger.error('设置音频输出设备失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('设置音频输出设备失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 设置语音参数
|
|
Future<bool> setSpeechParams({int rate = 0, int pitch = 0, int volume = 100}) async {
|
|
if (!_isInitialized) await initialize();
|
|
|
|
try {
|
|
final result = await _channel.invokeMethod('setSpeechParams', {
|
|
'rate': rate,
|
|
'pitch': pitch,
|
|
'volume': volume,
|
|
});
|
|
|
|
if (result) {
|
|
_currentRate = rate;
|
|
_currentPitch = pitch;
|
|
_currentVolume = volume;
|
|
}
|
|
|
|
return result;
|
|
} catch (e) {
|
|
Logger.error('设置语音参数失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('设置语音参数失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<bool> speakOnce(String text) async {
|
|
if (!_isInitialized) await initialize();
|
|
if (!isEnabled.value || text.isEmpty) return false;
|
|
|
|
try {
|
|
_isSpeaking.value = true;
|
|
|
|
// 发送开始事件
|
|
_eventController.add(TtsEvent(type: TtsEventType.started));
|
|
|
|
final result = await _channel.invokeMethod('speakText', {'text': text});
|
|
|
|
_isSpeaking.value = false;
|
|
|
|
// 发送完成事件
|
|
_eventController.add(TtsEvent(type: TtsEventType.completed));
|
|
|
|
return result == "OK";
|
|
} catch (e) {
|
|
_isSpeaking.value = false;
|
|
Logger.error('语音合成失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('语音合成失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 播放SSML
|
|
Future<bool> speakSsmlOnce(String ssml) async {
|
|
if (!_isInitialized) await initialize();
|
|
if (!isEnabled.value || ssml.isEmpty) return false;
|
|
|
|
try {
|
|
_isSpeaking.value = true;
|
|
|
|
// 发送开始事件
|
|
_eventController.add(TtsEvent(type: TtsEventType.started));
|
|
|
|
final result = await _channel.invokeMethod('speakSsml', {'ssml': ssml});
|
|
|
|
_isSpeaking.value = false;
|
|
|
|
// 发送完成事件
|
|
_eventController.add(TtsEvent(type: TtsEventType.completed));
|
|
|
|
return result == "OK";
|
|
} catch (e) {
|
|
_isSpeaking.value = false;
|
|
Logger.error('SSML语音合成失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('SSML语音合成失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<bool> speakStream(String text) async {
|
|
if (!isEnabled.value || text.isEmpty) return false;
|
|
|
|
try {
|
|
// 添加文本到缓冲区
|
|
_streamBuffer += text;
|
|
|
|
// 如果缓冲区为空,直接返回
|
|
if (_streamBuffer.isEmpty) return true;
|
|
|
|
// 使用正则表达式匹配句子,包括结束符号
|
|
// 匹配任意字符,直到遇到句子结束符号
|
|
final sentenceRegex = RegExp(
|
|
r'([^。.!!??;;:\n\r]+[。.!!??;;::\n\r]|[^。.!!??;;:\n\r]+(?:\.{3,}|…)|[^。.!!??;;:\n\r]+["」』])'
|
|
);
|
|
|
|
bool hasProcessed = false;
|
|
|
|
// 查找所有完整句子
|
|
final matches = sentenceRegex.allMatches(_streamBuffer);
|
|
final List<String> sentences = [];
|
|
int lastMatchEnd = 0;
|
|
|
|
for (final match in matches) {
|
|
// 提取完整句子(包含结束符号)
|
|
final sentence = match.group(1)?.trim();
|
|
if (sentence != null && sentence.isNotEmpty) {
|
|
sentences.add(sentence);
|
|
lastMatchEnd = match.end;
|
|
}
|
|
}
|
|
|
|
// 处理找到的句子
|
|
for (final sentence in sentences) {
|
|
_textQueue.add(_SpeechItem(
|
|
text: sentence,
|
|
rate: _currentRate,
|
|
pitch: _currentPitch,
|
|
volume: _currentVolume,
|
|
));
|
|
hasProcessed = true;
|
|
}
|
|
|
|
// 更新缓冲区,只保留未完成的部分
|
|
if (lastMatchEnd > 0) {
|
|
_streamBuffer = _streamBuffer.substring(lastMatchEnd);
|
|
}
|
|
|
|
// 如果处理了文本并且队列未在处理中,开始处理队列
|
|
if (hasProcessed && !_isProcessingQueue) {
|
|
_processQueue();
|
|
}
|
|
|
|
return true;
|
|
} catch (e) {
|
|
Logger.error('处理流式文本失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('处理流式文本失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<bool> flushStream() async {
|
|
if (_textQueue.isEmpty && _streamBuffer.isEmpty) return true;
|
|
|
|
try {
|
|
// 处理缓冲区中可能的完整句子
|
|
await speakStream('');
|
|
|
|
// 如果缓冲区仍有剩余文本,将其添加到播放队列
|
|
if (_streamBuffer.isNotEmpty && _streamBuffer.trim().isNotEmpty) {
|
|
_textQueue.add(_SpeechItem(
|
|
text: _streamBuffer,
|
|
rate: _currentRate,
|
|
pitch: _currentPitch,
|
|
volume: _currentVolume,
|
|
));
|
|
|
|
// 清空缓冲区
|
|
_streamBuffer = '';
|
|
|
|
// 如果队列未在处理中,开始处理队列
|
|
if (!_isProcessingQueue) {
|
|
_processQueue();
|
|
}
|
|
}
|
|
|
|
Logger.info('等待TTS队列播放完成,剩余${_textQueue.length}条');
|
|
|
|
// 等待队列处理完成
|
|
while (_isProcessingQueue || _isSpeaking.value) {
|
|
await Future.delayed(const Duration(milliseconds: 100));
|
|
}
|
|
|
|
Logger.info('TTS队列播放完成');
|
|
return true;
|
|
} catch (e) {
|
|
Logger.error('刷新TTS流失败: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('刷新TTS流失败: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 处理语音合成队列
|
|
Future<void> _processQueue() async {
|
|
if (_textQueue.isEmpty || _isProcessingQueue) return;
|
|
|
|
_isProcessingQueue = true;
|
|
|
|
try {
|
|
while (_textQueue.isNotEmpty) {
|
|
if (!isEnabled.value) {
|
|
_textQueue.clear();
|
|
break;
|
|
}
|
|
|
|
final item = _textQueue.removeAt(0);
|
|
|
|
if (item.rate != _currentRate || item.pitch != _currentPitch || item.volume != _currentVolume) {
|
|
await setSpeechParams(
|
|
rate: item.rate,
|
|
pitch: item.pitch,
|
|
volume: item.volume,
|
|
);
|
|
}
|
|
|
|
// 使用speakOnce播放当前项
|
|
await speakOnce(item.text);
|
|
}
|
|
} catch (e) {
|
|
Logger.error('处理语音队列出错: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('处理语音队列出错: $e'));
|
|
} finally {
|
|
_isProcessingQueue = false;
|
|
}
|
|
}
|
|
|
|
@override
|
|
Future<bool> stop() async {
|
|
try {
|
|
_textQueue.clear();
|
|
|
|
if (_isSpeaking.value) {
|
|
final result = await _channel.invokeMethod('stopSpeaking');
|
|
_isSpeaking.value = false;
|
|
|
|
// 发送取消事件
|
|
_eventController.add(TtsEvent(type: TtsEventType.canceled));
|
|
|
|
return result;
|
|
}
|
|
|
|
return true;
|
|
} catch (e) {
|
|
Logger.error('停止语音合成出错: $e');
|
|
|
|
// 发送错误事件
|
|
_eventController.add(TtsEvent.error('停止语音合成出错: $e'));
|
|
|
|
return false;
|
|
}
|
|
}
|
|
|
|
/// 生成SSML文本
|
|
String generateSsml({
|
|
required String text,
|
|
String? voiceName,
|
|
int? rate,
|
|
int? pitch,
|
|
int? volume,
|
|
}) {
|
|
final voice = voiceName ?? _currentVoice;
|
|
final rateValue = (rate ?? _currentRate).clamp(-100, 100);
|
|
final pitchValue = (pitch ?? _currentPitch).clamp(-100, 100);
|
|
final volumeValue = (volume ?? _currentVolume).clamp(0, 100);
|
|
|
|
final rateStr = rateValue == 0 ? '0%' : rateValue < 0 ? '${(rateValue * 0.9).round()}%' : '$rateValue%';
|
|
final pitchStr = pitchValue == 0 ? '0%' : '${(pitchValue * 0.5).round()}%';
|
|
final volumeStr = "$volumeValue%";
|
|
|
|
return '''
|
|
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xmlns:mstts="https://www.w3.org/2001/mstts" xml:lang="zh-CN">
|
|
<voice name="$voice">
|
|
<prosody rate="$rateStr" pitch="$pitchStr" volume="$volumeStr">
|
|
$text
|
|
</prosody>
|
|
</voice>
|
|
</speak>
|
|
''';
|
|
}
|
|
|
|
/// 切换启用状态
|
|
void toggleEnabled() {
|
|
isEnabled.toggle();
|
|
if (!isEnabled.value) stop();
|
|
}
|
|
|
|
/// 切换音频输出设备类型
|
|
Future<bool> toggleAudioOutputType() async {
|
|
AudioOutputType newType;
|
|
|
|
switch (_currentAudioOutputType) {
|
|
case AudioOutputType.speaker:
|
|
newType = AudioOutputType.earpiece;
|
|
break;
|
|
case AudioOutputType.earpiece:
|
|
newType = AudioOutputType.auto;
|
|
break;
|
|
case AudioOutputType.auto:
|
|
newType = AudioOutputType.speaker;
|
|
break;
|
|
}
|
|
|
|
return await setAudioOutputType(newType);
|
|
}
|
|
|
|
@override
|
|
Future<void> dispose() async {
|
|
if (!_isInitialized) return;
|
|
|
|
try {
|
|
await stop();
|
|
await _channel.invokeMethod('dispose');
|
|
|
|
// 关闭事件流控制器
|
|
await _eventController.close();
|
|
|
|
_isInitialized = false;
|
|
} catch (e) {
|
|
Logger.error('释放资源失败: $e');
|
|
}
|
|
}
|
|
|
|
/// 处理AI返回的文本流,将其转换为语音
|
|
///
|
|
/// [textStream] 文本流
|
|
/// 返回一个Future,当流处理完成时完成
|
|
Future<bool> processAiTextStream(Stream<String> textStream) async {
|
|
if (!isEnabled.value) {
|
|
Logger.error('TTS服务未启用');
|
|
return false;
|
|
}
|
|
|
|
try {
|
|
await for (final text in textStream) {
|
|
if (text.isEmpty) continue;
|
|
await speakStream(text);
|
|
}
|
|
|
|
// 处理完流后,刷新缓冲区中的剩余文本
|
|
await flushStream();
|
|
return true;
|
|
} catch (e) {
|
|
Logger.error('处理文本流时出错: $e');
|
|
return false;
|
|
}
|
|
}
|
|
}
|
|
|
|
/// 语音合成项目
|
|
class _SpeechItem {
|
|
final String text;
|
|
final int rate;
|
|
final int pitch;
|
|
final int volume;
|
|
|
|
_SpeechItem({
|
|
required this.text,
|
|
this.rate = 0,
|
|
this.pitch = 0,
|
|
this.volume = 100,
|
|
});
|
|
}
|