You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
590 lines
19 KiB
590 lines
19 KiB
import 'package:get/get.dart';
|
|
import 'dart:async';
|
|
import 'volcano_ai_service.dart';
|
|
import 'azure_asr_service.dart';
|
|
import 'volcano_tts_api_service.dart';
|
|
import '../../modules/chat/models/message_model.dart';
|
|
import '../../core/utils/logger.dart';
|
|
import 'package:flutter_dotenv/flutter_dotenv.dart';
|
|
|
|
class BackgroundAgentService extends GetxService {
|
|
static BackgroundAgentService get to => Get.find();
|
|
|
|
final VolcanoAIService _aiService;
|
|
final VolcanoTtsApiService _ttsService;
|
|
final AzureAsrService _voiceRecognitionService;
|
|
final List<Message> _messageHistory = [];
|
|
String _pendingTtsText = '';
|
|
static const int _minTtsLength = 20;
|
|
bool _isProcessing = false;
|
|
bool _isListening = false;
|
|
StreamSubscription? _recognitionSubscription;
|
|
|
|
// TTS相关订阅
|
|
StreamSubscription? _ttsEventSubscription;
|
|
bool _isTtsSpeaking = false;
|
|
|
|
// 可观察的状态
|
|
final RxBool isListening = false.obs;
|
|
final RxString recognizedText = ''.obs;
|
|
|
|
// 添加一个标志,表示是否已经识别到语音
|
|
bool _hasRecognizedSpeech = false;
|
|
|
|
// 添加一个标志,表示是否应该继续循环交互
|
|
bool _shouldContinueInteraction = false;
|
|
|
|
// 添加一个 Completer 用于在识别到最终结果时完成
|
|
Completer<String>? _recognitionCompleter;
|
|
|
|
// 添加一个标志,表示是否已经收到最终结果
|
|
bool _hasFinalResult = false;
|
|
|
|
// 添加一个计时器,用于在一段时间没有新的识别结果时提交当前结果
|
|
Timer? _silenceTimer;
|
|
|
|
// 添加一个计时器,用于检测用户长时间没有说话
|
|
Timer? _noSpeechTimer;
|
|
|
|
// 最后一次识别到语音的时间
|
|
// DateTime? _lastSpeechTime; // Removing unused field
|
|
|
|
// 添加一个变量来跟踪当前的AI响应流订阅
|
|
StreamSubscription? _aiResponseSubscription;
|
|
|
|
// 添加一个标志,表示是否应该取消当前的AI响应
|
|
bool _shouldCancelAiResponse = false;
|
|
|
|
// 添加计数器,用于跟踪连续无输入的次数
|
|
// int _noSpeechCount = 0; // Removing unused field
|
|
|
|
BackgroundAgentService()
|
|
: _aiService = VolcanoAIService(),
|
|
_ttsService = Get.find<VolcanoTtsApiService>(),
|
|
_voiceRecognitionService = Get.find<AzureAsrService>() {
|
|
// 设置TTS事件监听
|
|
_setupTtsEventListener();
|
|
}
|
|
|
|
// 设置TTS事件监听
|
|
void _setupTtsEventListener() {
|
|
_ttsEventSubscription?.cancel();
|
|
_ttsEventSubscription = _ttsService.eventStream.listen(_handleTtsEvent);
|
|
}
|
|
|
|
// 处理TTS事件
|
|
void _handleTtsEvent(Map<String, dynamic> event) {
|
|
final eventType = event['eventType'];
|
|
|
|
switch (eventType) {
|
|
case 'ttsSentenceStart':
|
|
_isTtsSpeaking = true;
|
|
break;
|
|
case 'ttsSentenceEnd':
|
|
case 'sessionFinished':
|
|
case 'error':
|
|
_isTtsSpeaking = false;
|
|
break;
|
|
}
|
|
}
|
|
|
|
// 启动无语音超时计时器
|
|
void _startNoSpeechTimer() {
|
|
// 取消之前的计时器
|
|
_noSpeechTimer?.cancel();
|
|
|
|
// 设置30秒的超时计时器
|
|
_noSpeechTimer = Timer(const Duration(seconds: 30), () {
|
|
// 检查TTS是否正在播放
|
|
if (_isTtsSpeaking) {
|
|
// 如果TTS正在播放,重置定时器
|
|
Logger.info('TTS正在播放,重置30秒超时计时器');
|
|
_startNoSpeechTimer();
|
|
} else {
|
|
// 如果TTS不在播放,且30秒内没有检测到用户输入,退出交互
|
|
Logger.info('30秒内没有检测到用户输入,退出交互');
|
|
_exitInteraction();
|
|
}
|
|
});
|
|
}
|
|
|
|
// 退出交互
|
|
Future<void> _exitInteraction() async {
|
|
// 完成当前的recognitionCompleter(如果有)
|
|
if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.complete('');
|
|
}
|
|
// 设置标志,停止循环交互
|
|
_shouldContinueInteraction = false;
|
|
_isProcessing = false;
|
|
// 播放退出提示
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
await _ttsService.synthesize("没有听到您说话,已退出语音交互", speaker);
|
|
} catch (e) {
|
|
print('播放退出提示失败: $e');
|
|
}
|
|
}
|
|
|
|
// 处理蓝牙耳机按钮触发的交互
|
|
Future<void> handleAgentInteraction(String systemPrompt) async {
|
|
if (_isProcessing) {
|
|
print('已经在处理交互,忽略此次请求');
|
|
return;
|
|
}
|
|
|
|
// 确保TTS服务已连接
|
|
if (!_ttsService.isConnected.value) {
|
|
try {
|
|
await _ttsService.connect();
|
|
} catch (e) {
|
|
print('连接TTS服务失败: $e');
|
|
return;
|
|
}
|
|
}
|
|
|
|
// 设置循环交互标志为 true
|
|
_shouldContinueInteraction = true;
|
|
|
|
try {
|
|
// 确保之前的语音识别已经停止
|
|
if (_isListening) {
|
|
await stopVoiceRecognition();
|
|
}
|
|
|
|
// 循环进行交互,直到用户 30 秒没有说话或手动停止
|
|
while (_shouldContinueInteraction) {
|
|
await _processSingleInteraction(systemPrompt);
|
|
}
|
|
} finally {
|
|
// 确保在交互结束时停止语音识别
|
|
if (_isListening) {
|
|
await stopVoiceRecognition();
|
|
}
|
|
|
|
// 重置语音识别服务
|
|
try {
|
|
await _voiceRecognitionService.initialize(
|
|
subscriptionKey: dotenv.env['AZURE_ASR_SUBSCRIPTION_KEY'] ?? '',
|
|
serviceRegion: dotenv.env['AZURE_ASR_SERVICE_REGION'] ?? 'eastasia',
|
|
language: dotenv.env['AZURE_ASR_LANGUAGE'] ?? 'zh-CN',
|
|
);
|
|
} catch (e) {
|
|
print('重置语音识别服务失败: $e');
|
|
}
|
|
}
|
|
}
|
|
|
|
// 处理单次交互
|
|
Future<void> _processSingleInteraction(String systemPrompt) async {
|
|
_isProcessing = true;
|
|
_hasRecognizedSpeech = false;
|
|
_hasFinalResult = false;
|
|
|
|
try {
|
|
// 先播放一个简短的提示音或提示语,表示开始监听
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
await _ttsService.synthesize("我在听", speaker);
|
|
} catch (e) {
|
|
print('播放提示音失败: $e');
|
|
// 继续执行,不要因为提示音失败而中断整个流程
|
|
}
|
|
|
|
// 开始语音识别
|
|
String userInput = '';
|
|
|
|
try {
|
|
// 确保之前的语音识别已经停止
|
|
if (_isListening) {
|
|
await stopVoiceRecognition();
|
|
}
|
|
|
|
// 尝试启动语音识别
|
|
await startVoiceRecognition();
|
|
|
|
// 创建一个 Completer 来处理语音识别完成
|
|
_recognitionCompleter = Completer<String>();
|
|
|
|
// 启动无语音超时计时器
|
|
_startNoSpeechTimer();
|
|
|
|
// 等待语音识别完成
|
|
userInput = await _recognitionCompleter!.future;
|
|
|
|
// 取消无语音超时计时器
|
|
_noSpeechTimer?.cancel();
|
|
_noSpeechTimer = null;
|
|
|
|
// 如果用户输入为空,重新启动超时计时器并返回
|
|
if (userInput.trim().isEmpty) {
|
|
_startNoSpeechTimer();
|
|
return;
|
|
}
|
|
} catch (e) {
|
|
print('语音识别过程出错: $e');
|
|
_startNoSpeechTimer(); // 出错时也启动超时计时器
|
|
return;
|
|
} finally {
|
|
// 清理资源,但保持语音识别状态
|
|
_recognitionCompleter = null;
|
|
_silenceTimer?.cancel();
|
|
_silenceTimer = null;
|
|
}
|
|
|
|
// 保存用户消息到历史记录
|
|
_messageHistory.add(Message(
|
|
role: 'user',
|
|
content: userInput,
|
|
timestamp: DateTime.now(),
|
|
));
|
|
|
|
// 构建用于生成回应的消息列表
|
|
final messages = [
|
|
{'role': 'system', 'content': systemPrompt},
|
|
{'role': 'user', 'content': userInput},
|
|
];
|
|
|
|
String fullResponse = '';
|
|
try {
|
|
// 重置取消标志
|
|
_shouldCancelAiResponse = false;
|
|
|
|
// 创建一个本地变量来跟踪是否已取消
|
|
bool isCancelled = false;
|
|
|
|
// 获取AI响应流
|
|
final responseStream = _aiService.sendMessageStream(
|
|
messages: messages,
|
|
systemPrompt: systemPrompt,
|
|
);
|
|
|
|
// 创建一个订阅来处理响应流
|
|
_aiResponseSubscription = responseStream.listen(
|
|
(chunk) {
|
|
// 如果已经设置了取消标志,则不处理这个块
|
|
if (_shouldCancelAiResponse) {
|
|
isCancelled = true;
|
|
return;
|
|
}
|
|
|
|
fullResponse += chunk;
|
|
|
|
// 累积文本并处理TTS
|
|
_pendingTtsText += chunk;
|
|
|
|
// 查找最后一个完整句子的结束位置
|
|
int lastSentenceEnd = _findLastSentenceEnd(_pendingTtsText);
|
|
if (lastSentenceEnd > 0) {
|
|
// 提取完整的句子
|
|
String sentenceToSpeak = _pendingTtsText.substring(0, lastSentenceEnd + 1);
|
|
// 只有当句子长度超过最小长度时才播放
|
|
if (sentenceToSpeak.length >= _minTtsLength) {
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
_ttsService.synthesize(sentenceToSpeak, speaker);
|
|
} catch (e) {
|
|
print('播放TTS失败: $e');
|
|
}
|
|
// 更新待处理文本,移除已播放的部分
|
|
_pendingTtsText = _pendingTtsText.substring(lastSentenceEnd + 1);
|
|
}
|
|
}
|
|
},
|
|
onError: (e) {
|
|
print('AI响应流错误: $e');
|
|
// 清理订阅
|
|
_aiResponseSubscription = null;
|
|
},
|
|
onDone: () {
|
|
// 如果已取消,不处理剩余的文本
|
|
if (!isCancelled && !_shouldCancelAiResponse) {
|
|
// 处理剩余的文本
|
|
if (_pendingTtsText.isNotEmpty) {
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
_ttsService.synthesize(_pendingTtsText, speaker);
|
|
} catch (e) {
|
|
print('播放剩余TTS失败: $e');
|
|
}
|
|
}
|
|
}
|
|
_pendingTtsText = '';
|
|
_aiResponseSubscription = null;
|
|
|
|
// AI响应完成后,重新启动超时计时器
|
|
_startNoSpeechTimer();
|
|
}
|
|
);
|
|
|
|
// 等待响应流完成
|
|
await _aiResponseSubscription!.asFuture();
|
|
|
|
} catch (e) {
|
|
print('AI响应生成失败: $e');
|
|
// 如果AI响应失败,使用默认回复
|
|
fullResponse = '抱歉,我现在无法回答您的问题。请稍后再试。';
|
|
|
|
// 播放错误提示
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
await _ttsService.synthesize(fullResponse, speaker);
|
|
} catch (_) {}
|
|
|
|
// 启动超时计时器
|
|
_startNoSpeechTimer();
|
|
} finally {
|
|
// 清理资源
|
|
_aiResponseSubscription?.cancel();
|
|
_aiResponseSubscription = null;
|
|
_shouldCancelAiResponse = false;
|
|
}
|
|
|
|
// 如果响应被取消,不保存到历史记录
|
|
if (!_shouldCancelAiResponse && fullResponse.isNotEmpty) {
|
|
// 保存助手回复到历史记录
|
|
_messageHistory.add(Message(
|
|
role: 'assistant',
|
|
content: fullResponse,
|
|
timestamp: DateTime.now(),
|
|
));
|
|
}
|
|
|
|
// 限制历史记录长度
|
|
if (_messageHistory.length > 20) {
|
|
_messageHistory.removeRange(0, _messageHistory.length - 20);
|
|
}
|
|
|
|
} catch (e) {
|
|
print('交互过程出错: $e');
|
|
try {
|
|
// 使用默认语音类型
|
|
const speaker = 'zh_female_shuangkuaisisi_moon_bigtts';
|
|
await _ttsService.synthesize("抱歉,出现了一些问题", speaker);
|
|
} catch (_) {}
|
|
|
|
// 发生错误时停止循环交互
|
|
_shouldContinueInteraction = false;
|
|
} finally {
|
|
_isProcessing = false;
|
|
}
|
|
}
|
|
|
|
// 停止循环交互
|
|
void stopContinuousInteraction() {
|
|
_shouldContinueInteraction = false;
|
|
print('手动停止循环交互');
|
|
}
|
|
|
|
// 开始语音识别
|
|
Future<void> startVoiceRecognition() async {
|
|
if (_isListening) {
|
|
await stopVoiceRecognition();
|
|
}
|
|
|
|
try {
|
|
// 确保语音识别服务已初始化
|
|
if (!await _voiceRecognitionService.initialize(
|
|
subscriptionKey: dotenv.env['AZURE_ASR_SUBSCRIPTION_KEY'] ?? '',
|
|
serviceRegion: dotenv.env['AZURE_ASR_SERVICE_REGION'] ?? 'eastasia',
|
|
language: dotenv.env['AZURE_ASR_LANGUAGE'] ?? 'zh-CN',
|
|
)) {
|
|
// 尝试重新初始化
|
|
await Future.delayed(const Duration(milliseconds: 500));
|
|
if (!await _voiceRecognitionService.initialize(
|
|
subscriptionKey: dotenv.env['AZURE_ASR_SUBSCRIPTION_KEY'] ?? '',
|
|
serviceRegion: dotenv.env['AZURE_ASR_SERVICE_REGION'] ?? 'eastasia',
|
|
language: dotenv.env['AZURE_ASR_LANGUAGE'] ?? 'zh-CN',
|
|
)) {
|
|
throw Exception('无法初始化语音识别服务');
|
|
}
|
|
}
|
|
|
|
// 开始连续识别
|
|
final success = await _voiceRecognitionService.startContinuousRecognition();
|
|
if (!success) {
|
|
if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.completeError(Exception('无法启动语音识别'));
|
|
}
|
|
return;
|
|
}
|
|
|
|
_isListening = true;
|
|
isListening.value = true;
|
|
recognizedText.value = '';
|
|
_hasRecognizedSpeech = false;
|
|
_hasFinalResult = false;
|
|
|
|
// 使用语音识别服务的recognitionStream而不是方法的返回值
|
|
_recognitionSubscription = _voiceRecognitionService.recognitionStream?.listen((event) {
|
|
if (event.type == RecognitionEventType.finalResult) {
|
|
recognizedText.value = event.text;
|
|
if (event.text.isNotEmpty) {
|
|
_hasRecognizedSpeech = true;
|
|
_hasFinalResult = true;
|
|
|
|
// 收到最终结果,立即完成识别过程
|
|
if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.complete(event.text);
|
|
}
|
|
} else {
|
|
// 如果最终结果为空,但有中间结果,使用最后的中间结果
|
|
if (_hasRecognizedSpeech && recognizedText.value.isNotEmpty) {
|
|
if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.complete(recognizedText.value);
|
|
}
|
|
} else {
|
|
// 如果最终结果为空,且没有中间结果,重新启动超时计时器
|
|
_startNoSpeechTimer();
|
|
}
|
|
}
|
|
} else if (event.type == RecognitionEventType.recognizing) {
|
|
recognizedText.value = event.text;
|
|
if (event.text.isNotEmpty) {
|
|
_hasRecognizedSpeech = true;
|
|
|
|
// 检测到用户说话,重置无语音计时器
|
|
_noSpeechTimer?.cancel();
|
|
_startNoSpeechTimer();
|
|
|
|
// 如果系统正在播放TTS或接收AI响应,检测到用户开始说话时立即中断
|
|
if (_isTtsSpeaking && event.text.trim().isNotEmpty) {
|
|
_ttsService.endSession(); // 结束当前TTS会话
|
|
|
|
// 设置标志,表示应该取消当前的AI响应
|
|
_shouldCancelAiResponse = true;
|
|
|
|
// 取消当前的AI响应流订阅
|
|
_aiResponseSubscription?.cancel();
|
|
_aiResponseSubscription = null;
|
|
|
|
// 清空待处理的TTS文本
|
|
_pendingTtsText = '';
|
|
}
|
|
|
|
// 取消之前的静默计时器
|
|
_silenceTimer?.cancel();
|
|
|
|
// 取消之前的无语音超时计时器,用户正在说话
|
|
_noSpeechTimer?.cancel();
|
|
_noSpeechTimer = null;
|
|
|
|
// 设置新的静默计时器,如果 2 秒内没有新的识别结果,则认为用户已经停止说话
|
|
_silenceTimer = Timer(const Duration(seconds: 2), () {
|
|
if (_hasRecognizedSpeech && !_hasFinalResult &&
|
|
_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.complete(recognizedText.value);
|
|
}
|
|
});
|
|
}
|
|
} else if (event.type == RecognitionEventType.error) {
|
|
print('识别错误: ${event.error}');
|
|
}
|
|
}, onError: (error) {
|
|
print('语音识别流错误: $error');
|
|
_isListening = false;
|
|
isListening.value = false;
|
|
|
|
// 发生错误时完成 completer
|
|
if (_recognitionCompleter != null && !_recognitionCompleter!.isCompleted) {
|
|
_recognitionCompleter!.completeError(error);
|
|
}
|
|
});
|
|
|
|
} catch (e) {
|
|
print('启动语音识别失败: $e');
|
|
_isListening = false;
|
|
isListening.value = false;
|
|
rethrow;
|
|
}
|
|
}
|
|
|
|
// 停止语音识别并返回识别的文本
|
|
Future<String> stopVoiceRecognition() async {
|
|
if (!_isListening) {
|
|
return '';
|
|
}
|
|
|
|
try {
|
|
// 取消静默计时器
|
|
_silenceTimer?.cancel();
|
|
_silenceTimer = null;
|
|
|
|
// 取消无语音超时计时器
|
|
_noSpeechTimer?.cancel();
|
|
_noSpeechTimer = null;
|
|
|
|
// 取消订阅
|
|
await _recognitionSubscription?.cancel();
|
|
_recognitionSubscription = null;
|
|
|
|
// 停止语音识别
|
|
await _voiceRecognitionService.stopContinuousRecognition();
|
|
|
|
// 获取最终识别结果
|
|
final result = recognizedText.value;
|
|
|
|
// 重置状态
|
|
_isListening = false;
|
|
isListening.value = false;
|
|
|
|
return result;
|
|
} catch (e) {
|
|
print('停止语音识别失败: $e');
|
|
_isListening = false;
|
|
isListening.value = false;
|
|
|
|
// 尝试强制重置语音识别服务
|
|
try {
|
|
await _voiceRecognitionService.initialize(
|
|
subscriptionKey: dotenv.env['AZURE_ASR_SUBSCRIPTION_KEY'] ?? '',
|
|
serviceRegion: dotenv.env['AZURE_ASR_SERVICE_REGION'] ?? 'eastasia',
|
|
language: dotenv.env['AZURE_ASR_LANGUAGE'] ?? 'zh-CN',
|
|
);
|
|
} catch (e) {
|
|
print('强制重置语音识别服务失败: $e');
|
|
}
|
|
|
|
return recognizedText.value; // 返回当前已识别的文本
|
|
}
|
|
}
|
|
|
|
int _findLastSentenceEnd(String text) {
|
|
final sentenceEnds = [
|
|
text.lastIndexOf('。'),
|
|
text.lastIndexOf('!'),
|
|
text.lastIndexOf('?'),
|
|
text.lastIndexOf('.'),
|
|
text.lastIndexOf('!'),
|
|
text.lastIndexOf('?'),
|
|
];
|
|
|
|
return sentenceEnds.reduce((max, pos) => pos > max ? pos : max);
|
|
}
|
|
|
|
List<Message> get messageHistory => List.unmodifiable(_messageHistory);
|
|
|
|
void clearHistory() {
|
|
_messageHistory.clear();
|
|
}
|
|
|
|
@override
|
|
void onClose() {
|
|
_recognitionSubscription?.cancel();
|
|
_silenceTimer?.cancel();
|
|
_noSpeechTimer?.cancel();
|
|
_aiResponseSubscription?.cancel();
|
|
_ttsEventSubscription?.cancel();
|
|
_shouldContinueInteraction = false;
|
|
|
|
// 断开TTS服务连接
|
|
_ttsService.disconnect();
|
|
|
|
super.onClose();
|
|
}
|
|
}
|