12 changed files with 516 additions and 1825 deletions
@ -1,618 +0,0 @@ |
|||
import Foundation |
|||
import AVFoundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
// 自定义语音处理组件,提供音频流处理等功能 |
|||
import speech |
|||
import os.log |
|||
|
|||
|
|||
/** |
|||
* Azure ASR Helper |
|||
* |
|||
* 基于微软Azure语音服务的ASR实现 |
|||
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-recognize-speech |
|||
*/ |
|||
public class AzureAsrHelper: NSObject { |
|||
private let tag = "AzureAsrHelper" |
|||
// 日志对象 |
|||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureAsrHelper") |
|||
|
|||
// 核心组件 |
|||
private var speechConfig: SPXSpeechConfiguration? |
|||
private var recognizer: SPXSpeechRecognizer? |
|||
private var audioConfig: SPXAudioConfiguration? |
|||
|
|||
// 状态管理 |
|||
private var _isContinuousRecognitionActive = false |
|||
|
|||
// 配置参数 |
|||
private var currentLanguage = "zh-CN" |
|||
private var supportedLanguages = ["zh-CN"] |
|||
private var isAutoDetectLanguage = false |
|||
private var subscriptionKey = "" |
|||
private var region = "" |
|||
|
|||
// 音频源配置 |
|||
public enum AudioSourceType { |
|||
/** 使用设备麦克风 */ |
|||
case microphone |
|||
|
|||
/** 使用外部提供的音频数据 */ |
|||
case external |
|||
} |
|||
|
|||
private var audioSourceType = AudioSourceType.microphone |
|||
|
|||
// 音频处理 |
|||
private var externalAudioStream: ExternalAudioPullStream? |
|||
|
|||
/** |
|||
* 初始化Azure语音服务 |
|||
* |
|||
* @param subscriptionKey Azure 订阅密钥 |
|||
* @param region Azure 区域 |
|||
* @param supportedLanguages 支持的语言数组 |
|||
* @param audioSourceType 音频源类型 |
|||
* @return 初始化是否成功 |
|||
*/ |
|||
public func initialize( |
|||
subscriptionKey: String, |
|||
region: String, |
|||
supportedLanguages: [String] = ["zh-CN"], |
|||
audioSourceType: AudioSourceType = .microphone |
|||
) -> Bool { |
|||
do { |
|||
// 检查配置是否为空 |
|||
if subscriptionKey.isEmpty || region.isEmpty { |
|||
os_log("Azure 配置信息不完整", log: log, type: .error) |
|||
return false |
|||
} |
|||
|
|||
// 释放之前的资源 |
|||
dispose() |
|||
|
|||
// 保存配置 |
|||
self.subscriptionKey = subscriptionKey |
|||
self.region = region |
|||
self.audioSourceType = audioSourceType |
|||
|
|||
// 设置语言 |
|||
if !supportedLanguages.isEmpty { |
|||
self.supportedLanguages = supportedLanguages |
|||
} |
|||
|
|||
// 根据支持的语言数量决定是否启用自动语言检测 |
|||
self.isAutoDetectLanguage = supportedLanguages.count >= 2 |
|||
|
|||
// 如果只有一种语言,设置为当前语言 |
|||
if !isAutoDetectLanguage && !supportedLanguages.isEmpty { |
|||
self.currentLanguage = supportedLanguages[0] |
|||
} |
|||
|
|||
// 创建语音配置 |
|||
speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region) |
|||
|
|||
if isAutoDetectLanguage { |
|||
// 直接启用语言检测模式 |
|||
speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") |
|||
os_log("启用语言检测模式", log: log, type: .info) |
|||
} else { |
|||
// 设置指定的识别语言 |
|||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|||
} |
|||
|
|||
// 创建识别器 |
|||
return setupRecognizer() |
|||
} catch { |
|||
os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 向音频流写入音频数据 |
|||
* 仅当音频源设置为external时有效 |
|||
* |
|||
* @param data 音频数据字节数组 |
|||
*/ |
|||
public func pushAudioData(data: Data) { |
|||
if audioSourceType != .external { |
|||
return |
|||
} |
|||
|
|||
// 使用拉流模式,将数据推入队列 |
|||
externalAudioStream?.pushAudio(data) |
|||
} |
|||
|
|||
/** |
|||
* 开始连续语音识别 |
|||
* |
|||
* @param callback 连续识别结果回调 |
|||
* @param audioSourceType 音频源类型 |
|||
* @return 是否成功开始识别 |
|||
*/ |
|||
public func startContinuousRecognition( |
|||
callback: ContinuousRecognizeCallback, |
|||
audioSourceType: AudioSourceType = .microphone |
|||
) -> Bool { |
|||
guard speechConfig != nil else { |
|||
callback.onError("语音服务未初始化") |
|||
return false |
|||
} |
|||
|
|||
if _isContinuousRecognitionActive { |
|||
return true |
|||
} |
|||
|
|||
self.audioSourceType = audioSourceType |
|||
|
|||
// 重置识别器 |
|||
if !setupRecognizer() { |
|||
callback.onError("重置识别器失败") |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 设置各种事件监听 |
|||
setupEventListeners(callback: callback) |
|||
|
|||
// 启动音频处理 |
|||
startAudioProcessing() |
|||
|
|||
// 开始连续识别 |
|||
try recognizer?.startContinuousRecognition() |
|||
_isContinuousRecognitionActive = true |
|||
|
|||
os_log("连续识别已启动,音频源: %{public}@", log: log, type: .info, audioSourceType == .microphone ? "麦克风" : "外部") |
|||
|
|||
return true |
|||
} catch { |
|||
stopAudioProcessing() |
|||
_isContinuousRecognitionActive = false |
|||
callback.onError("启动连续识别失败: \(error.localizedDescription)") |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 停止连续语音识别 |
|||
* |
|||
* @return 是否成功停止 |
|||
*/ |
|||
public func stopContinuousRecognition() -> Bool { |
|||
guard speechConfig != nil else { |
|||
os_log("语音服务未初始化", log: log, type: .error) |
|||
return false |
|||
} |
|||
|
|||
if !_isContinuousRecognitionActive { |
|||
return true |
|||
} |
|||
|
|||
do { |
|||
guard let recognizer = recognizer else { |
|||
os_log("识别器为空,重置状态", log: log, type: .info) |
|||
_isContinuousRecognitionActive = false |
|||
return true |
|||
} |
|||
|
|||
if audioSourceType == .external { |
|||
pushAudioData(data: Data()) |
|||
} |
|||
|
|||
// 停止连续识别 |
|||
try recognizer.stopContinuousRecognition() |
|||
|
|||
// 停止音频处理 |
|||
stopAudioProcessing() |
|||
|
|||
// 会话结束事件会设置_isContinuousRecognitionActive = false |
|||
return true |
|||
} catch { |
|||
// 强制重置状态 |
|||
_isContinuousRecognitionActive = false |
|||
os_log("停止连续识别失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
|
|||
// 停止音频处理 |
|||
stopAudioProcessing() |
|||
|
|||
// 尝试强制关闭识别器 |
|||
recognizer = nil |
|||
|
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 检查连续识别是否活跃 |
|||
*/ |
|||
public func isContinuousRecognitionActive() -> Bool { |
|||
return self._isContinuousRecognitionActive |
|||
} |
|||
|
|||
/** |
|||
* 释放所有资源 |
|||
*/ |
|||
public func dispose() { |
|||
do { |
|||
// 如果正在进行连续识别,先停止 |
|||
if _isContinuousRecognitionActive { |
|||
// 直接停止,不等待结果 |
|||
try? recognizer?.stopContinuousRecognition() |
|||
_isContinuousRecognitionActive = false |
|||
} |
|||
|
|||
// 停止音频处理 |
|||
stopAudioProcessing() |
|||
|
|||
// 释放资源 |
|||
recognizer = nil |
|||
speechConfig = nil |
|||
audioConfig = nil |
|||
|
|||
// 确保状态被重置 |
|||
_isContinuousRecognitionActive = false |
|||
externalAudioStream = nil |
|||
} catch { |
|||
// 确保状态被重置 |
|||
_isContinuousRecognitionActive = false |
|||
externalAudioStream = nil |
|||
audioConfig = nil |
|||
recognizer = nil |
|||
speechConfig = nil |
|||
} |
|||
} |
|||
|
|||
// MARK: - 私有方法 |
|||
|
|||
/** |
|||
* 设置识别器 |
|||
*/ |
|||
private func setupRecognizer() -> Bool { |
|||
do { |
|||
// 清理旧的识别器 |
|||
recognizer = nil |
|||
|
|||
// 设置音频配置 |
|||
switch audioSourceType { |
|||
case .microphone: |
|||
// 使用默认麦克风输入配置 |
|||
audioConfig = try SPXAudioConfiguration() |
|||
case .external: |
|||
// 使用拉流方式处理外部音频 |
|||
setupExternalAudioStream() |
|||
} |
|||
|
|||
// 创建识别器 |
|||
if isAutoDetectLanguage { |
|||
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) |
|||
recognizer = try SPXSpeechRecognizer( |
|||
speechConfiguration: speechConfig!, |
|||
autoDetectSourceLanguageConfiguration: autoDetectConfig, |
|||
audioConfiguration: audioConfig! |
|||
) |
|||
} else { |
|||
recognizer = try SPXSpeechRecognizer( |
|||
speechConfiguration: speechConfig!, |
|||
audioConfiguration: audioConfig! |
|||
) |
|||
} |
|||
|
|||
return true |
|||
} catch { |
|||
os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
stopAudioProcessing() |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置外部音频流 - 使用拉流方式 |
|||
*/ |
|||
private func setupExternalAudioStream() { |
|||
do { |
|||
// 创建外部音频拉流对象 |
|||
externalAudioStream = ExternalAudioPullStream() |
|||
|
|||
// 创建音频配置 |
|||
audioConfig = SPXAudioConfiguration(streamInput: externalAudioStream!.pullStream) |
|||
} catch { |
|||
os_log("设置外部音频流失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
externalAudioStream = nil |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置事件监听器 |
|||
*/ |
|||
private func setupEventListeners(callback: ContinuousRecognizeCallback) { |
|||
guard let recognizer = recognizer else { return } |
|||
|
|||
// 识别中事件 |
|||
recognizer.addRecognizingEventHandler { [weak self] (sender, event) in |
|||
guard let self = self else { return } |
|||
|
|||
let result = event.result |
|||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|||
// os_log("识别中: %{public}@", log: self.log, type: .info, result.text ?? "") |
|||
// 直接在当前线程调用回调 |
|||
callback.onRecognizing(result.text ?? "", detectedLanguage) |
|||
} |
|||
|
|||
// 识别完成事件 |
|||
recognizer.addRecognizedEventHandler { [weak self] (sender, event) in |
|||
guard let self = self else { return } |
|||
|
|||
let result = event.result |
|||
if result.reason == SPXResultReason.recognizedSpeech { |
|||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|||
// 直接在当前线程调用回调 |
|||
callback.onResult(result.text ?? "", detectedLanguage) |
|||
} |
|||
} |
|||
|
|||
// 会话开始事件 |
|||
recognizer.addSessionStartedEventHandler { (sender, event) in |
|||
// 直接在当前线程调用回调 |
|||
callback.onSessionStarted() |
|||
} |
|||
|
|||
// 会话结束事件 |
|||
recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in |
|||
guard let self = self else { return } |
|||
|
|||
// 直接在当前线程调用回调 |
|||
callback.onSessionStopped() |
|||
self._isContinuousRecognitionActive = false |
|||
self.stopAudioProcessing() |
|||
} |
|||
|
|||
// 取消事件 |
|||
recognizer.addCanceledEventHandler { [weak self] (sender, event) in |
|||
guard let self = self else { return } |
|||
|
|||
let errorDetails = event.errorDetails ?? "未知错误" |
|||
let reason = String(describing: event.reason.rawValue) |
|||
|
|||
os_log("识别取消: %{public}@", log: self.log, type: .error, errorDetails) |
|||
|
|||
callback.onCanceled(reason, errorDetails) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 获取检测到的语言 |
|||
*/ |
|||
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|||
if !isAutoDetectLanguage { |
|||
return "" |
|||
} |
|||
|
|||
// 尝试从属性中获取语言 |
|||
if let properties = result.properties, |
|||
let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage") { |
|||
return language |
|||
} |
|||
|
|||
// 尝试另一种方式获取语言 |
|||
do { |
|||
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|||
return langResult.language ?? "" |
|||
} catch { |
|||
return "" |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 启动音频处理 |
|||
*/ |
|||
private func startAudioProcessing() { |
|||
switch audioSourceType { |
|||
case .microphone: |
|||
// 拉流模式不需要额外启动,SDK会自动拉取数据 |
|||
break |
|||
case .external: |
|||
// 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData |
|||
break |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 停止音频处理 |
|||
*/ |
|||
private func stopAudioProcessing() { |
|||
if let stream = externalAudioStream { |
|||
stream.close() |
|||
externalAudioStream = nil |
|||
// os_log("外部音频流已关闭", log: log, type: .info) |
|||
} |
|||
} |
|||
|
|||
|
|||
/** |
|||
* 外部音频拉流 |
|||
* 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式 |
|||
*/ |
|||
private class ExternalAudioPullStream: NSObject { |
|||
private(set) var pullStream: SPXPullAudioInputStream! |
|||
private let queue = LinkedBlockingQueue<Data>() |
|||
private var closed = false |
|||
|
|||
override init() { |
|||
super.init() |
|||
|
|||
pullStream = SPXPullAudioInputStream( |
|||
readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in |
|||
guard let self = self else { return 0 } |
|||
return self.read(buffer: data, size: Int(size)) |
|||
}, |
|||
closeHandler: { [weak self] in |
|||
self?.close() |
|||
} |
|||
) |
|||
} |
|||
|
|||
/** |
|||
* 外部调用:推送音频数据到队列 |
|||
* @param data 音频数据 |
|||
*/ |
|||
func pushAudio(_ data: Data) { |
|||
if !closed { |
|||
queue.put(data) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* SDK调用:从队列中拉取数据 |
|||
* @param buffer SDK提供的缓冲区 |
|||
* @param size 缓冲区大小 |
|||
* @return 读取的字节数,0表示流结束 |
|||
*/ |
|||
private func read(buffer: NSMutableData, size: Int) -> Int { |
|||
// 阻塞等待下一块数据 |
|||
guard let chunk = queue.take() else { |
|||
return 0 // 队列已关闭 |
|||
} |
|||
|
|||
// 如果是空数据,表示流结束 |
|||
if chunk.isEmpty { |
|||
return 0 |
|||
} |
|||
|
|||
let toCopy = min(chunk.count, size) |
|||
buffer.append(chunk.prefix(toCopy)) |
|||
return toCopy |
|||
} |
|||
|
|||
/** |
|||
* SDK调用:关闭流 |
|||
*/ |
|||
func close() { |
|||
closed = true |
|||
queue.close() |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* iOS版LinkedBlockingQueue实现 |
|||
* 与Android LinkedBlockingQueue保持一致的API和行为 |
|||
*/ |
|||
private class LinkedBlockingQueue<T> { |
|||
private var queue: [T] = [] |
|||
private var isClosed = false |
|||
|
|||
// 使用DispatchSemaphore实现阻塞行为 |
|||
private let availableItems: DispatchSemaphore |
|||
private let queueLock = NSLock() |
|||
|
|||
init() { |
|||
self.availableItems = DispatchSemaphore(value: 0) |
|||
} |
|||
|
|||
/** |
|||
* 阻塞式获取元素(等价于Android的take()) |
|||
* @return 队列中的元素,如果队列已关闭则返回nil |
|||
*/ |
|||
func take() -> T? { |
|||
// 等待可用元素(阻塞直到有元素或队列关闭) |
|||
availableItems.wait() |
|||
|
|||
queueLock.lock() |
|||
defer { queueLock.unlock() } |
|||
|
|||
// 检查队列是否已关闭 |
|||
if isClosed && queue.isEmpty { |
|||
return nil |
|||
} |
|||
|
|||
// 获取第一个元素 |
|||
guard !queue.isEmpty else { |
|||
return nil |
|||
} |
|||
|
|||
return queue.removeFirst() |
|||
} |
|||
|
|||
/** |
|||
* 添加元素(等价于Android的put()) |
|||
* @param item 要添加的元素 |
|||
*/ |
|||
func put(_ item: T) { |
|||
queueLock.lock() |
|||
defer { queueLock.unlock() } |
|||
|
|||
// 检查队列是否已关闭 |
|||
if isClosed { |
|||
return |
|||
} |
|||
|
|||
// 添加元素(无容量限制,与Android一致) |
|||
queue.append(item) |
|||
|
|||
// 通知有新元素可用 |
|||
availableItems.signal() |
|||
} |
|||
|
|||
/** |
|||
* 关闭队列 |
|||
* 关闭后不能再添加新元素,但可以继续取出已有元素 |
|||
*/ |
|||
func close() { |
|||
queueLock.lock() |
|||
defer { queueLock.unlock() } |
|||
|
|||
isClosed = true |
|||
|
|||
// 唤醒所有等待的take()操作 |
|||
for _ in 0..<100 { // 假设最多100个等待者 |
|||
availableItems.signal() |
|||
} |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 连续识别回调接口 |
|||
*/ |
|||
public protocol ContinuousRecognizeCallback { |
|||
/** |
|||
* 返回识别结果 |
|||
* |
|||
* @param text 识别的文本 |
|||
* @param detectedLanguage 检测到的语言 |
|||
*/ |
|||
func onResult(_ text: String, _ detectedLanguage: String) |
|||
|
|||
/** |
|||
* 识别进行中调用 |
|||
* |
|||
* @param recognizing 正在识别的文本 |
|||
* @param detectedLanguage 检测到的语言 |
|||
*/ |
|||
func onRecognizing(_ recognizing: String, _ detectedLanguage: String) |
|||
|
|||
/** |
|||
* 会话开始时调用 |
|||
*/ |
|||
func onSessionStarted() |
|||
|
|||
/** |
|||
* 会话结束时调用 |
|||
*/ |
|||
func onSessionStopped() |
|||
|
|||
/** |
|||
* 识别取消时调用 |
|||
* |
|||
* @param reason 取消原因 |
|||
* @param errorDetails 错误详情 |
|||
*/ |
|||
func onCanceled(_ reason: String, _ errorDetails: String) |
|||
|
|||
/** |
|||
* 识别出错时调用 |
|||
* |
|||
* @param error 错误信息 |
|||
*/ |
|||
func onError(_ error: String) |
|||
} |
|||
} |
|||
@ -1,17 +0,0 @@ |
|||
//
|
|||
// AzureSpeechPlugin.h
|
|||
// azure_speech
|
|||
//
|
|||
// Created for azure_speech plugin compatibility.
|
|||
//
|
|||
|
|||
#import <Flutter/Flutter.h> |
|||
|
|||
/**
|
|||
* Azure Speech Plugin 头文件 |
|||
* |
|||
* 注:实际实现使用 Swift,此头文件仅用于 Flutter 框架兼容 |
|||
*/ |
|||
@interface AzureSpeechPlugin : NSObject <FlutterPlugin> |
|||
+ (void)registerWithRegistrar:(NSObject<FlutterPluginRegistrar>*)registrar; |
|||
@end |
|||
@ -1,422 +0,0 @@ |
|||
import Flutter |
|||
import UIKit |
|||
import AVFoundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
// 自定义语音处理组件,提供音频流处理等功能 |
|||
import speech |
|||
import os.log |
|||
|
|||
/** |
|||
* Azure Speech Plugin |
|||
* |
|||
* 基于微软Azure语音服务的Flutter插件 |
|||
* 提供语音识别(ASR)和语音合成(TTS)功能 |
|||
*/ |
|||
@objc public class AzureSpeechPlugin: NSObject, FlutterPlugin { |
|||
// 日志标签 |
|||
private let tag = "AzureSpeechPlugin" |
|||
// 日志对象 |
|||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureSpeechPlugin") |
|||
|
|||
// ASR相关 |
|||
private var asrChannel: FlutterMethodChannel? |
|||
private var asrEventChannel: FlutterEventChannel? |
|||
private var asrEventSink: FlutterEventSink? |
|||
private let azureAsrHelper = AzureAsrHelper() |
|||
|
|||
// TTS相关 |
|||
private var ttsChannel: FlutterMethodChannel? |
|||
private var ttsEventChannel: FlutterEventChannel? |
|||
private var ttsEventSink: FlutterEventSink? |
|||
private let azureTtsHelper = AzureTtsHelper() |
|||
|
|||
// 是否已添加TTS事件监听器 |
|||
private var isTtsListenerAdded = false |
|||
|
|||
// 当前的连续识别回调 |
|||
private var currentAsrCallback: AsrCallbackWrapper? |
|||
|
|||
// 插件注册 |
|||
public static func register(with registrar: FlutterPluginRegistrar) { |
|||
let instance = AzureSpeechPlugin() |
|||
|
|||
// 初始化ASR通道 |
|||
let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) |
|||
registrar.addMethodCallDelegate(instance, channel: asrChannel) |
|||
instance.asrChannel = asrChannel |
|||
|
|||
// 初始化TTS通道 |
|||
let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) |
|||
registrar.addMethodCallDelegate(instance, channel: ttsChannel) |
|||
instance.ttsChannel = ttsChannel |
|||
|
|||
// 初始化ASR事件通道 |
|||
let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) |
|||
asrEventChannel.setStreamHandler(instance) |
|||
instance.asrEventChannel = asrEventChannel |
|||
|
|||
// 初始化TTS事件通道 |
|||
let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger()) |
|||
ttsEventChannel.setStreamHandler(instance) |
|||
instance.ttsEventChannel = ttsEventChannel |
|||
} |
|||
|
|||
// 发送ASR事件方法 |
|||
internal func sendAsrEvent(_ event: [String: Any]) { |
|||
if asrEventSink == nil { |
|||
os_log("无法发送ASR事件:事件通道未准备好", log: log, type: .error) |
|||
return |
|||
} |
|||
|
|||
DispatchQueue.main.async { [weak self] in |
|||
guard let self = self else { return } |
|||
self.asrEventSink?(event) |
|||
} |
|||
} |
|||
|
|||
// 发送TTS事件方法 |
|||
private func sendTtsEvent(_ event: [String: Any]) { |
|||
if ttsEventSink == nil { |
|||
os_log("无法发送TTS事件:事件通道未准备好", log: log, type: .error) |
|||
return |
|||
} |
|||
|
|||
DispatchQueue.main.async { [weak self] in |
|||
guard let self = self else { return } |
|||
self.ttsEventSink?(event) |
|||
} |
|||
} |
|||
|
|||
// 设置TTS事件监听器 |
|||
private func setupTtsEventListener() { |
|||
if !isTtsListenerAdded { |
|||
azureTtsHelper.addListener(self) |
|||
isTtsListenerAdded = true |
|||
} |
|||
} |
|||
|
|||
|
|||
|
|||
// 处理Flutter方法调用 |
|||
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|||
if call.method.hasPrefix("tts_") { |
|||
handleTtsMethodCall(call, result) |
|||
} else { |
|||
handleAsrMethodCall(call, result) |
|||
} |
|||
} |
|||
|
|||
// MARK: - ASR 方法处理 |
|||
|
|||
private func handleAsrMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|||
switch call.method { |
|||
case "initialize": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let subscriptionKey = args["subscriptionKey"] as? String, |
|||
let region = args["region"] as? String else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
let supportedLanguages = args["supportedLanguages"] as? [String] ?? ["zh-CN"] |
|||
|
|||
// 初始化ASR引擎 |
|||
let success = azureAsrHelper.initialize( |
|||
subscriptionKey: subscriptionKey, |
|||
region: region, |
|||
supportedLanguages: supportedLanguages, |
|||
audioSourceType: .microphone |
|||
) |
|||
|
|||
result(success) |
|||
|
|||
case "recognizeOnce": |
|||
// iOS版本不支持recognizeOnce |
|||
result(FlutterError(code: "NOT_SUPPORTED", message: "iOS版本不支持recognizeOnce", details: nil)) |
|||
|
|||
case "startContinuousRecognition": |
|||
// 确保事件通道已准备好 |
|||
guard asrEventSink != nil else { |
|||
result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil)) |
|||
return |
|||
} |
|||
|
|||
// 解析音频源类型 |
|||
let audioSourceType: AzureAsrHelper.AudioSourceType |
|||
if let args = call.arguments as? [String: Any], |
|||
let audioSourceString = args["audioSourceType"] as? String { |
|||
switch audioSourceString.lowercased() { |
|||
case "external": |
|||
audioSourceType = .external |
|||
default: |
|||
audioSourceType = .microphone |
|||
} |
|||
} else { |
|||
audioSourceType = .microphone |
|||
} |
|||
|
|||
// 创建回调包装器 |
|||
currentAsrCallback = AsrCallbackWrapper(plugin: self) |
|||
|
|||
let success = azureAsrHelper.startContinuousRecognition( |
|||
callback: currentAsrCallback!, |
|||
audioSourceType: audioSourceType |
|||
) |
|||
result(success) |
|||
|
|||
case "stopContinuousRecognition": |
|||
let success = azureAsrHelper.stopContinuousRecognition() |
|||
currentAsrCallback = nil |
|||
result(success) |
|||
|
|||
case "isContinuousRecognitionActive": |
|||
result(azureAsrHelper.isContinuousRecognitionActive()) |
|||
|
|||
case "dispose": |
|||
azureAsrHelper.dispose() |
|||
currentAsrCallback = nil |
|||
result(true) |
|||
|
|||
case "pushAudioData": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let audioBytes = args["data"] as? FlutterStandardTypedData else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
azureAsrHelper.pushAudioData(data: audioBytes.data) |
|||
result(true) |
|||
|
|||
default: |
|||
result(FlutterMethodNotImplemented) |
|||
} |
|||
} |
|||
|
|||
// MARK: - TTS 方法处理 |
|||
|
|||
private func handleTtsMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|||
switch call.method { |
|||
case "tts_initialize": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let subscriptionKey = args["subscriptionKey"] as? String, |
|||
let region = args["region"] as? String else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
// 初始化TTS引擎 |
|||
let language = args["language"] as? String ?? "zh-CN" |
|||
let success = azureTtsHelper.initialize(ttsAppId: "", ttsAppToken: subscriptionKey, ttsResource: region, language: language) |
|||
|
|||
// 设置TTS事件监听器 |
|||
setupTtsEventListener() |
|||
|
|||
result(success) |
|||
|
|||
case "tts_set_voice": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let voiceName = args["voiceName"] as? String else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "语音名称不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
let success = azureTtsHelper.setVoice(voiceName) |
|||
result(success) |
|||
|
|||
case "tts_speak_once": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let text = args["text"] as? String else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
let success = azureTtsHelper.speakOnce(text) |
|||
result(success) |
|||
|
|||
case "tts_speak_stream": |
|||
guard let args = call.arguments as? [String: Any], |
|||
let text = args["text"] as? String else { |
|||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) |
|||
return |
|||
} |
|||
|
|||
let success = azureTtsHelper.speakStream(text) |
|||
result(success) |
|||
|
|||
case "tts_flush_stream": |
|||
let success = azureTtsHelper.flushStream() |
|||
result(success) |
|||
|
|||
case "tts_stop": |
|||
let success = azureTtsHelper.stop() |
|||
result(success) |
|||
|
|||
case "tts_isSpeaking": |
|||
result(azureTtsHelper.isSpeaking()) |
|||
|
|||
case "tts_release": |
|||
azureTtsHelper.dispose() |
|||
result(true) |
|||
|
|||
default: |
|||
result(FlutterMethodNotImplemented) |
|||
} |
|||
} |
|||
} |
|||
|
|||
// MARK: - ASR回调包装器 |
|||
private class AsrCallbackWrapper: AzureAsrHelper.ContinuousRecognizeCallback { |
|||
private weak var plugin: AzureSpeechPlugin? |
|||
|
|||
init(plugin: AzureSpeechPlugin) { |
|||
self.plugin = plugin |
|||
} |
|||
|
|||
func onResult(_ text: String, _ detectedLanguage: String) { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "result", |
|||
"text": text, |
|||
"language": detectedLanguage |
|||
]) |
|||
} |
|||
|
|||
func onRecognizing(_ recognizing: String, _ detectedLanguage: String) { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "recognizing", |
|||
"text": recognizing, |
|||
"language": detectedLanguage |
|||
]) |
|||
} |
|||
|
|||
func onSessionStarted() { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "sessionStarted" |
|||
]) |
|||
} |
|||
|
|||
func onSessionStopped() { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "sessionStopped" |
|||
]) |
|||
} |
|||
|
|||
func onCanceled(_ reason: String, _ errorDetails: String) { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "canceled", |
|||
"reason": reason, |
|||
"errorDetails": errorDetails |
|||
]) |
|||
} |
|||
|
|||
func onError(_ error: String) { |
|||
plugin?.sendAsrEvent([ |
|||
"type": "error", |
|||
"error": error |
|||
]) |
|||
} |
|||
} |
|||
|
|||
// MARK: - 事件处理 |
|||
extension AzureSpeechPlugin: FlutterStreamHandler { |
|||
public func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
|||
// 根据通道类型设置事件接收器 |
|||
if let args = arguments as? [String: Any], |
|||
let channel = args["channel"] as? String { |
|||
|
|||
if channel == "asr" { |
|||
asrEventSink = events |
|||
} else if channel == "tts" { |
|||
ttsEventSink = events |
|||
setupTtsEventListener() |
|||
} |
|||
} else if arguments == nil { |
|||
// 如果没有指定通道,尝试弄清楚是哪个通道在监听 |
|||
if asrEventSink == nil && ttsEventSink != nil { |
|||
asrEventSink = events |
|||
} else if asrEventSink != nil && ttsEventSink == nil { |
|||
ttsEventSink = events |
|||
setupTtsEventListener() |
|||
} else { |
|||
// 无法确定哪个通道,默认设置为ASR |
|||
asrEventSink = events |
|||
} |
|||
} |
|||
|
|||
return nil |
|||
} |
|||
|
|||
public func onCancel(withArguments arguments: Any?) -> FlutterError? { |
|||
// 根据通道类型清除事件接收器 |
|||
if let args = arguments as? [String: Any], |
|||
let channel = args["channel"] as? String { |
|||
|
|||
if channel == "asr" { |
|||
asrEventSink = nil |
|||
} else if channel == "tts" { |
|||
ttsEventSink = nil |
|||
} |
|||
} else { |
|||
// 如果没有指定通道,清除所有通道 |
|||
asrEventSink = nil |
|||
ttsEventSink = nil |
|||
} |
|||
|
|||
return nil |
|||
} |
|||
} |
|||
|
|||
|
|||
|
|||
// MARK: - TTS 事件监听实现 |
|||
extension AzureSpeechPlugin: TtsEventListener { |
|||
// 使用@objc特性为方法提供一个不同的Objective-C选择器名称 |
|||
@objc(onTtsEvent:) |
|||
public func onEvent(_ event: TtsEvent) { |
|||
var eventMap: [String: Any] = [:] |
|||
|
|||
// 根据事件类型转换 |
|||
switch event.type { |
|||
case .synthesisStarted: |
|||
eventMap["type"] = "synthesis_started" |
|||
case .synthesisCompleted: |
|||
eventMap["type"] = "synthesis_completed" |
|||
case .synthesisCanceled: |
|||
eventMap["type"] = "synthesis_canceled" |
|||
if let reason = event.params["reason"] { |
|||
eventMap["reason"] = reason |
|||
} |
|||
if let errorDetails = event.params["errorDetails"] { |
|||
eventMap["errorDetails"] = errorDetails |
|||
} |
|||
case .error: |
|||
eventMap["type"] = "error" |
|||
if let errorCode = event.params["errorCode"] { |
|||
eventMap["errorCode"] = errorCode |
|||
} |
|||
if let errorMessage = event.params["errorMessage"] { |
|||
eventMap["errorMessage"] = errorMessage |
|||
} |
|||
@unknown default: |
|||
eventMap["type"] = "unknown" |
|||
for (key, value) in event.params { |
|||
eventMap[key] = value |
|||
} |
|||
} |
|||
|
|||
sendTtsEvent(eventMap) |
|||
} |
|||
} |
|||
|
|||
// MARK: - 音频数据监听实现 |
|||
extension AzureSpeechPlugin: AudioDataListener { |
|||
public func onAudioData(_ data: Data) { |
|||
if ttsEventSink == nil { return } |
|||
|
|||
let eventMap: [String: Any] = [ |
|||
"type": "audio_data", |
|||
"data": FlutterStandardTypedData(bytes: data) |
|||
] |
|||
|
|||
sendTtsEvent(eventMap) |
|||
} |
|||
} |
|||
@ -1,604 +0,0 @@ |
|||
import Foundation |
|||
import AVFoundation |
|||
import MicrosoftCognitiveServicesSpeech |
|||
// 自定义语音处理组件,提供音频流处理等功能 |
|||
import speech |
|||
import os.log |
|||
|
|||
/** |
|||
* Azure TTS Helper |
|||
* |
|||
* 基于微软Azure语音服务的TTS实现 |
|||
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis |
|||
* |
|||
* 特性: |
|||
* - 对外接口保持同步,内部异步处理 |
|||
* - 使用专用队列保证合成顺序 |
|||
* - 避免阻塞主线程和其他后台线程 |
|||
* - 事件回调由调用方处理线程切换 |
|||
* |
|||
* 使用示例: |
|||
* let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理 |
|||
*/ |
|||
public class AzureTtsHelper: NSObject, ITtsService { |
|||
private let tag = "AzureTtsHelper" |
|||
// 日志对象 |
|||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") |
|||
|
|||
private static let DEFAULT_LANGUAGE = "zh-CN" |
|||
private static let DEFAULT_VOICE = "zh-CN-XiaoxiaoNeural" |
|||
|
|||
// Azure语音服务配置 |
|||
private var speechConfig: SPXSpeechConfiguration? |
|||
private var synthesizer: SPXSpeechSynthesizer? |
|||
private var isInitialized = false |
|||
private var speaking = false |
|||
|
|||
// 当前配置 |
|||
private var currentVoice = DEFAULT_VOICE |
|||
private var currentLanguage = DEFAULT_LANGUAGE |
|||
private var currentRate = "+0%" |
|||
private var currentPitch = "+0%" |
|||
private var currentVolume = "100%" |
|||
|
|||
// 事件监听器列表 |
|||
private var eventListeners = NSHashTable<AnyObject>.weakObjects() |
|||
|
|||
// 流式文本处理的缓冲区 |
|||
private var streamBuffer = "" |
|||
private var lastSpeakTime: TimeInterval = 0 |
|||
|
|||
// 内部异步处理队列,保证顺序执行 |
|||
private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) |
|||
private let synthesisGroup = DispatchGroup() |
|||
private var pendingTasks: [() -> Void] = [] |
|||
private let taskLock = NSLock() |
|||
|
|||
/** |
|||
* 初始化语音合成服务 |
|||
* |
|||
* @param appId 服务应用ID |
|||
* @param token 服务访问令牌/订阅密钥 |
|||
* @param resource 服务资源ID/区域(可选) |
|||
* @param language 语言代码,如"zh-CN" |
|||
* @return 是否初始化成功 |
|||
*/ |
|||
public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { |
|||
do { |
|||
// 创建语音配置 |
|||
if ttsResource.isEmpty { |
|||
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia") |
|||
} else { |
|||
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) |
|||
} |
|||
|
|||
// 设置语言 |
|||
speechConfig?.speechSynthesisLanguage = language |
|||
currentLanguage = language |
|||
|
|||
// 设置音频输出格式 - 使用16k、16位的PCM格式 |
|||
speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", |
|||
byName: "SpeechServiceConnection_SynthOutputFormat") |
|||
|
|||
// 创建合成器,使用默认音频输出(扬声器) |
|||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|||
|
|||
// 设置事件监听 |
|||
setupEventListeners() |
|||
|
|||
// 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny |
|||
if language.lowercased().starts(with: "zh") { |
|||
_ = setVoice("zh-CN-XiaoxiaoNeural") |
|||
} else { |
|||
_ = setVoice("en-US-JennyNeural") |
|||
} |
|||
|
|||
// 设置初始化完成 |
|||
isInitialized = true |
|||
|
|||
return true |
|||
} catch { |
|||
os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置语音角色 |
|||
* |
|||
* @param voiceName 语音角色名称(不同服务的语音角色命名可能不同) |
|||
* @return 是否设置成功 |
|||
*/ |
|||
public func setVoice(_ voiceName: String) -> Bool { |
|||
if !isInitialized { return false } |
|||
|
|||
do { |
|||
currentVoice = voiceName |
|||
|
|||
// 更新语音名称 |
|||
if let config = speechConfig { |
|||
config.speechSynthesisVoiceName = voiceName |
|||
} |
|||
|
|||
// 重新创建合成器 |
|||
recreateSynthesizer() |
|||
return true |
|||
} catch { |
|||
os_log("设置语音失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "VOICE_SET_FAILED", |
|||
"errorMessage": "设置语音失败: \(error.localizedDescription)" |
|||
]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 单次合成并播放 |
|||
* |
|||
* @param text 要合成的文本 |
|||
* @return 是否成功开始合成 |
|||
*/ |
|||
public func speakOnce(_ text: String) -> Bool { |
|||
if !isInitialized { |
|||
os_log("语音合成未初始化", log: log, type: .error) |
|||
return false |
|||
} |
|||
|
|||
// 立即返回成功,内部异步处理 |
|||
enqueueSynthesisTask { |
|||
self.performSynthesis(text: text) |
|||
} |
|||
|
|||
return true |
|||
} |
|||
|
|||
/** |
|||
* 将合成任务加入队列,保证顺序执行 |
|||
*/ |
|||
private func enqueueSynthesisTask(_ task: @escaping () -> Void) { |
|||
taskLock.lock() |
|||
defer { taskLock.unlock() } |
|||
|
|||
pendingTasks.append(task) |
|||
|
|||
// 如果当前没有任务在执行,开始处理队列 |
|||
if pendingTasks.count == 1 { |
|||
processNextTask() |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 处理队列中的下一个任务 |
|||
*/ |
|||
private func processNextTask() { |
|||
synthesisQueue.async { |
|||
self.synthesisGroup.enter() |
|||
|
|||
self.taskLock.lock() |
|||
guard !self.pendingTasks.isEmpty else { |
|||
self.taskLock.unlock() |
|||
self.synthesisGroup.leave() |
|||
return |
|||
} |
|||
let task = self.pendingTasks.removeFirst() |
|||
self.taskLock.unlock() |
|||
|
|||
// 执行任务 |
|||
task() |
|||
|
|||
self.synthesisGroup.leave() |
|||
|
|||
// 处理下一个任务 |
|||
self.taskLock.lock() |
|||
if !self.pendingTasks.isEmpty { |
|||
self.taskLock.unlock() |
|||
self.processNextTask() |
|||
} else { |
|||
self.taskLock.unlock() |
|||
} |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 实际执行合成的方法 |
|||
*/ |
|||
private func performSynthesis(text: String) { |
|||
// 重置状态 |
|||
speaking = true |
|||
|
|||
// 生成SSML |
|||
let ssml = generateSsml(text) |
|||
|
|||
do { |
|||
os_log("开始合成: %{public}@", log: log, type: .debug, ssml) |
|||
// 使用同步方法,在后台队列中执行 |
|||
let result = try synthesizer?.startSpeakingSsml(ssml) |
|||
|
|||
// 检查结果 |
|||
if let result = result { |
|||
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) |
|||
} |
|||
} catch { |
|||
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "SYNTHESIS_FAILED", |
|||
"errorMessage": error.localizedDescription |
|||
]) |
|||
speaking = false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 流式合成文本 |
|||
* |
|||
* @param text 要合成的文本片段 |
|||
* @return 是否成功处理 |
|||
*/ |
|||
public func speakStream(_ text: String) -> Bool { |
|||
if !isInitialized || text.isEmpty { |
|||
if !isInitialized { |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "NOT_INITIALIZED", |
|||
"errorMessage": "TTS引擎未初始化" |
|||
]) |
|||
} |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 添加新文本到缓冲区 |
|||
streamBuffer.append(text) |
|||
|
|||
// 增加500ms防抖逻辑 |
|||
let currentTime = Date().timeIntervalSince1970 |
|||
if currentTime - lastSpeakTime < 0.6 { |
|||
return true |
|||
} |
|||
lastSpeakTime = currentTime |
|||
|
|||
let currentText = streamBuffer |
|||
|
|||
// 定义标点符号列表 |
|||
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] |
|||
|
|||
// 查找最后一个标点符号的位置 |
|||
var lastPunctuationIndex = -1 |
|||
for (i, char) in currentText.enumerated().reversed() { |
|||
if punctuationMarks.contains(char) { |
|||
lastPunctuationIndex = i |
|||
break |
|||
} |
|||
} |
|||
|
|||
// 如果找到标点符号,则播放到该标点符号 |
|||
if lastPunctuationIndex >= 0 { |
|||
// 提取要播放的文本(包含标点符号) |
|||
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines) |
|||
|
|||
// 剩余的文本保存在缓冲区中 |
|||
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) |
|||
streamBuffer = String(currentText[startIndex...]) |
|||
|
|||
// 只有非空文本才播放 |
|||
if !textToSpeak.isEmpty { |
|||
return speakOnce(textToSpeak) |
|||
} |
|||
} |
|||
|
|||
// 如果没有找到标点符号,则等待更多文本 |
|||
return true |
|||
} catch { |
|||
os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "STREAM_FAILED", |
|||
"errorMessage": "流式语音合成失败: \(error.localizedDescription)" |
|||
]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 刷新并播放流式文本缓冲区中的剩余内容 |
|||
* |
|||
* @return 是否成功处理 |
|||
*/ |
|||
public func flushStream() -> Bool { |
|||
if !isInitialized { |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "NOT_INITIALIZED", |
|||
"errorMessage": "TTS引擎未初始化" |
|||
]) |
|||
return false |
|||
} |
|||
|
|||
do { |
|||
// 获取缓冲区中剩余的文本 |
|||
let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines) |
|||
|
|||
// 清空缓冲区 |
|||
streamBuffer = "" |
|||
|
|||
// 如果缓冲区为空,直接返回成功 |
|||
if remainingText.isEmpty { |
|||
return true |
|||
} |
|||
|
|||
// 播放剩余文本 |
|||
return speakOnce(remainingText) |
|||
} catch { |
|||
os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "FLUSH_FAILED", |
|||
"errorMessage": "刷新流式文本失败: \(error.localizedDescription)" |
|||
]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 停止语音合成和播放 |
|||
* |
|||
* @return 是否成功停止 |
|||
*/ |
|||
public func stop() -> Bool { |
|||
// 立即更新状态 |
|||
speaking = false |
|||
|
|||
// 清除流式缓冲区中的待播放内容 |
|||
streamBuffer = "" |
|||
|
|||
// 清空待处理的任务队列 |
|||
taskLock.lock() |
|||
pendingTasks.removeAll() |
|||
taskLock.unlock() |
|||
|
|||
// 在后台队列停止合成器,避免阻塞主线程 |
|||
synthesisQueue.async { |
|||
if let synthesizer = self.synthesizer { |
|||
do { |
|||
try synthesizer.stopSpeaking() |
|||
os_log("语音合成已停止", log: self.log, type: .info) |
|||
|
|||
// 直接通知停止完成 |
|||
self.notifyEvent(eventType: .synthesisCanceled) |
|||
} catch { |
|||
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) |
|||
self.notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "STOP_FAILED", |
|||
"errorMessage": error.localizedDescription |
|||
]) |
|||
} |
|||
} |
|||
} |
|||
|
|||
return true |
|||
} |
|||
|
|||
/** |
|||
* 释放资源 |
|||
* 在不再需要服务时调用,释放底层资源 |
|||
*/ |
|||
public func dispose() { |
|||
// 停止播放 |
|||
_ = stop() |
|||
|
|||
// 等待所有任务完成 |
|||
synthesisGroup.wait() |
|||
|
|||
// 清理资源 |
|||
synthesisQueue.async { |
|||
// 释放合成器 |
|||
self.synthesizer = nil |
|||
|
|||
// 释放配置 |
|||
self.speechConfig = nil |
|||
|
|||
// 清空流缓冲区 |
|||
self.streamBuffer = "" |
|||
|
|||
// 重置状态 |
|||
self.isInitialized = false |
|||
self.speaking = false |
|||
|
|||
// 清空任务队列 |
|||
self.taskLock.lock() |
|||
self.pendingTasks.removeAll() |
|||
self.taskLock.unlock() |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 添加TTS事件监听器 |
|||
* |
|||
* @param listener 事件监听器 |
|||
*/ |
|||
public func addListener(_ listener: TtsEventListener) { |
|||
eventListeners.add(listener as AnyObject) |
|||
} |
|||
|
|||
/** |
|||
* 移除TTS事件监听器 |
|||
* |
|||
* @param listener 要移除的事件监听器 |
|||
*/ |
|||
public func removeListener(_ listener: TtsEventListener) { |
|||
eventListeners.remove(listener as AnyObject) |
|||
} |
|||
|
|||
/** |
|||
* 添加音频数据监听器 |
|||
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|||
* |
|||
* @param listener 音频数据监听器 |
|||
*/ |
|||
public func addAudioDataListener(_ listener: AudioDataListener) { |
|||
os_log("警告:不支持音频数据监听器功能", log: log, type: .info) |
|||
} |
|||
|
|||
/** |
|||
* 移除音频数据监听器 |
|||
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|||
* |
|||
* @param listener 要移除的音频数据监听器 |
|||
*/ |
|||
public func removeAudioDataListener(_ listener: AudioDataListener) { |
|||
// 不做任何操作 |
|||
} |
|||
|
|||
/** |
|||
* 当前是否正在播放/合成 |
|||
*/ |
|||
public func isSpeaking() -> Bool { |
|||
return speaking |
|||
} |
|||
|
|||
// MARK: - 辅助方法 |
|||
|
|||
/** |
|||
* 设置事件监听器 |
|||
*/ |
|||
private func setupEventListeners() { |
|||
guard let synthesizer = synthesizer else { return } |
|||
|
|||
// 添加合成开始事件处理器 |
|||
synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in |
|||
guard let self = self else { return } |
|||
self.notifyEvent(eventType: .synthesisStarted) |
|||
} |
|||
|
|||
// 添加合成中事件处理器 |
|||
synthesizer.addSynthesizingEventHandler { [weak self] _, _ in |
|||
// 可以在这里处理合成中的事件,目前没有特别操作 |
|||
} |
|||
|
|||
// 添加合成完成事件处理器 |
|||
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in |
|||
guard let self = self else { return } |
|||
self.speaking = false |
|||
self.notifyEvent(eventType: .synthesisCompleted) |
|||
} |
|||
|
|||
// 添加合成取消事件处理器 |
|||
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in |
|||
guard let self = self else { return } |
|||
|
|||
self.speaking = false |
|||
|
|||
var params: [String: Any] = [:] |
|||
do { |
|||
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) |
|||
if cancellationDetails.reason == SPXCancellationReason.error { |
|||
params["reason"] = String(describing: cancellationDetails.reason.rawValue) |
|||
params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" |
|||
} |
|||
} catch { |
|||
params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)" |
|||
} |
|||
os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误")) |
|||
|
|||
self.notifyEvent(eventType: .synthesisCanceled, params: params) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 触发事件通知 |
|||
*/ |
|||
private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { |
|||
let event = TtsEvent(type: eventType, params: params) |
|||
|
|||
// 直接通知事件,由调用方处理线程切换 |
|||
for case let listener as TtsEventListener in self.eventListeners.allObjects { |
|||
listener.onEvent(event) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 重新创建合成器 |
|||
*/ |
|||
private func recreateSynthesizer() { |
|||
do { |
|||
// 使用默认音频输出配置创建合成器 |
|||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|||
|
|||
// 设置事件监听 |
|||
setupEventListeners() |
|||
} catch { |
|||
os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "RECREATE_FAILED", |
|||
"errorMessage": "重新创建合成器失败: \(error.localizedDescription)" |
|||
]) |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 设置语音参数 |
|||
*/ |
|||
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|||
if !isInitialized { return false } |
|||
|
|||
do { |
|||
currentRate = formatPercentage(rate) |
|||
currentPitch = formatPercentage(pitch) |
|||
currentVolume = "\(min(max(volume, 0), 100))%" |
|||
return true |
|||
} catch { |
|||
os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|||
notifyEvent(eventType: .error, params: [ |
|||
"errorCode": "PARAMS_SET_FAILED", |
|||
"errorMessage": "设置语音参数失败: \(error.localizedDescription)" |
|||
]) |
|||
return false |
|||
} |
|||
} |
|||
|
|||
/** |
|||
* 格式化百分比值 |
|||
*/ |
|||
private func formatPercentage(_ value: Int) -> String { |
|||
return value >= 0 ? "+\(value)%" : "\(value)%" |
|||
} |
|||
|
|||
/** |
|||
* 生成SSML |
|||
*/ |
|||
private func generateSsml(_ rawText: String) -> String { |
|||
// 1. 定义要静音的符号和表情符号列表 |
|||
let symbolsToMute = [ |
|||
"#", "*", |
|||
"😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" |
|||
] |
|||
|
|||
// 2. 转义 XML 保留字符 |
|||
var escapedText = rawText |
|||
.replacingOccurrences(of: "&", with: "&") |
|||
.replacingOccurrences(of: "<", with: "<") |
|||
.replacingOccurrences(of: ">", with: ">") |
|||
|
|||
// 3. 静音处理特殊符号和表情符号 |
|||
// 使用空白替换法,直接将符号替换为空字符串 |
|||
var processedText = escapedText |
|||
for symbol in symbolsToMute { |
|||
processedText = processedText.replacingOccurrences(of: symbol, with: "") |
|||
} |
|||
|
|||
// 4. 构造简化的SSML文档,减少嵌套层级 |
|||
let ssml = """ |
|||
<speak version="1.0" |
|||
xmlns="http://www.w3.org/2001/10/synthesis" |
|||
xmlns:mstts="https://www.w3.org/2001/mstts" |
|||
xml:lang="zh-CN"> |
|||
<voice name="\(currentVoice)"> |
|||
<mstts:express-as style="cheerful"> |
|||
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)"> |
|||
<say-as interpret-as="text">\(processedText)</say-as> |
|||
</prosody> |
|||
</mstts:express-as> |
|||
</voice> |
|||
</speak> |
|||
""" |
|||
|
|||
return ssml |
|||
} |
|||
} |
|||
Loading…
Reference in new issue