12 changed files with 516 additions and 1825 deletions
@ -1,618 +0,0 @@ |
|||||
import Foundation |
|
||||
import AVFoundation |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
// 自定义语音处理组件,提供音频流处理等功能 |
|
||||
import speech |
|
||||
import os.log |
|
||||
|
|
||||
|
|
||||
/** |
|
||||
* Azure ASR Helper |
|
||||
* |
|
||||
* 基于微软Azure语音服务的ASR实现 |
|
||||
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-recognize-speech |
|
||||
*/ |
|
||||
public class AzureAsrHelper: NSObject { |
|
||||
private let tag = "AzureAsrHelper" |
|
||||
// 日志对象 |
|
||||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureAsrHelper") |
|
||||
|
|
||||
// 核心组件 |
|
||||
private var speechConfig: SPXSpeechConfiguration? |
|
||||
private var recognizer: SPXSpeechRecognizer? |
|
||||
private var audioConfig: SPXAudioConfiguration? |
|
||||
|
|
||||
// 状态管理 |
|
||||
private var _isContinuousRecognitionActive = false |
|
||||
|
|
||||
// 配置参数 |
|
||||
private var currentLanguage = "zh-CN" |
|
||||
private var supportedLanguages = ["zh-CN"] |
|
||||
private var isAutoDetectLanguage = false |
|
||||
private var subscriptionKey = "" |
|
||||
private var region = "" |
|
||||
|
|
||||
// 音频源配置 |
|
||||
public enum AudioSourceType { |
|
||||
/** 使用设备麦克风 */ |
|
||||
case microphone |
|
||||
|
|
||||
/** 使用外部提供的音频数据 */ |
|
||||
case external |
|
||||
} |
|
||||
|
|
||||
private var audioSourceType = AudioSourceType.microphone |
|
||||
|
|
||||
// 音频处理 |
|
||||
private var externalAudioStream: ExternalAudioPullStream? |
|
||||
|
|
||||
/** |
|
||||
* 初始化Azure语音服务 |
|
||||
* |
|
||||
* @param subscriptionKey Azure 订阅密钥 |
|
||||
* @param region Azure 区域 |
|
||||
* @param supportedLanguages 支持的语言数组 |
|
||||
* @param audioSourceType 音频源类型 |
|
||||
* @return 初始化是否成功 |
|
||||
*/ |
|
||||
public func initialize( |
|
||||
subscriptionKey: String, |
|
||||
region: String, |
|
||||
supportedLanguages: [String] = ["zh-CN"], |
|
||||
audioSourceType: AudioSourceType = .microphone |
|
||||
) -> Bool { |
|
||||
do { |
|
||||
// 检查配置是否为空 |
|
||||
if subscriptionKey.isEmpty || region.isEmpty { |
|
||||
os_log("Azure 配置信息不完整", log: log, type: .error) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 释放之前的资源 |
|
||||
dispose() |
|
||||
|
|
||||
// 保存配置 |
|
||||
self.subscriptionKey = subscriptionKey |
|
||||
self.region = region |
|
||||
self.audioSourceType = audioSourceType |
|
||||
|
|
||||
// 设置语言 |
|
||||
if !supportedLanguages.isEmpty { |
|
||||
self.supportedLanguages = supportedLanguages |
|
||||
} |
|
||||
|
|
||||
// 根据支持的语言数量决定是否启用自动语言检测 |
|
||||
self.isAutoDetectLanguage = supportedLanguages.count >= 2 |
|
||||
|
|
||||
// 如果只有一种语言,设置为当前语言 |
|
||||
if !isAutoDetectLanguage && !supportedLanguages.isEmpty { |
|
||||
self.currentLanguage = supportedLanguages[0] |
|
||||
} |
|
||||
|
|
||||
// 创建语音配置 |
|
||||
speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region) |
|
||||
|
|
||||
if isAutoDetectLanguage { |
|
||||
// 直接启用语言检测模式 |
|
||||
speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") |
|
||||
os_log("启用语言检测模式", log: log, type: .info) |
|
||||
} else { |
|
||||
// 设置指定的识别语言 |
|
||||
speechConfig?.speechRecognitionLanguage = currentLanguage |
|
||||
} |
|
||||
|
|
||||
// 创建识别器 |
|
||||
return setupRecognizer() |
|
||||
} catch { |
|
||||
os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 向音频流写入音频数据 |
|
||||
* 仅当音频源设置为external时有效 |
|
||||
* |
|
||||
* @param data 音频数据字节数组 |
|
||||
*/ |
|
||||
public func pushAudioData(data: Data) { |
|
||||
if audioSourceType != .external { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 使用拉流模式,将数据推入队列 |
|
||||
externalAudioStream?.pushAudio(data) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 开始连续语音识别 |
|
||||
* |
|
||||
* @param callback 连续识别结果回调 |
|
||||
* @param audioSourceType 音频源类型 |
|
||||
* @return 是否成功开始识别 |
|
||||
*/ |
|
||||
public func startContinuousRecognition( |
|
||||
callback: ContinuousRecognizeCallback, |
|
||||
audioSourceType: AudioSourceType = .microphone |
|
||||
) -> Bool { |
|
||||
guard speechConfig != nil else { |
|
||||
callback.onError("语音服务未初始化") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if _isContinuousRecognitionActive { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
self.audioSourceType = audioSourceType |
|
||||
|
|
||||
// 重置识别器 |
|
||||
if !setupRecognizer() { |
|
||||
callback.onError("重置识别器失败") |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 设置各种事件监听 |
|
||||
setupEventListeners(callback: callback) |
|
||||
|
|
||||
// 启动音频处理 |
|
||||
startAudioProcessing() |
|
||||
|
|
||||
// 开始连续识别 |
|
||||
try recognizer?.startContinuousRecognition() |
|
||||
_isContinuousRecognitionActive = true |
|
||||
|
|
||||
os_log("连续识别已启动,音频源: %{public}@", log: log, type: .info, audioSourceType == .microphone ? "麦克风" : "外部") |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
stopAudioProcessing() |
|
||||
_isContinuousRecognitionActive = false |
|
||||
callback.onError("启动连续识别失败: \(error.localizedDescription)") |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 停止连续语音识别 |
|
||||
* |
|
||||
* @return 是否成功停止 |
|
||||
*/ |
|
||||
public func stopContinuousRecognition() -> Bool { |
|
||||
guard speechConfig != nil else { |
|
||||
os_log("语音服务未初始化", log: log, type: .error) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
if !_isContinuousRecognitionActive { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
guard let recognizer = recognizer else { |
|
||||
os_log("识别器为空,重置状态", log: log, type: .info) |
|
||||
_isContinuousRecognitionActive = false |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
if audioSourceType == .external { |
|
||||
pushAudioData(data: Data()) |
|
||||
} |
|
||||
|
|
||||
// 停止连续识别 |
|
||||
try recognizer.stopContinuousRecognition() |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopAudioProcessing() |
|
||||
|
|
||||
// 会话结束事件会设置_isContinuousRecognitionActive = false |
|
||||
return true |
|
||||
} catch { |
|
||||
// 强制重置状态 |
|
||||
_isContinuousRecognitionActive = false |
|
||||
os_log("停止连续识别失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopAudioProcessing() |
|
||||
|
|
||||
// 尝试强制关闭识别器 |
|
||||
recognizer = nil |
|
||||
|
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 检查连续识别是否活跃 |
|
||||
*/ |
|
||||
public func isContinuousRecognitionActive() -> Bool { |
|
||||
return self._isContinuousRecognitionActive |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 释放所有资源 |
|
||||
*/ |
|
||||
public func dispose() { |
|
||||
do { |
|
||||
// 如果正在进行连续识别,先停止 |
|
||||
if _isContinuousRecognitionActive { |
|
||||
// 直接停止,不等待结果 |
|
||||
try? recognizer?.stopContinuousRecognition() |
|
||||
_isContinuousRecognitionActive = false |
|
||||
} |
|
||||
|
|
||||
// 停止音频处理 |
|
||||
stopAudioProcessing() |
|
||||
|
|
||||
// 释放资源 |
|
||||
recognizer = nil |
|
||||
speechConfig = nil |
|
||||
audioConfig = nil |
|
||||
|
|
||||
// 确保状态被重置 |
|
||||
_isContinuousRecognitionActive = false |
|
||||
externalAudioStream = nil |
|
||||
} catch { |
|
||||
// 确保状态被重置 |
|
||||
_isContinuousRecognitionActive = false |
|
||||
externalAudioStream = nil |
|
||||
audioConfig = nil |
|
||||
recognizer = nil |
|
||||
speechConfig = nil |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 私有方法 |
|
||||
|
|
||||
/** |
|
||||
* 设置识别器 |
|
||||
*/ |
|
||||
private func setupRecognizer() -> Bool { |
|
||||
do { |
|
||||
// 清理旧的识别器 |
|
||||
recognizer = nil |
|
||||
|
|
||||
// 设置音频配置 |
|
||||
switch audioSourceType { |
|
||||
case .microphone: |
|
||||
// 使用默认麦克风输入配置 |
|
||||
audioConfig = try SPXAudioConfiguration() |
|
||||
case .external: |
|
||||
// 使用拉流方式处理外部音频 |
|
||||
setupExternalAudioStream() |
|
||||
} |
|
||||
|
|
||||
// 创建识别器 |
|
||||
if isAutoDetectLanguage { |
|
||||
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) |
|
||||
recognizer = try SPXSpeechRecognizer( |
|
||||
speechConfiguration: speechConfig!, |
|
||||
autoDetectSourceLanguageConfiguration: autoDetectConfig, |
|
||||
audioConfiguration: audioConfig! |
|
||||
) |
|
||||
} else { |
|
||||
recognizer = try SPXSpeechRecognizer( |
|
||||
speechConfiguration: speechConfig!, |
|
||||
audioConfiguration: audioConfig! |
|
||||
) |
|
||||
} |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
stopAudioProcessing() |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 设置外部音频流 - 使用拉流方式 |
|
||||
*/ |
|
||||
private func setupExternalAudioStream() { |
|
||||
do { |
|
||||
// 创建外部音频拉流对象 |
|
||||
externalAudioStream = ExternalAudioPullStream() |
|
||||
|
|
||||
// 创建音频配置 |
|
||||
audioConfig = SPXAudioConfiguration(streamInput: externalAudioStream!.pullStream) |
|
||||
} catch { |
|
||||
os_log("设置外部音频流失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
externalAudioStream = nil |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 设置事件监听器 |
|
||||
*/ |
|
||||
private func setupEventListeners(callback: ContinuousRecognizeCallback) { |
|
||||
guard let recognizer = recognizer else { return } |
|
||||
|
|
||||
// 识别中事件 |
|
||||
recognizer.addRecognizingEventHandler { [weak self] (sender, event) in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
let result = event.result |
|
||||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
||||
// os_log("识别中: %{public}@", log: self.log, type: .info, result.text ?? "") |
|
||||
// 直接在当前线程调用回调 |
|
||||
callback.onRecognizing(result.text ?? "", detectedLanguage) |
|
||||
} |
|
||||
|
|
||||
// 识别完成事件 |
|
||||
recognizer.addRecognizedEventHandler { [weak self] (sender, event) in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
let result = event.result |
|
||||
if result.reason == SPXResultReason.recognizedSpeech { |
|
||||
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
||||
// 直接在当前线程调用回调 |
|
||||
callback.onResult(result.text ?? "", detectedLanguage) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 会话开始事件 |
|
||||
recognizer.addSessionStartedEventHandler { (sender, event) in |
|
||||
// 直接在当前线程调用回调 |
|
||||
callback.onSessionStarted() |
|
||||
} |
|
||||
|
|
||||
// 会话结束事件 |
|
||||
recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
// 直接在当前线程调用回调 |
|
||||
callback.onSessionStopped() |
|
||||
self._isContinuousRecognitionActive = false |
|
||||
self.stopAudioProcessing() |
|
||||
} |
|
||||
|
|
||||
// 取消事件 |
|
||||
recognizer.addCanceledEventHandler { [weak self] (sender, event) in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
let errorDetails = event.errorDetails ?? "未知错误" |
|
||||
let reason = String(describing: event.reason.rawValue) |
|
||||
|
|
||||
os_log("识别取消: %{public}@", log: self.log, type: .error, errorDetails) |
|
||||
|
|
||||
callback.onCanceled(reason, errorDetails) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 获取检测到的语言 |
|
||||
*/ |
|
||||
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
||||
if !isAutoDetectLanguage { |
|
||||
return "" |
|
||||
} |
|
||||
|
|
||||
// 尝试从属性中获取语言 |
|
||||
if let properties = result.properties, |
|
||||
let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage") { |
|
||||
return language |
|
||||
} |
|
||||
|
|
||||
// 尝试另一种方式获取语言 |
|
||||
do { |
|
||||
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
||||
return langResult.language ?? "" |
|
||||
} catch { |
|
||||
return "" |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 启动音频处理 |
|
||||
*/ |
|
||||
private func startAudioProcessing() { |
|
||||
switch audioSourceType { |
|
||||
case .microphone: |
|
||||
// 拉流模式不需要额外启动,SDK会自动拉取数据 |
|
||||
break |
|
||||
case .external: |
|
||||
// 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData |
|
||||
break |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 停止音频处理 |
|
||||
*/ |
|
||||
private func stopAudioProcessing() { |
|
||||
if let stream = externalAudioStream { |
|
||||
stream.close() |
|
||||
externalAudioStream = nil |
|
||||
// os_log("外部音频流已关闭", log: log, type: .info) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
|
|
||||
/** |
|
||||
* 外部音频拉流 |
|
||||
* 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式 |
|
||||
*/ |
|
||||
private class ExternalAudioPullStream: NSObject { |
|
||||
private(set) var pullStream: SPXPullAudioInputStream! |
|
||||
private let queue = LinkedBlockingQueue<Data>() |
|
||||
private var closed = false |
|
||||
|
|
||||
override init() { |
|
||||
super.init() |
|
||||
|
|
||||
pullStream = SPXPullAudioInputStream( |
|
||||
readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in |
|
||||
guard let self = self else { return 0 } |
|
||||
return self.read(buffer: data, size: Int(size)) |
|
||||
}, |
|
||||
closeHandler: { [weak self] in |
|
||||
self?.close() |
|
||||
} |
|
||||
) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 外部调用:推送音频数据到队列 |
|
||||
* @param data 音频数据 |
|
||||
*/ |
|
||||
func pushAudio(_ data: Data) { |
|
||||
if !closed { |
|
||||
queue.put(data) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* SDK调用:从队列中拉取数据 |
|
||||
* @param buffer SDK提供的缓冲区 |
|
||||
* @param size 缓冲区大小 |
|
||||
* @return 读取的字节数,0表示流结束 |
|
||||
*/ |
|
||||
private func read(buffer: NSMutableData, size: Int) -> Int { |
|
||||
// 阻塞等待下一块数据 |
|
||||
guard let chunk = queue.take() else { |
|
||||
return 0 // 队列已关闭 |
|
||||
} |
|
||||
|
|
||||
// 如果是空数据,表示流结束 |
|
||||
if chunk.isEmpty { |
|
||||
return 0 |
|
||||
} |
|
||||
|
|
||||
let toCopy = min(chunk.count, size) |
|
||||
buffer.append(chunk.prefix(toCopy)) |
|
||||
return toCopy |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* SDK调用:关闭流 |
|
||||
*/ |
|
||||
func close() { |
|
||||
closed = true |
|
||||
queue.close() |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* iOS版LinkedBlockingQueue实现 |
|
||||
* 与Android LinkedBlockingQueue保持一致的API和行为 |
|
||||
*/ |
|
||||
private class LinkedBlockingQueue<T> { |
|
||||
private var queue: [T] = [] |
|
||||
private var isClosed = false |
|
||||
|
|
||||
// 使用DispatchSemaphore实现阻塞行为 |
|
||||
private let availableItems: DispatchSemaphore |
|
||||
private let queueLock = NSLock() |
|
||||
|
|
||||
init() { |
|
||||
self.availableItems = DispatchSemaphore(value: 0) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 阻塞式获取元素(等价于Android的take()) |
|
||||
* @return 队列中的元素,如果队列已关闭则返回nil |
|
||||
*/ |
|
||||
func take() -> T? { |
|
||||
// 等待可用元素(阻塞直到有元素或队列关闭) |
|
||||
availableItems.wait() |
|
||||
|
|
||||
queueLock.lock() |
|
||||
defer { queueLock.unlock() } |
|
||||
|
|
||||
// 检查队列是否已关闭 |
|
||||
if isClosed && queue.isEmpty { |
|
||||
return nil |
|
||||
} |
|
||||
|
|
||||
// 获取第一个元素 |
|
||||
guard !queue.isEmpty else { |
|
||||
return nil |
|
||||
} |
|
||||
|
|
||||
return queue.removeFirst() |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 添加元素(等价于Android的put()) |
|
||||
* @param item 要添加的元素 |
|
||||
*/ |
|
||||
func put(_ item: T) { |
|
||||
queueLock.lock() |
|
||||
defer { queueLock.unlock() } |
|
||||
|
|
||||
// 检查队列是否已关闭 |
|
||||
if isClosed { |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 添加元素(无容量限制,与Android一致) |
|
||||
queue.append(item) |
|
||||
|
|
||||
// 通知有新元素可用 |
|
||||
availableItems.signal() |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 关闭队列 |
|
||||
* 关闭后不能再添加新元素,但可以继续取出已有元素 |
|
||||
*/ |
|
||||
func close() { |
|
||||
queueLock.lock() |
|
||||
defer { queueLock.unlock() } |
|
||||
|
|
||||
isClosed = true |
|
||||
|
|
||||
// 唤醒所有等待的take()操作 |
|
||||
for _ in 0..<100 { // 假设最多100个等待者 |
|
||||
availableItems.signal() |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 连续识别回调接口 |
|
||||
*/ |
|
||||
public protocol ContinuousRecognizeCallback { |
|
||||
/** |
|
||||
* 返回识别结果 |
|
||||
* |
|
||||
* @param text 识别的文本 |
|
||||
* @param detectedLanguage 检测到的语言 |
|
||||
*/ |
|
||||
func onResult(_ text: String, _ detectedLanguage: String) |
|
||||
|
|
||||
/** |
|
||||
* 识别进行中调用 |
|
||||
* |
|
||||
* @param recognizing 正在识别的文本 |
|
||||
* @param detectedLanguage 检测到的语言 |
|
||||
*/ |
|
||||
func onRecognizing(_ recognizing: String, _ detectedLanguage: String) |
|
||||
|
|
||||
/** |
|
||||
* 会话开始时调用 |
|
||||
*/ |
|
||||
func onSessionStarted() |
|
||||
|
|
||||
/** |
|
||||
* 会话结束时调用 |
|
||||
*/ |
|
||||
func onSessionStopped() |
|
||||
|
|
||||
/** |
|
||||
* 识别取消时调用 |
|
||||
* |
|
||||
* @param reason 取消原因 |
|
||||
* @param errorDetails 错误详情 |
|
||||
*/ |
|
||||
func onCanceled(_ reason: String, _ errorDetails: String) |
|
||||
|
|
||||
/** |
|
||||
* 识别出错时调用 |
|
||||
* |
|
||||
* @param error 错误信息 |
|
||||
*/ |
|
||||
func onError(_ error: String) |
|
||||
} |
|
||||
} |
|
||||
@ -1,17 +0,0 @@ |
|||||
//
|
|
||||
// AzureSpeechPlugin.h
|
|
||||
// azure_speech
|
|
||||
//
|
|
||||
// Created for azure_speech plugin compatibility.
|
|
||||
//
|
|
||||
|
|
||||
#import <Flutter/Flutter.h> |
|
||||
|
|
||||
/**
|
|
||||
* Azure Speech Plugin 头文件 |
|
||||
* |
|
||||
* 注:实际实现使用 Swift,此头文件仅用于 Flutter 框架兼容 |
|
||||
*/ |
|
||||
@interface AzureSpeechPlugin : NSObject <FlutterPlugin> |
|
||||
+ (void)registerWithRegistrar:(NSObject<FlutterPluginRegistrar>*)registrar; |
|
||||
@end |
|
||||
@ -1,422 +0,0 @@ |
|||||
import Flutter |
|
||||
import UIKit |
|
||||
import AVFoundation |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
// 自定义语音处理组件,提供音频流处理等功能 |
|
||||
import speech |
|
||||
import os.log |
|
||||
|
|
||||
/** |
|
||||
* Azure Speech Plugin |
|
||||
* |
|
||||
* 基于微软Azure语音服务的Flutter插件 |
|
||||
* 提供语音识别(ASR)和语音合成(TTS)功能 |
|
||||
*/ |
|
||||
@objc public class AzureSpeechPlugin: NSObject, FlutterPlugin { |
|
||||
// 日志标签 |
|
||||
private let tag = "AzureSpeechPlugin" |
|
||||
// 日志对象 |
|
||||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureSpeechPlugin") |
|
||||
|
|
||||
// ASR相关 |
|
||||
private var asrChannel: FlutterMethodChannel? |
|
||||
private var asrEventChannel: FlutterEventChannel? |
|
||||
private var asrEventSink: FlutterEventSink? |
|
||||
private let azureAsrHelper = AzureAsrHelper() |
|
||||
|
|
||||
// TTS相关 |
|
||||
private var ttsChannel: FlutterMethodChannel? |
|
||||
private var ttsEventChannel: FlutterEventChannel? |
|
||||
private var ttsEventSink: FlutterEventSink? |
|
||||
private let azureTtsHelper = AzureTtsHelper() |
|
||||
|
|
||||
// 是否已添加TTS事件监听器 |
|
||||
private var isTtsListenerAdded = false |
|
||||
|
|
||||
// 当前的连续识别回调 |
|
||||
private var currentAsrCallback: AsrCallbackWrapper? |
|
||||
|
|
||||
// 插件注册 |
|
||||
public static func register(with registrar: FlutterPluginRegistrar) { |
|
||||
let instance = AzureSpeechPlugin() |
|
||||
|
|
||||
// 初始化ASR通道 |
|
||||
let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) |
|
||||
registrar.addMethodCallDelegate(instance, channel: asrChannel) |
|
||||
instance.asrChannel = asrChannel |
|
||||
|
|
||||
// 初始化TTS通道 |
|
||||
let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) |
|
||||
registrar.addMethodCallDelegate(instance, channel: ttsChannel) |
|
||||
instance.ttsChannel = ttsChannel |
|
||||
|
|
||||
// 初始化ASR事件通道 |
|
||||
let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) |
|
||||
asrEventChannel.setStreamHandler(instance) |
|
||||
instance.asrEventChannel = asrEventChannel |
|
||||
|
|
||||
// 初始化TTS事件通道 |
|
||||
let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger()) |
|
||||
ttsEventChannel.setStreamHandler(instance) |
|
||||
instance.ttsEventChannel = ttsEventChannel |
|
||||
} |
|
||||
|
|
||||
// 发送ASR事件方法 |
|
||||
internal func sendAsrEvent(_ event: [String: Any]) { |
|
||||
if asrEventSink == nil { |
|
||||
os_log("无法发送ASR事件:事件通道未准备好", log: log, type: .error) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
DispatchQueue.main.async { [weak self] in |
|
||||
guard let self = self else { return } |
|
||||
self.asrEventSink?(event) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 发送TTS事件方法 |
|
||||
private func sendTtsEvent(_ event: [String: Any]) { |
|
||||
if ttsEventSink == nil { |
|
||||
os_log("无法发送TTS事件:事件通道未准备好", log: log, type: .error) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
DispatchQueue.main.async { [weak self] in |
|
||||
guard let self = self else { return } |
|
||||
self.ttsEventSink?(event) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 设置TTS事件监听器 |
|
||||
private func setupTtsEventListener() { |
|
||||
if !isTtsListenerAdded { |
|
||||
azureTtsHelper.addListener(self) |
|
||||
isTtsListenerAdded = true |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
|
|
||||
|
|
||||
// 处理Flutter方法调用 |
|
||||
public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { |
|
||||
if call.method.hasPrefix("tts_") { |
|
||||
handleTtsMethodCall(call, result) |
|
||||
} else { |
|
||||
handleAsrMethodCall(call, result) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - ASR 方法处理 |
|
||||
|
|
||||
private func handleAsrMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|
||||
switch call.method { |
|
||||
case "initialize": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let subscriptionKey = args["subscriptionKey"] as? String, |
|
||||
let region = args["region"] as? String else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let supportedLanguages = args["supportedLanguages"] as? [String] ?? ["zh-CN"] |
|
||||
|
|
||||
// 初始化ASR引擎 |
|
||||
let success = azureAsrHelper.initialize( |
|
||||
subscriptionKey: subscriptionKey, |
|
||||
region: region, |
|
||||
supportedLanguages: supportedLanguages, |
|
||||
audioSourceType: .microphone |
|
||||
) |
|
||||
|
|
||||
result(success) |
|
||||
|
|
||||
case "recognizeOnce": |
|
||||
// iOS版本不支持recognizeOnce |
|
||||
result(FlutterError(code: "NOT_SUPPORTED", message: "iOS版本不支持recognizeOnce", details: nil)) |
|
||||
|
|
||||
case "startContinuousRecognition": |
|
||||
// 确保事件通道已准备好 |
|
||||
guard asrEventSink != nil else { |
|
||||
result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 解析音频源类型 |
|
||||
let audioSourceType: AzureAsrHelper.AudioSourceType |
|
||||
if let args = call.arguments as? [String: Any], |
|
||||
let audioSourceString = args["audioSourceType"] as? String { |
|
||||
switch audioSourceString.lowercased() { |
|
||||
case "external": |
|
||||
audioSourceType = .external |
|
||||
default: |
|
||||
audioSourceType = .microphone |
|
||||
} |
|
||||
} else { |
|
||||
audioSourceType = .microphone |
|
||||
} |
|
||||
|
|
||||
// 创建回调包装器 |
|
||||
currentAsrCallback = AsrCallbackWrapper(plugin: self) |
|
||||
|
|
||||
let success = azureAsrHelper.startContinuousRecognition( |
|
||||
callback: currentAsrCallback!, |
|
||||
audioSourceType: audioSourceType |
|
||||
) |
|
||||
result(success) |
|
||||
|
|
||||
case "stopContinuousRecognition": |
|
||||
let success = azureAsrHelper.stopContinuousRecognition() |
|
||||
currentAsrCallback = nil |
|
||||
result(success) |
|
||||
|
|
||||
case "isContinuousRecognitionActive": |
|
||||
result(azureAsrHelper.isContinuousRecognitionActive()) |
|
||||
|
|
||||
case "dispose": |
|
||||
azureAsrHelper.dispose() |
|
||||
currentAsrCallback = nil |
|
||||
result(true) |
|
||||
|
|
||||
case "pushAudioData": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let audioBytes = args["data"] as? FlutterStandardTypedData else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
azureAsrHelper.pushAudioData(data: audioBytes.data) |
|
||||
result(true) |
|
||||
|
|
||||
default: |
|
||||
result(FlutterMethodNotImplemented) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - TTS 方法处理 |
|
||||
|
|
||||
private func handleTtsMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { |
|
||||
switch call.method { |
|
||||
case "tts_initialize": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let subscriptionKey = args["subscriptionKey"] as? String, |
|
||||
let region = args["region"] as? String else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
// 初始化TTS引擎 |
|
||||
let language = args["language"] as? String ?? "zh-CN" |
|
||||
let success = azureTtsHelper.initialize(ttsAppId: "", ttsAppToken: subscriptionKey, ttsResource: region, language: language) |
|
||||
|
|
||||
// 设置TTS事件监听器 |
|
||||
setupTtsEventListener() |
|
||||
|
|
||||
result(success) |
|
||||
|
|
||||
case "tts_set_voice": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let voiceName = args["voiceName"] as? String else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "语音名称不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let success = azureTtsHelper.setVoice(voiceName) |
|
||||
result(success) |
|
||||
|
|
||||
case "tts_speak_once": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let text = args["text"] as? String else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let success = azureTtsHelper.speakOnce(text) |
|
||||
result(success) |
|
||||
|
|
||||
case "tts_speak_stream": |
|
||||
guard let args = call.arguments as? [String: Any], |
|
||||
let text = args["text"] as? String else { |
|
||||
result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) |
|
||||
return |
|
||||
} |
|
||||
|
|
||||
let success = azureTtsHelper.speakStream(text) |
|
||||
result(success) |
|
||||
|
|
||||
case "tts_flush_stream": |
|
||||
let success = azureTtsHelper.flushStream() |
|
||||
result(success) |
|
||||
|
|
||||
case "tts_stop": |
|
||||
let success = azureTtsHelper.stop() |
|
||||
result(success) |
|
||||
|
|
||||
case "tts_isSpeaking": |
|
||||
result(azureTtsHelper.isSpeaking()) |
|
||||
|
|
||||
case "tts_release": |
|
||||
azureTtsHelper.dispose() |
|
||||
result(true) |
|
||||
|
|
||||
default: |
|
||||
result(FlutterMethodNotImplemented) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - ASR回调包装器 |
|
||||
private class AsrCallbackWrapper: AzureAsrHelper.ContinuousRecognizeCallback { |
|
||||
private weak var plugin: AzureSpeechPlugin? |
|
||||
|
|
||||
init(plugin: AzureSpeechPlugin) { |
|
||||
self.plugin = plugin |
|
||||
} |
|
||||
|
|
||||
func onResult(_ text: String, _ detectedLanguage: String) { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "result", |
|
||||
"text": text, |
|
||||
"language": detectedLanguage |
|
||||
]) |
|
||||
} |
|
||||
|
|
||||
func onRecognizing(_ recognizing: String, _ detectedLanguage: String) { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "recognizing", |
|
||||
"text": recognizing, |
|
||||
"language": detectedLanguage |
|
||||
]) |
|
||||
} |
|
||||
|
|
||||
func onSessionStarted() { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "sessionStarted" |
|
||||
]) |
|
||||
} |
|
||||
|
|
||||
func onSessionStopped() { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "sessionStopped" |
|
||||
]) |
|
||||
} |
|
||||
|
|
||||
func onCanceled(_ reason: String, _ errorDetails: String) { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "canceled", |
|
||||
"reason": reason, |
|
||||
"errorDetails": errorDetails |
|
||||
]) |
|
||||
} |
|
||||
|
|
||||
func onError(_ error: String) { |
|
||||
plugin?.sendAsrEvent([ |
|
||||
"type": "error", |
|
||||
"error": error |
|
||||
]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 事件处理 |
|
||||
extension AzureSpeechPlugin: FlutterStreamHandler { |
|
||||
public func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { |
|
||||
// 根据通道类型设置事件接收器 |
|
||||
if let args = arguments as? [String: Any], |
|
||||
let channel = args["channel"] as? String { |
|
||||
|
|
||||
if channel == "asr" { |
|
||||
asrEventSink = events |
|
||||
} else if channel == "tts" { |
|
||||
ttsEventSink = events |
|
||||
setupTtsEventListener() |
|
||||
} |
|
||||
} else if arguments == nil { |
|
||||
// 如果没有指定通道,尝试弄清楚是哪个通道在监听 |
|
||||
if asrEventSink == nil && ttsEventSink != nil { |
|
||||
asrEventSink = events |
|
||||
} else if asrEventSink != nil && ttsEventSink == nil { |
|
||||
ttsEventSink = events |
|
||||
setupTtsEventListener() |
|
||||
} else { |
|
||||
// 无法确定哪个通道,默认设置为ASR |
|
||||
asrEventSink = events |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return nil |
|
||||
} |
|
||||
|
|
||||
public func onCancel(withArguments arguments: Any?) -> FlutterError? { |
|
||||
// 根据通道类型清除事件接收器 |
|
||||
if let args = arguments as? [String: Any], |
|
||||
let channel = args["channel"] as? String { |
|
||||
|
|
||||
if channel == "asr" { |
|
||||
asrEventSink = nil |
|
||||
} else if channel == "tts" { |
|
||||
ttsEventSink = nil |
|
||||
} |
|
||||
} else { |
|
||||
// 如果没有指定通道,清除所有通道 |
|
||||
asrEventSink = nil |
|
||||
ttsEventSink = nil |
|
||||
} |
|
||||
|
|
||||
return nil |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
|
|
||||
|
|
||||
// MARK: - TTS 事件监听实现 |
|
||||
extension AzureSpeechPlugin: TtsEventListener { |
|
||||
// 使用@objc特性为方法提供一个不同的Objective-C选择器名称 |
|
||||
@objc(onTtsEvent:) |
|
||||
public func onEvent(_ event: TtsEvent) { |
|
||||
var eventMap: [String: Any] = [:] |
|
||||
|
|
||||
// 根据事件类型转换 |
|
||||
switch event.type { |
|
||||
case .synthesisStarted: |
|
||||
eventMap["type"] = "synthesis_started" |
|
||||
case .synthesisCompleted: |
|
||||
eventMap["type"] = "synthesis_completed" |
|
||||
case .synthesisCanceled: |
|
||||
eventMap["type"] = "synthesis_canceled" |
|
||||
if let reason = event.params["reason"] { |
|
||||
eventMap["reason"] = reason |
|
||||
} |
|
||||
if let errorDetails = event.params["errorDetails"] { |
|
||||
eventMap["errorDetails"] = errorDetails |
|
||||
} |
|
||||
case .error: |
|
||||
eventMap["type"] = "error" |
|
||||
if let errorCode = event.params["errorCode"] { |
|
||||
eventMap["errorCode"] = errorCode |
|
||||
} |
|
||||
if let errorMessage = event.params["errorMessage"] { |
|
||||
eventMap["errorMessage"] = errorMessage |
|
||||
} |
|
||||
@unknown default: |
|
||||
eventMap["type"] = "unknown" |
|
||||
for (key, value) in event.params { |
|
||||
eventMap[key] = value |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
sendTtsEvent(eventMap) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// MARK: - 音频数据监听实现 |
|
||||
extension AzureSpeechPlugin: AudioDataListener { |
|
||||
public func onAudioData(_ data: Data) { |
|
||||
if ttsEventSink == nil { return } |
|
||||
|
|
||||
let eventMap: [String: Any] = [ |
|
||||
"type": "audio_data", |
|
||||
"data": FlutterStandardTypedData(bytes: data) |
|
||||
] |
|
||||
|
|
||||
sendTtsEvent(eventMap) |
|
||||
} |
|
||||
} |
|
||||
@ -1,604 +0,0 @@ |
|||||
import Foundation |
|
||||
import AVFoundation |
|
||||
import MicrosoftCognitiveServicesSpeech |
|
||||
// 自定义语音处理组件,提供音频流处理等功能 |
|
||||
import speech |
|
||||
import os.log |
|
||||
|
|
||||
/** |
|
||||
* Azure TTS Helper |
|
||||
* |
|
||||
* 基于微软Azure语音服务的TTS实现 |
|
||||
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis |
|
||||
* |
|
||||
* 特性: |
|
||||
* - 对外接口保持同步,内部异步处理 |
|
||||
* - 使用专用队列保证合成顺序 |
|
||||
* - 避免阻塞主线程和其他后台线程 |
|
||||
* - 事件回调由调用方处理线程切换 |
|
||||
* |
|
||||
* 使用示例: |
|
||||
* let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理 |
|
||||
*/ |
|
||||
public class AzureTtsHelper: NSObject, ITtsService { |
|
||||
private let tag = "AzureTtsHelper" |
|
||||
// 日志对象 |
|
||||
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") |
|
||||
|
|
||||
private static let DEFAULT_LANGUAGE = "zh-CN" |
|
||||
private static let DEFAULT_VOICE = "zh-CN-XiaoxiaoNeural" |
|
||||
|
|
||||
// Azure语音服务配置 |
|
||||
private var speechConfig: SPXSpeechConfiguration? |
|
||||
private var synthesizer: SPXSpeechSynthesizer? |
|
||||
private var isInitialized = false |
|
||||
private var speaking = false |
|
||||
|
|
||||
// 当前配置 |
|
||||
private var currentVoice = DEFAULT_VOICE |
|
||||
private var currentLanguage = DEFAULT_LANGUAGE |
|
||||
private var currentRate = "+0%" |
|
||||
private var currentPitch = "+0%" |
|
||||
private var currentVolume = "100%" |
|
||||
|
|
||||
// 事件监听器列表 |
|
||||
private var eventListeners = NSHashTable<AnyObject>.weakObjects() |
|
||||
|
|
||||
// 流式文本处理的缓冲区 |
|
||||
private var streamBuffer = "" |
|
||||
private var lastSpeakTime: TimeInterval = 0 |
|
||||
|
|
||||
// 内部异步处理队列,保证顺序执行 |
|
||||
private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) |
|
||||
private let synthesisGroup = DispatchGroup() |
|
||||
private var pendingTasks: [() -> Void] = [] |
|
||||
private let taskLock = NSLock() |
|
||||
|
|
||||
/** |
|
||||
* 初始化语音合成服务 |
|
||||
* |
|
||||
* @param appId 服务应用ID |
|
||||
* @param token 服务访问令牌/订阅密钥 |
|
||||
* @param resource 服务资源ID/区域(可选) |
|
||||
* @param language 语言代码,如"zh-CN" |
|
||||
* @return 是否初始化成功 |
|
||||
*/ |
|
||||
public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { |
|
||||
do { |
|
||||
// 创建语音配置 |
|
||||
if ttsResource.isEmpty { |
|
||||
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia") |
|
||||
} else { |
|
||||
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) |
|
||||
} |
|
||||
|
|
||||
// 设置语言 |
|
||||
speechConfig?.speechSynthesisLanguage = language |
|
||||
currentLanguage = language |
|
||||
|
|
||||
// 设置音频输出格式 - 使用16k、16位的PCM格式 |
|
||||
speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", |
|
||||
byName: "SpeechServiceConnection_SynthOutputFormat") |
|
||||
|
|
||||
// 创建合成器,使用默认音频输出(扬声器) |
|
||||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
||||
|
|
||||
// 设置事件监听 |
|
||||
setupEventListeners() |
|
||||
|
|
||||
// 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny |
|
||||
if language.lowercased().starts(with: "zh") { |
|
||||
_ = setVoice("zh-CN-XiaoxiaoNeural") |
|
||||
} else { |
|
||||
_ = setVoice("en-US-JennyNeural") |
|
||||
} |
|
||||
|
|
||||
// 设置初始化完成 |
|
||||
isInitialized = true |
|
||||
|
|
||||
return true |
|
||||
} catch { |
|
||||
os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 设置语音角色 |
|
||||
* |
|
||||
* @param voiceName 语音角色名称(不同服务的语音角色命名可能不同) |
|
||||
* @return 是否设置成功 |
|
||||
*/ |
|
||||
public func setVoice(_ voiceName: String) -> Bool { |
|
||||
if !isInitialized { return false } |
|
||||
|
|
||||
do { |
|
||||
currentVoice = voiceName |
|
||||
|
|
||||
// 更新语音名称 |
|
||||
if let config = speechConfig { |
|
||||
config.speechSynthesisVoiceName = voiceName |
|
||||
} |
|
||||
|
|
||||
// 重新创建合成器 |
|
||||
recreateSynthesizer() |
|
||||
return true |
|
||||
} catch { |
|
||||
os_log("设置语音失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "VOICE_SET_FAILED", |
|
||||
"errorMessage": "设置语音失败: \(error.localizedDescription)" |
|
||||
]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 单次合成并播放 |
|
||||
* |
|
||||
* @param text 要合成的文本 |
|
||||
* @return 是否成功开始合成 |
|
||||
*/ |
|
||||
public func speakOnce(_ text: String) -> Bool { |
|
||||
if !isInitialized { |
|
||||
os_log("语音合成未初始化", log: log, type: .error) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
// 立即返回成功,内部异步处理 |
|
||||
enqueueSynthesisTask { |
|
||||
self.performSynthesis(text: text) |
|
||||
} |
|
||||
|
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 将合成任务加入队列,保证顺序执行 |
|
||||
*/ |
|
||||
private func enqueueSynthesisTask(_ task: @escaping () -> Void) { |
|
||||
taskLock.lock() |
|
||||
defer { taskLock.unlock() } |
|
||||
|
|
||||
pendingTasks.append(task) |
|
||||
|
|
||||
// 如果当前没有任务在执行,开始处理队列 |
|
||||
if pendingTasks.count == 1 { |
|
||||
processNextTask() |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 处理队列中的下一个任务 |
|
||||
*/ |
|
||||
private func processNextTask() { |
|
||||
synthesisQueue.async { |
|
||||
self.synthesisGroup.enter() |
|
||||
|
|
||||
self.taskLock.lock() |
|
||||
guard !self.pendingTasks.isEmpty else { |
|
||||
self.taskLock.unlock() |
|
||||
self.synthesisGroup.leave() |
|
||||
return |
|
||||
} |
|
||||
let task = self.pendingTasks.removeFirst() |
|
||||
self.taskLock.unlock() |
|
||||
|
|
||||
// 执行任务 |
|
||||
task() |
|
||||
|
|
||||
self.synthesisGroup.leave() |
|
||||
|
|
||||
// 处理下一个任务 |
|
||||
self.taskLock.lock() |
|
||||
if !self.pendingTasks.isEmpty { |
|
||||
self.taskLock.unlock() |
|
||||
self.processNextTask() |
|
||||
} else { |
|
||||
self.taskLock.unlock() |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 实际执行合成的方法 |
|
||||
*/ |
|
||||
private func performSynthesis(text: String) { |
|
||||
// 重置状态 |
|
||||
speaking = true |
|
||||
|
|
||||
// 生成SSML |
|
||||
let ssml = generateSsml(text) |
|
||||
|
|
||||
do { |
|
||||
os_log("开始合成: %{public}@", log: log, type: .debug, ssml) |
|
||||
// 使用同步方法,在后台队列中执行 |
|
||||
let result = try synthesizer?.startSpeakingSsml(ssml) |
|
||||
|
|
||||
// 检查结果 |
|
||||
if let result = result { |
|
||||
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) |
|
||||
} |
|
||||
} catch { |
|
||||
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "SYNTHESIS_FAILED", |
|
||||
"errorMessage": error.localizedDescription |
|
||||
]) |
|
||||
speaking = false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 流式合成文本 |
|
||||
* |
|
||||
* @param text 要合成的文本片段 |
|
||||
* @return 是否成功处理 |
|
||||
*/ |
|
||||
public func speakStream(_ text: String) -> Bool { |
|
||||
if !isInitialized || text.isEmpty { |
|
||||
if !isInitialized { |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "NOT_INITIALIZED", |
|
||||
"errorMessage": "TTS引擎未初始化" |
|
||||
]) |
|
||||
} |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 添加新文本到缓冲区 |
|
||||
streamBuffer.append(text) |
|
||||
|
|
||||
// 增加500ms防抖逻辑 |
|
||||
let currentTime = Date().timeIntervalSince1970 |
|
||||
if currentTime - lastSpeakTime < 0.6 { |
|
||||
return true |
|
||||
} |
|
||||
lastSpeakTime = currentTime |
|
||||
|
|
||||
let currentText = streamBuffer |
|
||||
|
|
||||
// 定义标点符号列表 |
|
||||
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] |
|
||||
|
|
||||
// 查找最后一个标点符号的位置 |
|
||||
var lastPunctuationIndex = -1 |
|
||||
for (i, char) in currentText.enumerated().reversed() { |
|
||||
if punctuationMarks.contains(char) { |
|
||||
lastPunctuationIndex = i |
|
||||
break |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 如果找到标点符号,则播放到该标点符号 |
|
||||
if lastPunctuationIndex >= 0 { |
|
||||
// 提取要播放的文本(包含标点符号) |
|
||||
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines) |
|
||||
|
|
||||
// 剩余的文本保存在缓冲区中 |
|
||||
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) |
|
||||
streamBuffer = String(currentText[startIndex...]) |
|
||||
|
|
||||
// 只有非空文本才播放 |
|
||||
if !textToSpeak.isEmpty { |
|
||||
return speakOnce(textToSpeak) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
// 如果没有找到标点符号,则等待更多文本 |
|
||||
return true |
|
||||
} catch { |
|
||||
os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "STREAM_FAILED", |
|
||||
"errorMessage": "流式语音合成失败: \(error.localizedDescription)" |
|
||||
]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 刷新并播放流式文本缓冲区中的剩余内容 |
|
||||
* |
|
||||
* @return 是否成功处理 |
|
||||
*/ |
|
||||
public func flushStream() -> Bool { |
|
||||
if !isInitialized { |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "NOT_INITIALIZED", |
|
||||
"errorMessage": "TTS引擎未初始化" |
|
||||
]) |
|
||||
return false |
|
||||
} |
|
||||
|
|
||||
do { |
|
||||
// 获取缓冲区中剩余的文本 |
|
||||
let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines) |
|
||||
|
|
||||
// 清空缓冲区 |
|
||||
streamBuffer = "" |
|
||||
|
|
||||
// 如果缓冲区为空,直接返回成功 |
|
||||
if remainingText.isEmpty { |
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
// 播放剩余文本 |
|
||||
return speakOnce(remainingText) |
|
||||
} catch { |
|
||||
os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "FLUSH_FAILED", |
|
||||
"errorMessage": "刷新流式文本失败: \(error.localizedDescription)" |
|
||||
]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 停止语音合成和播放 |
|
||||
* |
|
||||
* @return 是否成功停止 |
|
||||
*/ |
|
||||
public func stop() -> Bool { |
|
||||
// 立即更新状态 |
|
||||
speaking = false |
|
||||
|
|
||||
// 清除流式缓冲区中的待播放内容 |
|
||||
streamBuffer = "" |
|
||||
|
|
||||
// 清空待处理的任务队列 |
|
||||
taskLock.lock() |
|
||||
pendingTasks.removeAll() |
|
||||
taskLock.unlock() |
|
||||
|
|
||||
// 在后台队列停止合成器,避免阻塞主线程 |
|
||||
synthesisQueue.async { |
|
||||
if let synthesizer = self.synthesizer { |
|
||||
do { |
|
||||
try synthesizer.stopSpeaking() |
|
||||
os_log("语音合成已停止", log: self.log, type: .info) |
|
||||
|
|
||||
// 直接通知停止完成 |
|
||||
self.notifyEvent(eventType: .synthesisCanceled) |
|
||||
} catch { |
|
||||
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) |
|
||||
self.notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "STOP_FAILED", |
|
||||
"errorMessage": error.localizedDescription |
|
||||
]) |
|
||||
} |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
return true |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 释放资源 |
|
||||
* 在不再需要服务时调用,释放底层资源 |
|
||||
*/ |
|
||||
public func dispose() { |
|
||||
// 停止播放 |
|
||||
_ = stop() |
|
||||
|
|
||||
// 等待所有任务完成 |
|
||||
synthesisGroup.wait() |
|
||||
|
|
||||
// 清理资源 |
|
||||
synthesisQueue.async { |
|
||||
// 释放合成器 |
|
||||
self.synthesizer = nil |
|
||||
|
|
||||
// 释放配置 |
|
||||
self.speechConfig = nil |
|
||||
|
|
||||
// 清空流缓冲区 |
|
||||
self.streamBuffer = "" |
|
||||
|
|
||||
// 重置状态 |
|
||||
self.isInitialized = false |
|
||||
self.speaking = false |
|
||||
|
|
||||
// 清空任务队列 |
|
||||
self.taskLock.lock() |
|
||||
self.pendingTasks.removeAll() |
|
||||
self.taskLock.unlock() |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 添加TTS事件监听器 |
|
||||
* |
|
||||
* @param listener 事件监听器 |
|
||||
*/ |
|
||||
public func addListener(_ listener: TtsEventListener) { |
|
||||
eventListeners.add(listener as AnyObject) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 移除TTS事件监听器 |
|
||||
* |
|
||||
* @param listener 要移除的事件监听器 |
|
||||
*/ |
|
||||
public func removeListener(_ listener: TtsEventListener) { |
|
||||
eventListeners.remove(listener as AnyObject) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 添加音频数据监听器 |
|
||||
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|
||||
* |
|
||||
* @param listener 音频数据监听器 |
|
||||
*/ |
|
||||
public func addAudioDataListener(_ listener: AudioDataListener) { |
|
||||
os_log("警告:不支持音频数据监听器功能", log: log, type: .info) |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 移除音频数据监听器 |
|
||||
* 由于不再支持自定义音频流,此方法实际上不再有效 |
|
||||
* |
|
||||
* @param listener 要移除的音频数据监听器 |
|
||||
*/ |
|
||||
public func removeAudioDataListener(_ listener: AudioDataListener) { |
|
||||
// 不做任何操作 |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 当前是否正在播放/合成 |
|
||||
*/ |
|
||||
public func isSpeaking() -> Bool { |
|
||||
return speaking |
|
||||
} |
|
||||
|
|
||||
// MARK: - 辅助方法 |
|
||||
|
|
||||
/** |
|
||||
* 设置事件监听器 |
|
||||
*/ |
|
||||
private func setupEventListeners() { |
|
||||
guard let synthesizer = synthesizer else { return } |
|
||||
|
|
||||
// 添加合成开始事件处理器 |
|
||||
synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in |
|
||||
guard let self = self else { return } |
|
||||
self.notifyEvent(eventType: .synthesisStarted) |
|
||||
} |
|
||||
|
|
||||
// 添加合成中事件处理器 |
|
||||
synthesizer.addSynthesizingEventHandler { [weak self] _, _ in |
|
||||
// 可以在这里处理合成中的事件,目前没有特别操作 |
|
||||
} |
|
||||
|
|
||||
// 添加合成完成事件处理器 |
|
||||
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in |
|
||||
guard let self = self else { return } |
|
||||
self.speaking = false |
|
||||
self.notifyEvent(eventType: .synthesisCompleted) |
|
||||
} |
|
||||
|
|
||||
// 添加合成取消事件处理器 |
|
||||
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in |
|
||||
guard let self = self else { return } |
|
||||
|
|
||||
self.speaking = false |
|
||||
|
|
||||
var params: [String: Any] = [:] |
|
||||
do { |
|
||||
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) |
|
||||
if cancellationDetails.reason == SPXCancellationReason.error { |
|
||||
params["reason"] = String(describing: cancellationDetails.reason.rawValue) |
|
||||
params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" |
|
||||
} |
|
||||
} catch { |
|
||||
params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)" |
|
||||
} |
|
||||
os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误")) |
|
||||
|
|
||||
self.notifyEvent(eventType: .synthesisCanceled, params: params) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 触发事件通知 |
|
||||
*/ |
|
||||
private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { |
|
||||
let event = TtsEvent(type: eventType, params: params) |
|
||||
|
|
||||
// 直接通知事件,由调用方处理线程切换 |
|
||||
for case let listener as TtsEventListener in self.eventListeners.allObjects { |
|
||||
listener.onEvent(event) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 重新创建合成器 |
|
||||
*/ |
|
||||
private func recreateSynthesizer() { |
|
||||
do { |
|
||||
// 使用默认音频输出配置创建合成器 |
|
||||
synthesizer = try SPXSpeechSynthesizer(speechConfig!) |
|
||||
|
|
||||
// 设置事件监听 |
|
||||
setupEventListeners() |
|
||||
} catch { |
|
||||
os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "RECREATE_FAILED", |
|
||||
"errorMessage": "重新创建合成器失败: \(error.localizedDescription)" |
|
||||
]) |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 设置语音参数 |
|
||||
*/ |
|
||||
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { |
|
||||
if !isInitialized { return false } |
|
||||
|
|
||||
do { |
|
||||
currentRate = formatPercentage(rate) |
|
||||
currentPitch = formatPercentage(pitch) |
|
||||
currentVolume = "\(min(max(volume, 0), 100))%" |
|
||||
return true |
|
||||
} catch { |
|
||||
os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
||||
notifyEvent(eventType: .error, params: [ |
|
||||
"errorCode": "PARAMS_SET_FAILED", |
|
||||
"errorMessage": "设置语音参数失败: \(error.localizedDescription)" |
|
||||
]) |
|
||||
return false |
|
||||
} |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 格式化百分比值 |
|
||||
*/ |
|
||||
private func formatPercentage(_ value: Int) -> String { |
|
||||
return value >= 0 ? "+\(value)%" : "\(value)%" |
|
||||
} |
|
||||
|
|
||||
/** |
|
||||
* 生成SSML |
|
||||
*/ |
|
||||
private func generateSsml(_ rawText: String) -> String { |
|
||||
// 1. 定义要静音的符号和表情符号列表 |
|
||||
let symbolsToMute = [ |
|
||||
"#", "*", |
|
||||
"😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" |
|
||||
] |
|
||||
|
|
||||
// 2. 转义 XML 保留字符 |
|
||||
var escapedText = rawText |
|
||||
.replacingOccurrences(of: "&", with: "&") |
|
||||
.replacingOccurrences(of: "<", with: "<") |
|
||||
.replacingOccurrences(of: ">", with: ">") |
|
||||
|
|
||||
// 3. 静音处理特殊符号和表情符号 |
|
||||
// 使用空白替换法,直接将符号替换为空字符串 |
|
||||
var processedText = escapedText |
|
||||
for symbol in symbolsToMute { |
|
||||
processedText = processedText.replacingOccurrences(of: symbol, with: "") |
|
||||
} |
|
||||
|
|
||||
// 4. 构造简化的SSML文档,减少嵌套层级 |
|
||||
let ssml = """ |
|
||||
<speak version="1.0" |
|
||||
xmlns="http://www.w3.org/2001/10/synthesis" |
|
||||
xmlns:mstts="https://www.w3.org/2001/mstts" |
|
||||
xml:lang="zh-CN"> |
|
||||
<voice name="\(currentVoice)"> |
|
||||
<mstts:express-as style="cheerful"> |
|
||||
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)"> |
|
||||
<say-as interpret-as="text">\(processedText)</say-as> |
|
||||
</prosody> |
|
||||
</mstts:express-as> |
|
||||
</voice> |
|
||||
</speak> |
|
||||
""" |
|
||||
|
|
||||
return ssml |
|
||||
} |
|
||||
} |
|
||||
Loading…
Reference in new issue