You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
618 lines
19 KiB
618 lines
19 KiB
import Foundation
|
|
import AVFoundation
|
|
import MicrosoftCognitiveServicesSpeech
|
|
// 自定义语音处理组件,提供音频流处理等功能
|
|
import speech
|
|
import os.log
|
|
|
|
|
|
/**
|
|
* Azure ASR Helper
|
|
*
|
|
* 基于微软Azure语音服务的ASR实现
|
|
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-recognize-speech
|
|
*/
|
|
public class AzureAsrHelper: NSObject {
|
|
private let tag = "AzureAsrHelper"
|
|
// 日志对象
|
|
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureAsrHelper")
|
|
|
|
// 核心组件
|
|
private var speechConfig: SPXSpeechConfiguration?
|
|
private var recognizer: SPXSpeechRecognizer?
|
|
private var audioConfig: SPXAudioConfiguration?
|
|
|
|
// 状态管理
|
|
private var _isContinuousRecognitionActive = false
|
|
|
|
// 配置参数
|
|
private var currentLanguage = "zh-CN"
|
|
private var supportedLanguages = ["zh-CN"]
|
|
private var isAutoDetectLanguage = false
|
|
private var subscriptionKey = ""
|
|
private var region = ""
|
|
|
|
// 音频源配置
|
|
public enum AudioSourceType {
|
|
/** 使用设备麦克风 */
|
|
case microphone
|
|
|
|
/** 使用外部提供的音频数据 */
|
|
case external
|
|
}
|
|
|
|
private var audioSourceType = AudioSourceType.microphone
|
|
|
|
// 音频处理
|
|
private var externalAudioStream: ExternalAudioPullStream?
|
|
|
|
/**
|
|
* 初始化Azure语音服务
|
|
*
|
|
* @param subscriptionKey Azure 订阅密钥
|
|
* @param region Azure 区域
|
|
* @param supportedLanguages 支持的语言数组
|
|
* @param audioSourceType 音频源类型
|
|
* @return 初始化是否成功
|
|
*/
|
|
public func initialize(
|
|
subscriptionKey: String,
|
|
region: String,
|
|
supportedLanguages: [String] = ["zh-CN"],
|
|
audioSourceType: AudioSourceType = .microphone
|
|
) -> Bool {
|
|
do {
|
|
// 检查配置是否为空
|
|
if subscriptionKey.isEmpty || region.isEmpty {
|
|
os_log("Azure 配置信息不完整", log: log, type: .error)
|
|
return false
|
|
}
|
|
|
|
// 释放之前的资源
|
|
dispose()
|
|
|
|
// 保存配置
|
|
self.subscriptionKey = subscriptionKey
|
|
self.region = region
|
|
self.audioSourceType = audioSourceType
|
|
|
|
// 设置语言
|
|
if !supportedLanguages.isEmpty {
|
|
self.supportedLanguages = supportedLanguages
|
|
}
|
|
|
|
// 根据支持的语言数量决定是否启用自动语言检测
|
|
self.isAutoDetectLanguage = supportedLanguages.count >= 2
|
|
|
|
// 如果只有一种语言,设置为当前语言
|
|
if !isAutoDetectLanguage && !supportedLanguages.isEmpty {
|
|
self.currentLanguage = supportedLanguages[0]
|
|
}
|
|
|
|
// 创建语音配置
|
|
speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region)
|
|
|
|
if isAutoDetectLanguage {
|
|
// 直接启用语言检测模式
|
|
speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode")
|
|
os_log("启用语言检测模式", log: log, type: .info)
|
|
} else {
|
|
// 设置指定的识别语言
|
|
speechConfig?.speechRecognitionLanguage = currentLanguage
|
|
}
|
|
|
|
// 创建识别器
|
|
return setupRecognizer()
|
|
} catch {
|
|
os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription)
|
|
return false
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 向音频流写入音频数据
|
|
* 仅当音频源设置为external时有效
|
|
*
|
|
* @param data 音频数据字节数组
|
|
*/
|
|
public func pushAudioData(data: Data) {
|
|
if audioSourceType != .external {
|
|
return
|
|
}
|
|
|
|
// 使用拉流模式,将数据推入队列
|
|
externalAudioStream?.pushAudio(data)
|
|
}
|
|
|
|
/**
|
|
* 开始连续语音识别
|
|
*
|
|
* @param callback 连续识别结果回调
|
|
* @param audioSourceType 音频源类型
|
|
* @return 是否成功开始识别
|
|
*/
|
|
public func startContinuousRecognition(
|
|
callback: ContinuousRecognizeCallback,
|
|
audioSourceType: AudioSourceType = .microphone
|
|
) -> Bool {
|
|
guard speechConfig != nil else {
|
|
callback.onError("语音服务未初始化")
|
|
return false
|
|
}
|
|
|
|
if _isContinuousRecognitionActive {
|
|
return true
|
|
}
|
|
|
|
self.audioSourceType = audioSourceType
|
|
|
|
// 重置识别器
|
|
if !setupRecognizer() {
|
|
callback.onError("重置识别器失败")
|
|
return false
|
|
}
|
|
|
|
do {
|
|
// 设置各种事件监听
|
|
setupEventListeners(callback: callback)
|
|
|
|
// 启动音频处理
|
|
startAudioProcessing()
|
|
|
|
// 开始连续识别
|
|
try recognizer?.startContinuousRecognition()
|
|
_isContinuousRecognitionActive = true
|
|
|
|
os_log("连续识别已启动,音频源: %{public}@", log: log, type: .info, audioSourceType == .microphone ? "麦克风" : "外部")
|
|
|
|
return true
|
|
} catch {
|
|
stopAudioProcessing()
|
|
_isContinuousRecognitionActive = false
|
|
callback.onError("启动连续识别失败: \(error.localizedDescription)")
|
|
return false
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 停止连续语音识别
|
|
*
|
|
* @return 是否成功停止
|
|
*/
|
|
public func stopContinuousRecognition() -> Bool {
|
|
guard speechConfig != nil else {
|
|
os_log("语音服务未初始化", log: log, type: .error)
|
|
return false
|
|
}
|
|
|
|
if !_isContinuousRecognitionActive {
|
|
return true
|
|
}
|
|
|
|
do {
|
|
guard let recognizer = recognizer else {
|
|
os_log("识别器为空,重置状态", log: log, type: .info)
|
|
_isContinuousRecognitionActive = false
|
|
return true
|
|
}
|
|
|
|
if audioSourceType == .external {
|
|
pushAudioData(data: Data())
|
|
}
|
|
|
|
// 停止连续识别
|
|
try recognizer.stopContinuousRecognition()
|
|
|
|
// 停止音频处理
|
|
stopAudioProcessing()
|
|
|
|
// 会话结束事件会设置_isContinuousRecognitionActive = false
|
|
return true
|
|
} catch {
|
|
// 强制重置状态
|
|
_isContinuousRecognitionActive = false
|
|
os_log("停止连续识别失败: %{public}@", log: log, type: .error, error.localizedDescription)
|
|
|
|
// 停止音频处理
|
|
stopAudioProcessing()
|
|
|
|
// 尝试强制关闭识别器
|
|
recognizer = nil
|
|
|
|
return false
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 检查连续识别是否活跃
|
|
*/
|
|
public func isContinuousRecognitionActive() -> Bool {
|
|
return self._isContinuousRecognitionActive
|
|
}
|
|
|
|
/**
|
|
* 释放所有资源
|
|
*/
|
|
public func dispose() {
|
|
do {
|
|
// 如果正在进行连续识别,先停止
|
|
if _isContinuousRecognitionActive {
|
|
// 直接停止,不等待结果
|
|
try? recognizer?.stopContinuousRecognition()
|
|
_isContinuousRecognitionActive = false
|
|
}
|
|
|
|
// 停止音频处理
|
|
stopAudioProcessing()
|
|
|
|
// 释放资源
|
|
recognizer = nil
|
|
speechConfig = nil
|
|
audioConfig = nil
|
|
|
|
// 确保状态被重置
|
|
_isContinuousRecognitionActive = false
|
|
externalAudioStream = nil
|
|
} catch {
|
|
// 确保状态被重置
|
|
_isContinuousRecognitionActive = false
|
|
externalAudioStream = nil
|
|
audioConfig = nil
|
|
recognizer = nil
|
|
speechConfig = nil
|
|
}
|
|
}
|
|
|
|
// MARK: - 私有方法
|
|
|
|
/**
|
|
* 设置识别器
|
|
*/
|
|
private func setupRecognizer() -> Bool {
|
|
do {
|
|
// 清理旧的识别器
|
|
recognizer = nil
|
|
|
|
// 设置音频配置
|
|
switch audioSourceType {
|
|
case .microphone:
|
|
// 使用默认麦克风输入配置
|
|
audioConfig = try SPXAudioConfiguration()
|
|
case .external:
|
|
// 使用拉流方式处理外部音频
|
|
setupExternalAudioStream()
|
|
}
|
|
|
|
// 创建识别器
|
|
if isAutoDetectLanguage {
|
|
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages)
|
|
recognizer = try SPXSpeechRecognizer(
|
|
speechConfiguration: speechConfig!,
|
|
autoDetectSourceLanguageConfiguration: autoDetectConfig,
|
|
audioConfiguration: audioConfig!
|
|
)
|
|
} else {
|
|
recognizer = try SPXSpeechRecognizer(
|
|
speechConfiguration: speechConfig!,
|
|
audioConfiguration: audioConfig!
|
|
)
|
|
}
|
|
|
|
return true
|
|
} catch {
|
|
os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription)
|
|
stopAudioProcessing()
|
|
return false
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 设置外部音频流 - 使用拉流方式
|
|
*/
|
|
private func setupExternalAudioStream() {
|
|
do {
|
|
// 创建外部音频拉流对象
|
|
externalAudioStream = ExternalAudioPullStream()
|
|
|
|
// 创建音频配置
|
|
audioConfig = SPXAudioConfiguration(streamInput: externalAudioStream!.pullStream)
|
|
} catch {
|
|
os_log("设置外部音频流失败: %{public}@", log: log, type: .error, error.localizedDescription)
|
|
externalAudioStream = nil
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 设置事件监听器
|
|
*/
|
|
private func setupEventListeners(callback: ContinuousRecognizeCallback) {
|
|
guard let recognizer = recognizer else { return }
|
|
|
|
// 识别中事件
|
|
recognizer.addRecognizingEventHandler { [weak self] (sender, event) in
|
|
guard let self = self else { return }
|
|
|
|
let result = event.result
|
|
let detectedLanguage = self.getDetectedLanguage(from: result)
|
|
// os_log("识别中: %{public}@", log: self.log, type: .info, result.text ?? "")
|
|
// 直接在当前线程调用回调
|
|
callback.onRecognizing(result.text ?? "", detectedLanguage)
|
|
}
|
|
|
|
// 识别完成事件
|
|
recognizer.addRecognizedEventHandler { [weak self] (sender, event) in
|
|
guard let self = self else { return }
|
|
|
|
let result = event.result
|
|
if result.reason == SPXResultReason.recognizedSpeech {
|
|
let detectedLanguage = self.getDetectedLanguage(from: result)
|
|
// 直接在当前线程调用回调
|
|
callback.onResult(result.text ?? "", detectedLanguage)
|
|
}
|
|
}
|
|
|
|
// 会话开始事件
|
|
recognizer.addSessionStartedEventHandler { (sender, event) in
|
|
// 直接在当前线程调用回调
|
|
callback.onSessionStarted()
|
|
}
|
|
|
|
// 会话结束事件
|
|
recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in
|
|
guard let self = self else { return }
|
|
|
|
// 直接在当前线程调用回调
|
|
callback.onSessionStopped()
|
|
self._isContinuousRecognitionActive = false
|
|
self.stopAudioProcessing()
|
|
}
|
|
|
|
// 取消事件
|
|
recognizer.addCanceledEventHandler { [weak self] (sender, event) in
|
|
guard let self = self else { return }
|
|
|
|
let errorDetails = event.errorDetails ?? "未知错误"
|
|
let reason = String(describing: event.reason.rawValue)
|
|
|
|
os_log("识别取消: %{public}@", log: self.log, type: .error, errorDetails)
|
|
|
|
callback.onCanceled(reason, errorDetails)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 获取检测到的语言
|
|
*/
|
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String {
|
|
if !isAutoDetectLanguage {
|
|
return ""
|
|
}
|
|
|
|
// 尝试从属性中获取语言
|
|
if let properties = result.properties,
|
|
let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage") {
|
|
return language
|
|
}
|
|
|
|
// 尝试另一种方式获取语言
|
|
do {
|
|
let langResult = try SPXAutoDetectSourceLanguageResult(result)
|
|
return langResult.language ?? ""
|
|
} catch {
|
|
return ""
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 启动音频处理
|
|
*/
|
|
private func startAudioProcessing() {
|
|
switch audioSourceType {
|
|
case .microphone:
|
|
// 拉流模式不需要额外启动,SDK会自动拉取数据
|
|
break
|
|
case .external:
|
|
// 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData
|
|
break
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 停止音频处理
|
|
*/
|
|
private func stopAudioProcessing() {
|
|
if let stream = externalAudioStream {
|
|
stream.close()
|
|
externalAudioStream = nil
|
|
// os_log("外部音频流已关闭", log: log, type: .info)
|
|
}
|
|
}
|
|
|
|
|
|
/**
|
|
* 外部音频拉流
|
|
* 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式
|
|
*/
|
|
private class ExternalAudioPullStream: NSObject {
|
|
private(set) var pullStream: SPXPullAudioInputStream!
|
|
private let queue = LinkedBlockingQueue<Data>()
|
|
private var closed = false
|
|
|
|
override init() {
|
|
super.init()
|
|
|
|
pullStream = SPXPullAudioInputStream(
|
|
readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in
|
|
guard let self = self else { return 0 }
|
|
return self.read(buffer: data, size: Int(size))
|
|
},
|
|
closeHandler: { [weak self] in
|
|
self?.close()
|
|
}
|
|
)
|
|
}
|
|
|
|
/**
|
|
* 外部调用:推送音频数据到队列
|
|
* @param data 音频数据
|
|
*/
|
|
func pushAudio(_ data: Data) {
|
|
if !closed {
|
|
queue.put(data)
|
|
}
|
|
}
|
|
|
|
/**
|
|
* SDK调用:从队列中拉取数据
|
|
* @param buffer SDK提供的缓冲区
|
|
* @param size 缓冲区大小
|
|
* @return 读取的字节数,0表示流结束
|
|
*/
|
|
private func read(buffer: NSMutableData, size: Int) -> Int {
|
|
// 阻塞等待下一块数据
|
|
guard let chunk = queue.take() else {
|
|
return 0 // 队列已关闭
|
|
}
|
|
|
|
// 如果是空数据,表示流结束
|
|
if chunk.isEmpty {
|
|
return 0
|
|
}
|
|
|
|
let toCopy = min(chunk.count, size)
|
|
buffer.append(chunk.prefix(toCopy))
|
|
return toCopy
|
|
}
|
|
|
|
/**
|
|
* SDK调用:关闭流
|
|
*/
|
|
func close() {
|
|
closed = true
|
|
queue.close()
|
|
}
|
|
}
|
|
|
|
/**
|
|
* iOS版LinkedBlockingQueue实现
|
|
* 与Android LinkedBlockingQueue保持一致的API和行为
|
|
*/
|
|
private class LinkedBlockingQueue<T> {
|
|
private var queue: [T] = []
|
|
private var isClosed = false
|
|
|
|
// 使用DispatchSemaphore实现阻塞行为
|
|
private let availableItems: DispatchSemaphore
|
|
private let queueLock = NSLock()
|
|
|
|
init() {
|
|
self.availableItems = DispatchSemaphore(value: 0)
|
|
}
|
|
|
|
/**
|
|
* 阻塞式获取元素(等价于Android的take())
|
|
* @return 队列中的元素,如果队列已关闭则返回nil
|
|
*/
|
|
func take() -> T? {
|
|
// 等待可用元素(阻塞直到有元素或队列关闭)
|
|
availableItems.wait()
|
|
|
|
queueLock.lock()
|
|
defer { queueLock.unlock() }
|
|
|
|
// 检查队列是否已关闭
|
|
if isClosed && queue.isEmpty {
|
|
return nil
|
|
}
|
|
|
|
// 获取第一个元素
|
|
guard !queue.isEmpty else {
|
|
return nil
|
|
}
|
|
|
|
return queue.removeFirst()
|
|
}
|
|
|
|
/**
|
|
* 添加元素(等价于Android的put())
|
|
* @param item 要添加的元素
|
|
*/
|
|
func put(_ item: T) {
|
|
queueLock.lock()
|
|
defer { queueLock.unlock() }
|
|
|
|
// 检查队列是否已关闭
|
|
if isClosed {
|
|
return
|
|
}
|
|
|
|
// 添加元素(无容量限制,与Android一致)
|
|
queue.append(item)
|
|
|
|
// 通知有新元素可用
|
|
availableItems.signal()
|
|
}
|
|
|
|
/**
|
|
* 关闭队列
|
|
* 关闭后不能再添加新元素,但可以继续取出已有元素
|
|
*/
|
|
func close() {
|
|
queueLock.lock()
|
|
defer { queueLock.unlock() }
|
|
|
|
isClosed = true
|
|
|
|
// 唤醒所有等待的take()操作
|
|
for _ in 0..<100 { // 假设最多100个等待者
|
|
availableItems.signal()
|
|
}
|
|
}
|
|
}
|
|
|
|
/**
|
|
* 连续识别回调接口
|
|
*/
|
|
public protocol ContinuousRecognizeCallback {
|
|
/**
|
|
* 返回识别结果
|
|
*
|
|
* @param text 识别的文本
|
|
* @param detectedLanguage 检测到的语言
|
|
*/
|
|
func onResult(_ text: String, _ detectedLanguage: String)
|
|
|
|
/**
|
|
* 识别进行中调用
|
|
*
|
|
* @param recognizing 正在识别的文本
|
|
* @param detectedLanguage 检测到的语言
|
|
*/
|
|
func onRecognizing(_ recognizing: String, _ detectedLanguage: String)
|
|
|
|
/**
|
|
* 会话开始时调用
|
|
*/
|
|
func onSessionStarted()
|
|
|
|
/**
|
|
* 会话结束时调用
|
|
*/
|
|
func onSessionStopped()
|
|
|
|
/**
|
|
* 识别取消时调用
|
|
*
|
|
* @param reason 取消原因
|
|
* @param errorDetails 错误详情
|
|
*/
|
|
func onCanceled(_ reason: String, _ errorDetails: String)
|
|
|
|
/**
|
|
* 识别出错时调用
|
|
*
|
|
* @param error 错误信息
|
|
*/
|
|
func onError(_ error: String)
|
|
}
|
|
}
|