diff --git a/ios/Podfile.lock b/ios/Podfile.lock index 77f53258d..63f38bd9d 100644 --- a/ios/Podfile.lock +++ b/ios/Podfile.lock @@ -85,7 +85,7 @@ PODS: - SDWebImage/Core (~> 5.17) - spotify_sdk (0.0.1): - Flutter - - tencentcloud_cos_sdk_plugin (1.2.4): + - tencentcloud_cos_sdk_plugin (1.2.3): - Flutter - QCloudCOSXML (= 6.4.7) - WechatOpenSDK-XCFramework (2.0.4) @@ -203,7 +203,7 @@ SPEC CHECKSUMS: SDWebImage: f29024626962457f3470184232766516dee8dfea SDWebImageWebPCoder: e38c0a70396191361d60c092933e22c20d5b1380 spotify_sdk: a48400bb29f70c4fe251ebfdc9135c37097ac5ca - tencentcloud_cos_sdk_plugin: 7bed564dbe72df23e7b5cacb1e51b1a20cb1639c + tencentcloud_cos_sdk_plugin: c3d78b2c3bae42b5a20009dc047f3f1621e6272b WechatOpenSDK-XCFramework: 36fb2bea0754266c17184adf4963d7e6ff98b69f wifi_iot: f645260a2be8608517b2a9bf4c39b98e97003acc diff --git a/ios/Runner.xcodeproj/project.pbxproj b/ios/Runner.xcodeproj/project.pbxproj index 34f3271ab..c04843123 100644 --- a/ios/Runner.xcodeproj/project.pbxproj +++ b/ios/Runner.xcodeproj/project.pbxproj @@ -252,7 +252,7 @@ ); mainGroup = 97C146E51CF9000F007C117D; packageReferences = ( - 781AD8BC2B33823900A9FFBB /* XCLocalSwiftPackageReference "Flutter/ephemeral/Packages/FlutterGeneratedPluginSwiftPackage" */, + 781AD8BC2B33823900A9FFBB /* XCLocalSwiftPackageReference "FlutterGeneratedPluginSwiftPackage" */, D013B7972DFA83E400B2B69B /* XCRemoteSwiftPackageReference "flutterfire" */, ); productRefGroup = 97C146EF1CF9000F007C117D /* Products */; @@ -529,6 +529,7 @@ DEVELOPMENT_TEAM = 537SFY28XS; ENABLE_BITCODE = NO; INFOPLIST_FILE = Runner/Info.plist; + INFOPLIST_KEY_CFBundleDisplayName = "Deep Voice"; IPHONEOS_DEPLOYMENT_TARGET = 18.0; LD_RUNPATH_SEARCH_PATHS = ( "$(inherited)", @@ -720,6 +721,7 @@ DEVELOPMENT_TEAM = 537SFY28XS; ENABLE_BITCODE = NO; INFOPLIST_FILE = Runner/Info.plist; + INFOPLIST_KEY_CFBundleDisplayName = "Deep Voice"; IPHONEOS_DEPLOYMENT_TARGET = 18.0; LD_RUNPATH_SEARCH_PATHS = ( "$(inherited)", @@ -745,6 +747,7 @@ DEVELOPMENT_TEAM = 537SFY28XS; ENABLE_BITCODE = NO; INFOPLIST_FILE = Runner/Info.plist; + INFOPLIST_KEY_CFBundleDisplayName = "Deep Voice"; IPHONEOS_DEPLOYMENT_TARGET = 18.0; LD_RUNPATH_SEARCH_PATHS = ( "$(inherited)", @@ -794,7 +797,7 @@ /* End XCConfigurationList section */ /* Begin XCLocalSwiftPackageReference section */ - 781AD8BC2B33823900A9FFBB /* XCLocalSwiftPackageReference "Flutter/ephemeral/Packages/FlutterGeneratedPluginSwiftPackage" */ = { + 781AD8BC2B33823900A9FFBB /* XCLocalSwiftPackageReference "FlutterGeneratedPluginSwiftPackage" */ = { isa = XCLocalSwiftPackageReference; relativePath = Flutter/ephemeral/Packages/FlutterGeneratedPluginSwiftPackage; }; diff --git a/lib/modules/translation/controllers/translation_controller.dart b/lib/modules/translation/controllers/translation_controller.dart index 703c87575..98c9f6cf9 100644 --- a/lib/modules/translation/controllers/translation_controller.dart +++ b/lib/modules/translation/controllers/translation_controller.dart @@ -1,4 +1,3 @@ -import 'dart:ffi'; import 'dart:io'; import '../../../core/utils/permission_util.dart'; diff --git a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift index 251984c27..fd90f7047 100644 --- a/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift +++ b/local_plugins/agent_service/ios/agent_service/Sources/agent_service/AgentServiceImpl.swift @@ -206,6 +206,12 @@ class AgentServiceImpl: NSObject { sendError("初始化语音识别服务失败", code: "ASR_INIT_ERROR") return false } + guard let asrsetupSuccess = azureAsrHelper?.setupEventListeners( + callback: self, + ), asrsetupSuccess else { + sendError("初始化语音识别服务失败", code: "ASR_INIT_ERROR") + return false + } guard let ttsSuccess = azureTtsHelper?.initialize( ttsAppId: "", @@ -316,7 +322,7 @@ class AgentServiceImpl: NSObject { } guard let success = azureAsrHelper?.startContinuousRecognition( - callback: self, + // callback: self, audioSourceType: audioSourceType ), success else { sendError("启动语音识别失败", code: "RECOGNITION_START_ERROR") @@ -346,8 +352,12 @@ class AgentServiceImpl: NSObject { if !isRecognizing || !isInitialized { return false } - - azureAsrHelper?.pushAudioData(data: audioData) + guard let audioStream = azureAsrHelper?.audioStream else { + os_log("音频流未初始化", log: logger,type: .error) + return false +} +audioStream.saveAudioDataTo(data: audioData) + return true } @@ -743,6 +753,7 @@ class AgentServiceImpl: NSObject { } func dispose() -> Bool { + print("ai释放资源") if isRecognizing { stopRecognition() } diff --git a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt index 3adf31b59..bd6ffc884 100644 --- a/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt +++ b/local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt @@ -711,7 +711,7 @@ class AzureAsrHelper(private val context: Context) { if (bytesToWrite < data.size) data.copyOf(bytesToWrite) else data try { - continuousCallback?.onAudio(finalData) + //continuousCallback?.onAudio(finalData) pushAudioStream?.write(finalData) recordfile?.saveAudioDataToWav(finalData) } catch (e: Exception) { diff --git a/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift deleted file mode 100644 index ffb76605c..000000000 --- a/local_plugins/azure_speech/ios/Classes/AzureAsrHelper.swift +++ /dev/null @@ -1,618 +0,0 @@ -import Foundation -import AVFoundation -import MicrosoftCognitiveServicesSpeech -// 自定义语音处理组件,提供音频流处理等功能 -import speech -import os.log - - -/** - * Azure ASR Helper - * - * 基于微软Azure语音服务的ASR实现 - * 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-recognize-speech - */ -public class AzureAsrHelper: NSObject { - private let tag = "AzureAsrHelper" - // 日志对象 - private let log = OSLog(subsystem: "com.azure.speech", category: "AzureAsrHelper") - - // 核心组件 - private var speechConfig: SPXSpeechConfiguration? - private var recognizer: SPXSpeechRecognizer? - private var audioConfig: SPXAudioConfiguration? - - // 状态管理 - private var _isContinuousRecognitionActive = false - - // 配置参数 - private var currentLanguage = "zh-CN" - private var supportedLanguages = ["zh-CN"] - private var isAutoDetectLanguage = false - private var subscriptionKey = "" - private var region = "" - - // 音频源配置 - public enum AudioSourceType { - /** 使用设备麦克风 */ - case microphone - - /** 使用外部提供的音频数据 */ - case external - } - - private var audioSourceType = AudioSourceType.microphone - - // 音频处理 - private var externalAudioStream: ExternalAudioPullStream? - - /** - * 初始化Azure语音服务 - * - * @param subscriptionKey Azure 订阅密钥 - * @param region Azure 区域 - * @param supportedLanguages 支持的语言数组 - * @param audioSourceType 音频源类型 - * @return 初始化是否成功 - */ - public func initialize( - subscriptionKey: String, - region: String, - supportedLanguages: [String] = ["zh-CN"], - audioSourceType: AudioSourceType = .microphone - ) -> Bool { - do { - // 检查配置是否为空 - if subscriptionKey.isEmpty || region.isEmpty { - os_log("Azure 配置信息不完整", log: log, type: .error) - return false - } - - // 释放之前的资源 - dispose() - - // 保存配置 - self.subscriptionKey = subscriptionKey - self.region = region - self.audioSourceType = audioSourceType - - // 设置语言 - if !supportedLanguages.isEmpty { - self.supportedLanguages = supportedLanguages - } - - // 根据支持的语言数量决定是否启用自动语言检测 - self.isAutoDetectLanguage = supportedLanguages.count >= 2 - - // 如果只有一种语言,设置为当前语言 - if !isAutoDetectLanguage && !supportedLanguages.isEmpty { - self.currentLanguage = supportedLanguages[0] - } - - // 创建语音配置 - speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region) - - if isAutoDetectLanguage { - // 直接启用语言检测模式 - speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") - os_log("启用语言检测模式", log: log, type: .info) - } else { - // 设置指定的识别语言 - speechConfig?.speechRecognitionLanguage = currentLanguage - } - - // 创建识别器 - return setupRecognizer() - } catch { - os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) - return false - } - } - - /** - * 向音频流写入音频数据 - * 仅当音频源设置为external时有效 - * - * @param data 音频数据字节数组 - */ - public func pushAudioData(data: Data) { - if audioSourceType != .external { - return - } - - // 使用拉流模式,将数据推入队列 - externalAudioStream?.pushAudio(data) - } - - /** - * 开始连续语音识别 - * - * @param callback 连续识别结果回调 - * @param audioSourceType 音频源类型 - * @return 是否成功开始识别 - */ - public func startContinuousRecognition( - callback: ContinuousRecognizeCallback, - audioSourceType: AudioSourceType = .microphone - ) -> Bool { - guard speechConfig != nil else { - callback.onError("语音服务未初始化") - return false - } - - if _isContinuousRecognitionActive { - return true - } - - self.audioSourceType = audioSourceType - - // 重置识别器 - if !setupRecognizer() { - callback.onError("重置识别器失败") - return false - } - - do { - // 设置各种事件监听 - setupEventListeners(callback: callback) - - // 启动音频处理 - startAudioProcessing() - - // 开始连续识别 - try recognizer?.startContinuousRecognition() - _isContinuousRecognitionActive = true - - os_log("连续识别已启动,音频源: %{public}@", log: log, type: .info, audioSourceType == .microphone ? "麦克风" : "外部") - - return true - } catch { - stopAudioProcessing() - _isContinuousRecognitionActive = false - callback.onError("启动连续识别失败: \(error.localizedDescription)") - return false - } - } - - /** - * 停止连续语音识别 - * - * @return 是否成功停止 - */ - public func stopContinuousRecognition() -> Bool { - guard speechConfig != nil else { - os_log("语音服务未初始化", log: log, type: .error) - return false - } - - if !_isContinuousRecognitionActive { - return true - } - - do { - guard let recognizer = recognizer else { - os_log("识别器为空,重置状态", log: log, type: .info) - _isContinuousRecognitionActive = false - return true - } - - if audioSourceType == .external { - pushAudioData(data: Data()) - } - - // 停止连续识别 - try recognizer.stopContinuousRecognition() - - // 停止音频处理 - stopAudioProcessing() - - // 会话结束事件会设置_isContinuousRecognitionActive = false - return true - } catch { - // 强制重置状态 - _isContinuousRecognitionActive = false - os_log("停止连续识别失败: %{public}@", log: log, type: .error, error.localizedDescription) - - // 停止音频处理 - stopAudioProcessing() - - // 尝试强制关闭识别器 - recognizer = nil - - return false - } - } - - /** - * 检查连续识别是否活跃 - */ - public func isContinuousRecognitionActive() -> Bool { - return self._isContinuousRecognitionActive - } - - /** - * 释放所有资源 - */ - public func dispose() { - do { - // 如果正在进行连续识别,先停止 - if _isContinuousRecognitionActive { - // 直接停止,不等待结果 - try? recognizer?.stopContinuousRecognition() - _isContinuousRecognitionActive = false - } - - // 停止音频处理 - stopAudioProcessing() - - // 释放资源 - recognizer = nil - speechConfig = nil - audioConfig = nil - - // 确保状态被重置 - _isContinuousRecognitionActive = false - externalAudioStream = nil - } catch { - // 确保状态被重置 - _isContinuousRecognitionActive = false - externalAudioStream = nil - audioConfig = nil - recognizer = nil - speechConfig = nil - } - } - - // MARK: - 私有方法 - - /** - * 设置识别器 - */ - private func setupRecognizer() -> Bool { - do { - // 清理旧的识别器 - recognizer = nil - - // 设置音频配置 - switch audioSourceType { - case .microphone: - // 使用默认麦克风输入配置 - audioConfig = try SPXAudioConfiguration() - case .external: - // 使用拉流方式处理外部音频 - setupExternalAudioStream() - } - - // 创建识别器 - if isAutoDetectLanguage { - let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) - recognizer = try SPXSpeechRecognizer( - speechConfiguration: speechConfig!, - autoDetectSourceLanguageConfiguration: autoDetectConfig, - audioConfiguration: audioConfig! - ) - } else { - recognizer = try SPXSpeechRecognizer( - speechConfiguration: speechConfig!, - audioConfiguration: audioConfig! - ) - } - - return true - } catch { - os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) - stopAudioProcessing() - return false - } - } - - /** - * 设置外部音频流 - 使用拉流方式 - */ - private func setupExternalAudioStream() { - do { - // 创建外部音频拉流对象 - externalAudioStream = ExternalAudioPullStream() - - // 创建音频配置 - audioConfig = SPXAudioConfiguration(streamInput: externalAudioStream!.pullStream) - } catch { - os_log("设置外部音频流失败: %{public}@", log: log, type: .error, error.localizedDescription) - externalAudioStream = nil - } - } - - /** - * 设置事件监听器 - */ - private func setupEventListeners(callback: ContinuousRecognizeCallback) { - guard let recognizer = recognizer else { return } - - // 识别中事件 - recognizer.addRecognizingEventHandler { [weak self] (sender, event) in - guard let self = self else { return } - - let result = event.result - let detectedLanguage = self.getDetectedLanguage(from: result) - // os_log("识别中: %{public}@", log: self.log, type: .info, result.text ?? "") - // 直接在当前线程调用回调 - callback.onRecognizing(result.text ?? "", detectedLanguage) - } - - // 识别完成事件 - recognizer.addRecognizedEventHandler { [weak self] (sender, event) in - guard let self = self else { return } - - let result = event.result - if result.reason == SPXResultReason.recognizedSpeech { - let detectedLanguage = self.getDetectedLanguage(from: result) - // 直接在当前线程调用回调 - callback.onResult(result.text ?? "", detectedLanguage) - } - } - - // 会话开始事件 - recognizer.addSessionStartedEventHandler { (sender, event) in - // 直接在当前线程调用回调 - callback.onSessionStarted() - } - - // 会话结束事件 - recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in - guard let self = self else { return } - - // 直接在当前线程调用回调 - callback.onSessionStopped() - self._isContinuousRecognitionActive = false - self.stopAudioProcessing() - } - - // 取消事件 - recognizer.addCanceledEventHandler { [weak self] (sender, event) in - guard let self = self else { return } - - let errorDetails = event.errorDetails ?? "未知错误" - let reason = String(describing: event.reason.rawValue) - - os_log("识别取消: %{public}@", log: self.log, type: .error, errorDetails) - - callback.onCanceled(reason, errorDetails) - } - } - - /** - * 获取检测到的语言 - */ - private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { - if !isAutoDetectLanguage { - return "" - } - - // 尝试从属性中获取语言 - if let properties = result.properties, - let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage") { - return language - } - - // 尝试另一种方式获取语言 - do { - let langResult = try SPXAutoDetectSourceLanguageResult(result) - return langResult.language ?? "" - } catch { - return "" - } - } - - /** - * 启动音频处理 - */ - private func startAudioProcessing() { - switch audioSourceType { - case .microphone: - // 拉流模式不需要额外启动,SDK会自动拉取数据 - break - case .external: - // 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData - break - } - } - - /** - * 停止音频处理 - */ - private func stopAudioProcessing() { - if let stream = externalAudioStream { - stream.close() - externalAudioStream = nil - // os_log("外部音频流已关闭", log: log, type: .info) - } - } - - - /** - * 外部音频拉流 - * 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式 - */ - private class ExternalAudioPullStream: NSObject { - private(set) var pullStream: SPXPullAudioInputStream! - private let queue = LinkedBlockingQueue() - private var closed = false - - override init() { - super.init() - - pullStream = SPXPullAudioInputStream( - readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in - guard let self = self else { return 0 } - return self.read(buffer: data, size: Int(size)) - }, - closeHandler: { [weak self] in - self?.close() - } - ) - } - - /** - * 外部调用:推送音频数据到队列 - * @param data 音频数据 - */ - func pushAudio(_ data: Data) { - if !closed { - queue.put(data) - } - } - - /** - * SDK调用:从队列中拉取数据 - * @param buffer SDK提供的缓冲区 - * @param size 缓冲区大小 - * @return 读取的字节数,0表示流结束 - */ - private func read(buffer: NSMutableData, size: Int) -> Int { - // 阻塞等待下一块数据 - guard let chunk = queue.take() else { - return 0 // 队列已关闭 - } - - // 如果是空数据,表示流结束 - if chunk.isEmpty { - return 0 - } - - let toCopy = min(chunk.count, size) - buffer.append(chunk.prefix(toCopy)) - return toCopy - } - - /** - * SDK调用:关闭流 - */ - func close() { - closed = true - queue.close() - } - } - - /** - * iOS版LinkedBlockingQueue实现 - * 与Android LinkedBlockingQueue保持一致的API和行为 - */ - private class LinkedBlockingQueue { - private var queue: [T] = [] - private var isClosed = false - - // 使用DispatchSemaphore实现阻塞行为 - private let availableItems: DispatchSemaphore - private let queueLock = NSLock() - - init() { - self.availableItems = DispatchSemaphore(value: 0) - } - - /** - * 阻塞式获取元素(等价于Android的take()) - * @return 队列中的元素,如果队列已关闭则返回nil - */ - func take() -> T? { - // 等待可用元素(阻塞直到有元素或队列关闭) - availableItems.wait() - - queueLock.lock() - defer { queueLock.unlock() } - - // 检查队列是否已关闭 - if isClosed && queue.isEmpty { - return nil - } - - // 获取第一个元素 - guard !queue.isEmpty else { - return nil - } - - return queue.removeFirst() - } - - /** - * 添加元素(等价于Android的put()) - * @param item 要添加的元素 - */ - func put(_ item: T) { - queueLock.lock() - defer { queueLock.unlock() } - - // 检查队列是否已关闭 - if isClosed { - return - } - - // 添加元素(无容量限制,与Android一致) - queue.append(item) - - // 通知有新元素可用 - availableItems.signal() - } - - /** - * 关闭队列 - * 关闭后不能再添加新元素,但可以继续取出已有元素 - */ - func close() { - queueLock.lock() - defer { queueLock.unlock() } - - isClosed = true - - // 唤醒所有等待的take()操作 - for _ in 0..<100 { // 假设最多100个等待者 - availableItems.signal() - } - } - } - - /** - * 连续识别回调接口 - */ - public protocol ContinuousRecognizeCallback { - /** - * 返回识别结果 - * - * @param text 识别的文本 - * @param detectedLanguage 检测到的语言 - */ - func onResult(_ text: String, _ detectedLanguage: String) - - /** - * 识别进行中调用 - * - * @param recognizing 正在识别的文本 - * @param detectedLanguage 检测到的语言 - */ - func onRecognizing(_ recognizing: String, _ detectedLanguage: String) - - /** - * 会话开始时调用 - */ - func onSessionStarted() - - /** - * 会话结束时调用 - */ - func onSessionStopped() - - /** - * 识别取消时调用 - * - * @param reason 取消原因 - * @param errorDetails 错误详情 - */ - func onCanceled(_ reason: String, _ errorDetails: String) - - /** - * 识别出错时调用 - * - * @param error 错误信息 - */ - func onError(_ error: String) - } -} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.h b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.h deleted file mode 100644 index 7f8606465..000000000 --- a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.h +++ /dev/null @@ -1,17 +0,0 @@ -// -// AzureSpeechPlugin.h -// azure_speech -// -// Created for azure_speech plugin compatibility. -// - -#import - -/** - * Azure Speech Plugin 头文件 - * - * 注:实际实现使用 Swift,此头文件仅用于 Flutter 框架兼容 - */ -@interface AzureSpeechPlugin : NSObject -+ (void)registerWithRegistrar:(NSObject*)registrar; -@end \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift b/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift deleted file mode 100644 index 66e375647..000000000 --- a/local_plugins/azure_speech/ios/Classes/AzureSpeechPlugin.swift +++ /dev/null @@ -1,422 +0,0 @@ -import Flutter -import UIKit -import AVFoundation -import MicrosoftCognitiveServicesSpeech -// 自定义语音处理组件,提供音频流处理等功能 -import speech -import os.log - -/** - * Azure Speech Plugin - * - * 基于微软Azure语音服务的Flutter插件 - * 提供语音识别(ASR)和语音合成(TTS)功能 - */ -@objc public class AzureSpeechPlugin: NSObject, FlutterPlugin { - // 日志标签 - private let tag = "AzureSpeechPlugin" - // 日志对象 - private let log = OSLog(subsystem: "com.azure.speech", category: "AzureSpeechPlugin") - - // ASR相关 - private var asrChannel: FlutterMethodChannel? - private var asrEventChannel: FlutterEventChannel? - private var asrEventSink: FlutterEventSink? - private let azureAsrHelper = AzureAsrHelper() - - // TTS相关 - private var ttsChannel: FlutterMethodChannel? - private var ttsEventChannel: FlutterEventChannel? - private var ttsEventSink: FlutterEventSink? - private let azureTtsHelper = AzureTtsHelper() - - // 是否已添加TTS事件监听器 - private var isTtsListenerAdded = false - - // 当前的连续识别回调 - private var currentAsrCallback: AsrCallbackWrapper? - - // 插件注册 - public static func register(with registrar: FlutterPluginRegistrar) { - let instance = AzureSpeechPlugin() - - // 初始化ASR通道 - let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) - registrar.addMethodCallDelegate(instance, channel: asrChannel) - instance.asrChannel = asrChannel - - // 初始化TTS通道 - let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) - registrar.addMethodCallDelegate(instance, channel: ttsChannel) - instance.ttsChannel = ttsChannel - - // 初始化ASR事件通道 - let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger()) - asrEventChannel.setStreamHandler(instance) - instance.asrEventChannel = asrEventChannel - - // 初始化TTS事件通道 - let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger()) - ttsEventChannel.setStreamHandler(instance) - instance.ttsEventChannel = ttsEventChannel - } - - // 发送ASR事件方法 - internal func sendAsrEvent(_ event: [String: Any]) { - if asrEventSink == nil { - os_log("无法发送ASR事件:事件通道未准备好", log: log, type: .error) - return - } - - DispatchQueue.main.async { [weak self] in - guard let self = self else { return } - self.asrEventSink?(event) - } - } - - // 发送TTS事件方法 - private func sendTtsEvent(_ event: [String: Any]) { - if ttsEventSink == nil { - os_log("无法发送TTS事件:事件通道未准备好", log: log, type: .error) - return - } - - DispatchQueue.main.async { [weak self] in - guard let self = self else { return } - self.ttsEventSink?(event) - } - } - - // 设置TTS事件监听器 - private func setupTtsEventListener() { - if !isTtsListenerAdded { - azureTtsHelper.addListener(self) - isTtsListenerAdded = true - } - } - - - - // 处理Flutter方法调用 - public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { - if call.method.hasPrefix("tts_") { - handleTtsMethodCall(call, result) - } else { - handleAsrMethodCall(call, result) - } - } - - // MARK: - ASR 方法处理 - - private func handleAsrMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - switch call.method { - case "initialize": - guard let args = call.arguments as? [String: Any], - let subscriptionKey = args["subscriptionKey"] as? String, - let region = args["region"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) - return - } - - let supportedLanguages = args["supportedLanguages"] as? [String] ?? ["zh-CN"] - - // 初始化ASR引擎 - let success = azureAsrHelper.initialize( - subscriptionKey: subscriptionKey, - region: region, - supportedLanguages: supportedLanguages, - audioSourceType: .microphone - ) - - result(success) - - case "recognizeOnce": - // iOS版本不支持recognizeOnce - result(FlutterError(code: "NOT_SUPPORTED", message: "iOS版本不支持recognizeOnce", details: nil)) - - case "startContinuousRecognition": - // 确保事件通道已准备好 - guard asrEventSink != nil else { - result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil)) - return - } - - // 解析音频源类型 - let audioSourceType: AzureAsrHelper.AudioSourceType - if let args = call.arguments as? [String: Any], - let audioSourceString = args["audioSourceType"] as? String { - switch audioSourceString.lowercased() { - case "external": - audioSourceType = .external - default: - audioSourceType = .microphone - } - } else { - audioSourceType = .microphone - } - - // 创建回调包装器 - currentAsrCallback = AsrCallbackWrapper(plugin: self) - - let success = azureAsrHelper.startContinuousRecognition( - callback: currentAsrCallback!, - audioSourceType: audioSourceType - ) - result(success) - - case "stopContinuousRecognition": - let success = azureAsrHelper.stopContinuousRecognition() - currentAsrCallback = nil - result(success) - - case "isContinuousRecognitionActive": - result(azureAsrHelper.isContinuousRecognitionActive()) - - case "dispose": - azureAsrHelper.dispose() - currentAsrCallback = nil - result(true) - - case "pushAudioData": - guard let args = call.arguments as? [String: Any], - let audioBytes = args["data"] as? FlutterStandardTypedData else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil)) - return - } - - azureAsrHelper.pushAudioData(data: audioBytes.data) - result(true) - - default: - result(FlutterMethodNotImplemented) - } - } - - // MARK: - TTS 方法处理 - - private func handleTtsMethodCall(_ call: FlutterMethodCall, _ result: @escaping FlutterResult) { - switch call.method { - case "tts_initialize": - guard let args = call.arguments as? [String: Any], - let subscriptionKey = args["subscriptionKey"] as? String, - let region = args["region"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) - return - } - - // 初始化TTS引擎 - let language = args["language"] as? String ?? "zh-CN" - let success = azureTtsHelper.initialize(ttsAppId: "", ttsAppToken: subscriptionKey, ttsResource: region, language: language) - - // 设置TTS事件监听器 - setupTtsEventListener() - - result(success) - - case "tts_set_voice": - guard let args = call.arguments as? [String: Any], - let voiceName = args["voiceName"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "语音名称不能为空", details: nil)) - return - } - - let success = azureTtsHelper.setVoice(voiceName) - result(success) - - case "tts_speak_once": - guard let args = call.arguments as? [String: Any], - let text = args["text"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) - return - } - - let success = azureTtsHelper.speakOnce(text) - result(success) - - case "tts_speak_stream": - guard let args = call.arguments as? [String: Any], - let text = args["text"] as? String else { - result(FlutterError(code: "INVALID_ARGUMENTS", message: "文本不能为空", details: nil)) - return - } - - let success = azureTtsHelper.speakStream(text) - result(success) - - case "tts_flush_stream": - let success = azureTtsHelper.flushStream() - result(success) - - case "tts_stop": - let success = azureTtsHelper.stop() - result(success) - - case "tts_isSpeaking": - result(azureTtsHelper.isSpeaking()) - - case "tts_release": - azureTtsHelper.dispose() - result(true) - - default: - result(FlutterMethodNotImplemented) - } - } -} - -// MARK: - ASR回调包装器 -private class AsrCallbackWrapper: AzureAsrHelper.ContinuousRecognizeCallback { - private weak var plugin: AzureSpeechPlugin? - - init(plugin: AzureSpeechPlugin) { - self.plugin = plugin - } - - func onResult(_ text: String, _ detectedLanguage: String) { - plugin?.sendAsrEvent([ - "type": "result", - "text": text, - "language": detectedLanguage - ]) - } - - func onRecognizing(_ recognizing: String, _ detectedLanguage: String) { - plugin?.sendAsrEvent([ - "type": "recognizing", - "text": recognizing, - "language": detectedLanguage - ]) - } - - func onSessionStarted() { - plugin?.sendAsrEvent([ - "type": "sessionStarted" - ]) - } - - func onSessionStopped() { - plugin?.sendAsrEvent([ - "type": "sessionStopped" - ]) - } - - func onCanceled(_ reason: String, _ errorDetails: String) { - plugin?.sendAsrEvent([ - "type": "canceled", - "reason": reason, - "errorDetails": errorDetails - ]) - } - - func onError(_ error: String) { - plugin?.sendAsrEvent([ - "type": "error", - "error": error - ]) - } -} - -// MARK: - 事件处理 -extension AzureSpeechPlugin: FlutterStreamHandler { - public func onListen(withArguments arguments: Any?, eventSink events: @escaping FlutterEventSink) -> FlutterError? { - // 根据通道类型设置事件接收器 - if let args = arguments as? [String: Any], - let channel = args["channel"] as? String { - - if channel == "asr" { - asrEventSink = events - } else if channel == "tts" { - ttsEventSink = events - setupTtsEventListener() - } - } else if arguments == nil { - // 如果没有指定通道,尝试弄清楚是哪个通道在监听 - if asrEventSink == nil && ttsEventSink != nil { - asrEventSink = events - } else if asrEventSink != nil && ttsEventSink == nil { - ttsEventSink = events - setupTtsEventListener() - } else { - // 无法确定哪个通道,默认设置为ASR - asrEventSink = events - } - } - - return nil - } - - public func onCancel(withArguments arguments: Any?) -> FlutterError? { - // 根据通道类型清除事件接收器 - if let args = arguments as? [String: Any], - let channel = args["channel"] as? String { - - if channel == "asr" { - asrEventSink = nil - } else if channel == "tts" { - ttsEventSink = nil - } - } else { - // 如果没有指定通道,清除所有通道 - asrEventSink = nil - ttsEventSink = nil - } - - return nil - } -} - - - -// MARK: - TTS 事件监听实现 -extension AzureSpeechPlugin: TtsEventListener { - // 使用@objc特性为方法提供一个不同的Objective-C选择器名称 - @objc(onTtsEvent:) - public func onEvent(_ event: TtsEvent) { - var eventMap: [String: Any] = [:] - - // 根据事件类型转换 - switch event.type { - case .synthesisStarted: - eventMap["type"] = "synthesis_started" - case .synthesisCompleted: - eventMap["type"] = "synthesis_completed" - case .synthesisCanceled: - eventMap["type"] = "synthesis_canceled" - if let reason = event.params["reason"] { - eventMap["reason"] = reason - } - if let errorDetails = event.params["errorDetails"] { - eventMap["errorDetails"] = errorDetails - } - case .error: - eventMap["type"] = "error" - if let errorCode = event.params["errorCode"] { - eventMap["errorCode"] = errorCode - } - if let errorMessage = event.params["errorMessage"] { - eventMap["errorMessage"] = errorMessage - } - @unknown default: - eventMap["type"] = "unknown" - for (key, value) in event.params { - eventMap[key] = value - } - } - - sendTtsEvent(eventMap) - } -} - -// MARK: - 音频数据监听实现 -extension AzureSpeechPlugin: AudioDataListener { - public func onAudioData(_ data: Data) { - if ttsEventSink == nil { return } - - let eventMap: [String: Any] = [ - "type": "audio_data", - "data": FlutterStandardTypedData(bytes: data) - ] - - sendTtsEvent(eventMap) - } -} \ No newline at end of file diff --git a/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift b/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift deleted file mode 100644 index 5a390d707..000000000 --- a/local_plugins/azure_speech/ios/Classes/AzureTtsHelper.swift +++ /dev/null @@ -1,604 +0,0 @@ -import Foundation -import AVFoundation -import MicrosoftCognitiveServicesSpeech -// 自定义语音处理组件,提供音频流处理等功能 -import speech -import os.log - -/** - * Azure TTS Helper - * - * 基于微软Azure语音服务的TTS实现 - * 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis - * - * 特性: - * - 对外接口保持同步,内部异步处理 - * - 使用专用队列保证合成顺序 - * - 避免阻塞主线程和其他后台线程 - * - 事件回调由调用方处理线程切换 - * - * 使用示例: - * let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理 - */ - public class AzureTtsHelper: NSObject, ITtsService { - private let tag = "AzureTtsHelper" - // 日志对象 - private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") - - private static let DEFAULT_LANGUAGE = "zh-CN" - private static let DEFAULT_VOICE = "zh-CN-XiaoxiaoNeural" - - // Azure语音服务配置 - private var speechConfig: SPXSpeechConfiguration? - private var synthesizer: SPXSpeechSynthesizer? - private var isInitialized = false - private var speaking = false - - // 当前配置 - private var currentVoice = DEFAULT_VOICE - private var currentLanguage = DEFAULT_LANGUAGE - private var currentRate = "+0%" - private var currentPitch = "+0%" - private var currentVolume = "100%" - - // 事件监听器列表 - private var eventListeners = NSHashTable.weakObjects() - - // 流式文本处理的缓冲区 - private var streamBuffer = "" - private var lastSpeakTime: TimeInterval = 0 - - // 内部异步处理队列,保证顺序执行 - private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) - private let synthesisGroup = DispatchGroup() - private var pendingTasks: [() -> Void] = [] - private let taskLock = NSLock() - - /** - * 初始化语音合成服务 - * - * @param appId 服务应用ID - * @param token 服务访问令牌/订阅密钥 - * @param resource 服务资源ID/区域(可选) - * @param language 语言代码,如"zh-CN" - * @return 是否初始化成功 - */ - public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { - do { - // 创建语音配置 - if ttsResource.isEmpty { - speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia") - } else { - speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource) - } - - // 设置语言 - speechConfig?.speechSynthesisLanguage = language - currentLanguage = language - - // 设置音频输出格式 - 使用16k、16位的PCM格式 - speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", - byName: "SpeechServiceConnection_SynthOutputFormat") - - // 创建合成器,使用默认音频输出(扬声器) - synthesizer = try SPXSpeechSynthesizer(speechConfig!) - - // 设置事件监听 - setupEventListeners() - - // 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny - if language.lowercased().starts(with: "zh") { - _ = setVoice("zh-CN-XiaoxiaoNeural") - } else { - _ = setVoice("en-US-JennyNeural") - } - - // 设置初始化完成 - isInitialized = true - - return true - } catch { - os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) - return false - } - } - - /** - * 设置语音角色 - * - * @param voiceName 语音角色名称(不同服务的语音角色命名可能不同) - * @return 是否设置成功 - */ - public func setVoice(_ voiceName: String) -> Bool { - if !isInitialized { return false } - - do { - currentVoice = voiceName - - // 更新语音名称 - if let config = speechConfig { - config.speechSynthesisVoiceName = voiceName - } - - // 重新创建合成器 - recreateSynthesizer() - return true - } catch { - os_log("设置语音失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "VOICE_SET_FAILED", - "errorMessage": "设置语音失败: \(error.localizedDescription)" - ]) - return false - } - } - - /** - * 单次合成并播放 - * - * @param text 要合成的文本 - * @return 是否成功开始合成 - */ - public func speakOnce(_ text: String) -> Bool { - if !isInitialized { - os_log("语音合成未初始化", log: log, type: .error) - return false - } - - // 立即返回成功,内部异步处理 - enqueueSynthesisTask { - self.performSynthesis(text: text) - } - - return true - } - - /** - * 将合成任务加入队列,保证顺序执行 - */ - private func enqueueSynthesisTask(_ task: @escaping () -> Void) { - taskLock.lock() - defer { taskLock.unlock() } - - pendingTasks.append(task) - - // 如果当前没有任务在执行,开始处理队列 - if pendingTasks.count == 1 { - processNextTask() - } - } - - /** - * 处理队列中的下一个任务 - */ - private func processNextTask() { - synthesisQueue.async { - self.synthesisGroup.enter() - - self.taskLock.lock() - guard !self.pendingTasks.isEmpty else { - self.taskLock.unlock() - self.synthesisGroup.leave() - return - } - let task = self.pendingTasks.removeFirst() - self.taskLock.unlock() - - // 执行任务 - task() - - self.synthesisGroup.leave() - - // 处理下一个任务 - self.taskLock.lock() - if !self.pendingTasks.isEmpty { - self.taskLock.unlock() - self.processNextTask() - } else { - self.taskLock.unlock() - } - } - } - - /** - * 实际执行合成的方法 - */ - private func performSynthesis(text: String) { - // 重置状态 - speaking = true - - // 生成SSML - let ssml = generateSsml(text) - - do { - os_log("开始合成: %{public}@", log: log, type: .debug, ssml) - // 使用同步方法,在后台队列中执行 - let result = try synthesizer?.startSpeakingSsml(ssml) - - // 检查结果 - if let result = result { - os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason)) - } - } catch { - os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "SYNTHESIS_FAILED", - "errorMessage": error.localizedDescription - ]) - speaking = false - } - } - - /** - * 流式合成文本 - * - * @param text 要合成的文本片段 - * @return 是否成功处理 - */ - public func speakStream(_ text: String) -> Bool { - if !isInitialized || text.isEmpty { - if !isInitialized { - notifyEvent(eventType: .error, params: [ - "errorCode": "NOT_INITIALIZED", - "errorMessage": "TTS引擎未初始化" - ]) - } - return false - } - - do { - // 添加新文本到缓冲区 - streamBuffer.append(text) - - // 增加500ms防抖逻辑 - let currentTime = Date().timeIntervalSince1970 - if currentTime - lastSpeakTime < 0.6 { - return true - } - lastSpeakTime = currentTime - - let currentText = streamBuffer - - // 定义标点符号列表 - let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"] - - // 查找最后一个标点符号的位置 - var lastPunctuationIndex = -1 - for (i, char) in currentText.enumerated().reversed() { - if punctuationMarks.contains(char) { - lastPunctuationIndex = i - break - } - } - - // 如果找到标点符号,则播放到该标点符号 - if lastPunctuationIndex >= 0 { - // 提取要播放的文本(包含标点符号) - let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines) - - // 剩余的文本保存在缓冲区中 - let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1) - streamBuffer = String(currentText[startIndex...]) - - // 只有非空文本才播放 - if !textToSpeak.isEmpty { - return speakOnce(textToSpeak) - } - } - - // 如果没有找到标点符号,则等待更多文本 - return true - } catch { - os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "STREAM_FAILED", - "errorMessage": "流式语音合成失败: \(error.localizedDescription)" - ]) - return false - } - } - - /** - * 刷新并播放流式文本缓冲区中的剩余内容 - * - * @return 是否成功处理 - */ - public func flushStream() -> Bool { - if !isInitialized { - notifyEvent(eventType: .error, params: [ - "errorCode": "NOT_INITIALIZED", - "errorMessage": "TTS引擎未初始化" - ]) - return false - } - - do { - // 获取缓冲区中剩余的文本 - let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines) - - // 清空缓冲区 - streamBuffer = "" - - // 如果缓冲区为空,直接返回成功 - if remainingText.isEmpty { - return true - } - - // 播放剩余文本 - return speakOnce(remainingText) - } catch { - os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "FLUSH_FAILED", - "errorMessage": "刷新流式文本失败: \(error.localizedDescription)" - ]) - return false - } - } - - /** - * 停止语音合成和播放 - * - * @return 是否成功停止 - */ - public func stop() -> Bool { - // 立即更新状态 - speaking = false - - // 清除流式缓冲区中的待播放内容 - streamBuffer = "" - - // 清空待处理的任务队列 - taskLock.lock() - pendingTasks.removeAll() - taskLock.unlock() - - // 在后台队列停止合成器,避免阻塞主线程 - synthesisQueue.async { - if let synthesizer = self.synthesizer { - do { - try synthesizer.stopSpeaking() - os_log("语音合成已停止", log: self.log, type: .info) - - // 直接通知停止完成 - self.notifyEvent(eventType: .synthesisCanceled) - } catch { - os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) - self.notifyEvent(eventType: .error, params: [ - "errorCode": "STOP_FAILED", - "errorMessage": error.localizedDescription - ]) - } - } - } - - return true - } - - /** - * 释放资源 - * 在不再需要服务时调用,释放底层资源 - */ - public func dispose() { - // 停止播放 - _ = stop() - - // 等待所有任务完成 - synthesisGroup.wait() - - // 清理资源 - synthesisQueue.async { - // 释放合成器 - self.synthesizer = nil - - // 释放配置 - self.speechConfig = nil - - // 清空流缓冲区 - self.streamBuffer = "" - - // 重置状态 - self.isInitialized = false - self.speaking = false - - // 清空任务队列 - self.taskLock.lock() - self.pendingTasks.removeAll() - self.taskLock.unlock() - } - } - - /** - * 添加TTS事件监听器 - * - * @param listener 事件监听器 - */ - public func addListener(_ listener: TtsEventListener) { - eventListeners.add(listener as AnyObject) - } - - /** - * 移除TTS事件监听器 - * - * @param listener 要移除的事件监听器 - */ - public func removeListener(_ listener: TtsEventListener) { - eventListeners.remove(listener as AnyObject) - } - - /** - * 添加音频数据监听器 - * 由于不再支持自定义音频流,此方法实际上不再有效 - * - * @param listener 音频数据监听器 - */ - public func addAudioDataListener(_ listener: AudioDataListener) { - os_log("警告:不支持音频数据监听器功能", log: log, type: .info) - } - - /** - * 移除音频数据监听器 - * 由于不再支持自定义音频流,此方法实际上不再有效 - * - * @param listener 要移除的音频数据监听器 - */ - public func removeAudioDataListener(_ listener: AudioDataListener) { - // 不做任何操作 - } - - /** - * 当前是否正在播放/合成 - */ - public func isSpeaking() -> Bool { - return speaking - } - - // MARK: - 辅助方法 - - /** - * 设置事件监听器 - */ - private func setupEventListeners() { - guard let synthesizer = synthesizer else { return } - - // 添加合成开始事件处理器 - synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in - guard let self = self else { return } - self.notifyEvent(eventType: .synthesisStarted) - } - - // 添加合成中事件处理器 - synthesizer.addSynthesizingEventHandler { [weak self] _, _ in - // 可以在这里处理合成中的事件,目前没有特别操作 - } - - // 添加合成完成事件处理器 - synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in - guard let self = self else { return } - self.speaking = false - self.notifyEvent(eventType: .synthesisCompleted) - } - - // 添加合成取消事件处理器 - synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in - guard let self = self else { return } - - self.speaking = false - - var params: [String: Any] = [:] - do { - let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) - if cancellationDetails.reason == SPXCancellationReason.error { - params["reason"] = String(describing: cancellationDetails.reason.rawValue) - params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" - } - } catch { - params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)" - } - os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误")) - - self.notifyEvent(eventType: .synthesisCanceled, params: params) - } - } - - /** - * 触发事件通知 - */ - private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { - let event = TtsEvent(type: eventType, params: params) - - // 直接通知事件,由调用方处理线程切换 - for case let listener as TtsEventListener in self.eventListeners.allObjects { - listener.onEvent(event) - } - } - - /** - * 重新创建合成器 - */ - private func recreateSynthesizer() { - do { - // 使用默认音频输出配置创建合成器 - synthesizer = try SPXSpeechSynthesizer(speechConfig!) - - // 设置事件监听 - setupEventListeners() - } catch { - os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "RECREATE_FAILED", - "errorMessage": "重新创建合成器失败: \(error.localizedDescription)" - ]) - } - } - - /** - * 设置语音参数 - */ - private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { - if !isInitialized { return false } - - do { - currentRate = formatPercentage(rate) - currentPitch = formatPercentage(pitch) - currentVolume = "\(min(max(volume, 0), 100))%" - return true - } catch { - os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) - notifyEvent(eventType: .error, params: [ - "errorCode": "PARAMS_SET_FAILED", - "errorMessage": "设置语音参数失败: \(error.localizedDescription)" - ]) - return false - } - } - - /** - * 格式化百分比值 - */ - private func formatPercentage(_ value: Int) -> String { - return value >= 0 ? "+\(value)%" : "\(value)%" - } - - /** - * 生成SSML - */ - private func generateSsml(_ rawText: String) -> String { - // 1. 定义要静音的符号和表情符号列表 - let symbolsToMute = [ - "#", "*", - "😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" - ] - - // 2. 转义 XML 保留字符 - var escapedText = rawText - .replacingOccurrences(of: "&", with: "&") - .replacingOccurrences(of: "<", with: "<") - .replacingOccurrences(of: ">", with: ">") - - // 3. 静音处理特殊符号和表情符号 - // 使用空白替换法,直接将符号替换为空字符串 - var processedText = escapedText - for symbol in symbolsToMute { - processedText = processedText.replacingOccurrences(of: symbol, with: "") - } - - // 4. 构造简化的SSML文档,减少嵌套层级 - let ssml = """ - - - - - \(processedText) - - - - - """ - - return ssml - } -} diff --git a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift index b0f4bcb02..f35a6561e 100644 --- a/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift +++ b/local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift @@ -44,8 +44,12 @@ public class AzureAsrHelper: NSObject { private var audioSourceType = AudioSourceType.microphone // 音频处理 - private var externalAudioStream: ExternalAudioPullStream? + // private var externalAudioStream: ExternalAudioPullStream? + // 音频处理 + public var audioStream: AudioStream? + + public var recordfile: RecordFile? /** * 初始化Azure语音服务 * @@ -70,7 +74,7 @@ public class AzureAsrHelper: NSObject { // 释放之前的资源 dispose() - + print("w[w[w]]") // 保存配置 self.subscriptionKey = subscriptionKey self.region = region @@ -91,11 +95,11 @@ public class AzureAsrHelper: NSObject { // 创建语音配置 speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region) - + speechConfig?.setPropertyTo("800", byName: "SpeechServiceConnection_EndSilenceTimeoutMs") speechConfig?.setPropertyTo("800", byName: "Speech_SegmentationSilenceTimeoutMs") speechConfig?.setPropertyTo("Time", byName: "Speech_SegmentationStrategy") - + if isAutoDetectLanguage { // 直接启用语言检测模式 speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") @@ -104,29 +108,16 @@ public class AzureAsrHelper: NSObject { // 设置指定的识别语言 speechConfig?.speechRecognitionLanguage = currentLanguage } - + // 录音文件类 + recordfile = RecordFile() // 创建识别器 - return setupRecognizer() + return true } catch { os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) return false } } - /** - * 向音频流写入音频数据 - * 仅当音频源设置为external时有效 - * - * @param data 音频数据字节数组 - */ - public func pushAudioData(data: Data) { - if audioSourceType != .external { - return - } - - // 使用拉流模式,将数据推入队列 - externalAudioStream?.pushAudio(data) - } /** * 开始连续语音识别 @@ -136,11 +127,10 @@ public class AzureAsrHelper: NSObject { * @return 是否成功开始识别 */ public func startContinuousRecognition( - callback: ContinuousRecognizeCallback, audioSourceType: AudioSourceType = .microphone ) -> Bool { guard speechConfig != nil else { - callback.onError("语音服务未初始化") + //callback.onError("语音服务未初始化") return false } @@ -150,18 +140,19 @@ public class AzureAsrHelper: NSObject { self.audioSourceType = audioSourceType - // 重置识别器 - if !setupRecognizer() { - callback.onError("重置识别器失败") - return false - } do { - // 设置各种事件监听 - setupEventListeners(callback: callback) + guard recognizer != nil else { + if !setupRecognizer() { + // callback.onError("重置识别器失败") + return false + } + return false + } - // 启动音频处理 - startAudioProcessing() + // 启动音频处理 + audioStream?.startAudioRecord() + // 开始连续识别 try recognizer?.startContinuousRecognition() @@ -171,9 +162,10 @@ public class AzureAsrHelper: NSObject { return true } catch { - stopAudioProcessing() - _isContinuousRecognitionActive = false - callback.onError("启动连续识别失败: \(error.localizedDescription)") + audioStream?.stopMicrophoneCapture() + // stopAudioProcessing() + _isContinuousRecognitionActive = false + //callback.onError("启动连续识别失败: \(error.localizedDescription)") return false } } @@ -184,44 +176,42 @@ public class AzureAsrHelper: NSObject { * @return 是否成功停止 */ public func stopContinuousRecognition() -> Bool { + print("stopContinuousRecognition") guard speechConfig != nil else { os_log("语音服务未初始化", log: log, type: .error) return false } - + guard let recognizer = recognizer else { + os_log("识别器为空,重置状态", log: log, type: .info) + return true + } + if !_isContinuousRecognitionActive { return true } do { - guard let recognizer = recognizer else { - os_log("识别器为空,重置状态", log: log, type: .info) - _isContinuousRecognitionActive = false - return true - } - + _isContinuousRecognitionActive = false + if audioSourceType == .external { - pushAudioData(data: Data()) + //pushAudioData(data: Data()) } // 停止连续识别 try recognizer.stopContinuousRecognition() // 停止音频处理 - stopAudioProcessing() - + //stopAudioProcessing() + audioStream?.stopMicrophoneCapture() // 会话结束事件会设置_isContinuousRecognitionActive = false return true } catch { // 强制重置状态 _isContinuousRecognitionActive = false - os_log("停止连续识别失败: %{public}@", log: log, type: .error, error.localizedDescription) - + os_log("强制停止识别失败", log: log, type: .error) + // 停止音频处理 - stopAudioProcessing() - - // 尝试强制关闭识别器 - recognizer = nil + audioStream?.stopMicrophoneCapture() return false } @@ -239,6 +229,7 @@ public class AzureAsrHelper: NSObject { */ public func dispose() { do { + print("释放所有资源:") // 如果正在进行连续识别,先停止 if _isContinuousRecognitionActive { // 直接停止,不等待结果 @@ -256,11 +247,11 @@ public class AzureAsrHelper: NSObject { // 确保状态被重置 _isContinuousRecognitionActive = false - externalAudioStream = nil + //externalAudioStream = nil } catch { // 确保状态被重置 _isContinuousRecognitionActive = false - externalAudioStream = nil + //externalAudioStream = nil audioConfig = nil recognizer = nil speechConfig = nil @@ -274,19 +265,17 @@ public class AzureAsrHelper: NSObject { */ private func setupRecognizer() -> Bool { do { + print("设置识别器${_isContinuousRecognitionActive}") + // 如果正在进行连续识别,先停止 + if _isContinuousRecognitionActive { + // 直接停止,不等待结果 + try? recognizer?.stopContinuousRecognition() + _isContinuousRecognitionActive = false + } + setupMicrophoneStream() // 清理旧的识别器 recognizer = nil - // 设置音频配置 - switch audioSourceType { - case .microphone: - // 使用默认麦克风输入配置 - audioConfig = try SPXAudioConfiguration() - case .external: - // 使用拉流方式处理外部音频 - setupExternalAudioStream() - } - // 创建识别器 if isAutoDetectLanguage { let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) @@ -310,32 +299,46 @@ public class AzureAsrHelper: NSObject { } } + /** - * 设置外部音频流 - 使用拉流方式 + * 设置麦克风流 - 使用推流方式 */ - private func setupExternalAudioStream() { + private func setupMicrophoneStream() { do { - // 创建外部音频拉流对象 - externalAudioStream = ExternalAudioPullStream() - + + if (audioStream == nil) { + // 创建外部音频拉流对象 + // 明确指定使用外部音频源(推流模式) + audioStream = AudioStream(audioSourceType: audioSourceType) + audioStream?.initAudioRecord() + } + // 创建音频配置 - audioConfig = SPXAudioConfiguration(streamInput: externalAudioStream!.pullStream) + if let pushStream = audioStream?.pushAudioStream { // Safely unwrap optional + audioConfig = SPXAudioConfiguration(streamInput: pushStream) + os_log("设置麦克风流:") + } } catch { - os_log("设置外部音频流失败: %{public}@", log: log, type: .error, error.localizedDescription) - externalAudioStream = nil + os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription) + audioStream = nil } } /** * 设置事件监听器 */ - private func setupEventListeners(callback: ContinuousRecognizeCallback) { - guard let recognizer = recognizer else { return } - + public func setupEventListeners(callback: ContinuousRecognizeCallback) -> Bool{ + print("设置ssssss监听器:${speechConfig}") + + // 重设识别器 + if (!setupRecognizer()) { + return false + } + guard let recognizer = recognizer else { return false} // 识别中事件 recognizer.addRecognizingEventHandler { [weak self] (sender, event) in guard let self = self else { return } - + print("识别中事件:") let result = event.result let detectedLanguage = self.getDetectedLanguage(from: result) // os_log("识别中: %{public}@", log: self.log, type: .info, result.text ?? "") @@ -346,7 +349,7 @@ public class AzureAsrHelper: NSObject { // 识别完成事件 recognizer.addRecognizedEventHandler { [weak self] (sender, event) in guard let self = self else { return } - + print("识别完成事件:") let result = event.result if result.reason == SPXResultReason.recognizedSpeech { let detectedLanguage = self.getDetectedLanguage(from: result) @@ -359,22 +362,23 @@ public class AzureAsrHelper: NSObject { recognizer.addSessionStartedEventHandler { (sender, event) in // 直接在当前线程调用回调 callback.onSessionStarted() + print("会话开始事件:") } // 会话结束事件 recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in guard let self = self else { return } - + print("会话结束事件:") // 直接在当前线程调用回调 - callback.onSessionStopped() - self._isContinuousRecognitionActive = false - self.stopAudioProcessing() + // callback.onSessionStopped() + // self._isContinuousRecognitionActive = false + //self.stopAudioProcessing() } // 取消事件 recognizer.addCanceledEventHandler { [weak self] (sender, event) in guard let self = self else { return } - + print("取消事件:") let errorDetails = event.errorDetails ?? "未知错误" let reason = String(describing: event.reason.rawValue) @@ -382,6 +386,7 @@ public class AzureAsrHelper: NSObject { callback.onCanceled(reason, errorDetails) } + return true } /** @@ -407,32 +412,273 @@ public class AzureAsrHelper: NSObject { } } + /** - * 启动音频处理 + * 停止音频处理 */ - private func startAudioProcessing() { - switch audioSourceType { - case .microphone: - // 拉流模式不需要额外启动,SDK会自动拉取数据 - break - case .external: - // 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData - break - } + private func stopAudioProcessing() { +print("stopAudioProcessing") + // if let stream = externalAudioStream { + // stream.close() + // externalAudioStream = nil + // // os_log("外部音频流已关闭", log: log, type: .info) + // } } - + //录音音频文件 /** - * 停止音频处理 + * 开启录音 */ - private func stopAudioProcessing() { - if let stream = externalAudioStream { - stream.close() - externalAudioStream = nil - // os_log("外部音频流已关闭", log: log, type: .info) + public func enableRecord(filePath: String) { + + + recordfile?.closeFile(isSave: true) + recordfile?.creatingFiles(atPath: filePath) // Fixed method call + + } + + /** + * 移动文件到新路径 + */ + public func moveFile(sourcePath: String, destPath: String)-> Bool { + + + recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call + return true + } + + + /** + * 重命名指定路径的音频文件 + */ + public func renameFile(filePath: String, newName: String)-> Bool { + + + + recordfile?.renameFile(at: filePath, to: newName) // Fixed method call + return true + } + + /** + * 停止录音 + */ + public func pauseRecord() { + + + recordfile?.isPause = true + } + + + /** + * 关闭录音 + */ + public func stopRecord(isSave: Bool) { + + + recordfile?.isPause = false + recordfile?.closeFile(isSave: true) + } + + public class AudioStream: NSObject { + public private(set) var pushAudioStream: SPXPushAudioInputStream? + private let writeQueue = LinkedBlockingQueue() + private var audioEngine: AVAudioEngine? + private var audioFormat: AVAudioFormat? + private let audioSourceType: AudioSourceType // 添加引用 + private var isRunning = false + private var isWriting = false + private let bufferSize: Int = 4096 + private let writeThread = DispatchQueue(label: "audio.stream.writer") + init(audioSourceType: AudioSourceType) { + self.audioSourceType = audioSourceType + } + public func initAudioRecord() { + audioFormat = getOptimalAudioFormat() + pushAudioStream = SPXPushAudioInputStream() + isRunning=true + startWriteThread() + } + /// 获取最佳音频格式 (iOS 通常支持标准采样率) + public func getOptimalAudioFormat() -> AVAudioFormat? { + let sampleRate: Double = 16000 // iOS 通常支持 16kHz + return AVAudioFormat( + commonFormat: .pcmFormatInt16, + sampleRate: sampleRate, + channels: 1, + interleaved: true + ) + } + public func startAudioRecord() { + isWriting=true + print("startAudioRecord=audioSourceType\(audioSourceType)") + switch audioSourceType { + case .microphone: + startMicrophoneCapture() + case .external: + startExternalCapture() + } + } + + public func startWriteThread() { + writeThread.async { [weak self] in + guard let self = self else { return } + + while self.isRunning { + if !self.isWriting { + usleep(10_000) + continue + } + + guard let dataToWrite = self.writeQueue.take() else { continue } + + print("写入数据长度: \(dataToWrite.count)") + self.pushAudioStream?.write(dataToWrite) } } - +} + + /** + * 向音频流写入音频数据 + * 仅当音频源设置为external时有效 + * + * @param data 音频数据字节数组 + */ + public func saveAudioDataTo(data: Data) { + if audioSourceType != .external { + return + } + + // 放入队列,由写线程写入 + writeQueue.put(data) + } + private func startMicrophoneCapture() { + do { + let audioSession = AVAudioSession.sharedInstance() + // 改用 playAndRecord 模式(同时支持播放和录音) + try audioSession.setCategory( + .playAndRecord, + mode: .default, + options: [.allowBluetooth, .defaultToSpeaker] + ) + try audioSession.setActive(true, options: .notifyOthersOnDeactivation) + + audioEngine = AVAudioEngine() + guard let inputNode = audioEngine?.inputNode else { + throw NSError(domain: "AudioSetup", code: 1) + } + + // Use hardware's native format + let hardwareFormat = inputNode.inputFormat(forBus: 0) + + // Create converter to target format + guard let targetFormat = audioFormat, + let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { + throw NSError(domain: "AudioSetup", code: 2) + } + + inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) { + [weak self] buffer, time in + guard let self = self, self.isWriting else { return } + + // Convert to target format + let convertedBuffer = AVAudioPCMBuffer( + pcmFormat: targetFormat, + frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate) +)! +///print("进入 tap 回调,frameLength: \(buffer.frameLength)") + var error: NSError? + let status = converter.convert( + to: convertedBuffer, + error: &error, + withInputFrom: { inNumPackets, outStatus in + outStatus.pointee = .haveData + return buffer + } + ) + //print("转换状态: \(status.rawValue), 错误: \(String(describing: error))") + + if status == .haveData, error == nil { + let data = self.audioBufferToData(convertedBuffer) + // print("准备写入数据,大小: \(data.count)") + + self.writeQueue.put(data) + + } + } + + try audioEngine?.start() + } catch { + print("麦克风启动失败: \(error)") + } +} + + private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { + let frameLength = Int(buffer.frameLength) + let channelCount = 1 + let dataLength = frameLength * channelCount * MemoryLayout.size + + // Handle 16-bit integer format + if let int16Data = buffer.int16ChannelData { + return Data( + bytes: int16Data.pointee, + count: dataLength + ) + } + // Handle float format + else if let floatData = buffer.floatChannelData { + var int16Array = [Int16](repeating: 0, count: frameLength) + let floatBuffer = floatData.pointee + + for i in 0.. Bool { + public func renameFile(at sourcePath: String, to newName: String) -> Bool { let sourceURL = URL(fileURLWithPath: sourcePath) guard FileManager.default.fileExists(atPath: sourcePath) else { @@ -162,7 +162,7 @@ class RecordFile { } // 移动文件 - func moveFile(from sourcePath: String, to destPath: String) -> Bool { + public func moveFile(from sourcePath: String, to destPath: String) -> Bool { let sourceURL = URL(fileURLWithPath: sourcePath) let destURL = URL(fileURLWithPath: destPath) @@ -192,7 +192,7 @@ class RecordFile { } // 关闭文件 - func closeFile(isSave: Bool) -> Bool { + public func closeFile(isSave: Bool) -> Bool { defer { fileHandle = nil currentAudioFile = nil