Browse Source

拆分ios录音功能和微软sdk

newdev_shunjiawei
fdp 1 year ago
parent
commit
b3eed32e11
  1. 28
      lib/modules/meeting/controllers/meeting_record_controller.dart
  2. 23
      lib/modules/translation/controllers/translation_controller.dart
  3. 343
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt
  4. 5
      local_plugins/azure_speech/ios/azure_speech/Package.swift
  5. 644
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift
  6. 54
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift
  7. 0
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/RecordFile.swift
  8. 534
      local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

28
lib/modules/meeting/controllers/meeting_record_controller.dart

@ -78,7 +78,6 @@ class MeetingRecordController extends GetxController
_initializeComponents();
_setupFromArguments();
generateWaveData();
_reinitializeAsrService();
}
@override
@ -87,18 +86,6 @@ class MeetingRecordController extends GetxController
super.onClose();
}
// 重新初始化ASR服务以支持新的语言
Future<void> _reinitializeAsrService() async {
try {
// 重新初始化ASR服务
await _asrService.initialize();
Logger.info('ASR服务重新初始化成功');
} catch (e) {
Logger.error('重新初始化ASR服务失败: ${e.toString()}');
}
}
// 请求录音权限
Future<bool> _requestRecordPermission() async {
try {
@ -164,7 +151,6 @@ class MeetingRecordController extends GetxController
_bleManager.closeCodec();
}
_asrService.stopRecord(true);
_asrService.stopContinuousRecognition();
isRecording.value = false;
}
@ -197,12 +183,6 @@ class MeetingRecordController extends GetxController
/// Toggles visibility of transcribed text content
Future<void> toggleTextDisplay() async {
final recognitionStream = await _asrService.recognizeCallback();
_recognitionSubscription?.cancel();
_recognitionSubscription =
recognitionStream.listen(_handleRecognitionEvent);
await _asrService.startContinuousRecognition(_audioSourceType);
showTextContent.toggle();
}
@ -242,10 +222,6 @@ class MeetingRecordController extends GetxController
/// Starts recording session
Future<void> _startRecording() async {
// Initialize speech recognition
final recognitionStream =
await _asrService.startContinuousRecognition(_audioSourceType);
// Prepare storage directory
final appDir = await appDocDir;
final dir = Directory("${appDir.path}/MeetingAudio/");
@ -256,7 +232,7 @@ class MeetingRecordController extends GetxController
final fullFileName = "${fileName.value}_$formattedTime";
// Start audio recording
await _asrService.enableRecord("${dir.path}/$fullFileName.wav",
await _asrService.enableRecord("${dir.path}$fullFileName.wav",
acceptAudioData: true);
// 监听音频数据流
@ -302,7 +278,6 @@ class MeetingRecordController extends GetxController
/// Pauses current recording session
Future<void> _pauseRecording() async {
await _asrService.stopContinuousRecognition();
await _asrService.pauseRecord();
pauseTimer();
}
@ -475,7 +450,6 @@ class MeetingRecordController extends GetxController
}
}
await _asrService.stopContinuousRecognition();
_resetToDefaultState();
isRecording.value = false;
}

23
lib/modules/translation/controllers/translation_controller.dart

@ -587,6 +587,7 @@ class TranslationController extends GetxController {
scrollController.dispose();
restoreOtherServices();
_asrService.dispose();
_astService.dispose();
super.onClose();
}
@ -753,7 +754,9 @@ class TranslationController extends GetxController {
try {
await stopRecording();
await _asrService.stopContinuousRecognition();
await _astService.stopContinuousTranslation();
if (currentMode.value == "call") {
await _astService.stopContinuousTranslation();
}
// 停止ASR活跃时长计时
_stopAsrActiveTracking();
@ -1214,23 +1217,19 @@ class TranslationController extends GetxController {
await _asrService.initialize(
supportedLanguages: asrSupportedLanguages,
);
// Future<bool> initialize({
// required String subscriptionKey,
// required String region,
// required List<String> supportedLanguages,
// required String audioSourceType,
// required String translationAccessKey,
// required String translationSecretKey,
// String translationRegion = 'cn-north-1',
// });
// 重新初始化ASR服务以支持通话音频源
await _astService.initialize(supportedLanguages: asrSupportedLanguages);
// 获取识别事件流 - 这启动了异步识别过程
var recognitionStream = await _asrService.recognizeCallback();
// 监听识别结果
_recognitionSubscription?.cancel();
_recognitionSubscription =
recognitionStream.listen(_handleRecognitionEvent);
//判断是否进入通话
if (currentMode.value == "call") {
// 重新初始化ASR服务以支持通话音频源
_initializeCallModeTranslationService();
}
Logger.info('ASR服务重新初始化成功');
} catch (e) {
Logger.error('重新初始化ASR服务失败: ${e.toString()}');

343
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrToAsr.kt

@ -735,6 +735,107 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) {
eventCallback?.onStateChanged("AudioPlayback", enabled)
}
/**
* 动态修改语音识别语言
* @param newLanguage 新的识别语言代码,如 "zh-CN", "en-US", "ja-JP" 等
* @param restartRecognition 是否重启识别服务,默认为true
*/
fun changeRecognitionLanguage(newLanguage: String, restartRecognition: Boolean = true) {
try {
val wasRecognizing = serviceState.isRecognizing.get()
// 如果正在识别,先停止
if (wasRecognizing && restartRecognition) {
stopContinuousTranslation()
}
// 更新配置
serviceConfig.sourceLanguage = newLanguage
speechConfig?.speechRecognitionLanguage = newLanguage
// 重新设置识别器
setupSpeechRecognizer()
Log.d(TAG, "识别语言已更改为: $newLanguage")
eventCallback?.onStateChanged("LanguageChanged", true)
// 如果之前在识别且需要重启,则重新开始
if (wasRecognizing && restartRecognition) {
startContinuousTranslation()
}
} catch (e: Exception) {
Log.e(TAG, "更改识别语言失败", e)
eventCallback?.onError("LanguageChange", "更改识别语言失败: ${e.message}")
}
}
/**
* 批量设置多语言识别
* @param languages 支持的语言列表
* @param enableAutoDetection 是否启用自动语言检测
*/
fun setupMultiLanguageRecognition(languages: List<String>, enableAutoDetection: Boolean = true) {
try {
serviceConfig.enableAutoLanguageDetection = enableAutoDetection
if (enableAutoDetection && languages.isNotEmpty()) {
// 设置自动语言检测的候选语言
speechConfig?.setProperty("SpeechServiceConnection_LanguageIdMode", "Continuous")
// 构建语言候选列表
val languageList = languages.joinToString(",")
speechConfig?.setProperty("SpeechServiceConnection_ContinuousLanguageIdPriority", languageList)
Log.d(TAG, "多语言识别已设置,支持语言: $languageList")
}
// 重新设置识别器
setupSpeechRecognizer()
} catch (e: Exception) {
Log.e(TAG, "设置多语言识别失败", e)
eventCallback?.onError("MultiLanguageSetup", "设置多语言识别失败: ${e.message}")
}
}
/**
* 获取当前支持的语言列表
*/
fun getSupportedLanguages(): List<String> {
return listOf(
"zh-CN", "zh-TW", "en-US", "en-GB", "ja-JP", "ko-KR",
"fr-FR", "de-DE", "es-ES", "ru-RU", "ar-SA", "pt-BR",
"it-IT", "th-TH", "vi-VN", "hi-IN"
)
}
/**
* 设置识别参数
* @param endSilenceTimeout 结束静音超时时间(毫秒)
* @param segmentationTimeout 分段静音超时时间(毫秒)
* @param initialSilenceTimeout 初始静音超时时间(毫秒)
*/
fun setRecognitionParameters(
endSilenceTimeout: Int = 300,
segmentationTimeout: Int = 300,
initialSilenceTimeout: Int = 200
) {
try {
speechConfig?.apply {
setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", endSilenceTimeout.toString())
setProperty("Speech_SegmentationSilenceTimeoutMs", segmentationTimeout.toString())
setProperty("SpeechServiceConnection_InitialSilenceTimeoutMs", initialSilenceTimeout.toString())
}
Log.d(TAG, "识别参数已更新: 结束静音=${endSilenceTimeout}ms, 分段静音=${segmentationTimeout}ms, 初始静音=${initialSilenceTimeout}ms")
} catch (e: Exception) {
Log.e(TAG, "设置识别参数失败", e)
eventCallback?.onError("ParameterSetup", "设置识别参数失败: ${e.message}")
}
}
/**
* 获取当前音频播放状态
* @return true表示启用音频播放,false表示禁用
@ -743,6 +844,54 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) {
return serviceConfig.enableAudioPlayback
}
/**
* 暂停服务(保留资源)
*/
fun pause() {
try {
if (serviceState.isRecognizing.get()) {
recognizer?.stopContinuousRecognitionAsync()
}
audioProcessor?.stopRecording()
Log.d(TAG, "服务已暂停")
} catch (e: Exception) {
Log.e(TAG, "暂停服务失败", e)
}
}
/**
* 恢复服务
*/
fun resume() {
try {
if (serviceState.isInitialized.get() && serviceConfig.enableContinuousRecognition) {
startContinuousTranslation()
Log.d(TAG, "服务已恢复")
}
} catch (e: Exception) {
Log.e(TAG, "恢复服务失败", e)
}
}
/**
* 检查服务健康状态
*/
fun checkServiceHealth(): Map<String, Any> {
return mapOf(
"initialized" to serviceState.isInitialized.get(),
"recognizing" to serviceState.isRecognizing.get(),
"translating" to serviceState.isTranslating.get(),
"synthesizing" to serviceState.isSynthesizing.get(),
"speechConfig" to (speechConfig != null),
"recognizer" to (recognizer != null),
"synthesizer" to (synthesizer != null),
"audioProcessor" to (audioProcessor != null),
"translationService" to (translationService != null),
"currentLanguage" to serviceConfig.sourceLanguage,
"targetLanguage" to serviceConfig.targetLanguage
)
}
/**
* 单独的语音合成方法,只合成不播放,返回音频数据
* @param text 要合成的文本
@ -801,37 +950,125 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) {
}
/**
* 释放资源
* 完善的资源释放方法
*/
fun dispose() {
try {
// 停止所有活动
launch {
stopContinuousTranslation()
Log.d(TAG, "开始释放服务资源...")
// 1. 设置释放标志,防止新的操作
serviceState.isInitialized.set(false)
// 2. 停止所有活动(使用runBlocking确保同步完成)
runBlocking {
try {
withTimeout(5000) { // 5秒超时
stopContinuousTranslation()
}
} catch (e: TimeoutCancellationException) {
Log.w(TAG, "停止识别超时,强制继续释放资源")
}
}
// 释放Azure资源
recognizer?.close()
synthesizer?.close()
speechConfig?.close()
// 释放音频处理器
audioProcessor?.dispose()
// 释放翻译服务
translationService?.dispose()
// 取消协程
job.cancel()
// 清理缓存
// 3. 等待当前处理完成(带超时)
val maxWaitTime = 3000L // 3秒
val startTime = System.currentTimeMillis()
while ((serviceState.isTranslating.get() || serviceState.isSynthesizing.get() || serviceState.isRecognizing.get())
&& (System.currentTimeMillis() - startTime) < maxWaitTime) {
Thread.sleep(100)
}
// 4. 强制停止识别器
try {
recognizer?.stopContinuousRecognitionAsync()?.get(2, TimeUnit.SECONDS)
} catch (e: Exception) {
Log.w(TAG, "停止识别器超时,强制关闭")
}
// 5. 释放Azure资源(按顺序释放)
try {
recognizer?.close()
recognizer = null
Log.d(TAG, "语音识别器已释放")
} catch (e: Exception) {
Log.e(TAG, "释放识别器失败", e)
}
try {
synthesizer?.close()
synthesizer = null
Log.d(TAG, "语音合成器已释放")
} catch (e: Exception) {
Log.e(TAG, "释放合成器失败", e)
}
try {
audioConfig?.close()
audioConfig = null
Log.d(TAG, "音频配置已释放")
} catch (e: Exception) {
Log.e(TAG, "释放音频配置失败", e)
}
try {
speechConfig?.close()
speechConfig = null
Log.d(TAG, "语音配置已释放")
} catch (e: Exception) {
Log.e(TAG, "释放语音配置失败", e)
}
// 6. 释放音频处理器
try {
audioProcessor?.dispose()
audioProcessor = null
Log.d(TAG, "音频处理器已释放")
} catch (e: Exception) {
Log.e(TAG, "释放音频处理器失败", e)
}
// 7. 释放录音文件
try {
recordfile1?.closeFile(true)
recordfile1 = null
Log.d(TAG, "录音文件已释放")
} catch (e: Exception) {
Log.e(TAG, "释放录音文件失败", e)
}
// 8. 释放翻译服务
try {
translationService?.dispose()
translationService = null
Log.d(TAG, "翻译服务已释放")
} catch (e: Exception) {
Log.e(TAG, "释放翻译服务失败", e)
}
// 9. 取消协程(使用cancelAndJoin确保完全停止)
try {
runBlocking {
job.cancelAndJoin()
}
Log.d(TAG, "协程已取消")
} catch (e: Exception) {
Log.e(TAG, "取消协程失败", e)
}
// 10. 清理缓存和回调
voiceCache.clear()
serviceState.isInitialized.set(false)
Log.d(TAG, "服务资源已释放")
eventCallback = null
filePath = null
// 11. 重置状态
serviceState.isRecognizing.set(false)
serviceState.isSynthesizing.set(false)
serviceState.isTranslating.set(false)
Log.d(TAG, "服务资源释放完成")
} catch (e: Exception) {
Log.e(TAG, "释放资源失败", e)
Log.e(TAG, "释放资源过程中发生异常", e)
}
}
@ -895,11 +1132,59 @@ fun setAudioOutputDevice(device: com.deep_voice.speech.tts.AudioOutputDevice) {
}
}
/**
* 改进的资源释放方法
*/
fun dispose() {
isRunning.set(false)
processingThread?.interrupt()
pushAudioStream?.close()
audioQueue.clear()
try {
Log.d(TAG, "开始释放音频处理器资源...")
// 1. 停止运行标志
isRunning.set(false)
// 2. 中断处理线程
processingThread?.let { thread ->
if (thread.isAlive) {
thread.interrupt()
try {
// 等待线程结束,最多等待2秒
thread.join(2000)
if (thread.isAlive) {
Log.w(TAG, "音频处理线程未能正常结束")
} else {
Log.d(TAG, "音频处理线程已结束")
}
} catch (e: InterruptedException) {
Log.w(TAG, "等待音频处理线程结束时被中断")
Thread.currentThread().interrupt()
}
}
}
processingThread = null
// 3. 关闭音频流
try {
pushAudioStream?.close()
pushAudioStream = null
Log.d(TAG, "音频流已关闭")
} catch (e: Exception) {
Log.e(TAG, "关闭音频流失败", e)
}
// 4. 清理音频队列
try {
val remainingData = audioQueue.size
audioQueue.clear()
Log.d(TAG, "音频队列已清理,丢弃 $remainingData 个数据包")
} catch (e: Exception) {
Log.e(TAG, "清理音频队列失败", e)
}
Log.d(TAG, "音频处理器资源释放完成")
} catch (e: Exception) {
Log.e(TAG, "释放音频处理器资源时发生异常", e)
}
}
}
}

5
local_plugins/azure_speech/ios/azure_speech/Package.swift

@ -18,7 +18,8 @@ let package = Package(
.product(name: "speech", package: "speech"),
"MicrosoftCognitiveServicesSpeech"
],
path: "Sources/azure_speech"
path: "Sources", // 修改为包含整个 Sources 目录
sources: ["azure_speech/", "tools/"] // 明确指定包含的子目录
),
// 使用binaryTarget引用Azure Speech SDK的.xcframework
.binaryTarget(
@ -26,4 +27,4 @@ let package = Package(
path: "MicrosoftCognitiveServicesSpeech.xcframework"
)
]
)
)

644
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

@ -1,7 +1,6 @@
import Foundation
import AVFoundation
import MicrosoftCognitiveServicesSpeech
// 自定义语音处理组件,提供音频流处理等功能
import speech
import os.log
@ -32,7 +31,6 @@ public class AzureAsrHelper: NSObject {
private var subscriptionKey = ""
private var region = ""
// 音频源配置
public enum AudioSourceType {
/** 使用设备麦克风 */
@ -48,9 +46,8 @@ public class AzureAsrHelper: NSObject {
// private var externalAudioStream: ExternalAudioPullStream?
// 音频处理
public var audioStream: AudioStream?
//录音文件
public var recordfile: RecordFile?
public var audioStream: SimpleAudioReceiver?
/**
* 初始化Azure语音服务
*
@ -110,11 +107,6 @@ public class AzureAsrHelper: NSObject {
os_log("固定语言模式,当前语言: %{public}@", log: log, type: .info, currentLanguage)
}
// 录音文件类
recordfile = RecordFile()
// 预初始化音频组件
preInitializeAudioComponents()
@ -131,18 +123,16 @@ public class AzureAsrHelper: NSObject {
*/
private func preInitializeAudioComponents() {
if audioStream == nil {
audioStream = AudioStream(parentHelper: self)
audioStream = SimpleAudioReceiver(parentHelper: self)
audioStream?.initAudioRecord()
audioStream?.onAudioData = { [weak self] data in
self?.recordfile?.saveAudioDataToWav(data)
}
}
// 预创建音频配置
if let pushStream = audioStream?.pushAudioStream {
audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("预初始化音频组件完成", log: log, type: .info)
}
}
}
@ -154,7 +144,8 @@ public class AzureAsrHelper: NSObject {
* @return 是否成功开始识别
*/
public func startContinuousRecognition(
audioSourceType: AudioSourceType = .microphone
audioSourceType: AudioSourceType = .microphone,
audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = nil
) -> Bool {
guard speechConfig != nil else {
print("语音服务未初始化")
@ -180,7 +171,7 @@ public class AzureAsrHelper: NSObject {
return false
}
// 启动音频处理
audioStream?.startAudioInput(audioSourceType: audioSourceType)
audioStream?.startAudioRecord(audioSourceType: audioSourceType == .microphone ? .microphone : .external, audioDataCallback: audioDataCallback)
// 开始连续识别
try recognizer?.startContinuousRecognition()
@ -190,7 +181,7 @@ public class AzureAsrHelper: NSObject {
return true
} catch {
audioStream?.stopAudioCapture()
audioStream?.stopMicrophoneCapture()
_isContinuousRecognitionActive = false
//callback.onError("启动连续识别失败: \(error.localizedDescription)")
return false
@ -200,8 +191,12 @@ public class AzureAsrHelper: NSObject {
/**
* 停止连续语音识别
*
* @return 是否成功停止
*
* @return 停止操作是否成功
*/
/**
* 停止连续语音识别
* @return 停止操作是否成功
*/
public func stopContinuousRecognition() -> Bool {
guard speechConfig != nil else {
@ -224,11 +219,10 @@ public class AzureAsrHelper: NSObject {
//pushAudioData(data: Data())
}
audioStream?.stopAudioCapture()
audioStream?.stopMicrophoneCapture()
// 停止连续识别
try recognizer.stopContinuousRecognition()
// 会话结束事件会设置_isContinuousRecognitionActive = false
print("stopContinuousRecognition")
return true
} catch {
// 强制重置状态
@ -236,7 +230,15 @@ public class AzureAsrHelper: NSObject {
os_log("强制停止识别失败", log: log, type: .error)
// 停止音频处理
audioStream?.stopAudioCapture()
audioStream?.stopMicrophoneCapture()
// 【修复】在 catch 块中安全地停止识别器,避免再次抛出异常
do {
try recognizer.stopContinuousRecognition()
} catch {
// 如果强制停止也失败,记录错误但不再抛出异常
os_log("强制停止识别器失败: %@", log: log, type: .error, error.localizedDescription)
}
return false
}
@ -264,8 +266,6 @@ public class AzureAsrHelper: NSObject {
// 停止音频处理
audioStream?.releaseAudioResources()
// 停止录音
recordfile?.closeFile(isSave: true)
// 释放资源
recognizer = nil
speechConfig = nil
@ -295,12 +295,9 @@ public class AzureAsrHelper: NSObject {
do {
print("设置音频配置 - 采样率: \(sampleRate), 声道: \(channels)")
// 设置音频流的配置
// 设置音频接收器的配置
audioStream?.setAudioConfig(sampleRate: sampleRate, channels: channels)
// 设置录音文件的配置
recordfile?.setAudioConfig(sampleRate: sampleRate, channels: channels)
print("音频配置设置成功")
return true
} catch {
@ -313,10 +310,6 @@ public class AzureAsrHelper: NSObject {
/**
* 设置识别器
*/
/**
* 设置识别器
* 优化:避免重复创建音频流,复用已初始化的组件
*/
private func setupRecognizer() -> Bool {
do {
print("设置识别器\(_isContinuousRecognitionActive)")
@ -332,6 +325,17 @@ public class AzureAsrHelper: NSObject {
setupMicrophoneStream()
}
// 安全检查:确保 audioConfig 不为 nil
guard let audioConfig = audioConfig else {
os_log("音频配置未初始化,无法创建识别器", log: log, type: .error)
return false
}
guard let speechConfig = speechConfig else {
os_log("语音配置未初始化,无法创建识别器", log: log, type: .error)
return false
}
// 清理旧的识别器
recognizer = nil
@ -342,22 +346,21 @@ public class AzureAsrHelper: NSObject {
print("创建自动语言检测配置失败,使用默认语言配置")
// 如果创建失败,使用默认的单语言配置
recognizer = try SPXSpeechRecognizer(
speechConfiguration: speechConfig!,
audioConfiguration: audioConfig!
speechConfiguration: speechConfig,
audioConfiguration: audioConfig
)
return true
}
print("autoDetectConfig: \(autoDetectSourceLanguageConfig)")
print("创建自动语言识别器")
recognizer = try SPXSpeechRecognizer(
speechConfiguration: speechConfig!,
speechConfiguration: speechConfig,
autoDetectSourceLanguageConfiguration: autoDetectSourceLanguageConfig,
audioConfiguration: audioConfig!
audioConfiguration: audioConfig
)
} else {
recognizer = try SPXSpeechRecognizer(
speechConfiguration: speechConfig!,
audioConfiguration: audioConfig!
speechConfiguration: speechConfig,
audioConfiguration: audioConfig
)
}
@ -374,27 +377,20 @@ public class AzureAsrHelper: NSObject {
* 设置麦克风流 - 使用推流方式
* 优化:避免重复初始化已存在的音频流
*/
/**
* 设置麦克风音频流
*/
private func setupMicrophoneStream() {
do {
// 【优化】检查是否已经初始化,避免重复创建
if audioStream == nil {
audioStream = AudioStream(parentHelper: self)
audioStream?.initAudioRecord()
audioStream?.onAudioData = { [weak self] data in
self?.recordfile?.saveAudioDataToWav(data)
}
}
// 【优化】检查音频配置是否已存在
if audioConfig == nil, let pushStream = audioStream?.pushAudioStream {
audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("设置麦克风流完成", log: log, type: .info)
}
} catch {
os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription)
audioStream = nil
}
audioStream = SimpleAudioReceiver(parentHelper: self)
// 【优化】检查音频配置是否已存在
if audioConfig == nil, let pushStream = audioStream?.pushAudioStream {
audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("设置麦克风流完成", log: log, type: .info)
}
}
/**
* 从语音识别结果中获取检测到的语言
* @param result 语音识别结果
@ -409,26 +405,12 @@ public class AzureAsrHelper: NSObject {
// 备用方案:从属性中获取语言
if let properties = result.properties,
let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage"),
let language = properties.getPropertyByName("speechServiceResponse_RecognitionLanguage"),
!language.isEmpty {
print("从属性中检测到语言: \(language)")
return language
}
// 最后的备用方案:从JSON响应中解析
if let properties = result.properties,
let jsonString = properties.getPropertyByName("SpeechServiceResponse_Json"),
let jsonData = jsonString.data(using: .utf8) {
do {
if let json = try JSONSerialization.jsonObject(with: jsonData, options: []) as? [String: Any],
let language = json["Language"] as? String {
print("从JSON中检测到语言: \(language)")
return language
}
} catch {
print("JSON解析失败: \(error)")
}
}
// 使用正确的SPXAutoDetectSourceLanguageResult初始化方法
do {
let langResult = try SPXAutoDetectSourceLanguageResult(result)
@ -499,11 +481,6 @@ public class AzureAsrHelper: NSObject {
let text = result.text
print("正在识别事件=检测到语言: \(detectedLanguage), 识别中: \(text)")
// // 检测文本语言,避免重复调用
// let detectedLanguage = self.detectTextLanguage(result.text ?? "") ?? "zh-CN"
// // 输出检测到的语言信息
// print("检测到的语言: \(detectedLanguage)")
DispatchQueue.main.async {
callback.onRecognizing(result.text ?? "", detectedLanguage)
@ -522,11 +499,6 @@ public class AzureAsrHelper: NSObject {
let text = result.text
print("识别完成事件=检测到语言: \(detectedLanguage), 识别中: \(text)")
// 检测文本语言,避免重复调用
// let detectedLanguage = self.detectTextLanguage(result.text ?? "") ?? "zh-CN"
// // 输出检测到的语言信息
// print("检测到的语言: \(detectedLanguage)")
// 在主线程回调结果
DispatchQueue.main.async {
@ -573,7 +545,7 @@ public class AzureAsrHelper: NSObject {
do {
print("禁用蓝牙音频功能,切换回正常音频模式")
audioStream?.setAudioOutputRoute(.speaker)
// audioStream?.disableBluetoothAudio()
} catch {
print("disableBluetoothAudio")
}
@ -585,7 +557,7 @@ public class AzureAsrHelper: NSObject {
do {
print("恢复原始音频设备状态(通常是重新启用蓝牙)")
audioStream?.setAudioOutputRoute(.receiver)
//audioStream?.restoreOriginalAudioState()
} catch {
print("restoreOriginalAudioState")
}
@ -595,13 +567,31 @@ public class AzureAsrHelper: NSObject {
/**
* 开启录音
* @param filePath 录音文件路径
* @param audioDataCallback 音频数据回调接口
*/
public func enableRecord(filePath: String) {
public func enableRecord(filePath: String, audioDataCallback: SimpleAudioReceiver.AudioDataCallback? = nil) {
os_log("开启录音: %{public}@", log: log, type: .info, filePath)
if audioStream == nil {
// 创建外部音频拉流对象
audioStream = SimpleAudioReceiver(parentHelper: self)
audioStream?.initAudioRecord()
}
recordfile?.closeFile(isSave: true)
recordfile?.creatingFiles(atPath: filePath) // Fixed method call
if audioStream?.recordfile == nil {
audioStream?.recordfile = RecordFile()
}
// 修复:移除多余的 audioDataCallback 参数
audioStream?.startAudioRecord(
audioSourceType: audioSourceType == .microphone ? .microphone : .external,
audioDataCallback: audioDataCallback
)
audioStream?.recordfile?.closeFile(isSave: true)
audioStream?.recordfile?.creatingFiles(atPath: filePath)
}
/**
@ -610,7 +600,7 @@ public class AzureAsrHelper: NSObject {
public func moveFile(sourcePath: String, destPath: String)-> Bool {
recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call
audioStream?.recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call
return true
}
@ -622,461 +612,47 @@ public class AzureAsrHelper: NSObject {
recordfile?.renameFile(at: filePath, to: newName) // Fixed method call
audioStream?.recordfile?.renameFile(at: filePath, to: newName) // Fixed method call
return true
}
/**
* 停止录音
/**
* 暂停录音
*/
public func pauseRecord() {
recordfile?.isPause = true
os_log("暂停录音", log: log, type: .info)
guard let audioStream = audioStream, audioStream.recordfile != nil else {
return
}
audioStream.stopMicrophoneCapture()
}
/**
* 关闭录音
* 继续录音
*/
public func stopRecord(isSave: Bool) {
recordfile?.isPause = false
recordfile?.closeFile(isSave: true)
}
// MARK: - 音频流类
public class AudioStream: NSObject {
public private(set) var pushAudioStream: SPXPushAudioInputStream?
private let writeQueue = LinkedBlockingQueue<Data>()
private var audioEngine: AVAudioEngine?
private var audioFormat: AVAudioFormat?
private let audioSession = AVAudioSession.sharedInstance()
private var audioSourceType = AudioSourceType.microphone
private var isRunning = false
public var isWriting = false
private let bufferSize: Int = 4096
private var writeThread: DispatchQueue?
private var currentRoute: AudioOutputRoute?
public var onAudioData: ((Data) -> Void)?
// 添加对外部类的弱引用
private weak var parentHelper: AzureAsrHelper?
// 添加初始化方法,接收外部类引用
init(parentHelper: AzureAsrHelper) {
self.parentHelper = parentHelper
super.init()
}
/**
* 初始化
*/
/**
* 初始化音频录制组件
* 包括音频格式、推流、音频引擎等核心组件的初始化
*/
public func initAudioRecord() {
print("初始化了")
audioFormat = getOptimalAudioFormat()
pushAudioStream = SPXPushAudioInputStream()
// 初始化音频引擎
audioEngine = AVAudioEngine()
isRunning = true
// 创建新的写线程
writeThread = DispatchQueue(label: "audio.stream.writer")
startWriteThread()
//self.setAudioOutputRoute(.receiver)
}
/// 获取最佳音频格式 (iOS 通常支持标准采样率)
public func getOptimalAudioFormat() -> AVAudioFormat? {
let sampleRate: Double = 16000 // iOS 通常支持 16kHz
return AVAudioFormat(
commonFormat: .pcmFormatInt16,
sampleRate: sampleRate,
channels: 1,
interleaved: true
)
}
/**
* 设置音频配置
* @param sampleRate 采样率,默认16000
* @param channels 声道数,默认1
*/
public func setAudioConfig(sampleRate: Int, channels: Int) {
print("AudioStream设置音频配置 - 采样率: \(sampleRate), 声道: \(channels)")
// 更新音频格式
audioFormat = AVAudioFormat(
commonFormat: .pcmFormatInt16,
sampleRate: Double(sampleRate),
channels: AVAudioChannelCount(channels),
interleaved: true
)
// 如果已经有推流,重新创建
if pushAudioStream != nil {
let audioStreamFormat = SPXAudioStreamFormat()
audioStreamFormat?.initUsingPCM(withSampleRate: UInt(sampleRate), bitsPerSample: 16, channels: UInt(channels))
pushAudioStream = SPXPushAudioInputStream(audioFormat: audioStreamFormat)
}
print("AudioStream音频配置设置完成")
}
/**
* 开始音频输入
*/
public func startAudioInput(audioSourceType: AudioSourceType = .microphone) {
isWriting=true
self.audioSourceType = audioSourceType;
print("startAudioInput=audioSourceType\(audioSourceType)")
switch audioSourceType {
case .microphone:
runMicrophoneCapture()
case .external:
runExternalCapture()
}
public func resumeRecord() {
os_log("继续录音", log: log, type: .info)
guard let audioStream = audioStream, audioStream.recordfile != nil else {
return
}
/**
* 开启音频写入线程
*/
public func startWriteThread() {
writeThread?.async { [weak self] in
guard let self = self else { return }
while self.isRunning {
// print("是否写入: \(self.isWriting)")
if !self.isWriting {
usleep(10_000)
continue
}
guard let dataToWrite = self.writeQueue.take() else { continue }
// print("写入数据长度: \(dataToWrite.count)")
do {
try self.pushAudioStream?.write(dataToWrite)
self.onAudioData?(dataToWrite)
} catch {
print("推送音频数据失败: \(error.localizedDescription)")
}
}
}
}
/**
* 向音频流写入音频数据
* 仅当音频源设置为external时有效
*
* @param data 音频数据字节数组
*/
public func saveAudioDataTo(data: Data) {
// 通过父类引用调用方法
if audioSourceType != .external || !(parentHelper?.isContinuousRecognitionActive() ?? false) {
return
}
// print("外部data=\(data)")
// 放入队列,由写线程写入
writeQueue.put(data)
}
// 私有方法:启动麦克风捕获
/**
* 启动麦克风捕获
* 优化:复用已初始化的音频引擎,减少启动延迟
*/
private func runMicrophoneCapture() {
do {
// 【优化】检查音频引擎是否已初始化,避免重复创建
if audioEngine == nil {
audioEngine = AVAudioEngine()
}
// 如果音频引擎正在运行,先停止
if audioEngine?.isRunning == true {
audioEngine?.stop()
}
// 获取音频输入节点(麦克风)
guard let inputNode = audioEngine?.inputNode else {
throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"])
}
// 移除之前的音频处理块,避免重复添加
inputNode.removeTap(onBus: 0)
// 获取硬件支持的原始音频格式
let hardwareFormat = inputNode.inputFormat(forBus: 0)
// iOS 13+ 启用语音处理
if #available(iOS 13.0, *) {
try inputNode.setVoiceProcessingEnabled(true)
}
// 检查目标音频格式和转换器是否可用
guard let targetFormat = audioFormat, // 外部定义的期望音频格式
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2)
}
// 在输入节点上安装录音回调
inputNode.installTap(onBus: 0,
bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小
format: hardwareFormat) { // 使用原始硬件格式
[weak self] buffer, time in // 弱引用避免循环引用
// 确保实例存在且正在写入状态
guard let self = self, self.isWriting else { return }
// 创建目标格式的音频缓冲区
let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
// 计算转换后的帧容量(考虑采样率差异)
frameCapacity: AVAudioFrameCount(
targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate
)
)!
var error: NSError?
// 执行音频格式转换
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
outStatus.pointee = .haveData // 标记有数据可用
return buffer // 返回原始音频数据
}
)
// 转换成功且无错误
if status == .haveData, error == nil {
// 将音频缓冲区转换为二进制数据
let data = self.audioBufferToData(convertedBuffer)
// 将数据放入写入队列(后续处理)
self.writeQueue.put(data)
}
}
// 激活音频会话(允许录音)
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
// 启动音频引擎
try audioEngine?.start()
} catch {
print("麦克风启动失败: \(error)")
audioStream.resumeRecord()
}
}
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
let frameLength = Int(buffer.frameLength)
let channelCount = 1
let dataLength = frameLength * channelCount * MemoryLayout<Int16>.size
// Handle 16-bit integer format
if let int16Data = buffer.int16ChannelData {
return Data(
bytes: int16Data.pointee,
count: dataLength
)
}
// Handle float format
else if let floatData = buffer.floatChannelData {
var int16Array = [Int16](repeating: 0, count: frameLength)
let floatBuffer = floatData.pointee
for i in 0..<frameLength {
let sample = floatBuffer[i]
let clamped = max(-1.0, min(sample, 1.0))
let scaled = clamped * Float(Int16.max)
int16Array[i] = Int16(scaled)
}
return Data(
bytes: int16Array,
count: dataLength
)
}
return Data() // Fallback for unsupported formats
}
private func runExternalCapture() {
if audioSourceType == .external {
//pushAudioData(data: Data())
audioEngine?.pause()
//audioEngine?.inputNode.removeTap(onBus: 0)
}
}
public func stopAudioCapture() {
guard isWriting==true else { return }
print("stopContinuousRecognition")
isWriting=false
audioEngine?.pause()
audioEngine?.inputNode.removeTap(onBus: 0)
writeThread?.async {
self.writeQueue.close()
}
}
public func releaseAudioResources() {
print("释放了")
stopAudioCapture()
audioEngine?.stop()
writeThread = nil
audioEngine?.inputNode.removeTap(onBus: 0)
isRunning=false
audioEngine = nil
}
// 音频路由管理
public enum AudioOutputRoute {
case speaker
case receiver
case bluetooth
}
public func setAudioOutputRoute(_ route: AudioOutputRoute) {
do {
//
if self.isWriting {
print("当前正在识别")
try audioEngine?.pause()
// 先停用以避免冲突
try audioSession.setActive(false)
}
print("setAudioOutputRoutecurrent,route=\(route)")
switch route {
case .speaker:
print("setAudioOutputRoutecurrent,speaker")
// 使用扬声器时必须用videoChat模式
try audioSession.setCategory(
.playAndRecord,
mode: .videoChat,
options: [.defaultToSpeaker]
)
try audioSession.overrideOutputAudioPort(.speaker)
case .receiver:
// 听筒模式使用voiceChat节省资源
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth]
)
try audioSession.overrideOutputAudioPort(.none)
case .bluetooth:
// 完整蓝牙设备支持
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth, .allowBluetoothA2DP]
)
try audioSession.overrideOutputAudioPort(.none)
// 不需要override,系统自动路由
}
self.currentRoute = route
if self.isWriting {
// 重新激活
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
try audioEngine?.start()
writeThread?.async {
self.writeQueue.close()
}
}
} catch {
print("路由切换失败: \(error)")
}
}
}
// MARK: - iOS版LinkedBlockingQueue实现
/**
* iOS版LinkedBlockingQueue实现
* 与Android LinkedBlockingQueue保持一致的API和行为
* 关闭录音
*/
private class LinkedBlockingQueue<T> {
private var queue: [T] = []
private var isClosed = false
// 使用DispatchSemaphore实现阻塞行为
private let availableItems: DispatchSemaphore
private let queueLock = NSLock()
init() {
self.availableItems = DispatchSemaphore(value: 0)
}
/**
* 阻塞式获取元素(等价于Android的take())
* @return 队列中的元素,如果队列已关闭则返回nil
*/
func take() -> T? {
// 等待可用元素(阻塞直到有元素或队列关闭)
availableItems.wait()
queueLock.lock()
defer { queueLock.unlock() }
// 检查队列是否已关闭
if isClosed && queue.isEmpty {
return nil
}
// 获取第一个元素
guard !queue.isEmpty else {
return nil
}
return queue.removeFirst()
}
/**
* 添加元素(等价于Android的put())
* @param item 要添加的元素
*/
func put(_ item: T) {
queueLock.lock()
defer { queueLock.unlock() }
// 检查队列是否已关闭
if isClosed {
return
}
// 添加元素(无容量限制,与Android一致)
queue.append(item)
// 通知有新元素可用
availableItems.signal()
}
/**
* 关闭队列
* 关闭后不能再添加新元素,但可以继续取出已有元素
*/
func close() {
queueLock.lock()
defer { queueLock.unlock() }
isClosed = true
// 唤醒所有等待的take()操作
for _ in 0..<100 { // 假设最多100个等待者
availableItems.signal()
}
public func stopRecord(isSave: Bool) {
guard let audioStream = audioStream, audioStream.recordfile != nil else {
return
}
audioStream.stopMicrophoneCapture()
audioStream.recordfile?.isPause = false //
audioStream.recordfile?.closeFile(isSave: true) //
}
@ -1177,4 +753,4 @@ private func runMicrophoneCapture() {
// "languageName": languageName,
// "originalCode": topResult.language
// ]
// }
// }

54
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift

@ -258,16 +258,37 @@ import os.log
case "enableRecord":
guard let args = call.arguments as? [String: Any],
let filePath = args["filePath"] as? String else {
result(FlutterError(code: "INVALID_ARGUMENTS", message: "filePath 参数不能为空", details: nil))
result(FlutterError(code: "INVALID_ARGUMENTS", message: "文件路径不能为空", details: nil))
return
}
// 检查是否需要接受音频数据
let acceptAudioData = args["acceptAudioData"] as? Bool ?? false
do {
print("音频文件名称为: \(filePath)")
try azureAsrHelper.enableRecord(filePath: filePath)
if acceptAudioData {
// 创建音频数据回调实现类
let audioCallback = AudioDataCallbackImpl { [weak self] audioData in
// 构建音频数据事件映射
let audioEvent: [String: Any] = [
"type": "audioData",
"data": audioData,
"timestamp": Date().timeIntervalSince1970 * 1000,
"size": audioData.count
]
// 发送音频数据事件到 Flutter 层
self?.sendAsrEvent(audioEvent)
}
try azureAsrHelper.enableRecord(filePath: filePath, audioDataCallback: audioCallback)
} else {
try azureAsrHelper.enableRecord(filePath: filePath)
}
result(true)
} catch {
result(FlutterError(code: "ENABLERECORD_ERROR", message: error.localizedDescription, details: nil))
}
case "pauseRecord":
azureAsrHelper.pauseRecord()
@ -574,3 +595,30 @@ extension AzureSpeechPlugin: AudioDataListener {
sendTtsEvent(eventMap)
}
}
// MARK: - 音频数据回调实现
/**
* 音频数据回调实现类
* 实现 SimpleAudioReceiver.AudioDataCallback 协议
*/
private class AudioDataCallbackImpl: SimpleAudioReceiver.AudioDataCallback {
private let callback: (Data) -> Void
/**
* 初始化音频数据回调实现
* @param callback 音频数据处理闭包
*/
init(callback: @escaping (Data) -> Void) {
self.callback = callback
}
/**
* 音频数据回调方法
* @param data 音频数据
*/
func onAudio(_ data: Data) {
callback(data)
}
}

0
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/RecordFile.swift → local_plugins/azure_speech/ios/azure_speech/Sources/tools/RecordFile.swift

534
local_plugins/azure_speech/ios/azure_speech/Sources/tools/SimpleAudioReceiver.swift

@ -0,0 +1,534 @@
import Foundation
import AVFoundation
import MicrosoftCognitiveServicesSpeech
import os.log
/**
* 简单音频接收器类,用于处理音频录制和流传输
* 对应Android的SimpleAudioReceiver功能
*/
public class SimpleAudioReceiver: NSObject {
private let tag = "SimpleAudioReceiver"
private let log = OSLog(subsystem: "com.azure.speech", category: "SimpleAudioReceiver")
/**
* 音频来源类型
*/
public enum AudioSourceType {
/** 使用设备麦克风 */
case microphone
/** 使用外部提供的音频数据 */
case external
}
// MARK: - 音频流类
public private(set) var pushAudioStream: SPXPushAudioInputStream?
private let writeQueue = LinkedBlockingQueue<Data>()
private var audioEngine: AVAudioEngine?
private var audioFormat: AVAudioFormat?
private let audioSession = AVAudioSession.sharedInstance()
private var audioSourceType = AudioSourceType.microphone
private var audioDataCallback: AudioDataCallback?
private var isRunning = false
public var _isWriting = false
private let bufferSize: Int = 4096
private var writeThread: DispatchQueue?
private var currentRoute: AudioOutputRoute?
//public var onAudioData: ((Data) -> Void)?
public var recordfile: RecordFile?
// 添加对外部类的弱引用
private weak var parentHelper: AzureAsrHelper?
// 添加初始化方法,接收外部类引用
init(parentHelper: AzureAsrHelper) {
self.parentHelper = parentHelper
super.init()
}
/**
* 初始化
*/
/**
* 初始化音频录制组件
* 包括音频格式、推流、音频引擎等核心组件的初始化
*/
public func initAudioRecord() {
print("初始化了")
audioFormat = getOptimalAudioFormat()
pushAudioStream = SPXPushAudioInputStream()
// 初始化音频引擎
audioEngine = AVAudioEngine()
isRunning = true
// 创建新的写线程
writeThread = DispatchQueue(label: "audio.stream.writer")
startWriteThread()
}
/// 获取最佳音频格式 (iOS 通常支持标准采样率)
public func getOptimalAudioFormat() -> AVAudioFormat? {
let sampleRate: Double = 16000 // iOS 通常支持 16kHz
return AVAudioFormat(
commonFormat: .pcmFormatInt16,
sampleRate: sampleRate,
channels: 1,
interleaved: true
)
}
/**
* 设置音频配置
* @param sampleRate 采样率,默认16000
* @param channels 声道数,默认1
*/
/**
* 设置音频配置参数
* @param sampleRate 采样率
* @param channels 声道数
*/
public func setAudioConfig(sampleRate: Int, channels: Int) {
print(
"AudioStream设置音频配置: sampleRate=\(sampleRate), channels=\(channels)"
)
// 如果已经有推流,重新创建
if pushAudioStream != nil {
// 修复:安全解包 SPXAudioStreamFormat
guard let audioStreamFormat = SPXAudioStreamFormat.init(
usingPCMWithSampleRate: UInt(sampleRate),
bitsPerSample: 16,
channels: UInt(channels)
) else {
print("创建音频流格式失败")
return
}
pushAudioStream = SPXPushAudioInputStream(audioFormat: audioStreamFormat)
}
print("AudioStream音频配置设置完成")
}
/**
* 开始音频输入
*/
public func startAudioRecord(audioSourceType: AudioSourceType = .microphone, audioDataCallback: AudioDataCallback?) {
_isWriting=true
self.audioSourceType = audioSourceType;
self.audioDataCallback = audioDataCallback
print("startAudioRecord=audioSourceType\(audioSourceType)")
switch audioSourceType {
case .microphone:
runMicrophoneCapture()
case .external:
runExternalCapture()
}
}
/**
* 开启音频写入线程
* 修复:使用userInitiated QoS避免优先级反转
*/
public func startWriteThread() {
// 使用userInitiated QoS匹配音频录制线程的优先级
writeThread = DispatchQueue(label: "audio.stream.writer", qos: .userInitiated)
writeThread?.async { [weak self] in
guard let self = self else { return }
while self.isRunning {
if !self._isWriting {
print("startWriteThread 111")
usleep(10_000)
continue
}
print("startWriteThread 222")
guard let dataToWrite = self.writeQueue.take() else { continue }
do {
if let callback = self.audioDataCallback {
callback.onAudio(dataToWrite)
}
try self.pushAudioStream?.write(dataToWrite)
if self.recordfile != nil {
print("写入recordfile数据长度: \(dataToWrite.count)")
recordfile?.saveAudioDataToWav(dataToWrite)
}
print("写入数据长度: \(dataToWrite.count)")
} catch {
print("推送音频数据失败: \(error.localizedDescription)")
}
}
}
}
/**
* 向音频流写入音频数据
* 仅当音频源设置为external时有效
*
* @param data 音频数据字节数组
*/
public func saveAudioDataTo(data: Data) {
// 通过父类引用调用方法
if audioSourceType != .external || !(parentHelper?.isContinuousRecognitionActive() ?? false) {
return
}
// print("外部data=\(data)")
// 放入队列,由写线程写入
writeQueue.put(data)
}
// 私有方法:启动麦克风捕获
/**
* 启动麦克风捕获
* 优化:复用已初始化的音频引擎,减少启动延迟
*/
private func runMicrophoneCapture() {
do {
// 【新增】首先配置音频会话
try audioSession.setCategory(
.playAndRecord,
mode: .measurement, // 使用 measurement 模式获得最佳录音质量
options: [.allowBluetooth, .defaultToSpeaker]
)
// 【新增】请求麦克风权限(如果尚未授权)
if audioSession.recordPermission != .granted {
audioSession.requestRecordPermission { granted in
if !granted {
print("麦克风权限被拒绝")
}
}
}
// 【新增】激活音频会话(在配置音频引擎之前)
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
// 检查音频引擎是否已初始化,避免重复创建
if audioEngine == nil {
audioEngine = AVAudioEngine()
}
// 如果音频引擎正在运行,先停止
if audioEngine?.isRunning == true {
audioEngine?.stop()
}
// 获取音频输入节点(麦克风)
guard let inputNode = audioEngine?.inputNode else {
throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"])
}
// 移除之前的音频处理块,避免重复添加
inputNode.removeTap(onBus: 0)
// 获取硬件支持的原始音频格式
let hardwareFormat = inputNode.inputFormat(forBus: 0)
// iOS 13+ 启用语音处理
if #available(iOS 13.0, *) {
try inputNode.setVoiceProcessingEnabled(true)
}
// 检查目标音频格式和转换器是否可用
guard let targetFormat = audioFormat, // 外部定义的期望音频格式
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2)
}
// 在输入节点上安装录音回调
inputNode.installTap(onBus: 0,
bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小
format: hardwareFormat) { // 使用原始硬件格式
[weak self] buffer, time in // 弱引用避免循环引用
// 确保实例存在且正在写入状态
guard let self = self, self._isWriting else { return }
// 创建目标格式的音频缓冲区
let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
// 计算转换后的帧容量(考虑采样率差异)
frameCapacity: AVAudioFrameCount(
targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate
)
)!
var error: NSError?
// 执行音频格式转换
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
outStatus.pointee = .haveData // 标记有数据可用
return buffer // 返回原始音频数据
}
)
// 转换成功且无错误
if status == .haveData, error == nil {
// 将音频缓冲区转换为二进制数据
let data = self.audioBufferToData(convertedBuffer)
// 将数据放入写入队列(后续处理)
self.writeQueue.put(data)
}
}
// 激活音频会话(允许录音)
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
try audioEngine?.start()
} catch {
print("麦克风启动失败: \(error)")
// 【新增】添加详细错误处理
if let nsError = error as NSError? {
print("错误域: \(nsError.domain), 错误代码: \(nsError.code)")
print("错误描述: \(nsError.localizedDescription)")
}
}
}
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
let frameLength = Int(buffer.frameLength)
let channelCount = 1
let dataLength = frameLength * channelCount * MemoryLayout<Int16>.size
// Handle 16-bit integer format
if let int16Data = buffer.int16ChannelData {
return Data(
bytes: int16Data.pointee,
count: dataLength
)
}
// Handle float format
else if let floatData = buffer.floatChannelData {
var int16Array = [Int16](repeating: 0, count: frameLength)
let floatBuffer = floatData.pointee
for i in 0..<frameLength {
let sample = floatBuffer[i]
let clamped = max(-1.0, min(sample, 1.0))
let scaled = clamped * Float(Int16.max)
int16Array[i] = Int16(scaled)
}
return Data(
bytes: int16Array,
count: dataLength
)
}
return Data() // Fallback for unsupported formats
}
private func runExternalCapture() {
if audioSourceType == .external {
//pushAudioData(data: Data())
audioEngine?.pause()
//audioEngine?.inputNode.removeTap(onBus: 0)
}
}
/**
* 继续麦克风捕获
*/
/**
* 继续麦克风捕获
*/
public func resumeRecord() {
guard _isWriting==false else { return }
_isWriting = true
// 启动录音引擎
do {
print("resumeRecord")
try audioEngine?.start()
} catch {
os_log("Failed to start audio engine: %@", log: log, type: .error, error.localizedDescription)
}
}
public func stopMicrophoneCapture() {
guard _isWriting==true else { return }
_isWriting=false
audioEngine?.pause()
print("stopMicrophoneCapture")
}
public func releaseAudioResources() {
print("释放了")
stopMicrophoneCapture()
audioEngine?.stop()
writeThread?.async {
self.writeQueue.close()
}
writeThread = nil
audioEngine?.inputNode.removeTap(onBus: 0)
isRunning=false
audioEngine = nil
}
// 音频路由管理
public enum AudioOutputRoute {
case speaker
case receiver
case bluetooth
}
public func setAudioOutputRoute(_ route: AudioOutputRoute) {
do {
//
if self._isWriting {
print("当前正在识别")
try audioEngine?.pause()
// 先停用以避免冲突
try audioSession.setActive(false)
}
print("setAudioOutputRoutecurrent,route=\(route)")
switch route {
case .speaker:
print("setAudioOutputRoutecurrent,speaker")
// 使用扬声器时必须用videoChat模式
try audioSession.setCategory(
.playAndRecord,
mode: .videoChat,
options: [.defaultToSpeaker]
)
try audioSession.overrideOutputAudioPort(.speaker)
case .receiver:
// 听筒模式使用voiceChat节省资源
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth]
)
try audioSession.overrideOutputAudioPort(.none)
case .bluetooth:
// 完整蓝牙设备支持
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth, .allowBluetoothA2DP]
)
try audioSession.overrideOutputAudioPort(.none)
// 不需要override,系统自动路由
}
self.currentRoute = route
if self._isWriting {
// 重新激活
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
try audioEngine?.start()
// writeThread?.async {
// self.writeQueue.close()
// }
}
} catch {
print("路由切换失败: \(error)")
}
}
/**
* 音频数据回调接口
*/
public protocol AudioDataCallback {
/**
* 音频数据回调
* @param data 音频数据
*/
func onAudio(_ data: Data)
}
/**
* 检查是否正在写入
* @return 是否正在写入音频数据
*/
public func isWriting() -> Bool {
return _isWriting
}
}
// MARK: - iOS版LinkedBlockingQueue实现
private class LinkedBlockingQueue<T> {
private var queue: [T] = []
private var isClosed = false
// 使用DispatchSemaphore实现阻塞行为
private let availableItems: DispatchSemaphore
private let queueLock = NSLock()
init() {
self.availableItems = DispatchSemaphore(value: 0)
}
/**
* 阻塞式获取元素(等价于Android的take())
* @return 队列中的元素,如果队列已关闭则返回nil
*/
func take() -> T? {
// 等待可用元素(阻塞直到有元素或队列关闭)
availableItems.wait()
queueLock.lock()
defer { queueLock.unlock() }
// 检查队列是否已关闭
if isClosed && queue.isEmpty {
return nil
}
// 获取第一个元素
guard !queue.isEmpty else {
return nil
}
return queue.removeFirst()
}
/**
* 添加元素(等价于Android的put())
* @param item 要添加的元素
*/
func put(_ item: T) {
queueLock.lock()
defer { queueLock.unlock() }
// 检查队列是否已关闭
if isClosed {
return
}
// 添加元素(无容量限制,与Android一致)
queue.append(item)
// 通知有新元素可用
availableItems.signal()
}
/**
* 关闭队列
* 关闭后不能再添加新元素,但可以继续取出已有元素
*/
func close() {
queueLock.lock()
defer { queueLock.unlock() }
isClosed = true
// 唤醒所有等待的take()操作
for _ in 0..<100 { // 假设最多100个等待者
availableItems.signal()
}
}
}
// MARK: - 协议定义
Loading…
Cancel
Save