|
|
|
@ -32,6 +32,7 @@ public class AzureAsrHelper: NSObject { |
|
|
|
private var subscriptionKey = "" |
|
|
|
private var region = "" |
|
|
|
|
|
|
|
|
|
|
|
// 音频源配置 |
|
|
|
public enum AudioSourceType { |
|
|
|
/** 使用设备麦克风 */ |
|
|
|
@ -59,10 +60,14 @@ public class AzureAsrHelper: NSObject { |
|
|
|
* @param audioSourceType 音频源类型 |
|
|
|
* @return 初始化是否成功 |
|
|
|
*/ |
|
|
|
/** |
|
|
|
* 初始化Azure语音服务 |
|
|
|
* 预初始化关键组件以提升首次启动性能 |
|
|
|
*/ |
|
|
|
public func initialize( |
|
|
|
subscriptionKey: String, |
|
|
|
region: String, |
|
|
|
supportedLanguages: [String] = ["zh-CN"], |
|
|
|
supportedLanguages: [String] = ["zh-CN", "en-US"], |
|
|
|
audioSourceType: AudioSourceType = .microphone |
|
|
|
) -> Bool { |
|
|
|
do { |
|
|
|
@ -95,21 +100,23 @@ public class AzureAsrHelper: NSObject { |
|
|
|
// 创建语音配置 |
|
|
|
speechConfig = try SPXSpeechConfiguration(subscription: subscriptionKey, region: region) |
|
|
|
|
|
|
|
speechConfig?.setPropertyTo("800", byName: "SpeechServiceConnection_EndSilenceTimeoutMs") |
|
|
|
speechConfig?.setPropertyTo("800", byName: "Speech_SegmentationSilenceTimeoutMs") |
|
|
|
speechConfig?.setPropertyTo("Time", byName: "Speech_SegmentationStrategy") |
|
|
|
|
|
|
|
if isAutoDetectLanguage { |
|
|
|
// 直接启用语言检测模式 |
|
|
|
speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") |
|
|
|
os_log("启用语言检测模式", log: log, type: .info) |
|
|
|
// 移除过时的属性设置方式 |
|
|
|
// speechConfig?.setPropertyTo("Continuous", byName: "SpeechServiceConnection_LanguageIdMode") |
|
|
|
speechConfig?.setPropertyTo(supportedLanguages.joined(separator: ","), byName: "SpeechServiceConnection_AutoDetectSourceLanguages") |
|
|
|
|
|
|
|
// 不在这里设置语言,让SPXAutoDetectSourceLanguageConfiguration处理 |
|
|
|
os_log("启用语言检测模式,支持语言: %{public}@", log: log, type: .info, supportedLanguages.joined(separator: ",")) |
|
|
|
} else { |
|
|
|
// 设置指定的识别语言 |
|
|
|
speechConfig?.speechRecognitionLanguage = currentLanguage |
|
|
|
} |
|
|
|
|
|
|
|
// 录音文件类 |
|
|
|
recordfile = RecordFile() |
|
|
|
// 创建识别器 |
|
|
|
|
|
|
|
// 【优化】预初始化音频流,避免首次启动延迟 |
|
|
|
preInitializeAudioComponents() |
|
|
|
|
|
|
|
return true |
|
|
|
} catch { |
|
|
|
os_log("初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
@ -117,6 +124,26 @@ public class AzureAsrHelper: NSObject { |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 预初始化音频组件 |
|
|
|
* 在初始化阶段就准备好音频流和相关组件,减少首次启动延迟 |
|
|
|
*/ |
|
|
|
private func preInitializeAudioComponents() { |
|
|
|
if audioStream == nil { |
|
|
|
audioStream = AudioStream() |
|
|
|
audioStream?.initAudioRecord() |
|
|
|
audioStream?.onAudioData = { [weak self] data in |
|
|
|
self?.recordfile?.saveAudioDataToWav(data) |
|
|
|
} |
|
|
|
|
|
|
|
// 预创建音频配置 |
|
|
|
if let pushStream = audioStream?.pushAudioStream { |
|
|
|
audioConfig = SPXAudioConfiguration(streamInput: pushStream) |
|
|
|
os_log("预初始化音频组件完成", log: log, type: .info) |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 开始连续语音识别 |
|
|
|
@ -165,6 +192,28 @@ public class AzureAsrHelper: NSObject { |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
// /** |
|
|
|
// * 获取检测到的语言 |
|
|
|
// */ |
|
|
|
// private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
|
|
// if !isAutoDetectLanguage { |
|
|
|
// return "" |
|
|
|
// } |
|
|
|
|
|
|
|
// // 尝试从属性中获取语言 |
|
|
|
// if let properties = result.properties, |
|
|
|
// let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage") { |
|
|
|
// return language |
|
|
|
// } |
|
|
|
|
|
|
|
// // 尝试另一种方式获取语言 |
|
|
|
// do { |
|
|
|
// let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
|
|
// return langResult.language ?? "" |
|
|
|
// } catch { |
|
|
|
// return "" |
|
|
|
// } |
|
|
|
// } |
|
|
|
|
|
|
|
/** |
|
|
|
* 停止连续语音识别 |
|
|
|
@ -257,30 +306,32 @@ public class AzureAsrHelper: NSObject { |
|
|
|
/** |
|
|
|
* 设置识别器 |
|
|
|
*/ |
|
|
|
/** |
|
|
|
* 设置识别器 |
|
|
|
* 优化:避免重复创建音频流,复用已初始化的组件 |
|
|
|
*/ |
|
|
|
private func setupRecognizer() -> Bool { |
|
|
|
do { |
|
|
|
print("设置识别器${_isContinuousRecognitionActive}") |
|
|
|
|
|
|
|
// 如果正在进行连续识别,先停止 |
|
|
|
if _isContinuousRecognitionActive { |
|
|
|
// 直接停止,不等待结果 |
|
|
|
try? recognizer?.stopContinuousRecognition() |
|
|
|
_isContinuousRecognitionActive = false |
|
|
|
} |
|
|
|
setupMicrophoneStream() |
|
|
|
|
|
|
|
// 【优化】只有在音频流未初始化时才创建,避免重复初始化 |
|
|
|
if audioStream == nil || audioConfig == nil { |
|
|
|
setupMicrophoneStream() |
|
|
|
} |
|
|
|
|
|
|
|
// 清理旧的识别器 |
|
|
|
recognizer = nil |
|
|
|
|
|
|
|
// 创建识别器 |
|
|
|
if isAutoDetectLanguage { |
|
|
|
// 使用更精确的语言配置方式 |
|
|
|
var sourceLanguageConfigs: [SPXSourceLanguageConfiguration] = [] |
|
|
|
for language in supportedLanguages { |
|
|
|
if let config = try? SPXSourceLanguageConfiguration(language) { |
|
|
|
sourceLanguageConfigs.append(config) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(sourceLanguageConfigurations: sourceLanguageConfigs) |
|
|
|
// 使用正确的初始化方法 |
|
|
|
let autoDetectConfig = try SPXAutoDetectSourceLanguageConfiguration(supportedLanguages) |
|
|
|
recognizer = try SPXSpeechRecognizer( |
|
|
|
speechConfiguration: speechConfig!, |
|
|
|
autoDetectSourceLanguageConfiguration: autoDetectConfig, |
|
|
|
@ -296,7 +347,7 @@ public class AzureAsrHelper: NSObject { |
|
|
|
return true |
|
|
|
} catch { |
|
|
|
os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
try? recognizer?.stopContinuousRecognition() |
|
|
|
try? recognizer?.stopContinuousRecognition() |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
@ -304,31 +355,76 @@ public class AzureAsrHelper: NSObject { |
|
|
|
|
|
|
|
/** |
|
|
|
* 设置麦克风流 - 使用推流方式 |
|
|
|
* 优化:避免重复初始化已存在的音频流 |
|
|
|
*/ |
|
|
|
private func setupMicrophoneStream() { |
|
|
|
do { |
|
|
|
|
|
|
|
if (audioStream == nil) { |
|
|
|
// 创建外部音频拉流对象 |
|
|
|
// 明确指定使用外部音频源(推流模式) |
|
|
|
// 【优化】检查是否已经初始化,避免重复创建 |
|
|
|
if audioStream == nil { |
|
|
|
audioStream = AudioStream() |
|
|
|
audioStream?.initAudioRecord() |
|
|
|
audioStream?.onAudioData = { [weak self] data in |
|
|
|
self?.recordfile?.saveAudioDataToWav(data) |
|
|
|
} |
|
|
|
audioStream?.onAudioData = { [weak self] data in |
|
|
|
self?.recordfile?.saveAudioDataToWav(data) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 创建音频配置 |
|
|
|
if let pushStream = audioStream?.pushAudioStream { // Safely unwrap optional |
|
|
|
// 【优化】检查音频配置是否已存在 |
|
|
|
if audioConfig == nil, let pushStream = audioStream?.pushAudioStream { |
|
|
|
audioConfig = SPXAudioConfiguration(streamInput: pushStream) |
|
|
|
os_log("设置麦克风流:") |
|
|
|
os_log("设置麦克风流完成", log: log, type: .info) |
|
|
|
} |
|
|
|
} catch { |
|
|
|
os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription) |
|
|
|
audioStream = nil |
|
|
|
} |
|
|
|
} |
|
|
|
/** |
|
|
|
* 从语音识别结果中获取检测到的语言 |
|
|
|
* @param result 语音识别结果 |
|
|
|
* @return 检测到的语言代码,如果检测失败则返回当前语言 |
|
|
|
*/ |
|
|
|
private func getDetectedLanguage(from result: SPXSpeechRecognitionResult) -> String { |
|
|
|
guard isAutoDetectLanguage else { |
|
|
|
return currentLanguage |
|
|
|
} |
|
|
|
|
|
|
|
print("result: \(result)") |
|
|
|
|
|
|
|
// 备用方案:从属性中获取语言 |
|
|
|
if let properties = result.properties, |
|
|
|
let language = properties.getPropertyByName("SpeechServiceResponse_RecognitionLanguage"), |
|
|
|
!language.isEmpty { |
|
|
|
print("从属性中检测到语言: \(language)") |
|
|
|
return language |
|
|
|
} |
|
|
|
|
|
|
|
// 最后的备用方案:从JSON响应中解析 |
|
|
|
if let properties = result.properties, |
|
|
|
let jsonString = properties.getPropertyByName("SpeechServiceResponse_Json"), |
|
|
|
let jsonData = jsonString.data(using: .utf8) { |
|
|
|
do { |
|
|
|
if let json = try JSONSerialization.jsonObject(with: jsonData, options: []) as? [String: Any], |
|
|
|
let language = json["Language"] as? String { |
|
|
|
print("从JSON中检测到语言: \(language)") |
|
|
|
return language |
|
|
|
} |
|
|
|
} catch { |
|
|
|
print("JSON解析失败: \(error)") |
|
|
|
} |
|
|
|
} |
|
|
|
// 使用正确的SPXAutoDetectSourceLanguageResult初始化方法 |
|
|
|
do { |
|
|
|
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
|
|
if let language = langResult.language, !language.isEmpty { |
|
|
|
print("SDK方法检测到语言: \(language)") |
|
|
|
return language |
|
|
|
} |
|
|
|
} catch { |
|
|
|
print("SDK语言检测失败: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
print("无法检测语言,使用默认语言: \(currentLanguage)") |
|
|
|
return currentLanguage |
|
|
|
} |
|
|
|
/** |
|
|
|
* 设置事件监听器 |
|
|
|
*/ |
|
|
|
@ -342,68 +438,28 @@ public class AzureAsrHelper: NSObject { |
|
|
|
guard let recognizer = recognizer else { return false} |
|
|
|
// 正在识别事件 |
|
|
|
recognizer.addRecognizingEventHandler { [weak self] (_, event) in |
|
|
|
guard let self = self else { return } |
|
|
|
guard let self = self else { return } |
|
|
|
let result = event.result |
|
|
|
print("识别中文本: \(result.text)") |
|
|
|
// 优化:只处理非空结果 |
|
|
|
if !(result.text?.isEmpty ?? true) { |
|
|
|
let detectedLanguage: String |
|
|
|
if self.isAutoDetectLanguage { |
|
|
|
do { |
|
|
|
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
|
|
let rawLanguage = langResult.language |
|
|
|
print("[DEBUG] 原始检测语言: \(rawLanguage ?? "nil"), 文本: \(result.text ?? "nil"), 支持的语言: \(self.supportedLanguages)") |
|
|
|
|
|
|
|
// 应用启发式语言检测纠正 |
|
|
|
detectedLanguage = self.correctLanguageDetection(rawLanguage: rawLanguage ?? "", text: result.text ?? "") |
|
|
|
print("[DEBUG] 纠正后语言: \(detectedLanguage)") |
|
|
|
} catch { |
|
|
|
print("[DEBUG] 语言检测异常: \(error.localizedDescription)") |
|
|
|
detectedLanguage = "" |
|
|
|
} |
|
|
|
} else { |
|
|
|
detectedLanguage = self.currentLanguage |
|
|
|
print("[DEBUG] 使用当前语言: \(detectedLanguage)") |
|
|
|
} |
|
|
|
print("识别中事件: \(detectedLanguage)") |
|
|
|
DispatchQueue.main.async { |
|
|
|
callback.onRecognizing(result.text ?? "", detectedLanguage) |
|
|
|
} |
|
|
|
guard !(result.text?.isEmpty ?? true) else { return } |
|
|
|
|
|
|
|
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
|
|
DispatchQueue.main.async { |
|
|
|
callback.onRecognizing(result.text ?? "", detectedLanguage) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 识别完成事件 |
|
|
|
recognizer.addRecognizedEventHandler { [weak self] (_, event) in |
|
|
|
guard let self = self else { return } |
|
|
|
guard let self = self else { return } |
|
|
|
let result = event.result |
|
|
|
print("识别完成文本: \(result.text)") |
|
|
|
if result.reason == .recognizedSpeech && !(result.text?.isEmpty ?? true) { |
|
|
|
let detectedLanguage: String |
|
|
|
if self.isAutoDetectLanguage { |
|
|
|
do { |
|
|
|
let langResult = try SPXAutoDetectSourceLanguageResult(result) |
|
|
|
let rawLanguage = langResult.language |
|
|
|
print("[DEBUG] 识别完成 - 原始检测语言: \(rawLanguage ?? "nil"), 文本: \(result.text ?? "nil"), 支持的语言: \(self.supportedLanguages)") |
|
|
|
|
|
|
|
// 应用启发式语言检测纠正 |
|
|
|
let correctedLanguage = self.correctLanguageDetection(rawLanguage: rawLanguage ?? "", text: result.text ?? "") |
|
|
|
detectedLanguage = correctedLanguage.isEmpty ? (self.supportedLanguages.first ?? self.currentLanguage) : correctedLanguage |
|
|
|
print("[DEBUG] 识别完成 - 纠正后语言: \(detectedLanguage)") |
|
|
|
} catch { |
|
|
|
print("[DEBUG] 识别完成 - 语言检测异常: \(error.localizedDescription)") |
|
|
|
detectedLanguage = self.supportedLanguages.first ?? self.currentLanguage |
|
|
|
} |
|
|
|
} else { |
|
|
|
detectedLanguage = self.currentLanguage |
|
|
|
print("[DEBUG] 识别完成 - 使用当前语言: \(detectedLanguage)") |
|
|
|
} |
|
|
|
print("识别完成事件: \(detectedLanguage)") |
|
|
|
DispatchQueue.main.async { |
|
|
|
callback.onResult(result.text ?? "", detectedLanguage) |
|
|
|
} |
|
|
|
guard result.reason == .recognizedSpeech, |
|
|
|
!(result.text?.isEmpty ?? true) else { return } |
|
|
|
|
|
|
|
let detectedLanguage = self.getDetectedLanguage(from: result) |
|
|
|
DispatchQueue.main.async { |
|
|
|
callback.onResult(result.text ?? "", detectedLanguage) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 会话开始事件 |
|
|
|
recognizer.addSessionStartedEventHandler { (sender, event) in |
|
|
|
// 直接在当前线程调用回调 |
|
|
|
@ -532,15 +588,23 @@ public class AzureAsrHelper: NSObject { |
|
|
|
/** |
|
|
|
* 初始化 |
|
|
|
*/ |
|
|
|
public func initAudioRecord() { |
|
|
|
/** |
|
|
|
* 初始化音频录制组件 |
|
|
|
* 包括音频格式、推流、音频引擎等核心组件的初始化 |
|
|
|
*/ |
|
|
|
public func initAudioRecord() { |
|
|
|
print("初始化了") |
|
|
|
audioFormat = getOptimalAudioFormat() |
|
|
|
pushAudioStream = SPXPushAudioInputStream() |
|
|
|
isRunning=true |
|
|
|
|
|
|
|
// 初始化音频引擎 |
|
|
|
audioEngine = AVAudioEngine() |
|
|
|
|
|
|
|
isRunning = true |
|
|
|
// 创建新的写线程 |
|
|
|
writeThread = DispatchQueue(label: "audio.stream.writer") |
|
|
|
writeThread = DispatchQueue(label: "audio.stream.writer") |
|
|
|
startWriteThread() |
|
|
|
self.setAudioOutputRoute(.receiver) |
|
|
|
self.setAudioOutputRoute(.receiver) |
|
|
|
} |
|
|
|
|
|
|
|
/// 获取最佳音频格式 (iOS 通常支持标准采样率) |
|
|
|
@ -585,8 +649,13 @@ public class AzureAsrHelper: NSObject { |
|
|
|
guard let dataToWrite = self.writeQueue.take() else { continue } |
|
|
|
|
|
|
|
print("写入数据长度: \(dataToWrite.count)") |
|
|
|
self.pushAudioStream?.write(dataToWrite) |
|
|
|
self.onAudioData?(dataToWrite) |
|
|
|
do { |
|
|
|
try self.pushAudioStream?.write(dataToWrite) |
|
|
|
self.onAudioData?(dataToWrite) |
|
|
|
} catch { |
|
|
|
print("推送音频数据失败: \(error.localizedDescription)") |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
@ -606,65 +675,92 @@ public class AzureAsrHelper: NSObject { |
|
|
|
writeQueue.put(data) |
|
|
|
} |
|
|
|
|
|
|
|
private func runMicrophoneCapture() { |
|
|
|
do { |
|
|
|
audioEngine = AVAudioEngine() |
|
|
|
guard let inputNode = audioEngine?.inputNode else { |
|
|
|
throw NSError(domain: "AudioSetup", code: 1) |
|
|
|
} |
|
|
|
// 推荐直接用系统 format |
|
|
|
let hardwareFormat = inputNode.inputFormat(forBus: 0) |
|
|
|
// // ===== 新增:启用专业级语音处理 ===== |
|
|
|
if #available(iOS 13.0, *) { |
|
|
|
try inputNode.setVoiceProcessingEnabled(true) |
|
|
|
print("Voice processing enabled") |
|
|
|
// 私有方法:启动麦克风捕获 |
|
|
|
/** |
|
|
|
* 启动麦克风捕获 |
|
|
|
* 优化:复用已初始化的音频引擎,减少启动延迟 |
|
|
|
*/ |
|
|
|
private func runMicrophoneCapture() { |
|
|
|
do { |
|
|
|
// 【优化】检查音频引擎是否已初始化,避免重复创建 |
|
|
|
if audioEngine == nil { |
|
|
|
audioEngine = AVAudioEngine() |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
// 如果音频引擎正在运行,先停止 |
|
|
|
if audioEngine?.isRunning == true { |
|
|
|
audioEngine?.stop() |
|
|
|
} |
|
|
|
|
|
|
|
// 获取音频输入节点(麦克风) |
|
|
|
guard let inputNode = audioEngine?.inputNode else { |
|
|
|
throw NSError(domain: "AudioSetup", code: 1, userInfo: [NSLocalizedDescriptionKey: "无法获取音频输入节点"]) |
|
|
|
} |
|
|
|
|
|
|
|
// Create converter to target format |
|
|
|
guard let targetFormat = audioFormat, |
|
|
|
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { |
|
|
|
throw NSError(domain: "AudioSetup", code: 2) |
|
|
|
} |
|
|
|
// 移除之前的音频处理块,避免重复添加 |
|
|
|
inputNode.removeTap(onBus: 0) |
|
|
|
|
|
|
|
inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) { |
|
|
|
[weak self] buffer, time in |
|
|
|
guard let self = self, self.isWriting else { return } |
|
|
|
|
|
|
|
// Convert to target format |
|
|
|
let convertedBuffer = AVAudioPCMBuffer( |
|
|
|
pcmFormat: targetFormat, |
|
|
|
frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate) |
|
|
|
)! |
|
|
|
///print("进入 tap 回调,frameLength: \(buffer.frameLength)") |
|
|
|
var error: NSError? |
|
|
|
let status = converter.convert( |
|
|
|
to: convertedBuffer, |
|
|
|
error: &error, |
|
|
|
withInputFrom: { inNumPackets, outStatus in |
|
|
|
outStatus.pointee = .haveData |
|
|
|
return buffer |
|
|
|
} |
|
|
|
) |
|
|
|
//print("转换状态: \(status.rawValue), 错误: \(String(describing: error))") |
|
|
|
// 获取硬件支持的原始音频格式 |
|
|
|
let hardwareFormat = inputNode.inputFormat(forBus: 0) |
|
|
|
|
|
|
|
// iOS 13+ 启用语音处理 |
|
|
|
if #available(iOS 13.0, *) { |
|
|
|
try inputNode.setVoiceProcessingEnabled(true) |
|
|
|
} |
|
|
|
|
|
|
|
if status == .haveData, error == nil { |
|
|
|
let data = self.audioBufferToData(convertedBuffer) |
|
|
|
// print("准备写入数据,大小: \(data.count)") |
|
|
|
// 检查目标音频格式和转换器是否可用 |
|
|
|
guard let targetFormat = audioFormat, // 外部定义的期望音频格式 |
|
|
|
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else { |
|
|
|
throw NSError(domain: "AudioSetup", code: 2) |
|
|
|
} |
|
|
|
|
|
|
|
self.writeQueue.put(data) |
|
|
|
// 在输入节点上安装录音回调 |
|
|
|
inputNode.installTap(onBus: 0, |
|
|
|
bufferSize: UInt32(bufferSize), // 每次回调的缓冲区大小 |
|
|
|
format: hardwareFormat) { // 使用原始硬件格式 |
|
|
|
[weak self] buffer, time in // 弱引用避免循环引用 |
|
|
|
|
|
|
|
} |
|
|
|
// 确保实例存在且正在写入状态 |
|
|
|
guard let self = self, self.isWriting else { return } |
|
|
|
|
|
|
|
// 创建目标格式的音频缓冲区 |
|
|
|
let convertedBuffer = AVAudioPCMBuffer( |
|
|
|
pcmFormat: targetFormat, |
|
|
|
// 计算转换后的帧容量(考虑采样率差异) |
|
|
|
frameCapacity: AVAudioFrameCount( |
|
|
|
targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate |
|
|
|
) |
|
|
|
)! |
|
|
|
|
|
|
|
var error: NSError? |
|
|
|
// 执行音频格式转换 |
|
|
|
let status = converter.convert( |
|
|
|
to: convertedBuffer, |
|
|
|
error: &error, |
|
|
|
withInputFrom: { inNumPackets, outStatus in |
|
|
|
outStatus.pointee = .haveData // 标记有数据可用 |
|
|
|
return buffer // 返回原始音频数据 |
|
|
|
} |
|
|
|
// 重新激活 |
|
|
|
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) |
|
|
|
) |
|
|
|
|
|
|
|
try audioEngine?.start() |
|
|
|
} catch { |
|
|
|
print("麦克风启动失败: \(error)") |
|
|
|
// 转换成功且无错误 |
|
|
|
if status == .haveData, error == nil { |
|
|
|
// 将音频缓冲区转换为二进制数据 |
|
|
|
let data = self.audioBufferToData(convertedBuffer) |
|
|
|
// 将数据放入写入队列(后续处理) |
|
|
|
self.writeQueue.put(data) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 激活音频会话(允许录音) |
|
|
|
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation]) |
|
|
|
|
|
|
|
// 启动音频引擎 |
|
|
|
try audioEngine?.start() |
|
|
|
} catch { |
|
|
|
print("麦克风启动失败: \(error)") |
|
|
|
} |
|
|
|
} |
|
|
|
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { |
|
|
|
let frameLength = Int(buffer.frameLength) |
|
|
|
let channelCount = 1 |
|
|
|
@ -870,55 +966,7 @@ public class AzureAsrHelper: NSObject { |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 启发式语言检测纠正方法 |
|
|
|
* 基于文本内容特征来纠正Azure语音服务的语言检测结果 |
|
|
|
*/ |
|
|
|
private func correctLanguageDetection(rawLanguage: String, text: String) -> String { |
|
|
|
// 如果原始检测为空,返回空 |
|
|
|
if rawLanguage.isEmpty || text.isEmpty { |
|
|
|
return rawLanguage |
|
|
|
} |
|
|
|
|
|
|
|
let trimmedText = text.trimmingCharacters(in: .whitespacesAndNewlines).lowercased() |
|
|
|
|
|
|
|
// 英文特征检测 |
|
|
|
let englishPattern = "^[a-z\\s\\.,!?;:'-]+$" |
|
|
|
let englishRegex = try? NSRegularExpression(pattern: englishPattern, options: .caseInsensitive) |
|
|
|
let englishMatches = englishRegex?.numberOfMatches(in: trimmedText, options: [], range: NSRange(location: 0, length: trimmedText.count)) ?? 0 |
|
|
|
|
|
|
|
// 中文特征检测 |
|
|
|
let chinesePattern = "[\\u4e00-\\u9fff]" |
|
|
|
let chineseRegex = try? NSRegularExpression(pattern: chinesePattern, options: []) |
|
|
|
let chineseMatches = chineseRegex?.numberOfMatches(in: trimmedText, options: [], range: NSRange(location: 0, length: trimmedText.count)) ?? 0 |
|
|
|
|
|
|
|
// 常见英文单词列表 |
|
|
|
let commonEnglishWords = ["hello", "hi", "please", "thank", "you", "yes", "no", "good", "morning", "afternoon", "evening", "night", "how", "are", "what", "where", "when", "why", "who", "can", "could", "would", "should", "will", "the", "and", "or", "but", "if", "then", "this", "that", "these", "those"] |
|
|
|
|
|
|
|
let hasCommonEnglishWords = commonEnglishWords.contains { word in |
|
|
|
trimmedText.contains(word) |
|
|
|
} |
|
|
|
|
|
|
|
print("[DEBUG] 语言纠正分析 - 文本: '\(trimmedText)', 英文匹配: \(englishMatches > 0), 中文字符: \(chineseMatches), 常见英文词: \(hasCommonEnglishWords), 原始语言: \(rawLanguage)") |
|
|
|
|
|
|
|
// 纠正逻辑 |
|
|
|
if rawLanguage == "zh-CN" { |
|
|
|
// 如果Azure检测为中文,但文本明显是英文特征 |
|
|
|
if (englishMatches > 0 && chineseMatches == 0) || hasCommonEnglishWords { |
|
|
|
print("[DEBUG] 纠正: \(rawLanguage) -> en-US (检测到英文特征)") |
|
|
|
return "en-US" |
|
|
|
} |
|
|
|
} else if rawLanguage == "en-US" { |
|
|
|
// 如果Azure检测为英文,但包含中文字符 |
|
|
|
if chineseMatches > 0 { |
|
|
|
print("[DEBUG] 纠正: \(rawLanguage) -> zh-CN (检测到中文字符)") |
|
|
|
return "zh-CN" |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 如果没有明确的纠正理由,保持原始检测结果 |
|
|
|
return rawLanguage |
|
|
|
} |
|
|
|
|
|
|
|
// MARK: - 回调函数 |
|
|
|
/** |
|
|
|
|