Browse Source

加入降噪,加油面对面翻译功能

weicu
fdp 1 year ago
parent
commit
d19eb4bfb1
  1. 8
      ios/Podfile.lock
  2. 2
      ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme
  3. 7
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt
  4. 547
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift
  5. 248
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift
  6. 729
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift

8
ios/Podfile.lock

@ -77,9 +77,9 @@ PODS:
- QCloudTrack/Beacon (6.4.7) - QCloudTrack/Beacon (6.4.7)
- record_darwin (1.0.0): - record_darwin (1.0.0):
- Flutter - Flutter
- SDWebImage (5.21.1): - SDWebImage (5.21.0):
- SDWebImage/Core (= 5.21.1) - SDWebImage/Core (= 5.21.0)
- SDWebImage/Core (5.21.1) - SDWebImage/Core (5.21.0)
- SDWebImageWebPCoder (0.14.6): - SDWebImageWebPCoder (0.14.6):
- libwebp (~> 1.0) - libwebp (~> 1.0)
- SDWebImage/Core (~> 5.17) - SDWebImage/Core (~> 5.17)
@ -200,7 +200,7 @@ SPEC CHECKSUMS:
QCloudCOSXML: 7205e76aa9cf613468222615483c9d7c074e4298 QCloudCOSXML: 7205e76aa9cf613468222615483c9d7c074e4298
QCloudTrack: 3b53a7fc4fe3920e407f2aa73f2452992a61f7f3 QCloudTrack: 3b53a7fc4fe3920e407f2aa73f2452992a61f7f3
record_darwin: fb1f375f1d9603714f55b8708a903bbb91ffdb0a record_darwin: fb1f375f1d9603714f55b8708a903bbb91ffdb0a
SDWebImage: f29024626962457f3470184232766516dee8dfea SDWebImage: f84b0feeb08d2d11e6a9b843cb06d75ebf5b8868
SDWebImageWebPCoder: e38c0a70396191361d60c092933e22c20d5b1380 SDWebImageWebPCoder: e38c0a70396191361d60c092933e22c20d5b1380
spotify_sdk: a48400bb29f70c4fe251ebfdc9135c37097ac5ca spotify_sdk: a48400bb29f70c4fe251ebfdc9135c37097ac5ca
tencentcloud_cos_sdk_plugin: 7bed564dbe72df23e7b5cacb1e51b1a20cb1639c tencentcloud_cos_sdk_plugin: 7bed564dbe72df23e7b5cacb1e51b1a20cb1639c

2
ios/Runner.xcodeproj/xcshareddata/xcschemes/Runner.xcscheme

@ -1,7 +1,7 @@
<?xml version="1.0" encoding="UTF-8"?> <?xml version="1.0" encoding="UTF-8"?>
<Scheme <Scheme
LastUpgradeVersion = "1510" LastUpgradeVersion = "1510"
version = "1.3"> version = "1.7">
<BuildAction <BuildAction
parallelizeBuildables = "YES" parallelizeBuildables = "YES"
buildImplicitDependencies = "YES"> buildImplicitDependencies = "YES">

7
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AzureAsrHelper.kt

@ -669,7 +669,9 @@ class AzureAsrHelper(private val context: Context) {
} }
/**
* 开启音频写入线程
*/
private fun startWriteThread() { private fun startWriteThread() {
writeThread = Thread { writeThread = Thread {
try { try {
@ -745,6 +747,9 @@ class AzureAsrHelper(private val context: Context) {
} }
/**
* 开始音频输入
*/
fun startAudioRecord() { fun startAudioRecord() {
isWriting.set(true) isWriting.set(true)

547
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureAsrHelper.swift

@ -5,7 +5,7 @@ import MicrosoftCognitiveServicesSpeech
import speech import speech
import os.log import os.log
// MARK: - 基于微软Azure语音服务的ASR实现
/** /**
* Azure ASR Helper * Azure ASR Helper
* *
@ -48,8 +48,8 @@ public class AzureAsrHelper: NSObject {
// 音频处理 // 音频处理
public var audioStream: AudioStream? public var audioStream: AudioStream?
//录音文件
public var recordfile: RecordFile? public var recordfile: RecordFile?
/** /**
* 初始化Azure语音服务 * 初始化Azure语音服务
* *
@ -74,7 +74,6 @@ public class AzureAsrHelper: NSObject {
// 释放之前的资源 // 释放之前的资源
dispose() dispose()
print("w[w[w]]")
// 保存配置 // 保存配置
self.subscriptionKey = subscriptionKey self.subscriptionKey = subscriptionKey
self.region = region self.region = region
@ -108,8 +107,8 @@ public class AzureAsrHelper: NSObject {
// 设置指定的识别语言 // 设置指定的识别语言
speechConfig?.speechRecognitionLanguage = currentLanguage speechConfig?.speechRecognitionLanguage = currentLanguage
} }
// 录音文件类 // 录音文件类
recordfile = RecordFile() recordfile = RecordFile()
// 创建识别器 // 创建识别器
return true return true
} catch { } catch {
@ -147,12 +146,10 @@ public class AzureAsrHelper: NSObject {
// callback.onError("重置识别器失败") // callback.onError("重置识别器失败")
return false return false
} }
return false return false
} }
// 启动音频处理
// 启动音频处理 audioStream?.startAudioInput()
audioStream?.startAudioRecord()
// 开始连续识别 // 开始连续识别
try recognizer?.startContinuousRecognition() try recognizer?.startContinuousRecognition()
@ -162,9 +159,8 @@ public class AzureAsrHelper: NSObject {
return true return true
} catch { } catch {
audioStream?.stopMicrophoneCapture() audioStream?.stopAudioCapture()
// stopAudioProcessing() _isContinuousRecognitionActive = false
_isContinuousRecognitionActive = false
//callback.onError("启动连续识别失败: \(error.localizedDescription)") //callback.onError("启动连续识别失败: \(error.localizedDescription)")
return false return false
} }
@ -176,42 +172,40 @@ public class AzureAsrHelper: NSObject {
* @return 是否成功停止 * @return 是否成功停止
*/ */
public func stopContinuousRecognition() -> Bool { public func stopContinuousRecognition() -> Bool {
print("stopContinuousRecognition") print("stopContinuousRecognition")
guard speechConfig != nil else { guard speechConfig != nil else {
os_log("语音服务未初始化", log: log, type: .error) os_log("语音服务未初始化", log: log, type: .error)
return false return false
} }
guard let recognizer = recognizer else { guard let recognizer = recognizer else {
os_log("识别器为空,重置状态", log: log, type: .info) os_log("识别器为空,重置状态", log: log, type: .info)
return true return true
} }
if !_isContinuousRecognitionActive { if !_isContinuousRecognitionActive {
return true return true
} }
do { do {
_isContinuousRecognitionActive = false _isContinuousRecognitionActive = false
if audioSourceType == .external { if audioSourceType == .external {
//pushAudioData(data: Data()) //pushAudioData(data: Data())
} }
audioStream?.stopAudioCapture()
// 停止连续识别 // 停止连续识别
try recognizer.stopContinuousRecognition() try recognizer.stopContinuousRecognition()
// 停止音频处理
//stopAudioProcessing()
audioStream?.stopMicrophoneCapture()
// 会话结束事件会设置_isContinuousRecognitionActive = false // 会话结束事件会设置_isContinuousRecognitionActive = false
return true return true
} catch { } catch {
// 强制重置状态 // 强制重置状态
_isContinuousRecognitionActive = false _isContinuousRecognitionActive = false
os_log("强制停止识别失败", log: log, type: .error) os_log("强制停止识别失败", log: log, type: .error)
// 停止音频处理 // 停止音频处理
audioStream?.stopMicrophoneCapture() audioStream?.stopAudioCapture()
return false return false
} }
@ -229,7 +223,7 @@ public class AzureAsrHelper: NSObject {
*/ */
public func dispose() { public func dispose() {
do { do {
print("释放所有资源:") print("释放所有资源:")
// 如果正在进行连续识别,先停止 // 如果正在进行连续识别,先停止
if _isContinuousRecognitionActive { if _isContinuousRecognitionActive {
// 直接停止,不等待结果 // 直接停止,不等待结果
@ -238,7 +232,7 @@ public class AzureAsrHelper: NSObject {
} }
// 停止音频处理 // 停止音频处理
stopAudioProcessing() audioStream?.releaseAudioResources()
// 释放资源 // 释放资源
recognizer = nil recognizer = nil
@ -257,8 +251,7 @@ public class AzureAsrHelper: NSObject {
speechConfig = nil speechConfig = nil
} }
} }
// MARK: - 私有方法
/** /**
* 设置识别器 * 设置识别器
@ -294,29 +287,29 @@ public class AzureAsrHelper: NSObject {
return true return true
} catch { } catch {
os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription) os_log("创建识别器失败: %{public}@", log: log, type: .error, error.localizedDescription)
stopAudioProcessing() try? recognizer?.stopContinuousRecognition()
return false return false
} }
} }
/** /**
* 设置麦克风流 - 使用推流方式 * 设置麦克风流 - 使用推流方式
*/ */
private func setupMicrophoneStream() { private func setupMicrophoneStream() {
do { do {
if (audioStream == nil) { if (audioStream == nil) {
// 创建外部音频拉流对象 // 创建外部音频拉流对象
// 明确指定使用外部音频源(推流模式) // 明确指定使用外部音频源(推流模式)
audioStream = AudioStream(audioSourceType: audioSourceType) audioStream = AudioStream(audioSourceType: audioSourceType)
audioStream?.initAudioRecord() audioStream?.initAudioRecord()
} }
// 创建音频配置 // 创建音频配置
if let pushStream = audioStream?.pushAudioStream { // Safely unwrap optional if let pushStream = audioStream?.pushAudioStream { // Safely unwrap optional
audioConfig = SPXAudioConfiguration(streamInput: pushStream) audioConfig = SPXAudioConfiguration(streamInput: pushStream)
os_log("设置麦克风流:") os_log("设置麦克风流:")
} }
} catch { } catch {
os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription) os_log("设置麦克风流失败: %{public}@", log: log, type: .error, error.localizedDescription)
@ -328,13 +321,13 @@ public class AzureAsrHelper: NSObject {
* 设置事件监听器 * 设置事件监听器
*/ */
public func setupEventListeners(callback: ContinuousRecognizeCallback) -> Bool{ public func setupEventListeners(callback: ContinuousRecognizeCallback) -> Bool{
print("设置ssssss监听器:${speechConfig}") print("设置ssssss监听器:${speechConfig}")
// 重设识别器 // 重设识别器
if (!setupRecognizer()) { if (!setupRecognizer()) {
return false return false
} }
guard let recognizer = recognizer else { return false} guard let recognizer = recognizer else { return false}
// 识别中事件 // 识别中事件
recognizer.addRecognizingEventHandler { [weak self] (sender, event) in recognizer.addRecognizingEventHandler { [weak self] (sender, event) in
guard let self = self else { return } guard let self = self else { return }
@ -349,7 +342,7 @@ public class AzureAsrHelper: NSObject {
// 识别完成事件 // 识别完成事件
recognizer.addRecognizedEventHandler { [weak self] (sender, event) in recognizer.addRecognizedEventHandler { [weak self] (sender, event) in
guard let self = self else { return } guard let self = self else { return }
print("识别完成事件:") print("识别完成事件:")
let result = event.result let result = event.result
if result.reason == SPXResultReason.recognizedSpeech { if result.reason == SPXResultReason.recognizedSpeech {
let detectedLanguage = self.getDetectedLanguage(from: result) let detectedLanguage = self.getDetectedLanguage(from: result)
@ -362,13 +355,13 @@ public class AzureAsrHelper: NSObject {
recognizer.addSessionStartedEventHandler { (sender, event) in recognizer.addSessionStartedEventHandler { (sender, event) in
// 直接在当前线程调用回调 // 直接在当前线程调用回调
callback.onSessionStarted() callback.onSessionStarted()
print("会话开始事件:") print("会话开始事件:")
} }
// 会话结束事件 // 会话结束事件
recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in recognizer.addSessionStoppedEventHandler { [weak self] (sender, event) in
guard let self = self else { return } guard let self = self else { return }
print("会话结束事件:") print("会话结束事件:")
// 直接在当前线程调用回调 // 直接在当前线程调用回调
// callback.onSessionStopped() // callback.onSessionStopped()
// self._isContinuousRecognitionActive = false // self._isContinuousRecognitionActive = false
@ -378,7 +371,7 @@ public class AzureAsrHelper: NSObject {
// 取消事件 // 取消事件
recognizer.addCanceledEventHandler { [weak self] (sender, event) in recognizer.addCanceledEventHandler { [weak self] (sender, event) in
guard let self = self else { return } guard let self = self else { return }
print("取消事件:") print("取消事件:")
let errorDetails = event.errorDetails ?? "未知错误" let errorDetails = event.errorDetails ?? "未知错误"
let reason = String(describing: event.reason.rawValue) let reason = String(describing: event.reason.rawValue)
@ -411,73 +404,87 @@ public class AzureAsrHelper: NSObject {
return "" return ""
} }
} }
/** /**
* 停止音频处理 * 禁用蓝牙音频功能,切换回正常音频模式
*/ */
private func stopAudioProcessing() { public func disableBluetoothAudio() {
print("stopAudioProcessing")
// if let stream = externalAudioStream { do {
// stream.close() audioStream?.setAudioOutputRoute(.speaker)
// externalAudioStream = nil } catch {
// // os_log("外部音频流已关闭", log: log, type: .info) print("disableBluetoothAudio")
// } }
}
/**
* 恢复原始音频设备状态(通常是重新启用蓝牙)
*/
public func restoreOriginalAudioState() {
do {
audioStream?.setAudioOutputRoute(.receiver)
} catch {
print("restoreOriginalAudioState")
}
} }
//录音音频文件
// MARK: - 录音音频文件方法
/** /**
* 开启录音 * 开启录音
*/ */
public func enableRecord(filePath: String) { public func enableRecord(filePath: String) {
recordfile?.closeFile(isSave: true) recordfile?.closeFile(isSave: true)
recordfile?.creatingFiles(atPath: filePath) // Fixed method call recordfile?.creatingFiles(atPath: filePath) // Fixed method call
} }
/** /**
* 移动文件到新路径 * 移动文件到新路径
*/ */
public func moveFile(sourcePath: String, destPath: String)-> Bool { public func moveFile(sourcePath: String, destPath: String)-> Bool {
recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call recordfile?.moveFile(from: sourcePath, to: destPath) // Fixed method call
return true return true
} }
/** /**
* 重命名指定路径的音频文件 * 重命名指定路径的音频文件
*/ */
public func renameFile(filePath: String, newName: String)-> Bool { public func renameFile(filePath: String, newName: String)-> Bool {
recordfile?.renameFile(at: filePath, to: newName) // Fixed method call recordfile?.renameFile(at: filePath, to: newName) // Fixed method call
return true return true
} }
/** /**
* 停止录音 * 停止录音
*/ */
public func pauseRecord() { public func pauseRecord() {
recordfile?.isPause = true recordfile?.isPause = true
} }
/** /**
* 关闭录音 * 关闭录音
*/ */
public func stopRecord(isSave: Bool) { public func stopRecord(isSave: Bool) {
recordfile?.isPause = false recordfile?.isPause = false
recordfile?.closeFile(isSave: true) recordfile?.closeFile(isSave: true)
} }
// MARK: - 音频流类
public class AudioStream: NSObject { public class AudioStream: NSObject {
public private(set) var pushAudioStream: SPXPushAudioInputStream? public private(set) var pushAudioStream: SPXPushAudioInputStream?
private let writeQueue = LinkedBlockingQueue<Data>() private let writeQueue = LinkedBlockingQueue<Data>()
@ -485,18 +492,24 @@ print("stopAudioProcessing")
private var audioFormat: AVAudioFormat? private var audioFormat: AVAudioFormat?
private let audioSourceType: AudioSourceType // 添加引用 private let audioSourceType: AudioSourceType // 添加引用
private var isRunning = false private var isRunning = false
private var isWriting = false public var isWriting = false
private let bufferSize: Int = 4096 private let bufferSize: Int = 4096
private let writeThread = DispatchQueue(label: "audio.stream.writer") private let writeThread = DispatchQueue(label: "audio.stream.writer")
init(audioSourceType: AudioSourceType) { init(audioSourceType: AudioSourceType) {
self.audioSourceType = audioSourceType self.audioSourceType = audioSourceType
} }
/**
* 初始化
*/
public func initAudioRecord() { public func initAudioRecord() {
audioFormat = getOptimalAudioFormat() audioFormat = getOptimalAudioFormat()
pushAudioStream = SPXPushAudioInputStream() pushAudioStream = SPXPushAudioInputStream()
isRunning=true isRunning=true
startWriteThread() startWriteThread()
} }
/// 获取最佳音频格式 (iOS 通常支持标准采样率) /// 获取最佳音频格式 (iOS 通常支持标准采样率)
public func getOptimalAudioFormat() -> AVAudioFormat? { public func getOptimalAudioFormat() -> AVAudioFormat? {
let sampleRate: Double = 16000 // iOS 通常支持 16kHz let sampleRate: Double = 16000 // iOS 通常支持 16kHz
@ -507,35 +520,41 @@ print("stopAudioProcessing")
interleaved: true interleaved: true
) )
} }
public func startAudioRecord() { /**
* 开始音频输入
*/
public func startAudioInput() {
isWriting=true isWriting=true
print("startAudioRecord=audioSourceType\(audioSourceType)") print("startAudioInput=audioSourceType\(audioSourceType)")
switch audioSourceType { switch audioSourceType {
case .microphone: case .microphone:
startMicrophoneCapture() runMicrophoneCapture()
case .external: case .external:
startExternalCapture() runExternalCapture()
} }
} }
public func startWriteThread() { /**
writeThread.async { [weak self] in * 开启音频写入线程
guard let self = self else { return } */
public func startWriteThread() {
while self.isRunning { writeThread.async { [weak self] in
if !self.isWriting { guard let self = self else { return }
usleep(10_000)
continue while self.isRunning {
if !self.isWriting {
usleep(10_000)
continue
}
guard let dataToWrite = self.writeQueue.take() else { continue }
//print("写入数据长度: \(dataToWrite.count)")
self.pushAudioStream?.write(dataToWrite)
}
} }
guard let dataToWrite = self.writeQueue.take() else { continue }
print("写入数据长度: \(dataToWrite.count)")
self.pushAudioStream?.write(dataToWrite)
} }
}
}
/** /**
* 向音频流写入音频数据 * 向音频流写入音频数据
* 仅当音频源设置为external时有效 * 仅当音频源设置为external时有效
@ -550,198 +569,178 @@ print("stopAudioProcessing")
// 放入队列,由写线程写入 // 放入队列,由写线程写入
writeQueue.put(data) writeQueue.put(data)
} }
private func startMicrophoneCapture() {
do {
let audioSession = AVAudioSession.sharedInstance()
// 改用 playAndRecord 模式(同时支持播放和录音)
try audioSession.setCategory(
.playAndRecord,
mode: .default,
options: [.allowBluetooth, .defaultToSpeaker]
)
try audioSession.setActive(true, options: .notifyOthersOnDeactivation)
audioEngine = AVAudioEngine()
guard let inputNode = audioEngine?.inputNode else {
throw NSError(domain: "AudioSetup", code: 1)
}
// Use hardware's native format
let hardwareFormat = inputNode.inputFormat(forBus: 0)
// Create converter to target format
guard let targetFormat = audioFormat,
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2)
}
inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) {
[weak self] buffer, time in
guard let self = self, self.isWriting else { return }
// Convert to target format
let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate)
)!
///print("进入 tap 回调,frameLength: \(buffer.frameLength)")
var error: NSError?
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
outStatus.pointee = .haveData
return buffer
}
)
//print("转换状态: \(status.rawValue), 错误: \(String(describing: error))")
if status == .haveData, error == nil { private func runMicrophoneCapture() {
let data = self.audioBufferToData(convertedBuffer) do {
// print("准备写入数据,大小: \(data.count)") //self.setAudioOutputRoute(.speaker)
audioEngine = AVAudioEngine()
self.writeQueue.put(data) guard let inputNode = audioEngine?.inputNode else {
throw NSError(domain: "AudioSetup", code: 1)
}
// 推荐直接用系统 format
let hardwareFormat = inputNode.inputFormat(forBus: 0)
// // ===== 新增:启用专业级语音处理 =====
if #available(iOS 13.0, *) {
try inputNode.setVoiceProcessingEnabled(true)
print("Voice processing enabled")
}
// Create converter to target format
guard let targetFormat = audioFormat,
let converter = AVAudioConverter(from: hardwareFormat, to: targetFormat) else {
throw NSError(domain: "AudioSetup", code: 2)
}
inputNode.installTap(onBus: 0, bufferSize: UInt32(bufferSize), format: hardwareFormat) {
[weak self] buffer, time in
guard let self = self, self.isWriting else { return }
// Convert to target format
let convertedBuffer = AVAudioPCMBuffer(
pcmFormat: targetFormat,
frameCapacity: AVAudioFrameCount(targetFormat.sampleRate * Double(buffer.frameLength) / buffer.format.sampleRate)
)!
///print("进入 tap 回调,frameLength: \(buffer.frameLength)")
var error: NSError?
let status = converter.convert(
to: convertedBuffer,
error: &error,
withInputFrom: { inNumPackets, outStatus in
outStatus.pointee = .haveData
return buffer
}
)
//print("转换状态: \(status.rawValue), 错误: \(String(describing: error))")
if status == .haveData, error == nil {
let data = self.audioBufferToData(convertedBuffer)
// print("准备写入数据,大小: \(data.count)")
self.writeQueue.put(data)
}
}
try audioEngine?.start()
} catch {
print("麦克风启动失败: \(error)")
} }
} }
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data {
try audioEngine?.start() let frameLength = Int(buffer.frameLength)
} catch { let channelCount = 1
print("麦克风启动失败: \(error)") let dataLength = frameLength * channelCount * MemoryLayout<Int16>.size
}
} // Handle 16-bit integer format
if let int16Data = buffer.int16ChannelData {
private func audioBufferToData(_ buffer: AVAudioPCMBuffer) -> Data { return Data(
let frameLength = Int(buffer.frameLength) bytes: int16Data.pointee,
let channelCount = 1 count: dataLength
let dataLength = frameLength * channelCount * MemoryLayout<Int16>.size )
}
// Handle 16-bit integer format // Handle float format
if let int16Data = buffer.int16ChannelData { else if let floatData = buffer.floatChannelData {
return Data( var int16Array = [Int16](repeating: 0, count: frameLength)
bytes: int16Data.pointee, let floatBuffer = floatData.pointee
count: dataLength
) for i in 0..<frameLength {
} let sample = floatBuffer[i]
// Handle float format let clamped = max(-1.0, min(sample, 1.0))
else if let floatData = buffer.floatChannelData { let scaled = clamped * Float(Int16.max)
var int16Array = [Int16](repeating: 0, count: frameLength) int16Array[i] = Int16(scaled)
let floatBuffer = floatData.pointee }
for i in 0..<frameLength { return Data(
let sample = floatBuffer[i] bytes: int16Array,
let clamped = max(-1.0, min(sample, 1.0)) count: dataLength
let scaled = clamped * Float(Int16.max) )
int16Array[i] = Int16(scaled) }
}
return Data(
bytes: int16Array,
count: dataLength
)
}
return Data() // Fallback for unsupported formats
}
private func startExternalCapture() {
stopMicrophoneCapture() return Data() // Fallback for unsupported formats
}
private func runExternalCapture() {
if audioSourceType == .external {
//pushAudioData(data: Data())
audioEngine?.stop()
audioEngine?.inputNode.removeTap(onBus: 0)
}
} }
public func stopMicrophoneCapture() { public func stopAudioCapture() {
print("stopMicrophoneCapture\(isWriting)")
guard isWriting==true else { return } guard isWriting==true else { return }
print("stopMicrophoneCapture")
isWriting=false isWriting=false
audioEngine?.stop() audioEngine?.stop()
audioEngine?.inputNode.removeTap(onBus: 0) audioEngine?.inputNode.removeTap(onBus: 0)
writeThread.async { writeThread.async {
self.writeQueue.close() self.writeQueue.close()
} }
// 新增:恢复音频会话
do {
let audioSession = AVAudioSession.sharedInstance()
try audioSession.setActive(false, options: .notifyOthersOnDeactivation)
} catch {
print("恢复音频会话失败: \(error)")
}
} }
public func releaseAudioResources() { public func releaseAudioResources() {
isWriting=false
isRunning=false stopAudioCapture()
isRunning=false
stopMicrophoneCapture() audioEngine = nil
audioEngine = nil
writeThread.async {
self.writeQueue.close()
} }
}
} // 音频路由管理
/** public enum AudioOutputRoute {
* 外部音频拉流 case speaker
* 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式 case receiver
*/ case bluetooth
private class ExternalAudioPullStream: NSObject { }
private(set) var pullStream: SPXPullAudioInputStream! public func setAudioOutputRoute(_ route: AudioOutputRoute) {
private let queue = LinkedBlockingQueue<Data>() let audioSession = AVAudioSession.sharedInstance()
private var closed = false do {
isWriting=false
override init() { try audioEngine?.stop()
super.init() // 先停用以避免冲突
try audioSession.setActive(false)
pullStream = SPXPullAudioInputStream(
readHandler: { [weak self] (data: NSMutableData, size: UInt) -> Int in switch route {
guard let self = self else { return 0 } case .speaker:
return self.read(buffer: data, size: Int(size)) // 使用扬声器时必须用videoChat模式
}, try audioSession.setCategory(
closeHandler: { [weak self] in .playAndRecord,
self?.close() mode: .videoChat,
options: [.allowBluetooth, .mixWithOthers, .defaultToSpeaker]
)
try audioSession.overrideOutputAudioPort(.speaker)
case .receiver:
// 听筒模式使用voiceChat节省资源
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth]
)
try audioSession.overrideOutputAudioPort(.none)
case .bluetooth:
// 完整蓝牙设备支持
try audioSession.setCategory(
.playAndRecord,
mode: .voiceChat,
options: [.allowBluetooth, .allowBluetoothA2DP]
)
try audioSession.overrideOutputAudioPort(.none)
// 不需要override,系统自动路由
} }
)
} // 重新激活
try audioSession.setActive(true, options: [.notifyOthersOnDeactivation])
/**
* 外部调用:推送音频数据到队列
* @param data 音频数据
*/
func pushAudio(_ data: Data) {
if !closed {
queue.put(data)
}
}
/**
* SDK调用:从队列中拉取数据
* @param buffer SDK提供的缓冲区
* @param size 缓冲区大小
* @return 读取的字节数,0表示流结束
*/
private func read(buffer: NSMutableData, size: Int) -> Int {
// 阻塞等待下一块数据
guard let chunk = queue.take() else {
return 0 // 队列已关闭
}
// 如果是空数据,表示流结束 try audioEngine?.start()
if chunk.isEmpty { isWriting=true
return 0 } catch {
print("路由切换失败: \(error)")
} }
let toCopy = min(chunk.count, size)
buffer.append(chunk.prefix(toCopy))
return toCopy
}
/**
* SDK调用:关闭流
*/
func close() {
closed = true
queue.close()
} }
} }
// MARK: - iOS版LinkedBlockingQueue实现
/** /**
* iOS版LinkedBlockingQueue实现 * iOS版LinkedBlockingQueue实现
@ -819,7 +818,7 @@ print("stopAudioProcessing")
} }
} }
} }
// MARK: - 回调函数
/** /**
* 连续识别回调接口 * 连续识别回调接口
*/ */

248
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureSpeechPlugin.swift

@ -37,39 +37,39 @@ import os.log
private var currentAsrCallback: AsrCallbackWrapper? private var currentAsrCallback: AsrCallbackWrapper?
// 插件注册 // 插件注册
public static func register(with registrar: FlutterPluginRegistrar) { public static func register(with registrar: FlutterPluginRegistrar) {
let instance = AzureSpeechPlugin() let instance = AzureSpeechPlugin()
// 初始化ASR通道 // 初始化ASR通道
let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger()) let asrChannel = FlutterMethodChannel(name: "azure_speech/asr", binaryMessenger: registrar.messenger())
registrar.addMethodCallDelegate(instance, channel: asrChannel) registrar.addMethodCallDelegate(instance, channel: asrChannel)
instance.asrChannel = asrChannel instance.asrChannel = asrChannel
// 设置ASR通道处理器 - 使用实例方法 // 设置ASR通道处理器 - 使用实例方法
asrChannel.setMethodCallHandler { [weak instance] (call, result) in asrChannel.setMethodCallHandler { [weak instance] (call, result) in
instance?.handleAsrMethodCall(call, result: result) instance?.handleAsrMethodCall(call, result: result)
} }
// 初始化TTS通道 // 初始化TTS通道
let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger()) let ttsChannel = FlutterMethodChannel(name: "azure_speech/tts", binaryMessenger: registrar.messenger())
registrar.addMethodCallDelegate(instance, channel: ttsChannel) registrar.addMethodCallDelegate(instance, channel: ttsChannel)
instance.ttsChannel = ttsChannel instance.ttsChannel = ttsChannel
// 设置TTS通道处理器 - 使用实例方法 // 设置TTS通道处理器 - 使用实例方法
ttsChannel.setMethodCallHandler { [weak instance] (call, result) in ttsChannel.setMethodCallHandler { [weak instance] (call, result) in
instance?.handleTtsMethodCall(call, result: result) instance?.handleTtsMethodCall(call, result: result)
}
// 初始化ASR事件通道
let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger())
asrEventChannel.setStreamHandler(instance)
instance.asrEventChannel = asrEventChannel
// 初始化TTS事件通道
let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger())
ttsEventChannel.setStreamHandler(instance)
instance.ttsEventChannel = ttsEventChannel
} }
// 初始化ASR事件通道
let asrEventChannel = FlutterEventChannel(name: "azure_speech/asr_events", binaryMessenger: registrar.messenger())
asrEventChannel.setStreamHandler(instance)
instance.asrEventChannel = asrEventChannel
// 初始化TTS事件通道
let ttsEventChannel = FlutterEventChannel(name: "azure_speech/tts_events", binaryMessenger: registrar.messenger())
ttsEventChannel.setStreamHandler(instance)
instance.ttsEventChannel = ttsEventChannel
}
// 发送ASR事件方法 // 发送ASR事件方法
internal func sendAsrEvent(_ event: [String: Any]) { internal func sendAsrEvent(_ event: [String: Any]) {
if asrEventSink == nil { if asrEventSink == nil {
@ -104,7 +104,7 @@ import os.log
} }
} }
// // 处理Flutter方法调用 // // 处理Flutter方法调用
// public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) { // public func handle(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
// switch call.method { // switch call.method {
@ -120,7 +120,7 @@ import os.log
// MARK: - ASR 方法处理 // MARK: - ASR 方法处理
private func handleAsrMethodCall(_ call: FlutterMethodCall, result: @escaping FlutterResult) { private func handleAsrMethodCall(_ call: FlutterMethodCall, result: @escaping FlutterResult) {
switch call.method { switch call.method {
case "initialize": case "initialize":
guard let args = call.arguments as? [String: Any], guard let args = call.arguments as? [String: Any],
@ -171,7 +171,7 @@ import os.log
audioSourceType: audioSourceType audioSourceType: audioSourceType
) )
result(success) result(success)
case "recognizeCallback": case "recognizeCallback":
// 确保事件通道已准备好 // 确保事件通道已准备好
guard asrEventSink != nil else { guard asrEventSink != nil else {
result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil)) result(FlutterError(code: "EVENT_CHANNEL_NOT_READY", message: "事件通道未准备好,无法开始连续识别", details: nil))
@ -185,7 +185,6 @@ import os.log
) )
result(success) result(success)
case "stopContinuousRecognition": case "stopContinuousRecognition":
print("stopContinuousRecognition")
let success = azureAsrHelper.stopContinuousRecognition() let success = azureAsrHelper.stopContinuousRecognition()
currentAsrCallback = nil currentAsrCallback = nil
result(success) result(success)
@ -194,7 +193,7 @@ import os.log
result(azureAsrHelper.isContinuousRecognitionActive()) result(azureAsrHelper.isContinuousRecognitionActive())
case "dispose": case "dispose":
print("dispose") print("dispose")
azureAsrHelper.dispose() azureAsrHelper.dispose()
currentAsrCallback = nil currentAsrCallback = nil
result(true) result(true)
@ -205,81 +204,91 @@ import os.log
result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil)) result(FlutterError(code: "INVALID_ARGUMENTS", message: "音频数据不能为空", details: nil))
return return
} }
// 安全解包版本 // 安全解包版本
guard let audioStream = azureAsrHelper.audioStream else { guard let audioStream = azureAsrHelper.audioStream else {
os_log("音频流未初始化", type: .error) os_log("音频流未初始化", type: .error)
return return
} }
audioStream.saveAudioDataTo(data: audioBytes.data) audioStream.saveAudioDataTo(data: audioBytes.data)
result(true)
case "renameFile":
guard let args = call.arguments as? [String: Any],
let filePath = args["filePath"] as? String,
let newName = args["newName"] as? String else {
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil))
return
}
do {
print("音频文件名称为: \(filePath)")
try azureAsrHelper.renameFile(filePath: filePath, newName: newName)
result(true) result(true)
} catch {
result(FlutterError(code: "RENAMEFILE_ERROR", message: error.localizedDescription, details: nil)) case "renameFile":
} guard let args = call.arguments as? [String: Any],
let filePath = args["filePath"] as? String,
case "moveFile": let newName = args["newName"] as? String else {
guard let args = call.arguments as? [String: Any], result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil))
let sourcePath = args["sourcePath"] as? String, return
let destPath = args["destPath"] as? String else { }
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil)) do {
return print("音频文件名称为: \(filePath)")
} try azureAsrHelper.renameFile(filePath: filePath, newName: newName)
do { result(true)
print("音频文件名称为: \(sourcePath)") } catch {
try azureAsrHelper.moveFile(sourcePath: sourcePath, destPath: destPath) result(FlutterError(code: "RENAMEFILE_ERROR", message: error.localizedDescription, details: nil))
}
case "moveFile":
guard let args = call.arguments as? [String: Any],
let sourcePath = args["sourcePath"] as? String,
let destPath = args["destPath"] as? String else {
result(FlutterError(code: "INVALID_ARGUMENTS", message: "必要的参数不能为空", details: nil))
return
}
do {
print("音频文件名称为: \(sourcePath)")
try azureAsrHelper.moveFile(sourcePath: sourcePath, destPath: destPath)
result(true)
} catch {
result(FlutterError(code: "MOVEFILE_ERROR", message: error.localizedDescription, details: nil))
}
case "enableRecord":
guard let args = call.arguments as? [String: Any],
let filePath = args["filePath"] as? String else {
result(FlutterError(code: "INVALID_ARGUMENTS", message: "filePath 参数不能为空", details: nil))
return
}
do {
print("音频文件名称为: \(filePath)")
try azureAsrHelper.enableRecord(filePath: filePath)
result(true)
} catch {
result(FlutterError(code: "ENABLERECORD_ERROR", message: error.localizedDescription, details: nil))
}
case "pauseRecord":
azureAsrHelper.pauseRecord()
result(true) result(true)
} catch {
result(FlutterError(code: "MOVEFILE_ERROR", message: error.localizedDescription, details: nil)) case "stopRecord":
} guard let args = call.arguments as? [String: Any],
let isSave = args["isSave"] as? Bool else {
case "enableRecord": result(FlutterError(code: "INVALID_ARGUMENTS", message: "isSave 参数不能为空", details: nil))
guard let args = call.arguments as? [String: Any], return
let filePath = args["filePath"] as? String else { }
result(FlutterError(code: "INVALID_ARGUMENTS", message: "filePath 参数不能为空", details: nil)) do {
return try azureAsrHelper.stopRecord(isSave: isSave)
} result(true)
do { } catch {
print("音频文件名称为: \(filePath)") result(FlutterError(code: "STOPRECORD_ERROR", message: error.localizedDescription, details: nil))
try azureAsrHelper.enableRecord(filePath: filePath) }
case "disableBluetoothAudio":
print("disableBluetoothAudio")
azureAsrHelper.disableBluetoothAudio();
azureTtsHelper.setAudioOutputDevice()
result(true) result(true)
} catch {
result(FlutterError(code: "ENABLERECORD_ERROR", message: error.localizedDescription, details: nil))
}
case "restoreOriginalAudioState":
case "pauseRecord": print("restoreOriginalAudioState")
azureAsrHelper.pauseRecord() azureAsrHelper.restoreOriginalAudioState();
result(true) azureTtsHelper.setAudioOutputDevice()
case "stopRecord":
guard let args = call.arguments as? [String: Any],
let isSave = args["isSave"] as? Bool else {
result(FlutterError(code: "INVALID_ARGUMENTS", message: "isSave 参数不能为空", details: nil))
return
}
do {
try azureAsrHelper.stopRecord(isSave: isSave)
result(true) result(true)
} catch {
result(FlutterError(code: "STOPRECORD_ERROR", message: error.localizedDescription, details: nil))
}
case "restoreOriginalAudioState":
result(true)
default: default:
result(FlutterMethodNotImplemented) result(FlutterMethodNotImplemented)
} }
@ -347,8 +356,35 @@ case "restoreOriginalAudioState":
case "isSpeaking": case "isSpeaking":
result(azureTtsHelper.isSpeaking()) result(azureTtsHelper.isSpeaking())
case "release": case "dispose":
azureTtsHelper.dispose() azureTtsHelper.dispose()
result(true)
case "setAudioOutputDevice":
guard let args = call.arguments as? [String: Any],
let type = args["type"] as? Int else {
print("setAudioOutputDevice:type=nil")
return
}
print("setAudioOutputDevice:type=\(type)")
if (type == 0) {
// 默认(如果有耳机选耳机,否则使用系统扬声器)
azureAsrHelper.restoreOriginalAudioState();
azureTtsHelper.setAudioOutputDevice()
} else if (type == 1) {
// 强制使用声器
azureAsrHelper.disableBluetoothAudio();
azureTtsHelper.setAudioOutputDevice()
} else if (type == 2) {
// 强制使用耳机
azureAsrHelper.restoreOriginalAudioState();
azureTtsHelper.setAudioOutputDevice()
}
result(true) result(true)
default: default:

729
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AzureTtsHelper.swift

@ -1,28 +1,11 @@
import Foundation import Foundation
import AVFoundation import AVFoundation
import MicrosoftCognitiveServicesSpeech import MicrosoftCognitiveServicesSpeech
import os.log
// 自定义语音处理组件,提供音频流处理等功能 // 自定义语音处理组件,提供音频流处理等功能
import speech import speech
import os.log public class AzureTtsHelper: NSObject, ITtsService {
/**
* Azure TTS Helper
*
* 基于微软Azure语音服务的TTS实现
* 参考文档: https://learn.microsoft.com/en-us/azure/ai-services/speech-service/how-to-speech-synthesis
*
* 特性:
* - 对外接口保持同步,内部异步处理
* - 使用专用队列保证合成顺序
* - 避免阻塞主线程和其他后台线程
* - 事件回调由调用方处理线程切换
*
* 使用示例:
* let success = ttsHelper.speakOnce("Hello World") // 立即返回,后台异步处理
*/
public class AzureTtsHelper: NSObject, ITtsService {
private let tag = "AzureTtsHelper" private let tag = "AzureTtsHelper"
// 日志对象
private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper") private let log = OSLog(subsystem: "com.azure.speech", category: "AzureTtsHelper")
private static let DEFAULT_LANGUAGE = "zh-CN" private static let DEFAULT_LANGUAGE = "zh-CN"
@ -43,59 +26,51 @@ import os.log
// 事件监听器列表 // 事件监听器列表
private var eventListeners = NSHashTable<AnyObject>.weakObjects() private var eventListeners = NSHashTable<AnyObject>.weakObjects()
private var audioDataListeners = NSHashTable<AnyObject>.weakObjects()
// 流式文本处理的缓冲区 // 流式文本处理的缓冲区
private var streamBuffer = "" private var streamBuffer = ""
private var lastSpeakTime: TimeInterval = 0 private var lastSpeakTime: TimeInterval = 0
// 内部异步处理队列,保证顺序执行 // 内部异步处理队列
private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated) private let synthesisQueue = DispatchQueue(label: "com.azure.tts.synthesis", qos: .userInitiated)
private var isProcessing = false // 添加处理状态标志
private let synthesisGroup = DispatchGroup() private let synthesisGroup = DispatchGroup()
private var pendingTasks: [() -> Void] = [] private var pendingTasks: [() -> Void] = []
private let taskLock = NSLock() private let taskLock = NSLock()
// 自定义音频输出流
private var customAudioOutputStream: SPXPushAudioOutputStream?
private var useInternalPlayer = true
// 记录最后播放的文本
private var lastSpokenText: String?
/** /**
* 初始化语音合成服务 * 初始化语音合成服务
*
* @param appId 服务应用ID
* @param token 服务访问令牌/订阅密钥
* @param resource 服务资源ID/区域(可选)
* @param language 语言代码,如"zh-CN"
* @return 是否初始化成功
*/ */
public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool { public func initialize(ttsAppId: String, ttsAppToken: String, ttsResource: String, language: String) -> Bool {
do { do {
// 创建语音配置 // 创建语音配置
if ttsResource.isEmpty { speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource)
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: "eastasia")
} else {
speechConfig = try SPXSpeechConfiguration(subscription: ttsAppToken, region: ttsResource)
}
// 设置语言
speechConfig?.speechSynthesisLanguage = language speechConfig?.speechSynthesisLanguage = language
currentLanguage = language currentLanguage = language
// 设置音频输出格式 - 使用16k、16位的PCM格式 // 设置音频输出格式 - 与Android一致
speechConfig?.setPropertyTo("Audio16Khz16BitMonoPcm", speechConfig?.setPropertyTo("riff-16khz-16bit-mono-pcm",
byName: "SpeechServiceConnection_SynthOutputFormat") byName: "SpeechServiceConnection_SynthOutputFormat")
// 创建合成器,使用默认音频输出(扬声器) // 设置低延迟属性
synthesizer = try SPXSpeechSynthesizer(speechConfig!) speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_InitialSilenceTimeoutMs")
speechConfig?.setPropertyTo("300", byName: "SpeechServiceConnection_EndSilenceTimeoutMs")
// 设置事件监听
setupEventListeners()
// 设置默认音色 - 中文默认使用晓晓,英文默认使用Jenny // 创建合成器
if language.lowercased().starts(with: "zh") { recreateSynthesizer()
_ = setVoice("zh-CN-XiaoxiaoNeural")
} else {
_ = setVoice("en-US-JennyNeural")
}
// 设置初始化完成 // 设置初始化完成
isInitialized = true isInitialized = true
// 预热TTS引擎
warmupSynthesizer()
return true return true
} catch { } catch {
os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription) os_log("语音合成服务初始化失败: %{public}@", log: log, type: .error, error.localizedDescription)
@ -103,24 +78,25 @@ import os.log
} }
} }
/**
* 设置自定义音频输出流
*/
public func setCustomAudioOutputStream(_ outputStream: SPXPushAudioOutputStream?) {
customAudioOutputStream = outputStream
if isInitialized {
recreateSynthesizer()
}
}
/** /**
* 设置语音角色 * 设置语音角色
*
* @param voiceName 语音角色名称(不同服务的语音角色命名可能不同)
* @return 是否设置成功
*/ */
public func setVoice(_ voiceName: String) -> Bool { public func setVoice(_ voiceName: String) -> Bool {
if !isInitialized { return false } if !isInitialized { return false }
do { do {
currentVoice = voiceName currentVoice = voiceName
speechConfig?.speechSynthesisVoiceName = voiceName
// 更新语音名称
if let config = speechConfig {
config.speechSynthesisVoiceName = voiceName
}
// 重新创建合成器
recreateSynthesizer() recreateSynthesizer()
return true return true
} catch { } catch {
@ -135,173 +111,88 @@ import os.log
/** /**
* 单次合成并播放 * 单次合成并播放
*
* @param text 要合成的文本
* @return 是否成功开始合成
*/ */
public func speakOnce(_ text: String) -> Bool { public func speakOnce(_ text: String) -> Bool {
if !isInitialized { if !isInitialized {
os_log("语音合成未初始化", log: log, type: .error) os_log("语音合成未初始化", log: log, type: .error)
notifyEvent(eventType: .error, params: [
"errorCode": "NOT_INITIALIZED",
"errorMessage": "TTS引擎未初始化"
])
return false return false
} }
// 立即返回成功,内部异步处理 // 清理文本
let cleanedText = cleanTextForTTS(text)
if cleanedText.isEmpty {
return false
}
// 异步处理
enqueueSynthesisTask { enqueueSynthesisTask {
self.performSynthesis(text: text) self.performSynthesis(text: cleanedText)
} }
return true return true
} }
/** /**
* 将合成任务加入队列,保证顺序执行 * 流式合成文本
*/ */
private func enqueueSynthesisTask(_ task: @escaping () -> Void) { public func speakStream(_ text: String) -> Bool {
taskLock.lock() if !isInitialized {
defer { taskLock.unlock() } notifyEvent(eventType: .error, params: [
"errorCode": "NOT_INITIALIZED",
pendingTasks.append(task) "errorMessage": "TTS引擎未初始化"
])
// 如果当前没有任务在执行,开始处理队列 return false
if pendingTasks.count == 1 {
processNextTask()
} }
}
if text.isEmpty {
/** return true
* 处理队列中的下一个任务
*/
private func processNextTask() {
synthesisQueue.async {
self.synthesisGroup.enter()
self.taskLock.lock()
guard !self.pendingTasks.isEmpty else {
self.taskLock.unlock()
self.synthesisGroup.leave()
return
}
let task = self.pendingTasks.removeFirst()
self.taskLock.unlock()
// 执行任务
task()
self.synthesisGroup.leave()
// 处理下一个任务
self.taskLock.lock()
if !self.pendingTasks.isEmpty {
self.taskLock.unlock()
self.processNextTask()
} else {
self.taskLock.unlock()
}
} }
}
/**
* 实际执行合成的方法
*/
private func performSynthesis(text: String) {
// 重置状态
speaking = true
// 生成SSML // 清理文本并添加到缓冲区
let ssml = generateSsml(text) let cleanedText = cleanTextForTTS(text)
streamBuffer.append(cleanedText)
do { // 防抖逻辑 (150ms)
os_log("开始合成: %{public}@", log: log, type: .debug, ssml) let currentTime = Date().timeIntervalSince1970
// 使用同步方法,在后台队列中执行 if currentTime - lastSpeakTime < 0.15 {
let result = try synthesizer?.startSpeakingSsml(ssml) return true
// 检查结果
if let result = result {
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason))
}
} catch {
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription)
notifyEvent(eventType: .error, params: [
"errorCode": "SYNTHESIS_FAILED",
"errorMessage": error.localizedDescription
])
speaking = false
} }
} lastSpeakTime = currentTime
/** let currentText = streamBuffer
* 流式合成文本
* // 标点符号列表 (与Android一致)
* @param text 要合成的文本片段 let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"]
* @return 是否成功处理
*/ // 从后往前查找最后一个标点符号
public func speakStream(_ text: String) -> Bool { var lastPunctuationIndex = -1
if !isInitialized || text.isEmpty { for (i, char) in currentText.enumerated().reversed() {
if !isInitialized { if punctuationMarks.contains(char) {
notifyEvent(eventType: .error, params: [ lastPunctuationIndex = i
"errorCode": "NOT_INITIALIZED", break
"errorMessage": "TTS引擎未初始化"
])
} }
return false
} }
do { // 播放到找到的标点符号
// 添加新文本到缓冲区 if lastPunctuationIndex >= 0 {
streamBuffer.append(text) let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1))
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1)
// 增加500ms防抖逻辑 streamBuffer = String(currentText[startIndex...])
let currentTime = Date().timeIntervalSince1970
if currentTime - lastSpeakTime < 0.6 {
return true
}
lastSpeakTime = currentTime
let currentText = streamBuffer
// 定义标点符号列表 if !textToSpeak.isEmpty {
let punctuationMarks: [Character] = [".", "。", "!", "!", "?", "?", ";", ";", ",", ",", ":", ":", "\n"]
// 查找最后一个标点符号的位置
var lastPunctuationIndex = -1
for (i, char) in currentText.enumerated().reversed() {
if punctuationMarks.contains(char) {
lastPunctuationIndex = i
break
}
}
// 如果找到标点符号,则播放到该标点符号
if lastPunctuationIndex >= 0 {
// 提取要播放的文本(包含标点符号)
let textToSpeak = String(currentText.prefix(lastPunctuationIndex + 1)).trimmingCharacters(in: .whitespacesAndNewlines)
// 剩余的文本保存在缓冲区中
let startIndex = currentText.index(currentText.startIndex, offsetBy: lastPunctuationIndex + 1)
streamBuffer = String(currentText[startIndex...])
// 只有非空文本才播放
if !textToSpeak.isEmpty {
return speakOnce(textToSpeak) return speakOnce(textToSpeak)
}
} }
// 如果没有找到标点符号,则等待更多文本
return true
} catch {
os_log("流式语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription)
notifyEvent(eventType: .error, params: [
"errorCode": "STREAM_FAILED",
"errorMessage": "流式语音合成失败: \(error.localizedDescription)"
])
return false
} }
return true
} }
/** /**
* 刷新并播放流式文本缓冲区中的剩余内容 * 刷新并播放流式文本
*
* @return 是否成功处理
*/ */
public func flushStream() -> Bool { public func flushStream() -> Bool {
if !isInitialized { if !isInitialized {
@ -312,115 +203,71 @@ import os.log
return false return false
} }
do { let remainingText = streamBuffer
// 获取缓冲区中剩余的文本 streamBuffer = ""
let remainingText = streamBuffer.trimmingCharacters(in: .whitespacesAndNewlines)
if remainingText.isEmpty {
// 清空缓冲区 return true
streamBuffer = ""
// 如果缓冲区为空,直接返回成功
if remainingText.isEmpty {
return true
}
// 播放剩余文本
return speakOnce(remainingText)
} catch {
os_log("刷新流式文本失败: %{public}@", log: log, type: .error, error.localizedDescription)
notifyEvent(eventType: .error, params: [
"errorCode": "FLUSH_FAILED",
"errorMessage": "刷新流式文本失败: \(error.localizedDescription)"
])
return false
} }
return speakOnce(remainingText)
} }
/** /**
* 停止语音合成和播放 * 停止语音合成
*
* @return 是否成功停止
*/ */
public func stop() -> Bool { public func stop() -> Bool {
// 立即更新状态
speaking = false speaking = false
// 清除流式缓冲区中的待播放内容
streamBuffer = "" streamBuffer = ""
// 清空待处理的任务队列
taskLock.lock() taskLock.lock()
pendingTasks.removeAll() pendingTasks.removeAll()
taskLock.unlock() taskLock.unlock()
// 在后台队列停止合成器,避免阻塞主线程
synthesisQueue.async { synthesisQueue.async {
if let synthesizer = self.synthesizer { do {
do { try self.synthesizer?.stopSpeaking()
try synthesizer.stopSpeaking() self.notifyEvent(eventType: .synthesisCanceled)
os_log("语音合成已停止", log: self.log, type: .info) } catch {
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription)
// 直接通知停止完成 self.notifyEvent(eventType: .error, params: [
self.notifyEvent(eventType: .synthesisCanceled) "errorCode": "STOP_FAILED",
} catch { "errorMessage": error.localizedDescription
os_log("停止语音合成失败: %{public}@", log: self.log, type: .error, error.localizedDescription) ])
self.notifyEvent(eventType: .error, params: [
"errorCode": "STOP_FAILED",
"errorMessage": error.localizedDescription
])
}
} }
} }
return true return true
} }
/** /**
* 释放资源 * 释放资源
* 在不再需要服务时调用,释放底层资源
*/ */
public func dispose() { public func dispose() {
// 停止播放
_ = stop() _ = stop()
// 等待所有任务完成
synthesisGroup.wait() synthesisGroup.wait()
// 清理资源 synthesisQueue.sync {
synthesisQueue.async { synthesizer = nil
// 释放合成器 speechConfig = nil
self.synthesizer = nil customAudioOutputStream = nil
isInitialized = false
// 释放配置 speaking = false
self.speechConfig = nil
// 清空流缓冲区
self.streamBuffer = ""
// 重置状态
self.isInitialized = false
self.speaking = false
// 清空任务队列 taskLock.lock()
self.taskLock.lock() pendingTasks.removeAll()
self.pendingTasks.removeAll() taskLock.unlock()
self.taskLock.unlock()
} }
} }
/** /**
* 添加TTS事件监听器 * 添加事件监听器
*
* @param listener 事件监听器
*/ */
public func addListener(_ listener: TtsEventListener) { public func addListener(_ listener: TtsEventListener) {
eventListeners.add(listener as AnyObject) eventListeners.add(listener as AnyObject)
} }
/** /**
* 移除TTS事件监听器 * 移除事件监听器
*
* @param listener 要移除的事件监听器
*/ */
public func removeListener(_ listener: TtsEventListener) { public func removeListener(_ listener: TtsEventListener) {
eventListeners.remove(listener as AnyObject) eventListeners.remove(listener as AnyObject)
@ -428,100 +275,179 @@ import os.log
/** /**
* 添加音频数据监听器 * 添加音频数据监听器
* 由于不再支持自定义音频流,此方法实际上不再有效
*
* @param listener 音频数据监听器
*/ */
public func addAudioDataListener(_ listener: AudioDataListener) { public func addAudioDataListener(_ listener: AudioDataListener) {
os_log("警告:不支持音频数据监听器功能", log: log, type: .info) audioDataListeners.add(listener as AnyObject)
} }
/** /**
* 移除音频数据监听器 * 移除音频数据监听器
* 由于不再支持自定义音频流,此方法实际上不再有效
*
* @param listener 要移除的音频数据监听器
*/ */
public func removeAudioDataListener(_ listener: AudioDataListener) { public func removeAudioDataListener(_ listener: AudioDataListener) {
// 不做任何操作 audioDataListeners.remove(listener as AnyObject)
} }
/** /**
* 当前是否正在播放/合成 * 设置是否使用内部播放器
*/ */
public func isSpeaking() -> Bool { public func setUseInternalPlayer(_ useInternalPlayer: Bool) {
return speaking self.useInternalPlayer = useInternalPlayer
if isInitialized {
recreateSynthesizer()
}
} }
// MARK: - 辅助方法
/** /**
* 设置事件监听器 * 设置音频输出设备
*/ */
private func setupEventListeners() { public func setAudioOutputDevice() -> Bool {
guard let synthesizer = synthesizer else { return } // 先暂停当前播放
let wasSpeaking = speaking
if wasSpeaking {
_ = stop()
}
// 添加合成开始事件处理器 // 重新创建合成器以应用新设备
synthesizer.addSynthesisStartedEventHandler { [weak self] _, _ in if isInitialized {
guard let self = self else { return } recreateSynthesizer()
self.notifyEvent(eventType: .synthesisStarted)
} }
// 添加合成中事件处理器 // 如果之前正在播放,恢复播放
synthesizer.addSynthesizingEventHandler { [weak self] _, _ in if wasSpeaking, let lastText = lastSpokenText {
// 可以在这里处理合成中的事件,目前没有特别操作 speakOnce(lastText)
} }
// 添加合成完成事件处理器 return true
synthesizer.addSynthesisCompletedEventHandler { [weak self] _, e in }
guard let self = self else { return } // /**
self.speaking = false // * 设置音频输出设备
self.notifyEvent(eventType: .synthesisCompleted) // */
// public func setAudioOutputDevice(_ device: AudioOutputDevice) -> Bool {
// do {
// let session = AVAudioSession.sharedInstance()
// try session.setCategory(.playAndRecord, options: [.defaultToSpeaker, .allowBluetooth])
//
// switch device {
// case .default:
// try session.overrideOutputAudioPort(.none)
// case .speaker:
// try session.overrideOutputAudioPort(.speaker)
// case .headphones:
// try session.overrideOutputAudioPort(.none)
// }
//
// try session.setActive(true)
// return true
// } catch {
// os_log("设置音频输出设备失败: %{public}@", log: log, type: .error, error.localizedDescription)
// return false
// }
// }
//
// MARK: - 私有方法
/**
* 加入合成任务队列
*/
private func enqueueSynthesisTask(_ task: @escaping () -> Void) {
taskLock.lock()
pendingTasks.append(task)
taskLock.unlock()
// 确保任务被处理
processTasksIfNeeded()
}
/**
* 处理任务队列
*/
private func processTasksIfNeeded() {
taskLock.lock()
// 如果已经在处理中或没有任务,则直接返回
guard !isProcessing && !pendingTasks.isEmpty else {
taskLock.unlock()
return
} }
// 添加合成取消事件处理器 isProcessing = true
synthesizer.addSynthesisCanceledEventHandler { [weak self] _, e in let task = pendingTasks.removeFirst()
guard let self = self else { return } taskLock.unlock()
synthesisQueue.async {
task()
self.speaking = false // 任务完成后,检查是否有更多任务
self.taskLock.lock()
self.isProcessing = false
var params: [String: Any] = [:] // 如果还有任务,递归处理
do { if !self.pendingTasks.isEmpty {
let cancellationDetails = try SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: e.result) self.taskLock.unlock()
if cancellationDetails.reason == SPXCancellationReason.error { self.processTasksIfNeeded()
params["reason"] = String(describing: cancellationDetails.reason.rawValue) } else {
params["errorDetails"] = cancellationDetails.errorDetails ?? "未知错误" self.taskLock.unlock()
}
} catch {
params["errorDetails"] = "获取取消详情失败: \(error.localizedDescription)"
} }
os_log("语音合成取消, %{public}@", log: self.log, type: .info, String(describing: params["errorDetails"] ?? "未知错误"))
self.notifyEvent(eventType: .synthesisCanceled, params: params)
} }
} }
/** /**
* 触发事件通知 * 处理下一个任务
*/ */
private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) { private func processNextTask() {
let event = TtsEvent(type: eventType, params: params) synthesisQueue.async {
self.taskLock.lock()
guard !self.pendingTasks.isEmpty else {
self.taskLock.unlock()
return
}
let task = self.pendingTasks.removeFirst()
self.taskLock.unlock()
task()
self.processNextTask()
}
}
/**
* 执行语音合成
*/
private func performSynthesis(text: String) {
speaking = true
lastSpokenText = text
// 生成SSML
let ssml = generateOptimizedSsml(text)
// 直接通知事件,由调用方处理线程切换 do {
for case let listener as TtsEventListener in self.eventListeners.allObjects { os_log("开始合成: %{public}@", log: log, type: .debug, ssml)
listener.onEvent(event) let result = try synthesizer?.startSpeakingSsml(ssml)
if let result = result {
os_log("合成完成,结果: %{public}@", log: log, type: .debug, String(describing: result.reason))
}
} catch {
speaking = false
os_log("语音合成失败: %{public}@", log: log, type: .error, error.localizedDescription)
notifyEvent(eventType: .error, params: [
"errorCode": "SYNTHESIS_FAILED",
"errorMessage": error.localizedDescription
])
} }
} }
/** /**
* 重新创建合成器 * 重新创建合成器
*/ */
private func recreateSynthesizer() { private func recreateSynthesizer() {
do { do {
// 使用默认音频输出配置创建合成器 // 创建音频配置
synthesizer = try SPXSpeechSynthesizer(speechConfig!) let audioConfig: SPXAudioConfiguration?
if !useInternalPlayer || customAudioOutputStream != nil {
audioConfig = try SPXAudioConfiguration(streamOutput: customAudioOutputStream ?? SPXPushAudioOutputStream())
} else {
audioConfig = nil // 使用默认扬声器
}
// 设置事件监听 // 创建合成器
synthesizer = try SPXSpeechSynthesizer(speechConfig!)
setupEventListeners() setupEventListeners()
} catch { } catch {
os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription) os_log("重新创建合成器失败: %{public}@", log: log, type: .error, error.localizedDescription)
@ -533,72 +459,139 @@ import os.log
} }
/** /**
* 设置语音参数 * 设置事件监听器
*/ */
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool { private func setupEventListeners() {
if !isInitialized { return false } synthesizer?.addSynthesisStartedEventHandler { [weak self] _, _ in
self?.notifyEvent(eventType: .synthesisStarted)
self?.notifyEvent(eventType: .playbackStarted)
}
do { synthesizer?.addSynthesizingEventHandler { [weak self] _, event in
currentRate = formatPercentage(rate) if let audioData = event.result.audioData {
currentPitch = formatPercentage(pitch) self?.notifyAudioData(audioData)
currentVolume = "\(min(max(volume, 0), 100))%" }
return true }
} catch {
os_log("设置语音参数失败: %{public}@", log: log, type: .error, error.localizedDescription) synthesizer?.addSynthesisCompletedEventHandler { [weak self] _, _ in
notifyEvent(eventType: .error, params: [ guard let self = self else { return }
"errorCode": "PARAMS_SET_FAILED", self.speaking = false
"errorMessage": "设置语音参数失败: \(error.localizedDescription)" self.notifyEvent(eventType: .synthesisCompleted)
]) self.notifyEvent(eventType: .playbackCompleted)
return false }
synthesizer?.addSynthesisCanceledEventHandler { [weak self] _, event in
guard let self = self else { return }
self.speaking = false
var params: [String: Any] = [:]
if let details = try? SPXSpeechSynthesisCancellationDetails(fromCanceledSynthesisResult: event.result) {
params["errorCode"] = details.errorCode
params["errorDetails"] = details.errorDetails
}
self.notifyEvent(eventType: .synthesisCanceled, params: params)
} }
} }
/** /**
* 格式化百分比值 * 通知事件
*/ */
private func formatPercentage(_ value: Int) -> String { private func notifyEvent(eventType: TtsEventType, params: [String: Any] = [:]) {
return value >= 0 ? "+\(value)%" : "\(value)%" let event = TtsEvent(type: eventType, params: params)
for case let listener as TtsEventListener in eventListeners.allObjects {
listener.onEvent(event)
}
} }
/** /**
* 生成SSML * 通知音频数据
*/ */
private func generateSsml(_ rawText: String) -> String { private func notifyAudioData(_ data: Data) {
// 1. 定义要静音的符号和表情符号列表 for case let listener as AudioDataListener in audioDataListeners.allObjects {
let symbolsToMute = [ listener.onAudioData(data)
"#", "*", }
"😀", "😂", "😊", "😍", "😢", "😎", "😉", "👍", "🙌", "🎉" }
]
/**
// 2. 转义 XML 保留字符 * 预热TTS引擎
var escapedText = rawText */
private func warmupSynthesizer() {
let warmupSsml = """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(currentLanguage)">
<voice name="\(currentVoice)">
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="0%">
.
</prosody>
</voice>
</speak>
"""
synthesisQueue.async {
_ = try? self.synthesizer?.startSpeakingSsml(warmupSsml)
}
}
/**
* 清理TTS文本
*/
private func cleanTextForTTS(_ text: String) -> String {
var cleaned = text
// 移除URL
if let regex = try? NSRegularExpression(pattern: "https?://\\S+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "")
}
// 移除emoji
if let regex = try? NSRegularExpression(pattern: "[\\uD83C-\\uDBFF\\uDC00-\\uDFFF]+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: "")
}
// 合并空格
if let regex = try? NSRegularExpression(pattern: "\\s+", options: .caseInsensitive) {
cleaned = regex.stringByReplacingMatches(in: cleaned, range: NSRange(location: 0, length: cleaned.count), withTemplate: " ")
}
return cleaned.trimmingCharacters(in: .whitespacesAndNewlines)
}
/**
* 生成优化的SSML
*/
private func generateOptimizedSsml(_ rawText: String) -> String {
// 转义XML保留字符
let escapedText = rawText
.replacingOccurrences(of: "&", with: "&amp;") .replacingOccurrences(of: "&", with: "&amp;")
.replacingOccurrences(of: "<", with: "&lt;") .replacingOccurrences(of: "<", with: "&lt;")
.replacingOccurrences(of: ">", with: "&gt;") .replacingOccurrences(of: ">", with: "&gt;")
// 3. 静音处理特殊符号和表情符号 // 简化SSML结构
// 使用空白替换法,直接将符号替换为空字符串 return """
var processedText = escapedText <speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="\(currentLanguage)">
for symbol in symbolsToMute {
processedText = processedText.replacingOccurrences(of: symbol, with: "")
}
// 4. 构造简化的SSML文档,减少嵌套层级
let ssml = """
<speak version="1.0"
xmlns="http://www.w3.org/2001/10/synthesis"
xmlns:mstts="https://www.w3.org/2001/mstts"
xml:lang="zh-CN">
<voice name="\(currentVoice)"> <voice name="\(currentVoice)">
<mstts:express-as style="cheerful"> <prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)">
<prosody rate="\(currentRate)" pitch="\(currentPitch)" volume="\(currentVolume)"> \(escapedText)
<say-as interpret-as="text">\(processedText)</say-as> </prosody>
</prosody>
</mstts:express-as>
</voice> </voice>
</speak> </speak>
""" """
}
return ssml
/**
* 设置语音参数
*/
private func setSpeechParams(rate: Int = 0, pitch: Int = 0, volume: Int = 100) -> Bool {
currentRate = rate >= 0 ? "+\(rate)%" : "\(rate)%"
currentPitch = pitch >= 0 ? "+\(pitch)%" : "\(pitch)%"
currentVolume = "\(min(max(volume, 0), 100))%"
return true
}
/**
* 是否正在播放
*/
public func isSpeaking() -> Bool {
return speaking
} }
} }

Loading…
Cancel
Save