|
|
|
@ -28,51 +28,56 @@ import java.util.concurrent.TimeUnit |
|
|
|
import java.io.File |
|
|
|
import java.io.FileOutputStream |
|
|
|
import java.io.RandomAccessFile |
|
|
|
|
|
|
|
/** |
|
|
|
* Azure语音识别辅助类,支持麦克风和外部音频源的单次和连续语音识别 |
|
|
|
* |
|
|
|
* |
|
|
|
* 针对回音消除的场景,使用拉流模式 (PullAudioInputStreamCallback) 实现,避免复杂的协程处理 |
|
|
|
* 通过检测扬声器和耳机状态,自动决定是否启用回音消除 |
|
|
|
*/ |
|
|
|
class AzureAsrHelper(private val context: Context) { |
|
|
|
private val tag = "AzureAsrHelper" |
|
|
|
|
|
|
|
|
|
|
|
// 核心组件 |
|
|
|
private var speechConfig: SpeechConfig? = null |
|
|
|
private var recognizer: SpeechRecognizer? = null |
|
|
|
private var audioConfig: AudioConfig? = null |
|
|
|
|
|
|
|
|
|
|
|
// 状态管理 |
|
|
|
private var isContinuousRecognitionActive = false |
|
|
|
|
|
|
|
|
|
|
|
// 配置参数 |
|
|
|
private var currentLanguage = "zh-CN" |
|
|
|
private var supportedLanguages = arrayOf("zh-CN") |
|
|
|
private var isAutoDetectLanguage = false |
|
|
|
private var subscriptionKey = "" |
|
|
|
private var region = "" |
|
|
|
|
|
|
|
|
|
|
|
// 音频源配置 |
|
|
|
private var audioSourceType = AudioSourceType.MICROPHONE |
|
|
|
|
|
|
|
var audioSourceType = AudioSourceType.MICROPHONE |
|
|
|
|
|
|
|
// 音频处理 |
|
|
|
private var microphoneStream: MicrophoneStream? = null |
|
|
|
private var externalAudioStream: ExternalAudioPullStream? = null |
|
|
|
|
|
|
|
|
|
|
|
// 录音文件处理 |
|
|
|
private var recordfile: RecordingFile? = null |
|
|
|
private var isRecording = false |
|
|
|
|
|
|
|
/** |
|
|
|
* 音频来源类型 |
|
|
|
*/ |
|
|
|
enum class AudioSourceType { |
|
|
|
/** 使用设备麦克风 */ |
|
|
|
MICROPHONE, |
|
|
|
|
|
|
|
|
|
|
|
/** 使用外部提供的音频数据 */ |
|
|
|
EXTERNAL |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 初始化Azure语音服务 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param subscriptionKey Azure 订阅密钥 |
|
|
|
* @param region Azure 区域 |
|
|
|
* @param supportedLanguages 支持的语言数组 |
|
|
|
@ -91,28 +96,28 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
Log.e(tag, "Azure 配置信息不完整") |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 释放之前的资源 |
|
|
|
dispose() |
|
|
|
|
|
|
|
|
|
|
|
// 保存配置 |
|
|
|
this.subscriptionKey = subscriptionKey |
|
|
|
this.region = region |
|
|
|
this.audioSourceType = audioSourceType |
|
|
|
|
|
|
|
|
|
|
|
// 设置语言 |
|
|
|
if (supportedLanguages.isNotEmpty()) { |
|
|
|
this.supportedLanguages = supportedLanguages |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 根据支持的语言数量决定是否启用自动语言检测 |
|
|
|
this.isAutoDetectLanguage = supportedLanguages.size >= 2 |
|
|
|
|
|
|
|
|
|
|
|
// 如果只有一种语言,设置为当前语言 |
|
|
|
if (!isAutoDetectLanguage && supportedLanguages.isNotEmpty()) { |
|
|
|
this.currentLanguage = supportedLanguages[0] |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 创建语音配置 |
|
|
|
speechConfig = SpeechConfig.fromSubscription(subscriptionKey, region).apply { |
|
|
|
if (isAutoDetectLanguage) { |
|
|
|
@ -122,15 +127,16 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
// 设置指定的识别语言 |
|
|
|
speechRecognitionLanguage = currentLanguage |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 设置静音超时时间(毫秒) |
|
|
|
setProperty("SpeechServiceConnection_EndSilenceTimeoutMs", "800") |
|
|
|
setProperty("Speech_SegmentationSilenceTimeoutMs", "800") |
|
|
|
|
|
|
|
|
|
|
|
// 设置分段策略为时间模式 |
|
|
|
setProperty("Speech_SegmentationStrategy", "Time") |
|
|
|
} |
|
|
|
|
|
|
|
// 录音文件类 |
|
|
|
recordfile = RecordingFile() |
|
|
|
// 创建识别器 |
|
|
|
return setupRecognizer() |
|
|
|
} catch (e: Exception) { |
|
|
|
@ -138,7 +144,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 设置识别器 |
|
|
|
*/ |
|
|
|
@ -147,17 +153,18 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
// 清理旧的识别器 |
|
|
|
recognizer?.close() |
|
|
|
recognizer = null |
|
|
|
|
|
|
|
|
|
|
|
// 设置音频配置 |
|
|
|
when (audioSourceType) { |
|
|
|
AudioSourceType.MICROPHONE -> { |
|
|
|
// 判断是否使用手机扬声器输出和耳机连接状态 |
|
|
|
val audioManager = context.getSystemService(Context.AUDIO_SERVICE) as AudioManager |
|
|
|
val audioManager = |
|
|
|
context.getSystemService(Context.AUDIO_SERVICE) as AudioManager |
|
|
|
val isSpeakerphoneOn = audioManager.isSpeakerphoneOn |
|
|
|
|
|
|
|
|
|
|
|
// 检查是否连接了有线耳机 |
|
|
|
val isWiredHeadsetOn = audioManager.isWiredHeadsetOn |
|
|
|
|
|
|
|
|
|
|
|
// 检查是否连接了蓝牙设备(需要注意兼容性) |
|
|
|
val isBluetoothConnected = try { |
|
|
|
val method = AudioManager::class.java.getMethod("isBluetoothA2dpOn") |
|
|
|
@ -165,12 +172,12 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
} catch (e: Exception) { |
|
|
|
false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
val isHeadsetConnected = isWiredHeadsetOn || isBluetoothConnected |
|
|
|
|
|
|
|
|
|
|
|
// 如果使用手机扬声器输出且没有连接耳机,可能会产生回音,需启动回音消除功能 |
|
|
|
val needEchoCancellation = !isHeadsetConnected |
|
|
|
|
|
|
|
|
|
|
|
if (needEchoCancellation) { |
|
|
|
// 使用拉流方式进行回音消除 |
|
|
|
setupMicrophoneStream() |
|
|
|
@ -178,20 +185,22 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
audioConfig = AudioConfig.fromDefaultMicrophoneInput() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
AudioSourceType.EXTERNAL -> { |
|
|
|
// 改用拉流方式处理外部音频 |
|
|
|
setupExternalAudioStream() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 创建识别器 |
|
|
|
recognizer = if (isAutoDetectLanguage) { |
|
|
|
val autoDetectConfig = AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
|
|
val autoDetectConfig = |
|
|
|
AutoDetectSourceLanguageConfig.fromLanguages(supportedLanguages.toList()) |
|
|
|
SpeechRecognizer(speechConfig, autoDetectConfig, audioConfig) |
|
|
|
} else { |
|
|
|
SpeechRecognizer(speechConfig, audioConfig) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
return true |
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "创建识别器失败: ${e.message}") |
|
|
|
@ -199,7 +208,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 设置麦克风流 - 使用拉流方式 |
|
|
|
*/ |
|
|
|
@ -207,7 +216,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
try { |
|
|
|
// 创建麦克风流 |
|
|
|
microphoneStream = MicrophoneStream() |
|
|
|
|
|
|
|
|
|
|
|
// 创建音频配置 - 正确使用fromStreamInput方法,只传入回调 |
|
|
|
audioConfig = AudioConfig.fromStreamInput(microphoneStream) |
|
|
|
} catch (e: Exception) { |
|
|
|
@ -215,7 +224,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
microphoneStream = null |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 设置外部音频流 - 使用拉流方式 |
|
|
|
*/ |
|
|
|
@ -223,7 +232,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
try { |
|
|
|
// 创建外部音频拉流对象 |
|
|
|
externalAudioStream = ExternalAudioPullStream() |
|
|
|
|
|
|
|
|
|
|
|
// 创建音频配置 |
|
|
|
audioConfig = AudioConfig.fromStreamInput(externalAudioStream) |
|
|
|
} catch (e: Exception) { |
|
|
|
@ -231,56 +240,61 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
externalAudioStream = null |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 向音频流写入音频数据 |
|
|
|
* 仅当音频源设置为EXTERNAL时有效 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param data 音频数据字节数组 |
|
|
|
*/ |
|
|
|
fun pushAudioData(data: ByteArray) { |
|
|
|
if (audioSourceType != AudioSourceType.EXTERNAL) { |
|
|
|
return |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 使用拉流模式,将数据推入队列 |
|
|
|
externalAudioStream?.pushAudio(data) |
|
|
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 执行一次性语音识别 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param callback 识别结果回调 |
|
|
|
*/ |
|
|
|
fun recognizeOnce(callback: RecognizeCallback, audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE) { |
|
|
|
fun recognizeOnce( |
|
|
|
callback: RecognizeCallback, |
|
|
|
audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE |
|
|
|
) { |
|
|
|
if (speechConfig == null) { |
|
|
|
callback.onError("语音服务未初始化") |
|
|
|
return |
|
|
|
} |
|
|
|
this.audioSourceType = audioSourceType |
|
|
|
this.audioSourceType = audioSourceType |
|
|
|
// 确保不在连续识别中 |
|
|
|
if (isContinuousRecognitionActive) { |
|
|
|
stopContinuousRecognition() |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 重置识别器 |
|
|
|
if (!setupRecognizer()) { |
|
|
|
callback.onError("重置识别器失败") |
|
|
|
return |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
try { |
|
|
|
// 启动音频处理 |
|
|
|
startAudioProcessing() |
|
|
|
|
|
|
|
|
|
|
|
// 执行同步识别 |
|
|
|
val result = recognizer?.recognizeOnceAsync()?.get() |
|
|
|
|
|
|
|
|
|
|
|
// 停止音频处理 |
|
|
|
stopAudioProcessing() |
|
|
|
|
|
|
|
|
|
|
|
if (result != null && result.reason == ResultReason.RecognizedSpeech) { |
|
|
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language ?: supportedLanguages[0] |
|
|
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(result)?.language |
|
|
|
?: supportedLanguages[0] |
|
|
|
callback.onResult(result.text, detectedLanguage) |
|
|
|
} else { |
|
|
|
callback.onError("未能识别语音") |
|
|
|
@ -290,19 +304,22 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
callback.onError("识别异常: ${e.message}") |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 开始连续语音识别 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param callback 连续识别结果回调 |
|
|
|
* @return 是否成功开始识别 |
|
|
|
*/ |
|
|
|
fun startContinuousRecognition(callback: ContinuousRecognizeCallback, audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE): Boolean { |
|
|
|
fun startContinuousRecognition( |
|
|
|
callback: ContinuousRecognizeCallback, |
|
|
|
audioSourceType: AudioSourceType = AudioSourceType.MICROPHONE |
|
|
|
): Boolean { |
|
|
|
if (speechConfig == null) { |
|
|
|
callback.onError("语音服务未初始化") |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if (isContinuousRecognitionActive) { |
|
|
|
return true |
|
|
|
} |
|
|
|
@ -312,18 +329,21 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
callback.onError("重置识别器失败") |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
try { |
|
|
|
// 设置各种事件监听 |
|
|
|
setupEventListeners(callback) |
|
|
|
|
|
|
|
|
|
|
|
// 启动音频处理 |
|
|
|
startAudioProcessing() |
|
|
|
|
|
|
|
|
|
|
|
// 开始连续识别 |
|
|
|
recognizer?.startContinuousRecognitionAsync() |
|
|
|
isContinuousRecognitionActive = true |
|
|
|
|
|
|
|
Log.e(tag, "开始连续语音识别 是否录音${isRecording}") |
|
|
|
if (isRecording) { |
|
|
|
enableRecord() |
|
|
|
} |
|
|
|
return true |
|
|
|
} catch (e: Exception) { |
|
|
|
stopAudioProcessing() |
|
|
|
@ -332,7 +352,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 设置事件监听器 |
|
|
|
*/ |
|
|
|
@ -340,7 +360,8 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
// 识别中事件 |
|
|
|
recognizer?.recognizing?.addEventListener( |
|
|
|
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|
|
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
|
|
val detectedLanguage = |
|
|
|
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
|
|
// 直接在当前线程调用回调 |
|
|
|
callback.onRecognizing(event.result.text, detectedLanguage) |
|
|
|
} |
|
|
|
@ -350,7 +371,8 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
recognizer?.recognized?.addEventListener( |
|
|
|
EventHandler<SpeechRecognitionEventArgs> { _, event -> |
|
|
|
if (event.result.reason == ResultReason.RecognizedSpeech) { |
|
|
|
val detectedLanguage = AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
|
|
val detectedLanguage = |
|
|
|
AutoDetectSourceLanguageResult.fromResult(event.result)?.language ?: "" |
|
|
|
// 直接在当前线程调用回调 |
|
|
|
callback.onResult(event.result.text, detectedLanguage) |
|
|
|
} |
|
|
|
@ -385,7 +407,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 停止连续语音识别 |
|
|
|
* |
|
|
|
@ -396,11 +418,11 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
Log.e(tag, "语音服务未初始化") |
|
|
|
return false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
if (!isContinuousRecognitionActive) { |
|
|
|
return true |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
try { |
|
|
|
if (recognizer == null) { |
|
|
|
Log.w(tag, "识别器为空,重置状态") |
|
|
|
@ -410,23 +432,24 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
if (audioSourceType == AudioSourceType.EXTERNAL) { |
|
|
|
pushAudioData(ByteArray(0)) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 直接停止连续识别(SDK内部已是异步操作) |
|
|
|
recognizer?.stopContinuousRecognitionAsync()?.get(1000, TimeUnit.MILLISECONDS) |
|
|
|
|
|
|
|
|
|
|
|
// 停止音频处理 |
|
|
|
stopAudioProcessing() |
|
|
|
|
|
|
|
// 停止录音 |
|
|
|
recordfile!!.closeFile() |
|
|
|
// 会话结束事件会设置isContinuousRecognitionActive = false |
|
|
|
return true |
|
|
|
} catch (e: Exception) { |
|
|
|
// 强制重置状态 |
|
|
|
isContinuousRecognitionActive = false |
|
|
|
Log.e(tag, "停止连续识别失败: ${e.message}") |
|
|
|
|
|
|
|
|
|
|
|
// 停止音频处理 |
|
|
|
stopAudioProcessing() |
|
|
|
|
|
|
|
recordfile!!.closeFile() |
|
|
|
// 尝试强制关闭识别器 |
|
|
|
try { |
|
|
|
recognizer?.close() |
|
|
|
@ -434,16 +457,16 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
} catch (ex: Exception) { |
|
|
|
Log.e(tag, "关闭识别器失败: ${ex.message}") |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
return false |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 检查连续识别是否活跃 |
|
|
|
*/ |
|
|
|
fun isContinuousRecognitionActive(): Boolean = isContinuousRecognitionActive |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 释放所有资源 |
|
|
|
*/ |
|
|
|
@ -455,22 +478,23 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
recognizer?.stopContinuousRecognitionAsync() |
|
|
|
isContinuousRecognitionActive = false |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 停止音频处理 |
|
|
|
stopAudioProcessing() |
|
|
|
|
|
|
|
// 停止录音 |
|
|
|
recordfile!!.closeFile() |
|
|
|
// 释放recognizer |
|
|
|
recognizer?.close() |
|
|
|
recognizer = null |
|
|
|
|
|
|
|
|
|
|
|
// 释放speechConfig |
|
|
|
speechConfig?.close() |
|
|
|
speechConfig = null |
|
|
|
|
|
|
|
|
|
|
|
// 释放音频配置 |
|
|
|
audioConfig?.close() |
|
|
|
audioConfig = null |
|
|
|
|
|
|
|
|
|
|
|
// 确保状态被重置 |
|
|
|
isContinuousRecognitionActive = false |
|
|
|
microphoneStream = null |
|
|
|
@ -488,9 +512,9 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
speechConfig = null |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 音频处理相关方法 |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 启动音频处理 |
|
|
|
*/ |
|
|
|
@ -500,12 +524,13 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
// 拉流模式不需要额外启动,SDK会自动拉取数据 |
|
|
|
// 无需执行任何操作 |
|
|
|
} |
|
|
|
|
|
|
|
AudioSourceType.EXTERNAL -> { |
|
|
|
// 外部音频数据模式下不需要启动处理,等待外部调用pushAudioData |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 停止音频处理 |
|
|
|
*/ |
|
|
|
@ -519,7 +544,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
e.printStackTrace() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
externalAudioStream?.let { |
|
|
|
try { |
|
|
|
it.close() |
|
|
|
@ -530,7 +555,31 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 开启录音 |
|
|
|
*/ |
|
|
|
fun enableRecord() { |
|
|
|
Log.i(tag, "开启录音:") |
|
|
|
isRecording = true |
|
|
|
recordfile!!.closeFile() |
|
|
|
|
|
|
|
|
|
|
|
if (isContinuousRecognitionActive) { |
|
|
|
recordfile!!.creatingFiles() |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 停止录音 |
|
|
|
*/ |
|
|
|
fun stopRecord() { |
|
|
|
Log.i(tag, "停止连续录音:") |
|
|
|
isRecording = false |
|
|
|
recordfile!!.closeFile() |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 麦克风流 - 拉流模式 |
|
|
|
* 实现PullAudioInputStreamCallback,为Azure SDK提供音频数据 |
|
|
|
@ -538,12 +587,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
private inner class MicrophoneStream : PullAudioInputStreamCallback() { |
|
|
|
private var audioRecord: AudioRecord? = null |
|
|
|
private var echoCanceler: AcousticEchoCanceler? = null |
|
|
|
private var currentAudioFile: File? = null |
|
|
|
private var fos: FileOutputStream? = null |
|
|
|
|
|
|
|
// 用于存储音频数据的缓冲区 |
|
|
|
private val dataBuffer = mutableListOf<ByteArray>() |
|
|
|
private var totalBytesWritten = 0 |
|
|
|
|
|
|
|
// 音频配置 |
|
|
|
private val sampleRate = 16000 |
|
|
|
private val channelConfig = AudioFormat.CHANNEL_IN_MONO |
|
|
|
@ -551,30 +595,16 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
private val bufferSize = AudioRecord.getMinBufferSize( |
|
|
|
sampleRate, channelConfig, audioFormat |
|
|
|
).let { if (it < 0) 4096 else it * 2 } |
|
|
|
|
|
|
|
|
|
|
|
init { |
|
|
|
initMicrophone() |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 初始化麦克风 |
|
|
|
*/ |
|
|
|
private fun initMicrophone() { |
|
|
|
try { |
|
|
|
// 创建新的音频文件 |
|
|
|
val timestamp = System.currentTimeMillis() |
|
|
|
val filePath = context.getExternalFilesDir(null)?.absolutePath + "/recorded_audio_$timestamp.wav" |
|
|
|
currentAudioFile = File(filePath) |
|
|
|
currentAudioFile?.createNewFile() |
|
|
|
|
|
|
|
// 仅写入初始文件头 |
|
|
|
fos = FileOutputStream(currentAudioFile).apply { |
|
|
|
write(generateWavHeader(0)) // 初始0长度 |
|
|
|
close() |
|
|
|
} |
|
|
|
|
|
|
|
// 追加音频数据到文件 |
|
|
|
fos = FileOutputStream(currentAudioFile, true) |
|
|
|
// 创建录音对象 |
|
|
|
if (android.os.Build.VERSION.SDK_INT >= android.os.Build.VERSION_CODES.M) { |
|
|
|
val format = AudioFormat.Builder() |
|
|
|
@ -582,7 +612,7 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
.setEncoding(audioFormat) |
|
|
|
.setChannelMask(channelConfig) |
|
|
|
.build() |
|
|
|
|
|
|
|
|
|
|
|
audioRecord = AudioRecord.Builder() |
|
|
|
.setAudioSource(MediaRecorder.AudioSource.VOICE_COMMUNICATION) |
|
|
|
.setAudioFormat(format) |
|
|
|
@ -597,18 +627,18 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
bufferSize |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 检查初始化状态 |
|
|
|
if (audioRecord?.state != AudioRecord.STATE_INITIALIZED) { |
|
|
|
throw IllegalStateException("AudioRecord初始化失败") |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 应用音频效果 |
|
|
|
applyAudioEffects() |
|
|
|
|
|
|
|
|
|
|
|
// 开始录音 |
|
|
|
audioRecord?.startRecording() |
|
|
|
|
|
|
|
|
|
|
|
// 检查录音状态 |
|
|
|
if (audioRecord?.recordingState != AudioRecord.RECORDSTATE_RECORDING) { |
|
|
|
throw IllegalStateException("录音启动失败") |
|
|
|
@ -617,23 +647,24 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
e.printStackTrace() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 应用音频效果 |
|
|
|
*/ |
|
|
|
private fun applyAudioEffects() { |
|
|
|
val sessionId = audioRecord?.audioSessionId ?: return |
|
|
|
|
|
|
|
|
|
|
|
// 回音消除 |
|
|
|
echoCanceler = AcousticEchoCanceler.create(sessionId) |
|
|
|
echoCanceler?.enabled = true; |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 获取音频格式 |
|
|
|
*/ |
|
|
|
fun getFormat(): AudioStreamFormat = AudioStreamFormat.getWaveFormatPCM(16000.toLong(), 16.toShort(), 1.toShort()) |
|
|
|
|
|
|
|
fun getFormat(): AudioStreamFormat = |
|
|
|
AudioStreamFormat.getWaveFormatPCM(16000.toLong(), 16.toShort(), 1.toShort()) |
|
|
|
|
|
|
|
/** |
|
|
|
* Azure SDK 调用此方法获取音频数据 |
|
|
|
* 这是拉流模式的核心方法,由SDK调用以获取音频数据 |
|
|
|
@ -645,82 +676,25 @@ class AzureAsrHelper(private val context: Context) { |
|
|
|
Log.e(tag, "读取音频数据失败: $result") |
|
|
|
return 0 |
|
|
|
} |
|
|
|
// 将音频数据保存成wav格式的音频文件 |
|
|
|
saveAudioDataToWav(buffer, result) |
|
|
|
Log.i(tag, "麦克风:") |
|
|
|
// 将音频数据保存成wav格式的音频文件 |
|
|
|
recordfile?.saveAudioDataToWav(buffer) |
|
|
|
return result |
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "读取音频异常: ${e.message}") |
|
|
|
return 0 |
|
|
|
} |
|
|
|
} |
|
|
|
/** |
|
|
|
* 保存音频数据到 WAV 文件 |
|
|
|
*/ |
|
|
|
private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
// 复制数据到新数组 |
|
|
|
val copy = ByteArray(length) |
|
|
|
System.arraycopy(buffer, 0, copy, 0, length) |
|
|
|
|
|
|
|
// 添加到缓冲区 |
|
|
|
dataBuffer.add(copy) |
|
|
|
totalBytesWritten += length |
|
|
|
|
|
|
|
// 写入文件 |
|
|
|
try { |
|
|
|
fos?.write(copy, 0, length) |
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "写入音频数据失败: ${e.message}") |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 更新文件头生成(修正RIFF长度计算) |
|
|
|
private fun generateWavHeader(dataLength: Int): ByteArray { |
|
|
|
val totalLength = 36 + dataLength // RIFF块总长度 = 头部36字节 + 音频数据 |
|
|
|
val byteRate = sampleRate * 2 * 1 // 采样率 * 字节/样本 * 通道数 |
|
|
|
|
|
|
|
return byteArrayOf( |
|
|
|
'R'.code.toByte(), 'I'.code.toByte(), 'F'.code.toByte(), 'F'.code.toByte(), |
|
|
|
(totalLength and 0xFF).toByte(), ((totalLength shr 8) and 0xFF).toByte(), |
|
|
|
((totalLength shr 16) and 0xFF).toByte(), ((totalLength shr 24) and 0xFF).toByte(), |
|
|
|
'W'.code.toByte(), 'A'.code.toByte(), 'V'.code.toByte(), 'E'.code.toByte(), |
|
|
|
'f'.code.toByte(), 'm'.code.toByte(), 't'.code.toByte(), ' '.code.toByte(), |
|
|
|
16, 0, 0, 0, // PCM头长度 |
|
|
|
1, 0, // PCM格式 |
|
|
|
1, 0, // 单声道 |
|
|
|
(sampleRate and 0xFF).toByte(), ((sampleRate shr 8) and 0xFF).toByte(), 0, 0, // 采样率 |
|
|
|
(byteRate and 0xFF).toByte(), ((byteRate shr 8) and 0xFF).toByte(), 0, 0, // 字节率 |
|
|
|
2, 0, // 块对齐 (通道数 * 样本位数/8) |
|
|
|
16, 0, // 样本位数 |
|
|
|
'd'.code.toByte(), 'a'.code.toByte(), 't'.code.toByte(), 'a'.code.toByte(), |
|
|
|
(dataLength and 0xFF).toByte(), ((dataLength shr 8) and 0xFF).toByte(), |
|
|
|
((dataLength shr 16) and 0xFF).toByte(), ((dataLength shr 24) and 0xFF).toByte() |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 关闭音频资源 |
|
|
|
*/ |
|
|
|
override fun close() { |
|
|
|
releaseAudioResources() |
|
|
|
try { |
|
|
|
// 关闭文件流 |
|
|
|
fos?.close() |
|
|
|
|
|
|
|
// 更新WAV文件头 |
|
|
|
currentAudioFile?.let { file -> |
|
|
|
RandomAccessFile(file, "rw").use { raf -> |
|
|
|
raf.seek(0) |
|
|
|
raf.write(generateWavHeader(totalBytesWritten)) |
|
|
|
} |
|
|
|
Log.d(tag, "音频文件保存完成: ${file.absolutePath}") |
|
|
|
} |
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "更新WAV文件头失败: ${e.message}") |
|
|
|
} finally { |
|
|
|
fos = null |
|
|
|
currentAudioFile = null |
|
|
|
} |
|
|
|
|
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 释放音频资源 |
|
|
|
*/ |
|
|
|
@ -730,11 +704,11 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
if (audioRecord?.recordingState == AudioRecord.RECORDSTATE_RECORDING) { |
|
|
|
audioRecord?.stop() |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
// 释放回音消除器 |
|
|
|
echoCanceler?.release() |
|
|
|
echoCanceler = null |
|
|
|
|
|
|
|
|
|
|
|
// 释放录音实例 |
|
|
|
audioRecord?.release() |
|
|
|
audioRecord = null |
|
|
|
@ -743,7 +717,7 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 外部音频拉流 |
|
|
|
* 实现PullAudioInputStreamCallback,将外部推送的音频数据转换为SDK可拉取的形式 |
|
|
|
@ -751,17 +725,19 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
private inner class ExternalAudioPullStream : PullAudioInputStreamCallback() { |
|
|
|
private val queue: BlockingQueue<ByteArray> = LinkedBlockingQueue() |
|
|
|
private var closed = false |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 外部调用:推送音频数据到队列 |
|
|
|
* @param data 音频数据 |
|
|
|
*/ |
|
|
|
fun pushAudio(data: ByteArray) { |
|
|
|
// 将音频数据保存成wav格式的音频文件 |
|
|
|
recordfile?.saveAudioDataToWav(data) |
|
|
|
if (!closed) { |
|
|
|
queue.offer(data) |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* SDK调用:从队列中拉取数据 |
|
|
|
* @param buffer SDK提供的缓冲区 |
|
|
|
@ -769,10 +745,12 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
*/ |
|
|
|
override fun read(buffer: ByteArray): Int { |
|
|
|
try { |
|
|
|
|
|
|
|
// 阻塞等待下一块数据 |
|
|
|
val chunk = queue.take() |
|
|
|
val toCopy = minOf(chunk.size, buffer.size) |
|
|
|
System.arraycopy(chunk, 0, buffer, 0, toCopy) |
|
|
|
Log.i(tag, "外部音频:") |
|
|
|
return toCopy |
|
|
|
} catch (e: InterruptedException) { |
|
|
|
Thread.currentThread().interrupt() |
|
|
|
@ -782,7 +760,7 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
return 0 |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* SDK调用:关闭流 |
|
|
|
*/ |
|
|
|
@ -791,70 +769,177 @@ private fun saveAudioDataToWav(buffer: ByteArray, length: Int) { |
|
|
|
queue.clear() |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 一次性识别回调接口 |
|
|
|
*/ |
|
|
|
interface RecognizeCallback { |
|
|
|
/** |
|
|
|
* 返回识别结果 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param text 识别的文本 |
|
|
|
* @param detectedLanguage 检测到的语言 |
|
|
|
*/ |
|
|
|
fun onResult(text: String, detectedLanguage: String) |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 识别错误时调用 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param error 错误信息 |
|
|
|
*/ |
|
|
|
fun onError(error: String) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 连续识别回调接口 |
|
|
|
*/ |
|
|
|
interface ContinuousRecognizeCallback { |
|
|
|
/** |
|
|
|
* 返回识别结果 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param text 识别的文本 |
|
|
|
* @param detectedLanguage 检测到的语言 |
|
|
|
*/ |
|
|
|
fun onResult(text: String, detectedLanguage: String) |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 识别进行中调用 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param recognizing 正在识别的文本 |
|
|
|
* @param detectedLanguage 检测到的语言 |
|
|
|
*/ |
|
|
|
fun onRecognizing(recognizing: String, detectedLanguage: String) |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 会话开始时调用 |
|
|
|
*/ |
|
|
|
fun onSessionStarted() |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 会话结束时调用 |
|
|
|
*/ |
|
|
|
fun onSessionStopped() |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 识别取消时调用 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param reason 取消原因 |
|
|
|
* @param errorDetails 错误详情 |
|
|
|
*/ |
|
|
|
fun onCanceled(reason: String, errorDetails: String) |
|
|
|
|
|
|
|
|
|
|
|
/** |
|
|
|
* 识别出错时调用 |
|
|
|
* |
|
|
|
* |
|
|
|
* @param error 错误信息 |
|
|
|
*/ |
|
|
|
fun onError(error: String) |
|
|
|
} |
|
|
|
|
|
|
|
|
|
|
|
private inner class RecordingFile { |
|
|
|
private var currentAudioFile: File? = null |
|
|
|
private var fos: FileOutputStream? = null |
|
|
|
|
|
|
|
// 用于存储音频数据的缓冲区 |
|
|
|
private val dataBuffer = mutableListOf<ByteArray>() |
|
|
|
private var totalBytesWritten = 0 |
|
|
|
|
|
|
|
// 音频配置 |
|
|
|
private val sampleRate = 16000 |
|
|
|
|
|
|
|
|
|
|
|
internal fun creatingFiles() { |
|
|
|
if (fos != null || currentAudioFile != null) { |
|
|
|
return |
|
|
|
} |
|
|
|
// 创建新的音频文件 |
|
|
|
val timestamp = System.currentTimeMillis() |
|
|
|
val filePath = |
|
|
|
context.getExternalFilesDir(null)?.absolutePath + "/recorded_audio_$timestamp.wav" |
|
|
|
currentAudioFile = File(filePath) |
|
|
|
currentAudioFile?.createNewFile() |
|
|
|
|
|
|
|
// 仅写入初始文件头 |
|
|
|
fos = FileOutputStream(currentAudioFile).apply { |
|
|
|
write(generateWavHeader(0)) // 初始0长度 |
|
|
|
close() |
|
|
|
} |
|
|
|
|
|
|
|
// 追加音频数据到文件 |
|
|
|
fos = FileOutputStream(currentAudioFile, true) |
|
|
|
} |
|
|
|
|
|
|
|
/** |
|
|
|
* 保存音频数据到 WAV 文件 |
|
|
|
*/ |
|
|
|
internal fun saveAudioDataToWav(buffer: ByteArray) { |
|
|
|
if (fos == null || currentAudioFile == null) { |
|
|
|
return |
|
|
|
} |
|
|
|
// 复制数据到新数组 |
|
|
|
val copy = ByteArray(buffer.size) |
|
|
|
System.arraycopy(buffer, 0, copy, 0, buffer.size) |
|
|
|
|
|
|
|
// 添加到缓冲区 |
|
|
|
dataBuffer.add(copy) |
|
|
|
totalBytesWritten += buffer.size |
|
|
|
|
|
|
|
// 写入文件 |
|
|
|
try { |
|
|
|
fos?.write(copy, 0, buffer.size) |
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "写入音频数据失败: ${e.message}") |
|
|
|
} |
|
|
|
} |
|
|
|
|
|
|
|
// 更新文件头生成(修正RIFF长度计算) |
|
|
|
private fun generateWavHeader(dataLength: Int): ByteArray { |
|
|
|
val totalLength = 36 + dataLength // RIFF块总长度 = 头部36字节 + 音频数据 |
|
|
|
val byteRate = sampleRate * 2 * 1 // 采样率 * 字节/样本 * 通道数 |
|
|
|
|
|
|
|
return byteArrayOf( |
|
|
|
'R'.code.toByte(), 'I'.code.toByte(), 'F'.code.toByte(), 'F'.code.toByte(), |
|
|
|
(totalLength and 0xFF).toByte(), ((totalLength shr 8) and 0xFF).toByte(), |
|
|
|
((totalLength shr 16) and 0xFF).toByte(), ((totalLength shr 24) and 0xFF).toByte(), |
|
|
|
'W'.code.toByte(), 'A'.code.toByte(), 'V'.code.toByte(), 'E'.code.toByte(), |
|
|
|
'f'.code.toByte(), 'm'.code.toByte(), 't'.code.toByte(), ' '.code.toByte(), |
|
|
|
16, 0, 0, 0, // PCM头长度 |
|
|
|
1, 0, // PCM格式 |
|
|
|
1, 0, // 单声道 |
|
|
|
(sampleRate and 0xFF).toByte(), ((sampleRate shr 8) and 0xFF).toByte(), 0, 0, // 采样率 |
|
|
|
(byteRate and 0xFF).toByte(), ((byteRate shr 8) and 0xFF).toByte(), 0, 0, // 字节率 |
|
|
|
2, 0, // 块对齐 (通道数 * 样本位数/8) |
|
|
|
16, 0, // 样本位数 |
|
|
|
'd'.code.toByte(), 'a'.code.toByte(), 't'.code.toByte(), 'a'.code.toByte(), |
|
|
|
(dataLength and 0xFF).toByte(), ((dataLength shr 8) and 0xFF).toByte(), |
|
|
|
((dataLength shr 16) and 0xFF).toByte(), ((dataLength shr 24) and 0xFF).toByte() |
|
|
|
) |
|
|
|
} |
|
|
|
|
|
|
|
internal fun closeFile() { |
|
|
|
try { |
|
|
|
if (fos == null || currentAudioFile == null) { |
|
|
|
return |
|
|
|
} |
|
|
|
// 关闭文件流 |
|
|
|
fos?.close() |
|
|
|
|
|
|
|
// 更新WAV文件头 |
|
|
|
currentAudioFile?.let { file -> |
|
|
|
RandomAccessFile(file, "rw").use { raf -> |
|
|
|
raf.seek(0) |
|
|
|
raf.write(generateWavHeader(totalBytesWritten)) |
|
|
|
} |
|
|
|
Log.d(tag, "音频文件保存完成: ${file.absolutePath}") |
|
|
|
} |
|
|
|
|
|
|
|
} catch (e: Exception) { |
|
|
|
Log.e(tag, "更新WAV文件头失败: ${e.message}") |
|
|
|
} finally { |
|
|
|
fos = null |
|
|
|
currentAudioFile = null |
|
|
|
} |
|
|
|
} |
|
|
|
} |
|
|
|
} |