Browse Source

提交端到端翻译

newdev_shunjiawei
liwei1dao 6 months ago
parent
commit
956674f0b8
  1. 88
      lib/data/services/language_configs/alibaba_language_config.dart
  2. 423
      lib/data/services/language_configs/ast_config.dart
  3. 1534
      lib/data/services/language_configs/azure_language_config.dart
  4. 296
      lib/data/services/language_configs/google_language_config.dart
  5. 807
      lib/data/services/language_configs/iflytek_language_config.dart
  6. 61
      lib/data/services/language_configs/volcano_language_config.dart
  7. 14
      local_plugins/azure_speech/android/.gitignore
  8. 44
      local_plugins/azure_speech/android/app/build.gradle.kts
  9. 7
      local_plugins/azure_speech/android/app/src/debug/AndroidManifest.xml
  10. 45
      local_plugins/azure_speech/android/app/src/main/AndroidManifest.xml
  11. 5
      local_plugins/azure_speech/android/app/src/main/kotlin/com/example/azure_speech/MainActivity.kt
  12. 12
      local_plugins/azure_speech/android/app/src/main/res/drawable-v21/launch_background.xml
  13. 12
      local_plugins/azure_speech/android/app/src/main/res/drawable/launch_background.xml
  14. BIN
      local_plugins/azure_speech/android/app/src/main/res/mipmap-hdpi/ic_launcher.png
  15. BIN
      local_plugins/azure_speech/android/app/src/main/res/mipmap-mdpi/ic_launcher.png
  16. BIN
      local_plugins/azure_speech/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png
  17. BIN
      local_plugins/azure_speech/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png
  18. BIN
      local_plugins/azure_speech/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png
  19. 18
      local_plugins/azure_speech/android/app/src/main/res/values-night/styles.xml
  20. 18
      local_plugins/azure_speech/android/app/src/main/res/values/styles.xml
  21. 7
      local_plugins/azure_speech/android/app/src/profile/AndroidManifest.xml
  22. 3
      local_plugins/azure_speech/android/gradle.properties
  23. 5
      local_plugins/azure_speech/android/gradle/wrapper/gradle-wrapper.properties
  24. 478
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AliyunBailianE2EHelper.kt
  25. 758
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AstCallbacks.kt
  26. 473
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/DoubaoE2ETranslateHelper.kt
  27. 376
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekAsrHelper.kt
  28. 659
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekAsrToAsr.kt
  29. 290
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekTranslationServiceImpl.kt
  30. 271
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekTtsWs.kt
  31. 250
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftAsrServiceImpl.kt
  32. 266
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTTSServiceImpl.kt
  33. 189
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTranslationAndTtsService.kt
  34. 268
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTranslationServiceImpl.kt
  35. 357
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/VolcanoTranslationServiceImpl.kt
  36. 235
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/ScreenCaptureForegroundService.kt
  37. 318
      local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/ScreenCaptureManager.kt
  38. 10
      local_plugins/azure_speech/android/src/main/protos/HOWTO.md
  39. 38
      local_plugins/azure_speech/android/src/main/protos/build_go.sh
  40. 133
      local_plugins/azure_speech/android/src/main/protos/common/events.proto
  41. 100
      local_plugins/azure_speech/android/src/main/protos/common/rpcmeta.proto
  42. 46
      local_plugins/azure_speech/android/src/main/protos/products/understanding/ast/ast_service.proto
  43. 195
      local_plugins/azure_speech/android/src/main/protos/products/understanding/base/au_base.proto
  44. 427
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift
  45. 650
      local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/DoubaoE2ETranslateHelper.swift
  46. 301
      local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/common/events.pb.swift
  47. 350
      local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/common/rpcmeta.pb.swift
  48. 461
      local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/products/understanding/ast/ast_service.pb.swift
  49. 1803
      local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/products/understanding/base/au_base.pb.swift
  50. 20
      local_plugins/azure_speech/lib/main.dart

88
lib/data/services/language_configs/alibaba_language_config.dart

@ -0,0 +1,88 @@
/// 返回阿里云通义千问的翻译语言配置表(Translation)
///
/// 数据来源:阿里云通义千问翻译支持语种
/// 参考: https://bailian.console.aliyun.com/cn-beijing/?spm=5176.12818093_47.overview_recent.2.3be916d0Dc1NWf&tab=doc#/doc/?type=model&url=2983281
List<Map<String, String>> getAlibabaTranslationLanguageSpecs() {
return [
// 中文系列
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
'translationCode': 'zh',
'ttsVoiceName': 'Cherry',
},
// 常见通用语种
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'translationCode': 'en',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'translationCode': 'ja',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
'translationCode': 'ko',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'translationCode': 'fr',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'translationCode': 'de',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'translationCode': 'es',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
'translationCode': 'it',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
'translationCode': 'pt',
'ttsVoiceName': 'Cherry',
},
{
'shortCode': 'ru',
'chineseName': '俄语',
'englishName': 'Russian',
'translationCode': 'ru',
'ttsVoiceName': 'Cherry',
},
// {
// 'shortCode': 'yue',
// 'chineseName': '粤语',
// 'englishName': 'Cantonese',
// 'translationCode': 'yue',
// },
];
}

423
lib/data/services/language_configs/ast_config.dart

@ -0,0 +1,423 @@
/// 返回AST配置表
List<Map<String, String>> getAstLanguageConfigs() {
return [
// 中文及方言
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
},
{
'shortCode': 'yue',
'chineseName': '粤语',
'englishName': 'Cantonese',
},
// 英语各地区
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
},
// 西班牙语各地区
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
},
// 其他欧洲语言
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
},
{
'shortCode': 'ru',
'chineseName': '俄语',
'englishName': 'Russian',
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
},
{
'shortCode': 'nl',
'chineseName': '荷兰语',
'englishName': 'Dutch',
},
{
'shortCode': 'nb',
'chineseName': '挪威语',
'englishName': 'Norwegian Bokmål',
},
{
'shortCode': 'da',
'chineseName': '丹麦语',
'englishName': 'Danish',
},
{
'shortCode': 'fi',
'chineseName': '芬兰语',
'englishName': 'Finnish',
},
{
'shortCode': 'sv',
'chineseName': '瑞典语',
'englishName': 'Swedish',
},
{
'shortCode': 'pl',
'chineseName': '波兰语',
'englishName': 'Polish',
},
{
'shortCode': 'tr',
'chineseName': '土耳其语',
'englishName': 'Turkish',
'asrCode': 'tr-TR',
},
{
'shortCode': 'el',
'chineseName': '希腊语',
'englishName': 'Greek',
},
{
'shortCode': 'cs',
'chineseName': '捷克语',
'englishName': 'Czech',
},
{
'shortCode': 'ca',
'chineseName': '加泰罗尼亚语',
'englishName': 'Catalan',
},
{
'shortCode': 'cy',
'chineseName': '威尔士语',
'englishName': 'Welsh',
},
{
'shortCode': 'bg',
'chineseName': '保加利亚语',
'englishName': 'Bulgarian',
},
// 亚洲语言
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
},
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
},
{
'shortCode': 'hi',
'chineseName': '印地语',
'englishName': 'Hindi',
},
{
'shortCode': 'th',
'chineseName': '泰语',
'englishName': 'Thai',
},
{
'shortCode': 'vi',
'chineseName': '越南语',
'englishName': 'Vietnamese',
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
},
{
'shortCode': 'ms',
'chineseName': '马来语',
'englishName': 'Malay',
},
{
'shortCode': 'bn',
'chineseName': '孟加拉语',
'englishName': 'Bengali',
},
{
'shortCode': 'am',
'chineseName': '阿姆哈拉语',
'englishName': 'Amharic',
'asrCode': 'am-ET',
},
{
'shortCode': 'as',
'chineseName': '阿萨姆语',
'englishName': 'Assamese',
},
// 中东与非洲
{
'shortCode': 'ar',
'chineseName': '阿拉伯语',
'englishName': 'Arabic',
},
{
'shortCode': 'af',
'chineseName': '南非荷兰语',
'englishName': 'Afrikaans',
},
// 其他
{
'shortCode': 'az',
'chineseName': '阿塞拜疆语',
'englishName': 'Azerbaijani',
},
{
'shortCode': 'bs',
'chineseName': '波斯尼亚语',
'englishName': 'Bosnian',
},
{
'shortCode': 'kn',
'chineseName': '卡纳达语',
'englishName': 'kn-IN',
},
{
'shortCode': 'et',
'chineseName': '爱沙尼亚语',
'englishName': 'Estonian',
},
{
'shortCode': 'eu',
'chineseName': '巴斯克语',
'englishName': 'Basque',
},
{
'shortCode': 'fa',
'chineseName': '波斯语',
'englishName': 'Persian',
},
{
'shortCode': 'fil',
'chineseName': '菲律宾语',
'englishName': 'Filipino',
},
{
'shortCode': 'ga',
'chineseName': '爱尔兰语',
'englishName': 'Ireland',
},
{
'shortCode': 'gl',
'chineseName': '加利西亚语',
'englishName': 'Galician',
},
{
'shortCode': 'gu',
'chineseName': '古吉拉特语',
'englishName': 'Gujarati',
},
{
'shortCode': 'he',
'chineseName': '希伯来语',
'englishName': 'Hebrew',
},
{
'shortCode': 'hr',
'chineseName': '克罗地亚语',
'englishName': 'Croatian',
},
{
'shortCode': 'hu',
'chineseName': '匈牙利语',
'englishName': 'Hungarian',
},
{
'shortCode': 'hy',
'chineseName': '亚美尼亚语',
'englishName': 'Armenian',
},
{
'shortCode': 'is',
'chineseName': '冰岛语',
'englishName': 'Icelandic',
},
{
'shortCode': 'jv',
'chineseName': '爪哇语',
'englishName': 'Javanese',
},
{
'shortCode': 'ka',
'chineseName': '格鲁吉亚语',
'englishName': 'Georgian',
},
{
'shortCode': 'kk',
'chineseName': '哈萨克语',
'englishName': 'Kazakh',
},
{
'shortCode': 'km',
'chineseName': '高棉语',
'englishName': 'Khmer',
},
{
'shortCode': 'lo',
'chineseName': '老挝语',
'englishName': 'Lao',
},
{
'shortCode': 'lt',
'chineseName': '立陶宛语',
'englishName': 'Lithuanian',
},
{
'shortCode': 'lv',
'chineseName': '拉脱维亚语',
'englishName': 'Latvian',
},
{
'shortCode': 'mk',
'chineseName': '马其顿语',
'englishName': 'Macedonian',
},
{
'shortCode': 'ml',
'chineseName': '马拉雅拉姆语',
'englishName': 'Malayalam',
},
{
'shortCode': 'mn',
'chineseName': '蒙古语',
'englishName': 'Mongolian',
},
{
'shortCode': 'mr',
'chineseName': '马拉地语',
'englishName': 'Marathi',
},
{
'shortCode': 'mt',
'chineseName': '马耳他语',
'englishName': 'Maltese',
},
{
'shortCode': 'my',
'chineseName': '缅甸语',
'englishName': 'Burmese',
},
{
'shortCode': 'nb',
'chineseName': '书面挪威语',
'englishName': 'Norwegian Bokmål',
},
{
'shortCode': 'ne',
'chineseName': '尼泊尔语',
'englishName': 'Nepali',
},
// {
// 'shortCode': 'or',
// 'chineseName': '奥里亚语',
// 'englishName': 'Oriya',
// },
// {
// 'shortCode': 'pa',
// 'chineseName': '旁遮普语',
// 'englishName': 'Punjabi',
// },
{
'shortCode': 'ps',
'chineseName': '普什图语',
'englishName': 'Pushto',
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
},
{
'shortCode': 'ro',
'chineseName': '罗马尼亚语',
'englishName': 'Romanian',
},
{
'shortCode': 'si',
'chineseName': '僧伽罗语',
'englishName': 'Sinhala',
},
{
'shortCode': 'sk',
'chineseName': '斯洛伐克语',
'englishName': 'Slovak',
},
{
'shortCode': 'sl',
'chineseName': '斯洛文尼亚语',
'englishName': 'Slovenian',
},
{
'shortCode': 'so',
'chineseName': '索马里语',
'englishName': 'Somali',
},
{
'shortCode': 'sq',
'chineseName': '阿尔巴尼亚语',
'englishName': 'Albanian',
},
{
'shortCode': 'sr',
'chineseName': '塞尔维亚语',
'englishName': 'Serbian',
},
{
'shortCode': 'sv',
'chineseName': '斯瓦希里语',
'englishName': 'Swedish',
},
{
'shortCode': 'ta',
'chineseName': '泰米尔语',
'englishName': 'Tamil',
},
{
'shortCode': 'te',
'chineseName': '泰卢固语',
'englishName': 'Tamil',
},
{
'shortCode': 'uk',
'chineseName': '乌克兰语',
'englishName': 'Ukrainian',
},
{
'shortCode': 'ur',
'chineseName': '乌尔都语',
'englishName': 'Urdu',
},
{
'shortCode': 'uz',
'chineseName': '乌兹别克语',
'englishName': 'Uzbek',
},
{
'shortCode': 'zu',
'chineseName': '祖鲁语',
'englishName': 'Zulu',
},
];
}

1534
lib/data/services/language_configs/azure_language_config.dart

File diff suppressed because it is too large

296
lib/data/services/language_configs/google_language_config.dart

@ -0,0 +1,296 @@
/// 返回谷歌的识别语言配置表(ASR)
List<Map<String, String>> getGoogleASRLanguageSpecs() {
return [
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese (simplified)',
'asrCode': 'zh-CN',
'ttsVoiceName': ''
},
{
'shortCode': 'zh-Hant',
'chineseName': '中文(繁体)',
'englishName': 'Chinese (traditional)',
'asrCode': 'zh-TW',
'ttsVoiceName': ''
},
{
'shortCode': 'zh-Hant-hk',
'chineseName': '中文(香港繁体)',
'englishName': 'Chinese (Hongkong traditional)',
'asrCode': 'zh-HK',
'ttsVoiceName': ''
},
{
'shortCode': 'yue',
'chineseName': '广东话',
'englishName': 'Cantonese',
'asrCode': 'yue-HK',
'ttsVoiceName': ''
},
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'asrCode': 'en-US',
'ttsVoiceName': ''
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'asrCode': 'ja-JP',
'ttsVoiceName': ''
},
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
'asrCode': 'ko-KR',
'ttsVoiceName': ''
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'asrCode': 'fr-FR',
'ttsVoiceName': ''
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'asrCode': 'es-ES',
'ttsVoiceName': ''
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
'asrCode': 'pt-BR',
'ttsVoiceName': ''
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
'asrCode': 'it-IT',
'ttsVoiceName': ''
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'asrCode': 'de-DE',
'ttsVoiceName': ''
},
{
'shortCode': 'ru',
'chineseName': '俄语',
'englishName': 'Russian',
'asrCode': 'ru-RU',
'ttsVoiceName': ''
},
{
'shortCode': 'pl',
'chineseName': '波兰语',
'englishName': 'Polish',
'asrCode': 'pl-PL',
'ttsVoiceName': ''
},
{
'shortCode': 'nl',
'chineseName': '荷兰语',
'englishName': 'Dutch',
'asrCode': 'nl-NL',
'ttsVoiceName': ''
},
{
'shortCode': 'sv',
'chineseName': '瑞典语',
'englishName': 'Swedish',
'asrCode': 'sv-SE',
'ttsVoiceName': ''
},
{
'shortCode': 'cs',
'chineseName': '捷克语',
'englishName': 'Czech',
'asrCode': 'cs-CZ',
'ttsVoiceName': ''
},
{
'shortCode': 'el',
'chineseName': '现代希腊语',
'englishName': 'Modern Greek',
'asrCode': 'el-GR',
'ttsVoiceName': ''
},
{
'shortCode': 'ro',
'chineseName': '罗马尼亚语',
'englishName': 'Romanian',
'asrCode': 'ro-RO',
'ttsVoiceName': ''
},
{
'shortCode': 'hu',
'chineseName': '匈牙利语',
'englishName': 'Hungarian',
'asrCode': 'hu-HU',
'ttsVoiceName': ''
},
{
'shortCode': 'uk',
'chineseName': '乌克兰语',
'englishName': 'Ukrainian',
'asrCode': 'uk-UA',
'ttsVoiceName': ''
},
{
'shortCode': 'da',
'chineseName': '丹麦语',
'englishName': 'Danish',
'asrCode': 'da-DK',
'ttsVoiceName': ''
},
{
'shortCode': 'fi',
'chineseName': '芬兰语',
'englishName': 'Finnish',
'asrCode': 'fi-FI',
'ttsVoiceName': ''
},
{
'shortCode': 'no',
'chineseName': '挪威语',
'englishName': 'Norwegian',
'asrCode': 'nb-NO',
'ttsVoiceName': ''
},
{
'shortCode': 'hr',
'chineseName': '克罗地亚语',
'englishName': 'Croatian',
'asrCode': 'hr-HR',
'ttsVoiceName': ''
},
{
'shortCode': 'ca',
'chineseName': '加泰隆语',
'englishName': 'Catalan',
'asrCode': 'ca-ES',
'ttsVoiceName': ''
},
{
'shortCode': 'ar',
'chineseName': '阿拉伯语',
'englishName': 'Arabic',
'asrCode': 'ar-EG',
'ttsVoiceName': ''
},
{
'shortCode': 'th',
'chineseName': '泰语',
'englishName': 'Thai',
'asrCode': 'th-TH',
'ttsVoiceName': ''
},
{
'shortCode': 'vi',
'chineseName': '越南语',
'englishName': 'Vietnamese',
'asrCode': 'vi-VN',
'ttsVoiceName': ''
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
'asrCode': 'id-ID',
'ttsVoiceName': ''
},
{
'shortCode': 'hi',
'chineseName': '印地语',
'englishName': 'Hindi',
'asrCode': 'hi-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'tr',
'chineseName': '土耳其语',
'englishName': 'Turkish',
'asrCode': 'tr-TR',
'ttsVoiceName': ''
},
{
'shortCode': 'ms',
'chineseName': '马来语',
'englishName': 'Malay',
'asrCode': 'ms-MY',
'ttsVoiceName': ''
},
{
'shortCode': 'bn',
'chineseName': '孟加拉语',
'englishName': 'Bengali',
'asrCode': 'bn-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'ta',
'chineseName': '泰米尔语',
'englishName': 'Tamil',
'asrCode': 'ta-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'te',
'chineseName': '泰卢固语',
'englishName': 'Telugu',
'asrCode': 'te-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'mr',
'chineseName': '马拉提语',
'englishName': 'Marathi',
'asrCode': 'mr-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'ur',
'chineseName': '乌尔都语',
'englishName': 'Urdu',
'asrCode': 'ur-IN',
'ttsVoiceName': ''
},
{
'shortCode': 'fa',
'chineseName': '波斯语',
'englishName': 'Persian',
'asrCode': 'fa-IR',
'ttsVoiceName': ''
},
{
'shortCode': 'sw',
'chineseName': '斯瓦希里语',
'englishName': 'Swahili',
'asrCode': 'sw-KE',
'ttsVoiceName': ''
},
];
}
/// 返回谷歌的翻译语言配置表(Translation)
List<Map<String, String>> getGoogleTranslationLanguageSpecs() {
// 谷歌翻译能力覆盖广泛,采用与ASR相同的语言集合
return getGoogleASRLanguageSpecs();
}
/// 返回谷歌的TTS语言配置表(TTS,保留空语音名以示未配置)
List<Map<String, String>> getGoogleTTSLanguageSpecs() {
return getGoogleASRLanguageSpecs();
}

807
lib/data/services/language_configs/iflytek_language_config.dart

@ -0,0 +1,807 @@
/// 返回科大讯飞的识别语言配置表(ASR,大模型)
///
/// 说明:基于“实时语音转写大模型”,`lang` 参数采用如下两种模式:
/// - `autodialect`:支持中文+英文及202种方言的免切识别
/// - `autominor`:支持37种多语种的免切识别(需开通对应能力)
///
/// 配置策略:
/// - 中文、英文及各中文方言的 `asrCode` 统一映射为 `autodialect`
/// - 其他小语种的 `asrCode` 统一映射为 `autominor`
List<Map<String, String>> getIflytekASRLanguageSpecs() {
return [
// 中文/英文(中英+方言免切)
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
'asrCode': 'zh-CN',
},
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'asrCode': 'en-US',
},
// 中文方言(示例列出常用的11种方言)
{
'shortCode': 'yue',
'chineseName': '粤语',
'englishName': 'Cantonese',
'asrCode': 'yue-HK',
},
{
'shortCode': 'sichuan',
'chineseName': '四川话',
'englishName': 'Sichuan Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'hunan',
'chineseName': '湖南话',
'englishName': 'Hunan Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'henan',
'chineseName': '河南话',
'englishName': 'Henan Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'dongbei',
'chineseName': '东北话',
'englishName': 'Northeastern Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'shaanxi',
'chineseName': '陕西话',
'englishName': 'Shaanxi Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'tw',
'chineseName': '台湾普通话',
'englishName': 'Taiwan Mandarin',
'asrCode': 'zh-TW',
},
{
'shortCode': 'shandong',
'chineseName': '山东话',
'englishName': 'Shandong Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'hubei',
'chineseName': '湖北话',
'englishName': 'Hubei Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'hefei',
'chineseName': '合肥话',
'englishName': 'Hefei Dialect',
'asrCode': 'zh-CN',
},
{
'shortCode': 'neimenggu',
'chineseName': '内蒙古方言',
'englishName': 'Inner Mongolia Dialect',
'asrCode': 'zh-CN',
},
// 37种多语种(免切识别)
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
'asrCode': 'ko-KR',
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'asrCode': 'ja-JP',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'asrCode': 'fr-FR',
},
{
'shortCode': 'ru',
'chineseName': '俄语',
'englishName': 'Russian',
'asrCode': 'ru-RU',
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'asrCode': 'es-ES',
},
{
'shortCode': 'hi',
'chineseName': '印地语',
'englishName': 'Hindi',
'asrCode': 'hi-IN',
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'asrCode': 'de-DE',
},
{
'shortCode': 'vi',
'chineseName': '越南语',
'englishName': 'Vietnamese',
'asrCode': 'vi-VN',
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
'asrCode': 'pt-PT',
},
{
'shortCode': 'pt-BR',
'chineseName': '巴西葡萄牙语',
'englishName': 'Brazilian Portuguese',
'asrCode': 'pt-BR',
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
'asrCode': 'it-IT',
},
{
'shortCode': 'th',
'chineseName': '泰语',
'englishName': 'Thai',
'asrCode': 'th-TH',
},
{
'shortCode': 'ur',
'chineseName': '乌尔都语',
'englishName': 'Urdu',
'asrCode': 'ur-PK',
},
{
'shortCode': 'pl',
'chineseName': '波兰语',
'englishName': 'Polish',
'asrCode': 'pl-PL',
},
{
'shortCode': 'ms',
'chineseName': '马来语',
'englishName': 'Malay',
'asrCode': 'ms-MY',
},
{
'shortCode': 'ar',
'chineseName': '阿拉伯语',
'englishName': 'Arabic',
'asrCode': 'ar-SA',
},
{
'shortCode': 'kk',
'chineseName': '哈萨克语',
'englishName': 'Kazakh',
'asrCode': 'kk-KZ',
},
{
'shortCode': 'sv',
'chineseName': '瑞典语',
'englishName': 'Swedish',
'asrCode': 'sv-SE',
},
{
'shortCode': 'fa',
'chineseName': '波斯语',
'englishName': 'Persian',
'asrCode': 'fa-IR',
},
{
'shortCode': 'bg',
'chineseName': '保加利亚语',
'englishName': 'Bulgarian',
'asrCode': 'bg-BG',
},
{
'shortCode': 'bo',
'chineseName': '藏语',
'englishName': 'Tibetan',
'asrCode': 'bo-CN',
},
{
'shortCode': 'uz',
'chineseName': '乌兹别克语',
'englishName': 'Uzbek',
'asrCode': 'uz-UZ',
},
{
'shortCode': 'nl',
'chineseName': '荷兰语',
'englishName': 'Dutch',
'asrCode': 'nl-NL',
},
{
'shortCode': 'uk',
'chineseName': '乌克兰语',
'englishName': 'Ukrainian',
'asrCode': 'uk-UA',
},
{
'shortCode': 'ta',
'chineseName': '泰米尔语',
'englishName': 'Tamil',
'asrCode': 'ta-IN',
},
{
'shortCode': 'sw',
'chineseName': '斯瓦希里语',
'englishName': 'Swahili',
'asrCode': 'sw-KE',
},
{
'shortCode': 'ro',
'chineseName': '罗马尼亚语',
'englishName': 'Romanian',
'asrCode': 'ro-RO',
},
{
'shortCode': 'ha',
'chineseName': '豪萨语',
'englishName': 'Hausa',
'asrCode': 'ha-NG',
},
{
'shortCode': 'cs',
'chineseName': '捷克语',
'englishName': 'Czech',
'asrCode': 'cs-CZ',
},
{
'shortCode': 'el',
'chineseName': '希腊语',
'englishName': 'Greek',
'asrCode': 'el-GR',
},
{
'shortCode': 'bn',
'chineseName': '孟加拉语',
'englishName': 'Bengali',
'asrCode': 'bn-IN',
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
'asrCode': 'id-ID',
},
{
'shortCode': 'fil',
'chineseName': '菲律宾语',
'englishName': 'Filipino',
'asrCode': 'fil-PH',
},
{
'shortCode': 'tr',
'chineseName': '土耳其语',
'englishName': 'Turkish',
'asrCode': 'tr-TR',
},
{
'shortCode': 'ug',
'chineseName': '维吾尔语',
'englishName': 'Uyghur',
'asrCode': 'ug-CN',
},
{
'shortCode': 'mn',
'chineseName': '蒙古语',
'englishName': 'Mongolian',
'asrCode': 'mn-MN',
},
];
}
/// 返回科大讯飞的翻译语言配置表(Translation)
List<Map<String, String>> getIflytekTranslationLanguageSpecs() {
return [
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
'translationCode': 'cn',
},
{
'shortCode': 'yue',
'chineseName': '粤语',
'englishName': 'Cantonese',
'translationCode': 'yue',
},
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'translationCode': 'en',
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'translationCode': 'es',
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'translationCode': 'de',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'translationCode': 'fr',
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
'translationCode': 'it',
},
{
'shortCode': 'nl',
'chineseName': '荷兰语',
'englishName': 'Dutch',
'translationCode': 'nl',
},
{
'shortCode': 'nb',
'chineseName': '挪威语',
'englishName': 'Norwegian Bokmål',
'translationCode': 'nb',
},
{
'shortCode': 'da',
'chineseName': '丹麦语',
'englishName': 'Danish',
'translationCode': 'da',
},
{
'shortCode': 'fi',
'chineseName': '芬兰语',
'englishName': 'Finnish',
'translationCode': 'fi',
},
{
'shortCode': 'sv',
'chineseName': '瑞典语',
'englishName': 'Swedish',
'translationCode': 'sv-SE',
},
{
'shortCode': 'pl',
'chineseName': '波兰语',
'englishName': 'Polish',
'translationCode': 'pl-PL',
},
{
'shortCode': 'tr',
'chineseName': '土耳其语',
'englishName': 'Turkish',
'translationCode': 'tr',
},
{
'shortCode': 'el',
'chineseName': '希腊语',
'englishName': 'Greek',
'translationCode': 'el',
},
{
'shortCode': 'cs',
'chineseName': '捷克语',
'englishName': 'Czech',
'translationCode': 'cs-CZ',
},
{
'shortCode': 'ca',
'chineseName': '加泰罗尼亚语',
'englishName': 'Catalan',
'translationCode': 'ca-ES',
},
{
'shortCode': 'cy',
'chineseName': '威尔士语',
'englishName': 'Welsh',
'translationCode': 'cy',
},
{
'shortCode': 'bg',
'chineseName': '保加利亚语',
'englishName': 'Bulgarian',
'translationCode': 'bg',
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'translationCode': 'ja',
},
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
'translationCode': 'ko',
},
{
'shortCode': 'hi',
'chineseName': '印地语',
'englishName': 'Hindi',
'translationCode': 'hi',
},
{
'shortCode': 'th',
'chineseName': '泰语',
'englishName': 'Thai',
'translationCode': 'th',
},
{
'shortCode': 'vi',
'chineseName': '越南语',
'englishName': 'Vietnamese',
'translationCode': 'vi',
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
'translationCode': 'id',
},
{
'shortCode': 'ms',
'chineseName': '马来语',
'englishName': 'Malay',
'translationCode': 'ms',
},
{
'shortCode': 'bn',
'chineseName': '孟加拉语',
'englishName': 'Bengali',
'translationCode': 'bn',
},
{
'shortCode': 'am',
'chineseName': '阿姆哈拉语',
'englishName': 'Amharic',
'translationCode': 'am',
},
{
'shortCode': 'as',
'chineseName': '阿萨姆语',
'englishName': 'Assamese',
'translationCode': 'as',
},
{
'shortCode': 'ar',
'chineseName': '阿拉伯语',
'englishName': 'Arabic',
'translationCode': 'ar',
},
{
'shortCode': 'bs',
'chineseName': '波斯尼亚语(波黑)',
'englishName': 'Bosnian (Bosnia and Herzegovina)',
'translationCode': 'bs',
},
];
}
/// 返回科大讯飞的TTS语言配置表(TTS)
List<Map<String, String>> getIflytekTTSLanguageSpecs() {
return [
// 基础语种
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
'ttsVoiceName': 'x4_yezi',
},
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'ttsVoiceName': 'x4_enus_luna_assist',
},
{
'shortCode': 'ko',
'chineseName': '韩语',
'englishName': 'Korean',
'ttsVoiceName': 'zhimin',
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'ttsVoiceName': 'qianhui',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'ttsVoiceName': '',
},
{
'shortCode': 'ru',
'chineseName': '俄语',
'englishName': 'Russian',
'ttsVoiceName': 'x2_RuRu_Keshu',
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'ttsVoiceName': 'gabriela',
},
{
'shortCode': 'hi',
'chineseName': '印地语',
'englishName': 'Hindi',
'ttsVoiceName': 'x2_HiIn_Mohita',
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'ttsVoiceName': 'x2_DeDe_Christiane',
},
{
'shortCode': 'vi',
'chineseName': '越南语',
'englishName': 'Vietnamese',
'ttsVoiceName': 'xiaoyun',
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
'ttsVoiceName': 'maria',
},
{
'shortCode': 'pt-BR',
'chineseName': '巴西葡萄牙语',
'englishName': 'Brazilian Portuguese',
'ttsVoiceName': 'sonia',
},
{
'shortCode': 'it',
'chineseName': '意大利语',
'englishName': 'Italian',
'ttsVoiceName': 'x2_ItIt_Anna',
},
{
'shortCode': 'th',
'chineseName': '泰语',
'englishName': 'Thai',
'ttsVoiceName': 'yingying',
},
{
'shortCode': 'ur',
'chineseName': '乌尔都语',
'englishName': 'Urdu',
'ttsVoiceName': 'x2_UrPk_Noreen',
},
{
'shortCode': 'pl',
'chineseName': '波兰语',
'englishName': 'Polish',
'ttsVoiceName': 'x2_PlPl_Malgorzata',
},
{
'shortCode': 'ms',
'chineseName': '马来语',
'englishName': 'Malay',
'ttsVoiceName': 'x2_MsMy_Hashim',
},
{
'shortCode': 'ar',
'chineseName': '阿拉伯语',
'englishName': 'Arabic',
'ttsVoiceName': 'x2_ArEn_Rania',
},
{
'shortCode': 'kk',
'chineseName': '哈萨克语',
'englishName': 'Kazakh',
'ttsVoiceName': 'x2_kzkz_Irina',
},
{
'shortCode': 'sv',
'chineseName': '瑞典语',
'englishName': 'Swedish',
'ttsVoiceName': 'x2_SvSe_Michaela',
},
{
'shortCode': 'fa',
'chineseName': '波斯语',
'englishName': 'Persian',
'ttsVoiceName': 'x2_FaIr_Saheli',
},
{
'shortCode': 'bg',
'chineseName': '保加利亚语',
'englishName': 'Bulgarian',
'ttsVoiceName': 'x2_BgBg_Zlati',
},
{
'shortCode': 'bo',
'chineseName': '藏语',
'englishName': 'Tibetan',
'ttsVoiceName': 'x2_BoCn_YangJin',
},
{
'shortCode': 'uz',
'chineseName': '乌兹别克语',
'englishName': 'Uzbek',
'ttsVoiceName': 'x2_UzUz_Nigina',
},
{
'shortCode': 'nl',
'chineseName': '荷兰语',
'englishName': 'Dutch',
'ttsVoiceName': 'x2_NlNl_Robin',
},
{
'shortCode': 'uk',
'chineseName': '乌克兰语',
'englishName': 'Ukrainian',
'ttsVoiceName': 'x2_UkUa_Halyna',
},
{
'shortCode': 'ta',
'chineseName': '泰米尔语',
'englishName': 'Tamil',
'ttsVoiceName': 'x2_TaIn_Udaya',
},
{
'shortCode': 'sw',
'chineseName': '斯瓦希里语',
'englishName': 'Swahili',
'ttsVoiceName': 'x2_SwTz_Hamdan',
},
{
'shortCode': 'ro',
'chineseName': '罗马尼亚语',
'englishName': 'Romanian',
'ttsVoiceName': 'x2_RoRo_Miruna',
},
{
'shortCode': 'ha',
'chineseName': '豪萨语',
'englishName': 'Hausa',
'ttsVoiceName': 'x2_HaNg_Zainab',
},
{
'shortCode': 'cs',
'chineseName': '捷克语',
'englishName': 'Czech',
'ttsVoiceName': 'x2_CsCz_Petra',
},
{
'shortCode': 'el',
'chineseName': '希腊语',
'englishName': 'Greek',
'ttsVoiceName': 'x2_ElGr_Dimitra',
},
{
'shortCode': 'bn',
'chineseName': '孟加拉语',
'englishName': 'Bengali',
'ttsVoiceName': 'x2_BnBd_Elmy',
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
'ttsVoiceName': 'x2_IdId_Fira',
},
{
'shortCode': 'fil',
'chineseName': '菲律宾语',
'englishName': 'Filipino',
'ttsVoiceName': 'x2_TlPh_Grace',
},
{
'shortCode': 'tr',
'chineseName': '土耳其语',
'englishName': 'Turkish',
'ttsVoiceName': 'x2_trtr_cenk',
},
{
'shortCode': 'ug',
'chineseName': '维吾尔语',
'englishName': 'Uyghur',
'ttsVoiceName': '',
},
{
'shortCode': 'mn',
'chineseName': '蒙古语',
'englishName': 'Mongolian',
'ttsVoiceName': '',
},
// 方言(部分方言在SDK中可能不支持,仅WebAPI支持)
{
'shortCode': 'yue',
'chineseName': '粤语',
'englishName': 'Cantonese',
'ttsVoiceName': 'x3_xiaoyue',
},
{
'shortCode': 'sichuan',
'chineseName': '四川话',
'englishName': 'Sichuan Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'hunan',
'chineseName': '湖南话',
'englishName': 'Hunan Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'henan',
'chineseName': '河南话',
'englishName': 'Henan Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'dongbei',
'chineseName': '东北话',
'englishName': 'Northeastern Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'shaanxi',
'chineseName': '陕西话',
'englishName': 'Shaanxi Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'tw',
'chineseName': '台湾普通话',
'englishName': 'Taiwan Mandarin',
'ttsVoiceName': '',
},
{
'shortCode': 'shandong',
'chineseName': '山东话',
'englishName': 'Shandong Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'hubei',
'chineseName': '湖北话',
'englishName': 'Hubei Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'hefei',
'chineseName': '合肥话',
'englishName': 'Hefei Dialect',
'ttsVoiceName': '',
},
{
'shortCode': 'neimenggu',
'chineseName': '内蒙古方言',
'englishName': 'Inner Mongolia Dialect',
'ttsVoiceName': '',
},
];
}

61
lib/data/services/language_configs/volcano_language_config.dart

@ -0,0 +1,61 @@
/// 返回火山引擎的翻译语言配置表(Translation)
///
/// 数据来源:火山引擎机器翻译支持语种
/// 参考: https://www.volcengine.com/docs/4640/35107?lang=zh
List<Map<String, String>> getVolcanoTranslationLanguageSpecs() {
return [
// 中文系列
{
'shortCode': 'zh',
'chineseName': '中文',
'englishName': 'Chinese',
'translationCode': 'zh',
},
// 常见通用语种
{
'shortCode': 'en',
'chineseName': '英语',
'englishName': 'English',
'translationCode': 'en',
},
{
'shortCode': 'ja',
'chineseName': '日语',
'englishName': 'Japanese',
'translationCode': 'ja',
},
{
'shortCode': 'id',
'chineseName': '印尼语',
'englishName': 'Indonesian',
'translationCode': 'id',
},
{
'shortCode': 'es',
'chineseName': '西班牙语',
'englishName': 'Spanish',
'translationCode': 'es',
},
{
'shortCode': 'pt',
'chineseName': '葡萄牙语',
'englishName': 'Portuguese',
'translationCode': 'pt',
},
{
'shortCode': 'fr',
'chineseName': '法语',
'englishName': 'French',
'translationCode': 'fr',
},
{
'shortCode': 'de',
'chineseName': '德语',
'englishName': 'German',
'translationCode': 'de',
},
];
}

14
local_plugins/azure_speech/android/.gitignore

@ -0,0 +1,14 @@
gradle-wrapper.jar
/.gradle
/captures/
/gradlew
/gradlew.bat
/local.properties
GeneratedPluginRegistrant.java
.cxx/
# Remember to never publicly share your keystore.
# See https://flutter.dev/to/reference-keystore
key.properties
**/*.keystore
**/*.jks

44
local_plugins/azure_speech/android/app/build.gradle.kts

@ -0,0 +1,44 @@
plugins {
id("com.android.application")
id("kotlin-android")
// The Flutter Gradle Plugin must be applied after the Android and Kotlin Gradle plugins.
id("dev.flutter.flutter-gradle-plugin")
}
android {
namespace = "com.example.azure_speech"
compileSdk = flutter.compileSdkVersion
ndkVersion = flutter.ndkVersion
compileOptions {
sourceCompatibility = JavaVersion.VERSION_11
targetCompatibility = JavaVersion.VERSION_11
}
kotlinOptions {
jvmTarget = JavaVersion.VERSION_11.toString()
}
defaultConfig {
// TODO: Specify your own unique Application ID (https://developer.android.com/studio/build/application-id.html).
applicationId = "com.example.azure_speech"
// You can update the following values to match your application needs.
// For more information, see: https://flutter.dev/to/review-gradle-config.
minSdk = flutter.minSdkVersion
targetSdk = flutter.targetSdkVersion
versionCode = flutter.versionCode
versionName = flutter.versionName
}
buildTypes {
release {
// TODO: Add your own signing config for the release build.
// Signing with the debug keys for now, so `flutter run --release` works.
signingConfig = signingConfigs.getByName("debug")
}
}
}
flutter {
source = "../.."
}

7
local_plugins/azure_speech/android/app/src/debug/AndroidManifest.xml

@ -0,0 +1,7 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
<!-- The INTERNET permission is required for development. Specifically,
the Flutter tool needs it to communicate with the running application
to allow setting breakpoints, to provide hot reload, etc.
-->
<uses-permission android:name="android.permission.INTERNET"/>
</manifest>

45
local_plugins/azure_speech/android/app/src/main/AndroidManifest.xml

@ -0,0 +1,45 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
<application
android:label="azure_speech"
android:name="${applicationName}"
android:icon="@mipmap/ic_launcher">
<activity
android:name=".MainActivity"
android:exported="true"
android:launchMode="singleTop"
android:taskAffinity=""
android:theme="@style/LaunchTheme"
android:configChanges="orientation|keyboardHidden|keyboard|screenSize|smallestScreenSize|locale|layoutDirection|fontScale|screenLayout|density|uiMode"
android:hardwareAccelerated="true"
android:windowSoftInputMode="adjustResize">
<!-- Specifies an Android theme to apply to this Activity as soon as
the Android process has started. This theme is visible to the user
while the Flutter UI initializes. After that, this theme continues
to determine the Window background behind the Flutter UI. -->
<meta-data
android:name="io.flutter.embedding.android.NormalTheme"
android:resource="@style/NormalTheme"
/>
<intent-filter>
<action android:name="android.intent.action.MAIN"/>
<category android:name="android.intent.category.LAUNCHER"/>
</intent-filter>
</activity>
<!-- Don't delete the meta-data below.
This is used by the Flutter tool to generate GeneratedPluginRegistrant.java -->
<meta-data
android:name="flutterEmbedding"
android:value="2" />
</application>
<!-- Required to query activities that can process text, see:
https://developer.android.com/training/package-visibility and
https://developer.android.com/reference/android/content/Intent#ACTION_PROCESS_TEXT.
In particular, this is used by the Flutter engine in io.flutter.plugin.text.ProcessTextPlugin. -->
<queries>
<intent>
<action android:name="android.intent.action.PROCESS_TEXT"/>
<data android:mimeType="text/plain"/>
</intent>
</queries>
</manifest>

5
local_plugins/azure_speech/android/app/src/main/kotlin/com/example/azure_speech/MainActivity.kt

@ -0,0 +1,5 @@
package com.example.azure_speech
import io.flutter.embedding.android.FlutterActivity
class MainActivity : FlutterActivity()

12
local_plugins/azure_speech/android/app/src/main/res/drawable-v21/launch_background.xml

@ -0,0 +1,12 @@
<?xml version="1.0" encoding="utf-8"?>
<!-- Modify this file to customize your launch splash screen -->
<layer-list xmlns:android="http://schemas.android.com/apk/res/android">
<item android:drawable="?android:colorBackground" />
<!-- You can insert your own image assets here -->
<!-- <item>
<bitmap
android:gravity="center"
android:src="@mipmap/launch_image" />
</item> -->
</layer-list>

12
local_plugins/azure_speech/android/app/src/main/res/drawable/launch_background.xml

@ -0,0 +1,12 @@
<?xml version="1.0" encoding="utf-8"?>
<!-- Modify this file to customize your launch splash screen -->
<layer-list xmlns:android="http://schemas.android.com/apk/res/android">
<item android:drawable="@android:color/white" />
<!-- You can insert your own image assets here -->
<!-- <item>
<bitmap
android:gravity="center"
android:src="@mipmap/launch_image" />
</item> -->
</layer-list>

BIN
local_plugins/azure_speech/android/app/src/main/res/mipmap-hdpi/ic_launcher.png

Binary file not shown.

After

Width:  |  Height:  |  Size: 544 B

BIN
local_plugins/azure_speech/android/app/src/main/res/mipmap-mdpi/ic_launcher.png

Binary file not shown.

After

Width:  |  Height:  |  Size: 442 B

BIN
local_plugins/azure_speech/android/app/src/main/res/mipmap-xhdpi/ic_launcher.png

Binary file not shown.

After

Width:  |  Height:  |  Size: 721 B

BIN
local_plugins/azure_speech/android/app/src/main/res/mipmap-xxhdpi/ic_launcher.png

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.0 KiB

BIN
local_plugins/azure_speech/android/app/src/main/res/mipmap-xxxhdpi/ic_launcher.png

Binary file not shown.

After

Width:  |  Height:  |  Size: 1.4 KiB

18
local_plugins/azure_speech/android/app/src/main/res/values-night/styles.xml

@ -0,0 +1,18 @@
<?xml version="1.0" encoding="utf-8"?>
<resources>
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is on -->
<style name="LaunchTheme" parent="@android:style/Theme.Black.NoTitleBar">
<!-- Show a splash screen on the activity. Automatically removed when
the Flutter engine draws its first frame -->
<item name="android:windowBackground">@drawable/launch_background</item>
</style>
<!-- Theme applied to the Android Window as soon as the process has started.
This theme determines the color of the Android Window while your
Flutter UI initializes, as well as behind your Flutter UI while its
running.
This Theme is only used starting with V2 of Flutter's Android embedding. -->
<style name="NormalTheme" parent="@android:style/Theme.Black.NoTitleBar">
<item name="android:windowBackground">?android:colorBackground</item>
</style>
</resources>

18
local_plugins/azure_speech/android/app/src/main/res/values/styles.xml

@ -0,0 +1,18 @@
<?xml version="1.0" encoding="utf-8"?>
<resources>
<!-- Theme applied to the Android Window while the process is starting when the OS's Dark Mode setting is off -->
<style name="LaunchTheme" parent="@android:style/Theme.Light.NoTitleBar">
<!-- Show a splash screen on the activity. Automatically removed when
the Flutter engine draws its first frame -->
<item name="android:windowBackground">@drawable/launch_background</item>
</style>
<!-- Theme applied to the Android Window as soon as the process has started.
This theme determines the color of the Android Window while your
Flutter UI initializes, as well as behind your Flutter UI while its
running.
This Theme is only used starting with V2 of Flutter's Android embedding. -->
<style name="NormalTheme" parent="@android:style/Theme.Light.NoTitleBar">
<item name="android:windowBackground">?android:colorBackground</item>
</style>
</resources>

7
local_plugins/azure_speech/android/app/src/profile/AndroidManifest.xml

@ -0,0 +1,7 @@
<manifest xmlns:android="http://schemas.android.com/apk/res/android">
<!-- The INTERNET permission is required for development. Specifically,
the Flutter tool needs it to communicate with the running application
to allow setting breakpoints, to provide hot reload, etc.
-->
<uses-permission android:name="android.permission.INTERNET"/>
</manifest>

3
local_plugins/azure_speech/android/gradle.properties

@ -0,0 +1,3 @@
org.gradle.jvmargs=-Xmx8G -XX:MaxMetaspaceSize=4G -XX:ReservedCodeCacheSize=512m -XX:+HeapDumpOnOutOfMemoryError
android.useAndroidX=true
android.enableJetifier=true

5
local_plugins/azure_speech/android/gradle/wrapper/gradle-wrapper.properties

@ -0,0 +1,5 @@
distributionBase=GRADLE_USER_HOME
distributionPath=wrapper/dists
zipStoreBase=GRADLE_USER_HOME
zipStorePath=wrapper/dists
distributionUrl=https\://services.gradle.org/distributions/gradle-8.12-all.zip

478
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AliyunBailianE2EHelper.kt

@ -0,0 +1,478 @@
package com.yunqiinnovation.azure_speech
import android.content.Context
import android.util.Log
import okhttp3.*
import okio.ByteString
import okio.ByteString.Companion.toByteString
import org.json.JSONObject
import java.util.UUID
import java.util.concurrent.TimeUnit
import java.util.concurrent.atomic.AtomicBoolean
import kotlinx.coroutines.*
/**
* 阿里云百炼(Bailian)实时音视频翻译助手 (qwen3-livetranslate-flash-realtime)。
* 参考阿里云 "实时音视频翻译-通义千问" WebSocket API 实现。
* 采用 Session-based 协议 (类似 OpenAI Realtime API)。
*/
class AliyunBailianE2EHelper(
private val context: Context
) {
private val TAG = "AliyunBailianHelper"
interface Callback {
/** 会话开始 */
fun onSessionStarted(sessionId: String)
/** 增量文本到达(大模型回复的文本/翻译结果) */
fun onPartialText(sessionId: String, text: String)
/** 增量源文本到达(用户语音的实时识别结果) */
fun onPartialSourceText(sessionId: String, text: String)
/** 源文本最终结果到达(用户一句话识别完成) */
fun onFinalSourceText(sessionId: String, finalText: String)
/** 翻译/回复最终结果到达(大模型回复完成) */
fun onFinalTranslatedText(sessionId: String, finalText: String)
/** 增量音频片段到达(服务端返回的 TTS 音频) */
fun onPartialAudio(sessionId: String, data: ByteArray)
/** 会话结束,返回完整文本与完整合成音频 */
fun onSessionFinished(sessionId: String, finalText: String, finalAudio: ByteArray)
/** 会话失败或取消 */
fun onSessionError(sessionId: String, code: Int, message: String)
}
data class Config(
// 阿里云百炼 WebSocket URL
// 实时音视频翻译通常使用: wss://dashscope.aliyuncs.com/api-ws/v1/realtime
val wsUrl: String = "wss://dashscope.aliyuncs.com/api-ws/v1/realtime",
// 阿里云 API Key
val apiKey: String = "",
// 模型名称,例如 "qwen3-livetranslate-flash-realtime"
val appId: String = "qwen3-livetranslate-flash-realtime",
// 采样率,默认 16000
val sampleRate: Int = 16000,
// 源语言 (e.g. "zh", "en", "ja")
val sourceLanguage: String = "zh",
// 目标语言 (e.g. "en", "zh", "ja")
val targetLanguage: String = "en",
// 音色 (e.g. "Cherry", "Kiki")
val voice: String = ""
)
private var conf: Config = Config()
private var callback: Callback? = null
private var client: OkHttpClient? = null
private var webSocket: WebSocket? = null
private var sessionId: String = ""
// 用于累积文本,以便在结束时返回完整内容
private val fullTextBuffer = StringBuilder()
// 用于累积音频
private val fullAudioBuffer = java.io.ByteArrayOutputStream()
// 当前句子的缓冲
private val recvTextBuffer = StringBuilder()
private val recvText = mutableListOf<String>()
// 音频分块缓冲
private var audioChunkBuffer = ByteArray(0)
private val isStarted = AtomicBoolean(false)
private val scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
/**
* 初始化助手,设置配置与回调。
*
* @param config 配置对象
* @param cb 回调接口
* @return 是否初始化成功
*/
fun initialize(config: Config, cb: Callback): Boolean {
// 注意:通义千问 LiveTranslate 通常使用 2 位语言代码,但部分方言可能不同。
// 这里做一个简单的处理,具体需参考文档支持的语种列表。
conf = config.copy(sourceLanguage = config.sourceLanguage, targetLanguage = config.targetLanguage)
callback = cb
Log.d(TAG, "initialize: wsUrl=${conf.wsUrl}, model=${conf.appId}, src=${conf.sourceLanguage}, tgt=${conf.targetLanguage}")
client = OkHttpClient.Builder()
.pingInterval(30, TimeUnit.SECONDS)
.readTimeout(0, TimeUnit.SECONDS)
.build()
startContinuousConversation()
return true
}
/**
* 启动会话,建立 WebSocket 连接。
*/
fun startContinuousConversation(): Boolean {
if (client == null) return false
if (isStarted.get()) return true
sessionId = UUID.randomUUID().toString()
Log.d(TAG, "startContinuousConversation: sessionId=${sessionId}")
recvTextBuffer.setLength(0)
fullTextBuffer.setLength(0)
fullAudioBuffer.reset()
// 构造 URL,必须包含 model 参数
// wss://dashscope.aliyuncs.com/api-ws/v1/realtime?model=qwen3-livetranslate-flash-realtime
val urlWithModel = if (conf.wsUrl.contains("?")) {
// 如果已经包含了参数,假设用户自己拼接了 model 或者其他参数
conf.wsUrl
} else {
// 强制拼接 model=qwen3-livetranslate-flash-realtime
"${conf.wsUrl}?model=qwen3-livetranslate-flash-realtime"
}
val request = Request.Builder()
.url(urlWithModel)
.header("Authorization", "Bearer ${conf.apiKey}")
.build()
val listener = object : WebSocketListener() {
override fun onOpen(ws: WebSocket, response: Response) {
Log.d(TAG, "onOpen: code=${response.code}")
webSocket = ws
isStarted.set(true)
// 发送 session.update 配置会话
sendSessionUpdate(ws)
callback?.onSessionStarted(sessionId)
}
override fun onMessage(ws: WebSocket, text: String) {
handleJsonMessage(text)
}
override fun onMessage(ws: WebSocket, bytes: ByteString) {
// 通常 LiveTranslate 协议使用 JSON 传输所有内容(包括 Base64 音频)
// 但如果协议支持二进制帧,可以在此处理
Log.w(TAG, "Received unexpected binary message: size=${bytes.size}")
}
override fun onFailure(ws: WebSocket, t: Throwable, response: Response?) {
val msg = t.message ?: "unknown"
val code = response?.code ?: -1
Log.e(TAG, "onFailure: ${msg} code=${code}")
// // 检查是否需要自动重连 (例如网络异常)
// // 注意:鉴权失败(401)不应重试,避免死循环或账号封禁
// if (isStarted.get() && code != 401 && code != 403) {
// Log.i(TAG, "onFailure: detected failure (code=$code), triggering auto-restart")
// restartSession()
// return
// }
callback?.onSessionError(sessionId, 1011, msg)
isStarted.set(false)
}
override fun onClosed(ws: WebSocket, code: Int, reason: String) {
Log.d(TAG, "onClosed: code=${code} reason=${reason}")
isStarted.set(false)
callback?.onSessionFinished(sessionId, fullTextBuffer.toString(), fullAudioBuffer.toByteArray())
}
}
Log.d(TAG, "connecting to $urlWithModel")
client!!.newWebSocket(request, listener)
return true
}
/**
* 重启会话(用于异常自动重连)
*/
private fun restartSession() {
Log.i(TAG, "restartSession: performing auto-restart...")
try { webSocket?.close(1000, "restarting") } catch (_: Exception) {}
webSocket = null
isStarted.set(false)
try {
java.util.Timer().schedule(object : java.util.TimerTask() {
override fun run() {
Log.i(TAG, "restartSession: timer task running")
if (!isStarted.get()) {
Log.i(TAG, "restartSession: calling startContinuousConversation")
val result = startContinuousConversation()
Log.i(TAG, "restartSession: startContinuousConversation result=$result")
} else {
Log.i(TAG, "restartSession: already started, skipping")
}
}
}, 200)
} catch (e: Exception) {
Log.e(TAG, "restartSession: timer failed", e)
}
}
private fun sendSessionUpdate(ws: WebSocket) {
try {
val json = JSONObject()
json.put("type", "session.update")
val session = JSONObject()
// 配置输入音频转写(ASR)
val transcription = JSONObject()
transcription.put("model", "qwen3-asr-flash-realtime") // 配合 LiveTranslate 使用的推荐 ASR 模型
transcription.put("language", conf.sourceLanguage) // 源语言
session.put("input_audio_transcription", transcription)
// 配置翻译
val translation = JSONObject()
translation.put("language", conf.targetLanguage) // 目标语言
session.put("translation", translation)
if (conf.voice.isNotEmpty()) {
session.put("voice", conf.voice)
} else {
if(conf.targetLanguage != "yue"){
session.put("voice", "Cherry")//音色
}else{
session.put("voice", "Kiki")//音色
}
}
session.put("input_audio_format", "pcm16")//输入音频格式
session.put("output_audio_format", "pcm24")//输出音频格式
// 配置模态 (文本 + 音频)
val modalities = org.json.JSONArray()
modalities.put("text")
modalities.put("audio")
session.put("modalities", modalities)
json.put("session", session)
ws.send(json.toString())
Log.d(TAG, "Sent session.update: $json")
} catch (e: Exception) {
Log.e(TAG, "Error sending session.update", e)
}
}
/**
* 将 24kHz PCM 音频重采样为 16kHz
*/
private fun resample24kTo16k(input: ByteArray): ByteArray {
// 16 bit PCM
val inputShorts = ShortArray(input.size / 2)
java.nio.ByteBuffer.wrap(input).order(java.nio.ByteOrder.LITTLE_ENDIAN).asShortBuffer().get(inputShorts)
// 24k -> 16k (ratio 1.5)
// Output size = Input size / 1.5 = Input size * 2 / 3
val outputSize = (inputShorts.size * 2) / 3
val outputShorts = ShortArray(outputSize)
for (i in 0 until outputSize) {
val inputIndex = (i * 1.5).toFloat()
val index = inputIndex.toInt()
val frac = inputIndex - index
if (index + 1 < inputShorts.size) {
val val1 = inputShorts[index]
val val2 = inputShorts[index + 1]
outputShorts[i] = (val1 * (1 - frac) + val2 * frac).toInt().toShort()
} else if (index < inputShorts.size) {
outputShorts[i] = inputShorts[index]
}
}
val outputBytes = ByteArray(outputShorts.size * 2)
java.nio.ByteBuffer.wrap(outputBytes).order(java.nio.ByteOrder.LITTLE_ENDIAN).asShortBuffer().put(outputShorts)
return outputBytes
}
private fun handleJsonMessage(text: String) {
try {
val json = JSONObject(text)
val type = json.optString("type")
// Log.d(TAG, "Received event: $type")
when (type) {
"error" -> {
val error = json.optJSONObject("error")
val msg = error?.optString("message") ?: "Unknown error"
val code = error?.optString("code") ?: ""
Log.e(TAG, "Error from server: $code - $msg")
callback?.onSessionError(sessionId, 1012, msg)
}
"session.created" -> {
Log.d(TAG, "Session created")
}
"session.updated" -> {
Log.d(TAG, "Session updated")
}
// 源语言识别结果 (Partial)
"conversation.item.input_audio_transcription.text" -> {
val txt = json.optString("text", "")
if (txt.isNotEmpty()) {
Log.d(TAG, "onPartialSourceText: $txt")
callback?.onPartialSourceText(sessionId, txt)
}
}
// 源语言识别结果 (Final)
"conversation.item.input_audio_transcription.completed" -> {
val item = json.optJSONObject("item")
val content = item?.optJSONArray("content")
// 提取 text
var finalTxt = ""
if (content != null && content.length() > 0) {
for (i in 0 until content.length()) {
val part = content.optJSONObject(i)
if (part?.optString("type") == "input_audio") {
finalTxt = part.optString("transcript", "")
break
}
}
}
if (finalTxt.isEmpty()) {
// 备用:有些事件格式可能直接在 payload 里
finalTxt = json.optString("transcript", "")
}
if (finalTxt.isNotEmpty()) {
Log.d(TAG, "sessionId: ${sessionId}, onFinalSourceText: $finalTxt")
callback?.onFinalSourceText(sessionId, finalTxt)
}
}
// 翻译/生成结果 (Partial Text)
"response.audio_transcript.delta" -> {
val delta = json.optString("delta", "")
if (delta.isNotEmpty()) {
recvTextBuffer.append(delta)
Log.d(TAG, "onPartialText: $delta")
callback?.onPartialText(sessionId, delta)
}
}
// 翻译/生成结果 (Final Text)
"response.audio_transcript.done" -> {
val transcript = json.optString("transcript", "")
val finalText = if (transcript.isNotEmpty()) transcript else recvTextBuffer.toString()
Log.d(TAG, "sessionId: ${sessionId}, onFinalTranslatedText: $finalText")
callback?.onFinalTranslatedText(sessionId, finalText)
if (fullTextBuffer.isNotEmpty()) fullTextBuffer.append(" ")
fullTextBuffer.append(finalText)
recvTextBuffer.setLength(0)
}
// 翻译/生成音频 (Delta)
"response.audio.delta" -> {
val base64Audio = json.optString("delta", "")
if (base64Audio.isNotEmpty()) {
// Log.d(TAG, "event_id: ${UUID.randomUUID().toString()}, onPartialAudio: $base64Audio")
try {
val audioBytes = android.util.Base64.decode(base64Audio, android.util.Base64.DEFAULT)
if (audioBytes != null && audioBytes.isNotEmpty()) {
// 默认输出 24k,需重采样到 16k
val resampled = resample24kTo16k(audioBytes)
try {
fullAudioBuffer.write(resampled)
} catch (_: Exception) {}
Log.d(TAG, "onPartialAudio: rawSize=${audioBytes.size} resampledSize=${resampled.size}")
// 使用 processAudioChunk 进行分块回调
processAudioChunk(resampled)
}
} catch (e: Exception) {
Log.e(TAG, "Base64 decode error", e)
}
}
}
"response.done" -> {
// 一次响应结束
Log.d(TAG, "Response done")
}
else -> {
// Log.d(TAG, "Unhandled event: $type")
}
}
} catch (e: Exception) {
Log.e(TAG, "handleJsonMessage error: ${e.message}")
}
}
private fun processAudioChunk(
incoming: ByteArray,
multiple: Int = 1280
) {
if (incoming.isNotEmpty()) {
audioChunkBuffer += incoming
}
val sendLen = (audioChunkBuffer.size / multiple) * multiple
if (sendLen > 0) {
val chunk = audioChunkBuffer.copyOfRange(0, sendLen)
callback?.onPartialAudio(sessionId, chunk)
audioChunkBuffer = audioChunkBuffer.copyOfRange(sendLen, audioChunkBuffer.size)
}
}
/**
* 推送音频数据。
* LiveTranslate 协议要求发送 input_audio_buffer.append 事件,
* 音频数据需 Base64 编码。
*/
fun pushAudioData(data: ByteArray): Boolean {
val ws = webSocket ?: return false
if (!isStarted.get()) return false
try {
// 将 PCM 数据转为 Base64
val base64Audio = data.toByteString().base64()
val json = JSONObject()
json.put("type", "input_audio_buffer.append")
json.put("audio", base64Audio)
ws.send(json.toString())
return true
} catch (e: Exception) {
Log.e(TAG, "pushAudioData error", e)
return false
}
}
fun stopContinuousConversation(): Boolean {
// LiveTranslate 没有明确的 "stop task" 指令,通常直接 close 连接即可
// 或者发送 commit 强制模型生成(如果处于等待状态)
// 这里简单处理为关闭连接
webSocket?.close(1000, "User stopped")
return true
}
fun dispose() {
try { webSocket?.close(1000, "dispose") } catch (_: Exception) {}
webSocket = null
isStarted.set(false)
scope.cancel()
}
}

758
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/AstCallbacks.kt

@ -0,0 +1,758 @@
package com.yunqiinnovation.azure_speech
import com.yunqiinnovation.azure_speech.utils.FileLogger
import com.example.astclient.DoubaoE2ETranslateHelper
/**
* Interface for sending events back to Flutter/Native
*/
fun interface AstEventSender {
fun send(event: Map<String, Any>)
}
/**
* Interface for writing audio data (PCM)
*/
fun interface AudioWriter {
fun write(data: ByteArray)
}
/**
* Azure AST Callback Implementation
*/
class AzureAstCallback(
private val serviceId: String,
private val direction: String,
private val eventSender: AstEventSender,
private val audioWriter: AudioWriter? = null
) : IntegratedSpeechTranslationService.ServiceEventCallback {
private val tag = "AzureAstCallback"
override fun onServiceInitialized() {
eventSender.send(
mapOf(
"type" to "serviceInitialized",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onRecognizing(
utteranceId: String,
text: String,
language: String,
confidence: Float
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 识别中: $text, 语言: $language, 置信度: $confidence"
)
eventSender.send(
mapOf(
"type" to "recognizing",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"language" to language,
"confidence" to confidence
)
)
}
override fun onRecognized(
utteranceId: String,
text: String,
language: String,
confidence: Float
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 识别到文本: $text, 语言: $language, 置信度: $confidence"
)
eventSender.send(
mapOf(
"type" to "recognized",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"language" to language,
"confidence" to confidence
)
)
}
override fun onTranslated(
utteranceId: String,
originalText: String,
translatedText: String,
targetLanguage: String
) {
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"originalText" to originalText,
"translatedText" to translatedText,
"targetLanguage" to targetLanguage
)
)
}
override fun onInterimTranslated(
utteranceId: String,
originalText: String,
translatedText: String,
targetLanguage: String
) {
eventSender.send(
mapOf(
"type" to "translatedInterim",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"originalText" to originalText,
"translatedText" to translatedText,
"targetLanguage" to targetLanguage
)
)
}
override fun onTranslationStarted(utteranceId: String, text: String) {
eventSender.send(
mapOf(
"type" to "translationStarted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onTranslationFailed(utteranceId: String, text: String, error: String) {
eventSender.send(
mapOf(
"type" to "translationFailed",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"error" to error
)
)
}
override fun onSynthesisStarted(utteranceId: String, text: String) {
eventSender.send(
mapOf(
"type" to "synthesisStarted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onSynthesisCompleted(utteranceId: String, text: String) {
FileLogger.d(tag, "[$serviceId/$direction] 语音合成完成,文本=${text}")
eventSender.send(
mapOf(
"type" to "synthesisCompleted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onSynthesisAudioGenerated(
utteranceId: String,
text: String,
audioData: ByteArray
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 合成音频生成,文本=${text},音频=${audioData.size}"
)
// val pcmData = extractPcmFromWav(audioData) // 原逻辑已注释,直接使用 audioData
val pcmData = audioData
if (pcmData.isNotEmpty()) {
audioWriter?.write(pcmData)
} else {
FileLogger.e(tag, "[$serviceId/$direction] 音频数据为空")
}
}
override fun onSynthesisFailed(utteranceId: String, text: String, error: String) {
eventSender.send(
mapOf(
"type" to "synthesisFailed",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"error" to error
)
)
}
override fun onSynthesisProgress(
utteranceId: String,
text: String,
progress: Float
) {
eventSender.send(
mapOf(
"type" to "synthesisProgress",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"progress" to progress
)
)
}
override fun onRecognitionStarted() {
eventSender.send(
mapOf(
"type" to "recognitionStarted",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onRecognitionStopped() {
eventSender.send(
mapOf(
"type" to "recognitionStopped",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onStateChanged(component: String, isActive: Boolean) {
eventSender.send(
mapOf(
"type" to "stateChanged",
"serviceId" to serviceId,
"direction" to direction,
"component" to component,
"isActive" to isActive
)
)
}
override fun onError(component: String, error: String) {
FileLogger.e(tag, "[$serviceId/$direction] 错误,组件=${component},错误=${error}")
eventSender.send(
mapOf(
"type" to "error",
"serviceId" to serviceId,
"direction" to direction,
"component" to component,
"error" to error
)
)
}
}
/**
* Iflytek AST Callback Implementation
*/
class IflytekAstCallback(
private val serviceId: String,
private val direction: String,
private val eventSender: AstEventSender,
private val audioWriter: AudioWriter? = null
) : IflytekIntegratedSpeechService.ServiceEventCallback {
private val tag = "IflytekAstCallback"
override fun onServiceInitialized() {
eventSender.send(
mapOf(
"type" to "serviceInitialized",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onRecognizing(
utteranceId: String,
text: String,
language: String,
confidence: Float
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 识别中: $text, 语言: $language, 置信度: $confidence"
)
eventSender.send(
mapOf(
"type" to "recognizing",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"language" to language,
"confidence" to confidence
)
)
}
override fun onRecognized(
utteranceId: String,
text: String,
language: String,
confidence: Float
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 识别到文本: $text, 语言: $language, 置信度: $confidence"
)
eventSender.send(
mapOf(
"type" to "recognized",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"language" to language,
"confidence" to confidence
)
)
}
override fun onTranslated(
utteranceId: String,
originalText: String,
translatedText: String,
targetLanguage: String
) {
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"originalText" to originalText,
"translatedText" to translatedText,
"targetLanguage" to targetLanguage
)
)
}
override fun onInterimTranslated(
utteranceId: String,
originalText: String,
translatedText: String,
targetLanguage: String
) {
eventSender.send(
mapOf(
"type" to "translatedInterim",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"originalText" to originalText,
"translatedText" to translatedText,
"targetLanguage" to targetLanguage
)
)
}
override fun onTranslationStarted(utteranceId: String, text: String) {
eventSender.send(
mapOf(
"type" to "translationStarted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onTranslationFailed(utteranceId: String, text: String, error: String) {
eventSender.send(
mapOf(
"type" to "translationFailed",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"error" to error
)
)
}
override fun onSynthesisStarted(utteranceId: String, text: String) {
eventSender.send(
mapOf(
"type" to "synthesisStarted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onSynthesisCompleted(utteranceId: String, text: String) {
FileLogger.d(tag, "[$serviceId/$direction] 语音合成完成,文本=${text}")
eventSender.send(
mapOf(
"type" to "synthesisCompleted",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text
)
)
}
override fun onSynthesisAudioGenerated(
utteranceId: String,
text: String,
audioData: ByteArray
) {
FileLogger.d(
tag,
"[$serviceId/$direction] 合成音频生成,文本=${text},音频=${audioData.size}"
)
val pcmData = audioData
if (pcmData.isNotEmpty()) {
audioWriter?.write(pcmData)
} else {
FileLogger.e(tag, "[$serviceId/$direction] 音频数据为空或不是WAV格式")
}
}
override fun onSynthesisFailed(utteranceId: String, text: String, error: String) {
eventSender.send(
mapOf(
"type" to "synthesisFailed",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"error" to error
)
)
}
override fun onSynthesisProgress(
utteranceId: String,
text: String,
progress: Float
) {
eventSender.send(
mapOf(
"type" to "synthesisProgress",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to utteranceId,
"text" to text,
"progress" to progress
)
)
}
override fun onRecognitionStarted() {
eventSender.send(
mapOf(
"type" to "recognitionStarted",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onRecognitionStopped() {
eventSender.send(
mapOf(
"type" to "recognitionStopped",
"serviceId" to serviceId,
"direction" to direction
)
)
}
override fun onStateChanged(component: String, isActive: Boolean) {
eventSender.send(
mapOf(
"type" to "stateChanged",
"serviceId" to serviceId,
"direction" to direction,
"component" to component,
"isActive" to isActive
)
)
}
override fun onError(component: String, error: String) {
eventSender.send(
mapOf(
"type" to "error",
"serviceId" to serviceId,
"direction" to direction,
"component" to component,
"error" to error
)
)
}
}
/**
* Doubao AST Callback Implementation
*/
class DoubaoAstCallback(
private val serviceId: String,
private val direction: String,
private val targetLanguage: String,
private val eventSender: AstEventSender,
private val audioWriter: AudioWriter? = null
) : DoubaoE2ETranslateHelper.Callback {
private val tag = "DoubaoAstCallback"
private val doubaoFinalSourceTextCache: MutableMap<String, String> = mutableMapOf()
override fun onSessionStarted(sessionId: String) {
eventSender.send(
mapOf(
"type" to "serviceInitialized",
"serviceId" to serviceId,
"direction" to direction,
"sessionId" to sessionId
)
)
}
override fun onPartialSourceText(sessionId: String, text: String) {
eventSender.send(
mapOf(
"type" to "recognizing",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"text" to text,
"language" to direction.split("->").firstOrNull().orEmpty()
)
)
}
override fun onFinalSourceText(sessionId: String, finalText: String) {
val key = "$serviceId:$sessionId"
doubaoFinalSourceTextCache[key] = finalText
eventSender.send(
mapOf(
"type" to "recognized",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"text" to finalText,
"language" to direction.split("->").firstOrNull().orEmpty()
)
)
}
override fun onPartialText(sessionId: String, text: String) {
eventSender.send(
mapOf(
"type" to "translatedInterim",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to "",
"translatedText" to text,
"targetLanguage" to targetLanguage
)
)
}
override fun onPartialAudio(sessionId: String, data: ByteArray) {
audioWriter?.write(data)
}
override fun onSessionFinished(
sessionId: String,
finalText: String,
finalAudio: ByteArray
) {
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to (doubaoFinalSourceTextCache.remove("$serviceId:$sessionId") ?: ""),
"translatedText" to finalText,
"targetLanguage" to targetLanguage
)
)
// if (finalAudio.isNotEmpty()) {
// audioWriter?.write(finalAudio)
// }
}
override fun onSessionError(sessionId: String, code: Int, message: String) {
eventSender.send(
mapOf(
"type" to "error",
"serviceId" to serviceId,
"direction" to direction,
"component" to "doubaoAst",
"error" to message,
"code" to code
)
)
}
override fun onFinalTranslatedText(sessionId: String, finalText: String) {
val key = "$serviceId:$sessionId"
val original = doubaoFinalSourceTextCache.remove(key) ?: ""
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to original,
"translatedText" to finalText,
"targetLanguage" to targetLanguage
)
)
}
}
/**
* Aliyun Bailian AST Callback Implementation
*/
class AliyunAstCallback(
private val serviceId: String,
private val direction: String,
private val targetLanguage: String,
private val eventSender: AstEventSender,
private val audioWriter: AudioWriter? = null
) : AliyunBailianE2EHelper.Callback {
private val tag = "AliyunAstCallback"
private val sourceTextCache: MutableMap<String, String> = mutableMapOf()
override fun onSessionStarted(sessionId: String) {
eventSender.send(
mapOf(
"type" to "serviceInitialized",
"serviceId" to serviceId,
"direction" to direction,
"sessionId" to sessionId
)
)
}
override fun onPartialSourceText(sessionId: String, text: String) {
eventSender.send(
mapOf(
"type" to "recognizing",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"text" to text,
"language" to direction.split("->").firstOrNull().orEmpty()
)
)
}
override fun onFinalSourceText(sessionId: String, finalText: String) {
val key = "$serviceId:$sessionId"
sourceTextCache[key] = finalText
eventSender.send(
mapOf(
"type" to "recognized",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"text" to finalText,
"language" to direction.split("->").firstOrNull().orEmpty()
)
)
}
override fun onPartialText(sessionId: String, text: String) {
eventSender.send(
mapOf(
"type" to "translatedInterim",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to "",
"translatedText" to text,
"targetLanguage" to targetLanguage
)
)
}
override fun onPartialAudio(sessionId: String, data: ByteArray) {
audioWriter?.write(data)
}
override fun onSessionFinished(
sessionId: String,
finalText: String,
finalAudio: ByteArray
) {
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to (sourceTextCache.remove("$serviceId:$sessionId") ?: ""),
"translatedText" to finalText,
"targetLanguage" to targetLanguage
)
)
// if (finalAudio.isNotEmpty()) {
// audioWriter?.write(finalAudio)
// }
}
override fun onSessionError(sessionId: String, code: Int, message: String) {
eventSender.send(
mapOf(
"type" to "error",
"serviceId" to serviceId,
"direction" to direction,
"component" to "aliyunAst",
"error" to message,
"code" to code
)
)
}
override fun onFinalTranslatedText(sessionId: String, finalText: String) {
val key = "$serviceId:$sessionId"
val original = sourceTextCache.remove(key) ?: ""
eventSender.send(
mapOf(
"type" to "translated",
"serviceId" to serviceId,
"direction" to direction,
"utteranceId" to sessionId,
"originalText" to original,
"translatedText" to finalText,
"targetLanguage" to targetLanguage
)
)
}
}

473
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/DoubaoE2ETranslateHelper.kt

@ -0,0 +1,473 @@
package com.example.astclient
import android.content.Context
import android.net.Uri
import android.util.Log
import okhttp3.*
import okio.ByteString
import java.io.ByteArrayOutputStream
import java.io.File
import java.io.InputStream
import java.util.UUID
import java.util.concurrent.TimeUnit
import java.util.concurrent.atomic.AtomicBoolean
import kotlinx.coroutines.*
import data.speech.ast.TranslateRequest
import data.speech.ast.TranslateResponse
import data.speech.ast.ReqParams
import data.speech.understanding.User
import data.speech.understanding.Audio
import data.speech.common.RequestMeta
import data.speech.event.Type
/**
* AST 流式翻译助手,提供与 `IntegratedSpeechTranslationService` 类似的接口,
* 以便在插件中通过 `pushAudioData` 输入音频并在会话结束时获取合成音频。
*/
class DoubaoE2ETranslateHelper(
private val context: Context
) {
private val TAG = "DoubaoE2ETranslateHelper"
interface Callback {
/** 会话开始 */
fun onSessionStarted(sessionId: String)
/** 增量文本到达 */
fun onPartialText(sessionId: String, text: String)
/** 增量源文本到达(实时识别的原文字幕) */
fun onPartialSourceText(sessionId: String, text: String)
/** 源文本最终结果到达(原文识别完成) */
fun onFinalSourceText(sessionId: String, finalText: String)
/** 翻译最终结果到达(译文生成完成) */
fun onFinalTranslatedText(sessionId: String, finalText: String)
/** 增量音频片段到达(服务端返回的目标音频) */
fun onPartialAudio(sessionId: String, data: ByteArray)
/** 会话结束,返回完整文本与完整合成音频 */
fun onSessionFinished(sessionId: String, finalText: String, finalAudio: ByteArray)
/** 会话失败或取消 */
fun onSessionError(sessionId: String, code: Int, message: String)
}
private var conf: Config = Config()
private var callback: Callback? = null
private var client: OkHttpClient? = null
private var webSocket: WebSocket? = null
private var sessionId: String = ""
private val sessionLock = Any()
private val recvAudio = ByteArrayOutputStream()
private val recvText = mutableListOf<String>()
private val recvSourceText = mutableListOf<String>()
private var audioChunkBuffer = ByteArray(0)
private val isStarted = AtomicBoolean(false)
private val scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
/**
* 初始化助手,设置配置与回调。
*/
fun initialize(config: Config, cb: Callback): Boolean {
Log.d(TAG, "initialize: wsUrl=${config.wsUrl}, resourceId=${config.resourceId}")
conf = config
callback = cb
client = OkHttpClient.Builder()
.pingInterval(30, TimeUnit.SECONDS)
.readTimeout(0, TimeUnit.SECONDS)
.build()
Log.d(TAG, "initialize: client created")
startContinuousTranslation()
return true
}
/**
* 重启会话(用于超时自动重连)
*/
private fun restartSession() {
Log.i(TAG, "restartSession: performing auto-restart...")
try { webSocket?.close(1000, "restarting") } catch (_: Exception) {}
webSocket = null
isStarted.set(false)
try {
java.util.Timer().schedule(object : java.util.TimerTask() {
override fun run() {
Log.i(TAG, "restartSession: timer task running")
if (!isStarted.get()) {
Log.i(TAG, "restartSession: calling startContinuousTranslation")
val result = startContinuousTranslation()
Log.i(TAG, "restartSession: startContinuousTranslation result=$result")
} else {
Log.i(TAG, "restartSession: already started, skipping")
}
}
}, 200)
} catch (e: Exception) {
Log.e(TAG, "restartSession: timer failed", e)
}
}
/**
* 启动会话,建立 WebSocket 并发送 StartSession。
*/
fun startContinuousTranslation(): Boolean {
Log.d(TAG, "startContinuousTranslation: isStarted=${isStarted.get()} clientIsNull=${client==null}")
if (client == null) return false
if (isStarted.get()) return true
synchronized(sessionLock) {
sessionId = UUID.randomUUID().toString()
Log.d(TAG, "startContinuousTranslation: sessionId=${sessionId}")
recvAudio.reset()
recvText.clear()
}
val connId = UUID.randomUUID().toString()
val request = Request.Builder()
.url(conf.wsUrl)
.header("X-Api-App-Key", conf.appKey)
.header("X-Api-Access-Key", conf.accessKey)
.header("X-Api-Resource-Id", conf.resourceId)
.header("X-Api-Connect-Id", connId)
.build()
val listener = object : WebSocketListener() {
/**
* WebSocket 连接建立时回调。
* 设置内部会话状态,构造并发送 `StartSession` 请求,
* 然后通过 `callback.onSessionStarted` 通知上层会话已开始。
*
* @param ws 当前的 WebSocket 连接
* @param response 握手响应,包含状态码与头信息
*/
override fun onOpen(ws: WebSocket, response: Response) {
Log.d(TAG, "onOpen: code=${response.code} logid=${response.header("X-Tt-Logid")}")
webSocket = ws
isStarted.set(true)
val startReq = makeStartRequest(sessionId)
ws.send(ByteString.of(*startReq.toByteArray()))
Log.d(TAG, "onOpen: StartSession sent")
callback?.onSessionStarted(sessionId)
}
/**
* 接收服务端消息的回调。
* 解析为 `TranslateResponse` 后依据 `event` 分类处理:
* - `UsageResponse`:仅记录日志后返回;
* - `SessionFailed`/`SessionCanceled`:触发错误回调并关闭连接;
* - `SessionFinished`:汇总累计的文本与音频,触发完成回调并正常关闭;
* - 其他:追加增量音频与文本,并通过 `callback` 将增量结果向上层反馈。
*
* @param ws 当前的 WebSocket 连接
* @param bytes 服务端下发的二进制消息
*/
override fun onMessage(ws: WebSocket, bytes: ByteString) {
Log.d(TAG, "onMessage: ${bytes.size} bytes")
val resp = TranslateResponse.parseFrom(bytes.toByteArray())
val event = resp.event
val text = resp.text
val data = resp.data.toByteArray()
Log.d(TAG, "onMessage: event=${event} textLen=${text?.length ?: 0} audioLen=${data.size}")
if (event == Type.UsageResponse) {
Log.d(TAG, "onMessage: UsageResponse")
return
}
if (event == Type.SessionFailed || event == Type.SessionCanceled) {
val msg = if (resp.hasResponseMeta()) resp.responseMeta.message else ""
Log.e(TAG, "onMessage: session error event=${event} msg=${msg}")
if (msg.contains("Timeout waiting next packet", ignoreCase = true)) {
Log.i(TAG, "onMessage: detected timeout error, triggering auto-restart")
restartSession()
return
}
callback?.onSessionError(sessionId, 1011, msg)
ws.close(1011, msg)
return
}
if (event == Type.SessionFinished) {
Log.d(TAG, "onMessage: SessionFinished")
val finalText = recvText.joinToString(" ")
val finalAudio = recvAudio.toByteArray()
callback?.onSessionFinished(sessionId, finalText, finalAudio)
ws.close(1000, "Completed")
return
}
if (data.isNotEmpty()) {
recvAudio.write(data)
Log.d(TAG, "onMessage: partial audio appended size=${data.size}")
processAudioChunk(data)
}
if (!text.isNullOrBlank()) {
when (event) {
Type.SourceSubtitleStart -> {
Log.d(TAG, "onMessage: SourceSubtitleStart")
recvSourceText.clear()
}
Type.SourceSubtitleResponse -> {
Log.d(TAG, "onMessage: SourceSubtitleResponse text='${text}'")
recvSourceText.add(text)
callback?.onPartialSourceText(sessionId, text)
}
Type.SourceSubtitleEnd -> {
Log.d(TAG, "onMessage: SourceSubtitleEnd")
val finalSrc = recvSourceText.joinToString(" ")
callback?.onFinalSourceText(sessionId, finalSrc)
}
Type.TranslationSubtitleStart -> {
Log.d(TAG, "onMessage: TranslationSubtitleStart")
recvText.clear()
}
Type.TranslationSubtitleResponse -> {
Log.d(TAG, "onMessage: TranslationSubtitleResponse text='${text}'")
recvText.add(text)
callback?.onPartialText(sessionId, text)
}
Type.TranslationSubtitleEnd -> {
Log.d(TAG, "onMessage: TranslationSubtitleEnd")
val finalTgt = recvText.joinToString(" ")
callback?.onFinalTranslatedText(sessionId, finalTgt)
// 一句话结束后,重新生成 sessionId 并在同一连接上发送新的 StartSession
startNewSubSession(ws)
}
else -> {
// 其他文本事件按译文处理以保持兼容
// recvText.add(text)
Log.d(TAG, "onMessage: partial text appended text='${text}' (compat)")
//callback?.onPartialText(sessionId, text)
}
}
}
}
/**
* WebSocket 连接发生错误时的回调。
* 读取异常信息并通过 `callback.onSessionError` 向上层报告,
* 不在此处执行重试逻辑,由上层决定后续策略。
*
* @param ws 当前的 WebSocket 连接
* @param t 抛出的异常
* @param response 可选的响应对象(可能为 null)
*/
override fun onFailure(ws: WebSocket, t: Throwable, response: Response?) {
val msg = t.message ?: "unknown"
Log.e(TAG, "onFailure: ${msg} code=${response?.code}")
callback?.onSessionError(sessionId, 1011, msg)
}
/**
* WebSocket 连接关闭时的回调(正常或异常关闭)。
* 释放音频缓冲并重置会话标志,便于下一次会话重新开始。
*
* @param ws 当前的 WebSocket 连接
* @param code 关闭码
* @param reason 关闭原因描述
*/
override fun onClosed(ws: WebSocket, code: Int, reason: String) {
Log.d(TAG, "onClosed: code=${code} reason=${reason}")
try { recvAudio.close() } catch (_: Exception) {}
isStarted.set(false)
}
}
Log.d(TAG, "startContinuousTranslation: connecting wsUrl=${conf.wsUrl}")
client!!.newWebSocket(request, listener)
Log.d(TAG, "startContinuousTranslation: newWebSocket invoked")
return true
}
/**
* 准备下一句话的接收缓冲区。
* 不再重新生成 SessionID,保持整个 WebSocket 连接使用同一个 SessionID。
*
* @param ws 当前的 WebSocket 连接
*/
private fun startNewSubSession(ws: WebSocket) {
synchronized(sessionLock) {
// sessionId 保持不变,避免服务端报错 "unexpected client event: current state=Started"
// sessionId = UUID.randomUUID().toString()
Log.d(TAG, "startNewSubSession: keeping sessionId=${sessionId} clearing buffers")
recvText.clear()
recvSourceText.clear()
}
}
private fun processAudioChunk(
incoming: ByteArray,
multiple: Int = 1280
) {
if (incoming.isNotEmpty()) {
audioChunkBuffer += incoming
}
val sendLen = (audioChunkBuffer.size / multiple) * multiple
if (sendLen > 0) {
val chunk = audioChunkBuffer.copyOfRange(0, sendLen)
callback?.onPartialAudio(sessionId, chunk)
audioChunkBuffer = audioChunkBuffer.copyOfRange(sendLen, audioChunkBuffer.size)
}
}
/**
* 推送一段 PCM/WAV 音频数据到服务端。
*/
fun pushAudioData(data: ByteArray): Boolean {
// Log.d(TAG, "pushAudioData: size=${data.size} isStarted=${isStarted.get()} wsIsNull=${webSocket==null}")
val ws = webSocket ?: return false
if (!isStarted.get()) return false
synchronized(sessionLock) {
val req = makeChunkRequest(sessionId, data)
val ok = ws.send(ByteString.of(*req.toByteArray()))
// Log.d(TAG, "pushAudioData: sent=${ok}")
return ok
}
}
/**
* 停止会话,发送 FinishSession,并等待服务端返回。
*/
fun stopContinuousTranslation(): Boolean {
Log.d(TAG, "stopContinuousTranslation: wsIsNull=${webSocket==null}")
val ws = webSocket ?: return false
synchronized(sessionLock) {
val finishReq = makeFinishRequest(sessionId)
val ok = ws.send(ByteString.of(*finishReq.toByteArray()))
Log.d(TAG, "stopContinuousTranslation: sent=${ok}")
return ok
}
}
/**
* 释放资源,关闭会话。
*/
fun dispose() {
Log.d(TAG, "dispose: begin")
try { webSocket?.close(1000, "dispose") } catch (_: Exception) {}
webSocket = null
isStarted.set(false)
try { recvAudio.close() } catch (_: Exception) {}
scope.cancel()
Log.d(TAG, "dispose: done")
}
/**
* 构造 StartSession 请求
*/
private fun makeStartRequest(sessionId: String): TranslateRequest {
Log.d(TAG, "makeStartRequest: sessionId=${sessionId}")
val user = User.newBuilder()
.setUid("ast_android_client")
.setDid("ast_android_client")
.setPlatform("android")
.build()
val sourceAudio = Audio.newBuilder()
.setFormat("wav")
.setRate(16000)
.setBits(16)
.setChannel(1)
.build()
val targetAudio = Audio.newBuilder()
.setFormat("pcm")
.setRate(16000)
.build()
val params = ReqParams.newBuilder()
.setMode("s2s")
.setSourceLanguage(conf.sourceLanguage)
.setTargetLanguage(conf.targetLanguage)
.build()
val meta = RequestMeta.newBuilder()
.setSessionID(sessionId)
.build()
return TranslateRequest.newBuilder()
.setRequestMeta(meta)
.setEvent(Type.StartSession)
.setUser(user)
.setSourceAudio(sourceAudio)
.setTargetAudio(targetAudio)
.setRequest(params)
.build()
}
/**
* 构造音频片段请求
*/
private fun makeChunkRequest(sessionId: String, chunk: ByteArray): TranslateRequest {
//Log.d(TAG, "makeChunkRequest: sessionId=${sessionId} chunkSize=${chunk.size}")
val sourceAudio = Audio.newBuilder()
.setFormat("wav")
.setRate(16000)
.setBits(16)
.setChannel(1)
.setBinaryData(com.google.protobuf.ByteString.copyFrom(chunk))
.build()
val meta = RequestMeta.newBuilder()
.setSessionID(sessionId)
.build()
return TranslateRequest.newBuilder()
.setRequestMeta(meta)
.setEvent(Type.TaskRequest)
.setUser(
User.newBuilder()
.setUid("ast_android_client")
.setDid("ast_android_client")
.build()
)
.setSourceAudio(sourceAudio)
.build()
}
/**
* 构造结束请求
*/
private fun makeFinishRequest(sessionId: String): TranslateRequest {
Log.d(TAG, "makeFinishRequest: sessionId=${sessionId}")
val meta = RequestMeta.newBuilder()
.setSessionID(sessionId)
.build()
return TranslateRequest.newBuilder()
.setRequestMeta(meta)
.setEvent(Type.FinishSession)
.setUser(
User.newBuilder()
.setUid("ast_android_client")
.setDid("ast_android_client")
.build()
)
.setSourceAudio(Audio.newBuilder().build())
.build()
}
}
/**
* 客户端配置,包含服务端地址与鉴权信息。
*/
data class Config(
val wsUrl: String = "wss://openspeech.bytedance.com/api/v4/ast/v2/translate",
val appKey: String = "",
val accessKey: String = "",
val resourceId: String = "volc.service_type.10053",
val sourceLanguage: String = "zh",
val targetLanguage: String = "en",
val clientWaitMs: Long = 60_000 // 等待会话结束的最大时间,可按需调整
)

376
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekAsrHelper.kt

@ -0,0 +1,376 @@
package com.yunqiinnovation.azure_speech
import android.content.Context
import android.os.Handler
import android.os.Looper
import android.util.Base64
import android.util.Log
import okhttp3.*
import okio.ByteString.Companion.toByteString
import org.json.JSONObject
import java.net.URLEncoder
import java.nio.charset.StandardCharsets
import java.security.MessageDigest
import java.util.UUID
import javax.crypto.Mac
import javax.crypto.spec.SecretKeySpec
// 讯飞语音识别助手
//调用讯飞开放平台的实时语音转写大模型接口
// 参考文档:https://www.xfyun.cn/doc/spark/asr_llm/rtasr_llm.html
/**
* 讯飞语音识别助手,封装与讯飞 AST 实时转写接口的交互。
* 负责建立 WebSocket 连接、发送音频数据并解析识别结果,
* 并通过回调向上层报告会话开始/识别中/最终结果/错误等事件。
*/
class IflytekAsrHelper(private val context: Context) {
private val tag = "IflytekAsrHelper"
private var appId: String = ""
private var accessKeyId: String = ""
private var accessKeySecret: String = ""
private var webSocket: WebSocket? = null
private val client = OkHttpClient()
private var currentSessionId = ""
private var sessionId = ""
private var lastIntermediateResult = ""
private var currentLanguage = "zh-CN"
private var isAutoDetectLanguage = false
// 是否第去除一个文字的标点符号
private var removeFirstPunctuation = true
/**
* 使用指定配置构造识别助手。
* @param context Android 上下文
* @param appId 讯飞应用 ID
* @param accessKeyId 访问密钥 ID
* @param accessKeySecret 访问密钥 Secret
*/
constructor(
context: Context,
appId: String,
accessKeyId: String,
accessKeySecret: String
) : this(context) {
updateConfig(appId = appId, accessKeyId = accessKeyId, accessKeySecret = accessKeySecret)
}
/**
* 更新讯飞接口鉴权配置。
* @param appId 讯飞应用 ID
* @param accessKeyId 访问密钥 ID
* @param accessKeySecret 访问密钥 Secret
*/
fun updateConfig(appId: String, accessKeyId: String, accessKeySecret: String) {
this.appId = appId
this.accessKeyId = accessKeyId
this.accessKeySecret = accessKeySecret
}
/**
* 校验当前配置是否完整可用。
* @return 若 `appId`、`accessKeyId`、`accessKeySecret` 均不为空返回 `true`,否则 `false`
*/
private fun isConfigValid(): Boolean {
return appId.isNotBlank() && accessKeyId.isNotBlank() && accessKeySecret.isNotBlank()
}
/**
* 启动实时识别会话并建立 WebSocket 连接。
* 会在连接建立后触发 `onSessionStarted`,在收到中间结果触发 `onRecognizing`,
* 收到最终结果触发 `onResult` 并重置会话 ID,连接关闭时触发 `onSessionStopped`。
* 若配置缺失或连接异常,将触发 `onError`。
* @param callback 识别回调接口
* @param language 指定语言代码(如 `zh-CN`、`en-US`),当 `isAutoDetect` 为 `false` 时生效
* @param isAutoDetect 是否自动检测中英文(根据结果文本是否包含中文字符判断)
* @param isRemoveFirstPunctuation 是否去除结果首字符的标点符号
*/
fun start(callback: AzureAsrHelper.ContinuousRecognizeCallback, language: String, isAutoDetect: Boolean, isRemoveFirstPunctuation: Boolean) {
try {
webSocket?.cancel()
webSocket = null
} catch (e: Exception) {
Log.e(tag, "Clear previous websocket failed: ${e.message}")
}
currentSessionId = UUID.randomUUID().toString()
lastIntermediateResult = ""
currentLanguage = language
isAutoDetectLanguage = isAutoDetect
removeFirstPunctuation = isRemoveFirstPunctuation
if (!isConfigValid()) {
callback.onError(currentSessionId, 1007, "Iflytek config missing")
return
}
val url = getWebSocketUrl()
val request = Request.Builder().url(url).build()
webSocket = client.newWebSocket(request, object : WebSocketListener() {
override fun onOpen(webSocket: WebSocket, response: Response) {
Log.d(tag, "WebSocket Opened")
callback.onSessionStarted(currentSessionId)
}
override fun onMessage(webSocket: WebSocket, text: String) {
renderResult(text, callback)
}
override fun onClosing(webSocket: WebSocket, code: Int, reason: String) {
Log.d(tag, "WebSocket Closing: $code / $reason")
webSocket.close(1000, null)
}
override fun onClosed(webSocket: WebSocket, code: Int, reason: String) {
Log.d(tag, "WebSocket Closed: $code / $reason")
if (currentSessionId.isNotEmpty() && lastIntermediateResult.isNotEmpty()) {
val detectedLanguage = if (isAutoDetectLanguage) {
if (lastIntermediateResult.contains(Regex("[\\u4e00-\\u9fa5]"))) {
"zh-CN"
} else {
"en-US"
}
} else {
currentLanguage
}
callback.onResult(currentSessionId, lastIntermediateResult, detectedLanguage)
}
callback.onSessionStopped(currentSessionId)
}
override fun onFailure(webSocket: WebSocket, t: Throwable, response: Response?) {
Log.e(tag, "WebSocket Error", t)
callback.onError(currentSessionId, 1004, "Iflytek connection failed: ${t.message}")
}
})
}
/**
* 发送一帧原始音频数据到服务端。
* @param frameBuffer 音频字节数组(例如 16kHz、16bit PCM)
*/
fun sendAudio(frameBuffer: ByteArray) {
webSocket?.let { ws: WebSocket ->
ws.send(frameBuffer.toByteString(0, frameBuffer.size))
}
}
/**
* 主动结束当前识别会话。
* 向服务端发送结束标记并在 1 秒后关闭连接。
* 若关闭失败会记录错误日志。
*/
fun stop() {
try {
webSocket?.let { ws: WebSocket ->
ws.send("{\"end\": true, \"sessionId\": \"$sessionId\"}")
Handler(Looper.getMainLooper()).postDelayed({
try {
ws.close(1000, "User stopped")
} catch (e: Exception) {
Log.e(tag, "Close Iflytek failed: ${e.message}")
}
}, 1000)
}
} catch (e: Exception) {
Log.e(tag, "Stop Iflytek failed: ${e.message}")
} finally {
webSocket = null
}
}
/**
* 解析服务端文本消息并根据类型触发相应回调。
* 支持 `action`(握手)、`result`(识别结果)、`error`(错误)三类消息。
* 在 `result` 中会根据 `st.type` 区分中间结果与最终结果,并可进行语言自动检测。
* @param resultData 服务端返回的 JSON 字符串
* @param callback 识别回调接口
*/
private fun renderResult(
resultData: String,
callback: AzureAsrHelper.ContinuousRecognizeCallback
) {
try {
val jsonData = JSONObject(resultData)
Log.d(tag, "-------------------$jsonData")
val msgType = jsonData.optString("msg_type")
when (msgType) {
"action" -> {
val sid = jsonData.optString("sessionId")
if (!sid.isNullOrEmpty()) {
sessionId = sid
}
Log.d(tag, "Handshake success")
}
"result" -> {
val dataStr = jsonData.optString("data")
val data = JSONObject(dataStr)
val ls = data.optBoolean("ls")
val cn = data.optJSONObject("cn") ?: return
val st = cn.optJSONObject("st") ?: return
val rt = st.optJSONArray("rt") ?: return
var resultTextTemp = ""
var punctuationCount = 0
var wordCount = 0
for (i in 0 until rt.length()) {
val j = rt.getJSONObject(i)
val ws = j.optJSONArray("ws") ?: continue
for (k in 0 until ws.length()) {
val kObj = ws.getJSONObject(k)
val cw = kObj.optJSONArray("cw") ?: continue
if(removeFirstPunctuation) {
// 如果resultTextTemp中第一个文字是标点符号,则去除第一个文字的标点符号
for (l in 0 until cw.length()) {
val lObj = cw.getJSONObject(l)
val w = lObj.optString("w")
val wp = lObj.optString("wp")
if (wp == "p") {
punctuationCount++
if (resultTextTemp.isEmpty()) {
continue
}
} else {
wordCount++
}
resultTextTemp += w
}
} else {
// 不去除resultTextTemp第一个文字的标点符号
for (l in 0 until cw.length()) {
val lObj = cw.getJSONObject(l)
resultTextTemp += lObj.optString("w")
if (lObj.optString("wp") == "p") {
punctuationCount++
} else {
wordCount++
}
}
}
}
}
if (ls && wordCount == 0 && punctuationCount == 1) {
callback.onSessionStopped(currentSessionId)
return
}
val detectedLanguage = if (isAutoDetectLanguage) {
if (resultTextTemp.contains(Regex("[\\u4e00-\\u9fa5]"))) {
"zh-CN"
} else {
"en-US"
}
} else {
currentLanguage
}
if (st.optInt("type") == 0) {
callback.onResult(currentSessionId, resultTextTemp, detectedLanguage)
currentSessionId = UUID.randomUUID().toString()
lastIntermediateResult = ""
} else {
lastIntermediateResult = resultTextTemp
callback.onRecognizing(currentSessionId, resultTextTemp, detectedLanguage)
}
}
"error" -> {
callback.onError(currentSessionId, 1005, "Iflytek error: $resultData")
Log.e(tag, "Error: $resultData")
}
else -> {
callback.onError(currentSessionId, 1006, "Iflytek unknown msg_type: $msgType")
Log.w(tag, "Unknown msg_type: $msgType")
}
}
} catch (e: Exception) {
Log.e(tag, "Parse error", e)
}
}
/**
* 生成用于连接讯飞 AST 的 WebSocket URL。
* 包含音频编码、采样率、语言、鉴权参数等,并对参数进行 HMAC-SHA1 签名。
* @return 可直接用于发起 WebSocket 请求的完整 URL
*/
private fun getWebSocketUrl(): String {
val baseWsUrl = "wss://office-api-ast-dx.iflyaisol.com/ast/communicate/v1"
val params = java.util.TreeMap<String, String>()
params["audio_encode"] = "pcm_s16le"
params["lang"] = "autodialect"
params["samplerate"] = "16000"
params["accessKeyId"] = accessKeyId
params["appId"] = appId
params["utc"] = getUtcTime()
val signature = calculateSignature(params)
params["signature"] = signature
val sb = StringBuilder()
var first = true
for ((key, value) in params) {
if (!first) {
sb.append("&")
}
sb.append(URLEncoder.encode(key, "UTF-8")).append("=").append(URLEncoder.encode(value, "UTF-8"))
first = false
}
return "$baseWsUrl?$sb"
}
/**
* 获取当前时间的字符串表示。
* 注意此处使用 `GMT+8` 时区格式化为 `yyyy-MM-dd'T'HH:mm:ssZ`。
* @return 当前时间字符串
*/
private fun getUtcTime(): String {
val sdf = java.text.SimpleDateFormat("yyyy-MM-dd'T'HH:mm:ssZ")
sdf.timeZone = java.util.TimeZone.getTimeZone("GMT+8")
return sdf.format(java.util.Date())
}
/**
* 计算请求参数的签名值。
* 使用 HMAC-SHA1 对未包含 `signature` 且非空的参数进行签名,并返回 Base64 编码的结果。
* @param params 待签名的参数映射
* @return Base64 编码的签名字符串
*/
private fun calculateSignature(params: Map<String, String>): String {
val baseStr = StringBuilder()
var first = true
for ((key, value) in params) {
if ("signature" == key) continue
if (value.isEmpty()) continue
if (!first) {
baseStr.append("&")
}
baseStr.append(URLEncoder.encode(key, "UTF-8")).append("=").append(URLEncoder.encode(value, "UTF-8"))
first = false
}
val mac = Mac.getInstance("HmacSHA1")
val keySpec = SecretKeySpec(accessKeySecret.toByteArray(StandardCharsets.UTF_8), "HmacSHA1")
mac.init(keySpec)
val signBytes = mac.doFinal(baseStr.toString().toByteArray(StandardCharsets.UTF_8))
return Base64.encodeToString(signBytes, Base64.NO_WRAP)
}
}

659
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekAsrToAsr.kt

@ -0,0 +1,659 @@
package com.yunqiinnovation.azure_speech
import android.content.Context
import android.util.Log
import kotlinx.coroutines.*
import java.util.concurrent.atomic.AtomicBoolean
import java.util.concurrent.LinkedBlockingQueue
import kotlin.coroutines.CoroutineContext
import org.json.JSONObject
/**
* 讯飞识别 + 讯飞翻译 + 讯飞TTS 的整合服务
* 音频输入 -> 讯飞ASR -> 讯飞翻译 -> 讯飞TTS -> 音频输出/回调
*/
class IflytekIntegratedSpeechService(
private val context: Context
) : CoroutineScope {
companion object {
private const val TAG = "IflytekIntegratedSpeechService"
private const val SAMPLE_RATE = 16000
private const val CHANNELS = 1
private const val BITS_PER_SAMPLE = 16
}
private var job = SupervisorJob()
override val coroutineContext: CoroutineContext
get() = Dispatchers.Main + job
// 讯飞 TTS 客户端
private var ttsClient: IflytekTtsWs? = null
// 讯飞 ASR / 翻译
private var asrHelper: IflytekAsrHelper? = null
private var translationService: IntegratedSpeechTranslationService.TranslationServiceInterface? = null
// 配置与状态
private var serviceConfig = ServiceConfiguration()
private val serviceState = ServiceState()
private var eventCallback: ServiceEventCallback? = null
// 识别中翻译任务管理
private var recognizingTranslationJob: Job? = null
// 录音/推送输入(用于保存/中转)
private var audioQueue = LinkedBlockingQueue<ByteArray>()
private val isAudioThreadRunning = AtomicBoolean(false)
private var audioThread: Thread? = null
private var ttsdata: ByteArray = ByteArray(0)
private var currentSynthesisId: String? = null
private var currentSynthesisText: String? = null
/**
* 服务事件回调接口
*/
interface ServiceEventCallback {
fun onServiceInitialized()
fun onRecognizing(utteranceId: String, text: String, language: String, confidence: Float)
fun onRecognized(utteranceId: String, text: String, language: String, confidence: Float)
fun onTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String)
fun onInterimTranslated(utteranceId: String, originalText: String, translatedText: String, targetLanguage: String)
fun onTranslationStarted(utteranceId: String, text: String)
fun onTranslationFailed(utteranceId: String, text: String, error: String)
fun onSynthesisStarted(utteranceId: String, text: String)
fun onSynthesisCompleted(utteranceId: String, text: String)
fun onSynthesisFailed(utteranceId: String, text: String, error: String)
fun onSynthesisProgress(utteranceId: String, text: String, progress: Float)
fun onRecognitionStarted()
fun onRecognitionStopped()
fun onStateChanged(component: String, isActive: Boolean)
fun onError(component: String, error: String)
fun onSynthesisAudioGenerated(utteranceId: String, text: String, audioData: ByteArray)
}
/**
* 服务配置
*/
data class ServiceConfiguration(
var sourceLanguage: String = "zh-CN",
var targetLanguage: String = "en-US",
var translationSourceLanguage: String = "",
var translationTargetLanguage: String = "",
var currentVoice: String = "x4_yezi",
var speechRate: String = "0%",
var speechPitch: String = "0%",
var speechVolume: String = "100%",
var enableAudioPlayback: Boolean = false,
var enableAutoLanguageDetection: Boolean = false,
var translationTimeout: Long = 10000L,
var removeFirstPunctuation: Boolean = true,
// 讯飞鉴权
var xfyunAppId: String = "",
var xfyunAccessKeyId: String = "",
var xfyunAccessKeySecret: String = ""
)
/**
* 服务状态
*/
data class ServiceState(
val isInitialized: AtomicBoolean = AtomicBoolean(false),
val isRecognizing: AtomicBoolean = AtomicBoolean(false),
val isTranslating: AtomicBoolean = AtomicBoolean(false),
val isSynthesizing: AtomicBoolean = AtomicBoolean(false)
)
/**
* 微软 TTS 配置(保留入参兼容,不再使用)
*/
data class AzureConfiguration(
val subscriptionKey: String,
val region: String
)
/**
* 讯飞翻译配置
*/
data class IflytekTranslationConfiguration(
val appId: String,
val apiKey: String,
val apiSecret: String,
val host: String = "ntrans.xfyun.cn",
val timeoutMs: Long = 8000
) {
fun toConfigMap(): Map<String, String> = mapOf(
"appId" to appId,
"apiKey" to apiKey,
"apiSecret" to apiSecret,
"host" to host,
"timeout" to timeoutMs.toString()
)
}
/**
* 初始化整合服务
*/
suspend fun initialize(
azureConfig: AzureConfiguration,
translationConfig: IflytekTranslationConfiguration,
serviceConfig: ServiceConfiguration? = null,
callback: ServiceEventCallback
): Boolean = withContext(Dispatchers.IO) {
try {
Log.i(TAG, "开始初始化整合服务")
dispose()
this@IflytekIntegratedSpeechService.eventCallback = callback
serviceConfig?.let { this@IflytekIntegratedSpeechService.serviceConfig = it }
// 不再初始化微软TTS,改用讯飞TTS WebSocket
translationService = IflytekTranslationServiceImpl().apply {
initialize(translationConfig.toConfigMap())
}
Log.i(TAG, "讯飞翻译服务初始化完成")
asrHelper = IflytekAsrHelper(
context,
this@IflytekIntegratedSpeechService.serviceConfig.xfyunAppId,
this@IflytekIntegratedSpeechService.serviceConfig.xfyunAccessKeyId,
this@IflytekIntegratedSpeechService.serviceConfig.xfyunAccessKeySecret
)
Log.i(TAG, "讯飞ASR助手创建完成")
startAudioThread()
setupTtsClient()
Log.i(TAG, "音频线程与TTS合成器已启动")
serviceState.isInitialized.set(true)
withContext(Dispatchers.Main) { callback.onServiceInitialized() }
Log.i(TAG, "整合服务初始化成功")
startContinuousTranslation()
true
} catch (e: Exception) {
Log.e(TAG, "初始化失败", e)
withContext(Dispatchers.Main) { callback.onError("Initialization", "初始化异常: ${e.message}") }
false
}
}
/**
* 启动连续识别与翻译
*/
fun startContinuousTranslation(): Boolean {
if (!serviceState.isInitialized.get()) {
eventCallback?.onError("Service", "服务未初始化")
return false
}
return try {
Log.i(TAG, "启动连续识别与翻译: autoDetect=${serviceConfig.enableAutoLanguageDetection}, src=${serviceConfig.sourceLanguage}, tgt=${serviceConfig.targetLanguage}")
asrHelper?.start(object : AzureAsrHelper.ContinuousRecognizeCallback {
override fun onResult(sessiond: String, text: String, detectedLanguage: String) {
val confidence = 0.9f
Log.d(TAG, "最终识别结果(${detectedLanguage}): ${text.take(100)}")
eventCallback?.onRecognized(sessiond, text, detectedLanguage, confidence)
launch { processTranslationAndSynthesis(sessiond, text) }
}
override fun onRecognizing(sessiond: String, recognizing: String, detectedLanguage: String) {
val confidence = 0.6f
Log.d(TAG, "识别中(${detectedLanguage}): ${recognizing.take(80)}")
eventCallback?.onRecognizing(sessiond, recognizing, detectedLanguage, confidence)
recognizingTranslationJob?.cancel()
recognizingTranslationJob = launch { processTranslationOnly(sessiond, recognizing) }
}
override fun onSessionStarted(sessiond: String) {
serviceState.isRecognizing.set(true)
Log.i(TAG, "识别会话开始: ${sessiond}")
eventCallback?.onRecognitionStarted()
eventCallback?.onStateChanged("Recognition", true)
}
override fun onSessionStopped(sessiond: String) {
serviceState.isRecognizing.set(false)
Log.i(TAG, "识别会话结束: ${sessiond}")
eventCallback?.onRecognitionStopped()
eventCallback?.onStateChanged("Recognition", false)
}
override fun onCanceled(sessiond: String, reason: String, errorDetails: String) {
Log.w(TAG, "识别取消: ${reason}, ${errorDetails}")
eventCallback?.onError("Recognition", "识别取消: $reason, $errorDetails")
}
override fun onError(sessiond: String, code: Int, error: String) {
Log.e(TAG, "讯飞识别错误(${code}): ${error}")
eventCallback?.onError("Recognition", "讯飞识别错误($code): $error")
}
}, serviceConfig.sourceLanguage, serviceConfig.enableAutoLanguageDetection, serviceConfig.removeFirstPunctuation)
true
} catch (e: Exception) {
Log.e(TAG, "启动识别失败", e)
eventCallback?.onError("Service", "启动失败: ${e.message}")
false
}
}
/**
* 停止连续识别与翻译
*/
fun stopContinuousTranslation() {
try {
Log.i(TAG, "停止连续识别与翻译")
serviceState.isRecognizing.set(false)
asrHelper?.stop()
} catch (e: Exception) {
Log.e(TAG, "停止失败", e)
}
}
/**
* 推送外部音频(转发到讯飞ASR)
*/
fun pushAudioData(audioData: ByteArray) {
// Log.d(TAG, "推送音频数据: ${audioData.size} 字节")
audioQueue.offer(audioData)
asrHelper?.sendAudio(audioData)
}
/**
* 仅翻译识别中的文本
*/
/**
* 仅翻译识别中的文本(不触发合成)
*/
private suspend fun processTranslationOnly(utteranceId: String, text: String) {
if (!serviceState.isRecognizing.get()) return
if (serviceState.isTranslating.get()) return
serviceState.isTranslating.set(true)
Log.d(TAG, "开始临时翻译: ${text.take(80)}")
eventCallback?.onTranslationStarted(utteranceId, text)
eventCallback?.onStateChanged("Translation", true)
try {
val (srcLang, tgtLang) = resolveTranslationLanguages()
val res = withTimeout(serviceConfig.translationTimeout) {
translationService?.translateText(text, srcLang, tgtLang)
}
when {
res?.success == true && !res.translatedText.isNullOrEmpty() -> {
Log.d(TAG, "临时翻译结果: ${res.translatedText!!.take(80)}")
val dst = extractDstFromTranslated(res.translatedText!!)
if (dst != null && dst.isNotEmpty()) {
Log.i(TAG, "使用 dst 进行合成: ${dst.take(120)}")
eventCallback?.onInterimTranslated(utteranceId, text, dst, tgtLang)
} else {
Log.w(TAG, "未能解析 dst,回退使用原始翻译文本进行合成")
}
}
res?.error != null -> {
Log.w(TAG, "临时翻译错误: ${res.error}")
eventCallback?.onTranslationFailed(utteranceId, text, res.error!!)
}
else -> {
Log.w(TAG, "临时翻译返回空结果")
eventCallback?.onTranslationFailed(utteranceId, text, "翻译服务返回空结果")
}
}
} catch (e: TimeoutCancellationException) {
Log.w(TAG, "临时翻译超时")
eventCallback?.onTranslationFailed(utteranceId, text, "翻译超时")
} catch (e: Exception) {
Log.e(TAG, "临时翻译异常: ${e.message}")
eventCallback?.onTranslationFailed(utteranceId, text, "翻译异常: ${e.message}")
} finally {
serviceState.isTranslating.set(false)
eventCallback?.onStateChanged("Translation", false)
}
}
/**
* 翻译并合成最终文本
*/
/**
* 翻译并合成最终文本
*/
private suspend fun processTranslationAndSynthesis(utteranceId: String, text: String) {
if (!serviceState.isRecognizing.get()) return
if (serviceState.isTranslating.get()) return
serviceState.isTranslating.set(true)
Log.d(TAG, "开始最终翻译与合成: ${text.take(120)}")
eventCallback?.onTranslationStarted(utteranceId, text)
eventCallback?.onStateChanged("Translation", true)
try {
val (srcLang, tgtLang) = resolveTranslationLanguages()
val res = withTimeout(serviceConfig.translationTimeout) {
translationService?.translateText(text, srcLang, tgtLang)
}
serviceState.isTranslating.set(false)
eventCallback?.onStateChanged("Translation", false)
when {
res?.success == true && !res.translatedText.isNullOrEmpty() -> {
Log.i(TAG, "翻译成功,触发合成: ${res.translatedText!!.take(120)}")
val dst = extractDstFromTranslated(res.translatedText!!)
if (dst != null && dst.isNotEmpty()) {
Log.i(TAG, "使用 dst 进行合成: ${dst.take(120)}")
eventCallback?.onTranslated(utteranceId, text, dst, tgtLang)
synthesizeText(utteranceId, dst)
} else {
Log.w(TAG, "未能解析 dst,回退使用原始翻译文本进行合成")
synthesizeText(utteranceId, res.translatedText!!)
}
}
res?.error != null -> eventCallback?.onTranslationFailed(utteranceId, text, res.error!!)
else -> eventCallback?.onTranslationFailed(utteranceId, text, "翻译服务返回空结果")
}
} catch (e: TimeoutCancellationException) {
serviceState.isTranslating.set(false)
eventCallback?.onStateChanged("Translation", false)
Log.w(TAG, "最终翻译超时")
eventCallback?.onTranslationFailed(utteranceId, text, "翻译超时")
} catch (e: Exception) {
serviceState.isTranslating.set(false)
eventCallback?.onStateChanged("Translation", false)
Log.e(TAG, "最终翻译异常: ${e.message}")
eventCallback?.onTranslationFailed(utteranceId, text, "翻译异常: ${e.message}")
}
}
/**
* 解析翻译所用的源/目标语言
* 若配置中提供 `translationSourceLanguage`/`translationTargetLanguage`,优先使用;否则回退到会话的 `sourceLanguage`/`targetLanguage`
*/
private fun resolveTranslationLanguages(): Pair<String, String> {
val src = if (serviceConfig.translationSourceLanguage.isNotEmpty()) serviceConfig.translationSourceLanguage else serviceConfig.sourceLanguage
val tgt = if (serviceConfig.translationTargetLanguage.isNotEmpty()) serviceConfig.translationTargetLanguage else serviceConfig.targetLanguage
return Pair(src, tgt)
}
/**
* 初始化成功后示例合成一句目标语言话语
* 功能:将固定中文短句翻译为目标语言并进行语音合成,便于确认语音配置与管线是否正常
*/
fun speakDemoSentence() {
if (!serviceState.isInitialized.get()) {
return
}
val srcLang = serviceConfig.sourceLanguage
val tgtLang = serviceConfig.targetLanguage
try {
val sourceText = "初始化成功"
val res = runBlocking { translationService?.translateText(sourceText, srcLang, tgtLang) }
val textToSpeak = res?.translatedText ?: run {
when {
tgtLang.lowercase().startsWith("en") -> "Initialization successful"
tgtLang.lowercase().startsWith("zh") -> "初始化成功"
tgtLang.lowercase().startsWith("ja") -> "初期化が完了しました"
tgtLang.lowercase().startsWith("ko") -> "초기화가 완료되었습니다"
tgtLang.lowercase().startsWith("fr") -> "Initialisation réussie"
tgtLang.lowercase().startsWith("es") -> "Inicialización completada"
else -> "Initialization completed"
}
}
val uttId = "utt-demo-${System.currentTimeMillis()}"
synthesizeText(uttId, textToSpeak)
} catch (e: Exception) {
Log.w(TAG, "演示合成失败: ${e.message}")
}
}
/**
* 从翻译服务返回的 JSON 文本中提取 `trans_result.dst`
*
* 参数:
* - translatedJson: 翻译服务返回的字符串(形如 {"trans_result":{"dst":"..."}})
* 返回:
* - 若解析成功返回 dst 字段,否则返回 null
*/
private fun extractDstFromTranslated(translatedJson: String): String? {
return try {
val root = JSONObject(translatedJson)
val trans = root.optJSONObject("trans_result")
when {
trans != null -> trans.optString("dst", null)
else -> root.optString("dst", null)
}
} catch (e: Exception) {
Log.w(TAG, "解析翻译结果JSON失败: ${e.message}")
null
}
}
/**
* 合成文本到音频
*/
/**
* 触发语音合成(异步)
*
* 参数:
* - utteranceId: 当前句子的唯一标识
* - text: 需要合成的目标文本
* 行为:
* - 仅设置当前合成上下文并调用异步合成;事件回调在合成器监听中统一分发。
*/
private fun synthesizeText(utteranceId: String, text: String) {
try {
currentSynthesisId = utteranceId
currentSynthesisText = text
val vcn = serviceConfig.currentVoice
val pitch = parsePercentToScale(serviceConfig.speechPitch)
val speed = parsePercentToScale(serviceConfig.speechRate)
ttsClient?.synthesize(
appId = serviceConfig.xfyunAppId,
apiKey = serviceConfig.xfyunAccessKeyId,
apiSecret = serviceConfig.xfyunAccessKeySecret,
text = text,
vcn = vcn,
tte = "UTF8",
pitch = pitch,
speed = speed,
aue = "raw",
callback = object : IflytekTtsWs.Callback {
override fun onStarted() {
ttsdata = ByteArray(0)
serviceState.isSynthesizing.set(true)
val uttId = currentSynthesisId ?: ""
val t = currentSynthesisText ?: ""
Log.i(TAG, "TTS合成开始: ${uttId}")
eventCallback?.onSynthesisStarted(uttId, t)
eventCallback?.onStateChanged("Synthesis", true)
}
override fun onAudioChunk(data: ByteArray) {
val uttId = currentSynthesisId ?: ""
val t = currentSynthesisText ?: ""
Log.i(TAG, "TTS合成数据: ${uttId} 数据长度=${data.size}")
if (data.isNotEmpty()) {
processSynthesisAudioChunk(uttId, t, data, 1280)
eventCallback?.onSynthesisProgress(uttId, t, 0.5f)
}
}
override fun onCompleted(sid: String?) {
val uttId = currentSynthesisId ?: ""
val t = currentSynthesisText ?: ""
Log.i(TAG, "TTS合成完成: ${uttId} sid=${sid ?: ""}")
eventCallback?.onSynthesisCompleted(uttId, t)
// 发送剩余缓存
if (ttsdata.isNotEmpty()) {
eventCallback?.onSynthesisAudioGenerated(uttId, t, ttsdata)
ttsdata = ByteArray(0)
}
currentSynthesisId = null
currentSynthesisText = null
serviceState.isSynthesizing.set(false)
eventCallback?.onStateChanged("Synthesis", false)
}
override fun onError(code: Int?, message: String) {
val uttId = currentSynthesisId ?: ""
val t = currentSynthesisText ?: ""
Log.w(TAG, "TTS合成错误: code=${code ?: -1} msg=$message")
eventCallback?.onSynthesisFailed(uttId, t, "语音合成错误: ${message}")
currentSynthesisId = null
currentSynthesisText = null
serviceState.isSynthesizing.set(false)
eventCallback?.onStateChanged("Synthesis", false)
}
}
)
} catch (e: Exception) {
eventCallback?.onSynthesisFailed(utteranceId, text, "合成异常: ${e.message}")
}
}
/**
* 生成基本 SSML
*/
/**
* 生成优化的 SSML 文本
* 清理与转义输入文本,并应用速率/音高/音量与当前语音
*/
private fun generateSsml(text: String): String {
val cleanText = cleanTextForSynthesis(text)
val escaped = escapeXmlText(cleanText)
return """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="${serviceConfig.targetLanguage}">
<voice name="${serviceConfig.currentVoice}">
<prosody rate="${serviceConfig.speechRate}" pitch="${serviceConfig.speechPitch}" volume="${serviceConfig.speechVolume}">
$escaped
</prosody>
</voice>
</speak>
""".trimIndent()
}
/**
* 清理文本用于语音合成
* 去除 URL、表情符号并规范空白字符
*/
private fun cleanTextForSynthesis(text: String): String {
return text
.replace(Regex("https?://[^\\s]+"), "")
.replace(Regex("[\\p{So}\\p{Sk}]"), "")
.replace(Regex("\\s+"), " ")
.trim()
}
/**
* 将文本进行 XML 转义
*/
private fun escapeXmlText(text: String): String {
return text
.replace("&", "&amp;")
.replace("<", "&lt;")
.replace(">", "&gt;")
.replace("\"", "&quot;")
.replace("'", "&apos;")
}
/**
* 设置微软 TTS 合成器
*/
/**
* 设置微软 TTS 合成器(事件驱动)
*
* 行为:
* - 播放启用:输出到系统扬声器
* - 播放禁用:创建推流输出以保持有效的 AudioConfig(丢弃数据),音频通过事件返回
* - 监听合成开始/进行中/完成/取消事件,分发状态与音频数据
*/
private fun setupTtsClient() {
ttsClient?.close()
ttsClient = IflytekTtsWs()
}
/**
* 将合成中的音频按 1280 字节对齐分片并回调
*
* 参数:
* - utteranceId: 当前句子的唯一标识
* - text: 合成文本内容
* - incoming: 新收到的音频数据
* - multiple: 分片对齐字节数(默认 1280)
* 行为:
* - 追加到缓存,仅当满足 multiple 的整数倍时发送;余量保留等待补齐。
*/
private fun processSynthesisAudioChunk(
utteranceId: String,
text: String,
incoming: ByteArray,
multiple: Int = 1280
) {
if (incoming.isNotEmpty()) {
ttsdata = ttsdata + incoming
}
val sendLen = (ttsdata.size / multiple) * multiple
if (sendLen > 0) {
val chunk = ttsdata.copyOfRange(0, sendLen)
Log.d(TAG, "TTS分片回调: 发送 ${sendLen} 字节,剩余 ${ttsdata.size - sendLen} 字节")
eventCallback?.onSynthesisAudioGenerated(utteranceId, text, chunk)
ttsdata = ttsdata.copyOfRange(sendLen, ttsdata.size)
}
}
private fun parsePercentToScale(percent: String): Int {
val v = percent.trim().removeSuffix("%")
val num = v.toIntOrNull() ?: 0
val scaled = 50 + num // 0% -> 50, -50% -> 0, +50% -> 100
return scaled.coerceIn(0, 100)
}
/**
* 启动音频线程(保存/中转用途)
*/
private fun startAudioThread() {
if (isAudioThreadRunning.get()) return
isAudioThreadRunning.set(true)
audioThread = Thread {
while (isAudioThreadRunning.get()) {
try {
val data = audioQueue.poll(100, java.util.concurrent.TimeUnit.MILLISECONDS)
if (data != null) {
// 此处可扩展:保存到文件或做能量检测
// Log.v(TAG, "音频队列取出: ${data.size} 字节")
}
} catch (_: InterruptedException) {
Thread.currentThread().interrupt()
break
} catch (e: Exception) {
Log.e(TAG, "音频线程异常", e)
}
}
}.apply { name = "IflytekAudioThread"; Log.i(TAG, "启动音频线程"); start() }
}
/**
* 释放资源
*/
fun dispose() {
try {
Log.i(TAG, "释放整合服务资源")
isAudioThreadRunning.set(false)
audioThread?.interrupt()
audioThread = null
recognizingTranslationJob?.cancel()
asrHelper?.stop()
asrHelper = null
ttsClient?.close()
ttsClient = null
serviceState.isInitialized.set(false)
serviceState.isRecognizing.set(false)
serviceState.isTranslating.set(false)
serviceState.isSynthesizing.set(false)
} catch (e: Exception) {
Log.e(TAG, "释放资源失败", e)
}
}
}

290
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekTranslationServiceImpl.kt

@ -0,0 +1,290 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.withContext
import okhttp3.MediaType.Companion.toMediaType
import okhttp3.OkHttpClient
import okhttp3.Request
import okhttp3.RequestBody
import okhttp3.RequestBody.Companion.toRequestBody
import org.json.JSONObject
import java.text.SimpleDateFormat
import java.util.Date
import java.util.Locale
import java.util.TimeZone
import java.util.concurrent.TimeUnit
import javax.crypto.Mac
import javax.crypto.spec.SecretKeySpec
import android.util.Base64
/**
* 讯飞机器翻译 OTS 服务实现
* 兼容 IntegratedSpeechTranslationService 的 TranslationServiceInterface
*/
class IflytekTranslationServiceImpl :
IntegratedSpeechTranslationService.TranslationServiceInterface {
companion object {
private const val TAG = "IflytekTranslationService"
private const val DEFAULT_HOST = "ntrans.xfyun.cn"
private const val REQUEST_URI = "/v2/ots"
private const val HTTP_METHOD = "POST"
private const val HTTP_PROTO = "HTTP/1.1"
private const val ALG = "hmac-sha256"
}
private var isInitialized = false
private var appId: String = ""
private var apiKey: String = ""
private var apiSecret: String = ""
private var host: String = DEFAULT_HOST
private var timeoutMs: Long = 8000
private val httpClient = OkHttpClient.Builder()
.connectTimeout(10, TimeUnit.SECONDS)
.readTimeout(30, TimeUnit.SECONDS)
.writeTimeout(30, TimeUnit.SECONDS)
.build()
/**
* 初始化服务配置
*/
override suspend fun initialize(config: Map<String, String>): Boolean {
return try {
appId = (config["appId"]?.trim() ?: "")
apiKey = (config["apiKey"]?.trim() ?: "")
apiSecret = (config["apiSecret"]?.trim() ?: "")
val cfgHost = config["host"]?.trim()
host = if (cfgHost.isNullOrEmpty()) DEFAULT_HOST else cfgHost
timeoutMs = config["timeout"]?.toLongOrNull() ?: 8000
Log.d(TAG, "初始化配置: appId=" + appId + ", apiKey=" + apiKey + ", apiSecret=" + apiSecret)
if (appId.isEmpty() || apiKey.isEmpty() || apiSecret.isEmpty()) {
Log.e(TAG, "缺少必要的鉴权参数")
false
} else {
isInitialized = true
Log.d(TAG, "讯飞 OTS 翻译服务初始化成功")
Log.d(TAG, "初始化配置: host=" + host + ", timeoutMs=" + timeoutMs)
true
}
} catch (e: Exception) {
Log.e(TAG, "初始化失败", e)
false
}
}
/**
* 文本翻译
*/
override suspend fun translateText(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
if (!isInitialized) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "翻译服务未初始化"
)
}
return withContext(Dispatchers.IO) {
try {
Log.d(TAG, "开始翻译: src=" + sourceLanguage + ", dst=" + targetLanguage + ", 文本长度=" + text.length)
if (host.isBlank()) {
Log.w(TAG, "host为空,使用默认host: " + DEFAULT_HOST)
host = DEFAULT_HOST
}
val bodyJson = buildBody(text, sourceLanguage, targetLanguage)
val digest = buildDigest(bodyJson)
val date = httpDate(Date())
val authorization = buildAuthorization(digest, date)
val headers = buildHeaders(digest, authorization, date)
Log.d(TAG, "请求体长度=" + bodyJson.length)
Log.d(TAG, "Digest长度=" + digest.length)
val mediaType = "application/json".toMediaType()
val reqBody: RequestBody = bodyJson.toRequestBody(mediaType)
val url = "https://$host$REQUEST_URI"
Log.d(TAG, "发起请求: url=" + url)
val request = Request.Builder()
.url(url)
.post(reqBody)
.apply {
headers.forEach { (k, v) -> addHeader(k, v) }
}
.build()
val startNanos = System.nanoTime()
val response = httpClient.newCall(request).execute()
val durationMs = java.util.concurrent.TimeUnit.NANOSECONDS.toMillis(System.nanoTime() - startNanos)
val code = response.code
val respStr = response.body?.string()
Log.d(TAG, "HTTP响应: code=" + code + ", bodyLen=" + (respStr?.length ?: -1) + ", 耗时=" + durationMs + "ms")
if (respStr == null) {
Log.w(TAG, "HTTP错误: code=" + code + ", 空响应体")
return@withContext IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "HTTP错误: $code"
)
}
if (code != 200) {
val snippet = if (respStr.length > 256) respStr.substring(0, 256) else respStr
Log.w(TAG, "HTTP错误: code=" + code + ", bodySnippet=" + snippet)
try {
val errJson = JSONObject(respStr)
val msg = errJson.optString("message")
val c = errJson.optInt("code", code)
val friendly = interpretError(c, msg)
return@withContext IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = friendly
)
} catch (_: Exception) {
return@withContext IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "HTTP错误: $code"
)
}
}
val json = JSONObject(respStr)
val errCode = json.optInt("code", -1)
Log.d(TAG, "服务返回码: " + errCode)
if (errCode == 0) {
val data = json.optJSONObject("data")
val result = when {
data == null -> null
data.has("result") -> data.optString("result")
data.has("translation") -> data.optString("translation")
data.has("trans_result") -> data.optString("trans_result")
else -> null
}
if (!result.isNullOrEmpty()) {
Log.d(TAG, "翻译成功: 结果长度=" + result.length)
return@withContext IntegratedSpeechTranslationService.TranslationResult(
success = true,
translatedText = result,
confidence = 0.9f
)
}
}
val message = json.optString("message", "翻译失败")
val friendly = interpretError(errCode, message)
Log.w(TAG, "翻译失败: message=" + message)
return@withContext IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = friendly
)
} catch (e: Exception) {
Log.e(TAG, "翻译请求异常", e)
IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "请求异常: ${e.message}"
)
}
}
}
/**
* 释放资源
*/
override fun dispose() {
isInitialized = false
try {
httpClient.dispatcher.executorService.shutdown()
httpClient.connectionPool.evictAll()
} catch (e: Exception) {
Log.e(TAG, "清理资源失败", e)
}
}
/**
* 构建请求体 JSON 字符串
*/
private fun buildBody(text: String, from: String, to: String): String {
val contentB64 = Base64.encodeToString(text.toByteArray(Charsets.UTF_8), Base64.NO_WRAP)
Log.v(TAG, "构建请求体: from=" + from + ", to=" + to + ", textB64Len=" + contentB64.length)
val root = JSONObject()
root.put("common", JSONObject().put("app_id", appId))
root.put("business", JSONObject().put("from", from).put("to", to))
root.put("data", JSONObject().put("text", contentB64))
return root.toString()
}
/**
* 构建 Digest 头(RFC 标准)
*/
private fun buildDigest(bodyJson: String): String {
val md = java.security.MessageDigest.getInstance("SHA-256")
val hash = md.digest(bodyJson.toByteArray(Charsets.UTF_8))
val b64 = Base64.encodeToString(hash, Base64.NO_WRAP)
return "SHA-256=$b64"
}
/**
* 构建 Authorization 头
*/
private fun buildAuthorization(digest: String, date: String): String {
val signStr = buildSignString(host, date, digest)
val mac = Mac.getInstance("HmacSHA256")
mac.init(SecretKeySpec(apiSecret.toByteArray(Charsets.UTF_8), "HmacSHA256"))
val signatureB64 = Base64.encodeToString(mac.doFinal(signStr.toByteArray(Charsets.UTF_8)), Base64.NO_WRAP)
Log.d(TAG, "构建鉴权: date=" + date + ", 签名长度=" + signatureB64.length)
return "api_key=\"$apiKey\", algorithm=\"$ALG\", headers=\"host date request-line digest\", signature=\"$signatureB64\""
}
/**
* 构建请求头映射
*/
private fun buildHeaders(digest: String, authorization: String, date: String): Map<String, String> {
Log.d(TAG, "构建请求头: host=" + host + ", date=" + date)
return mapOf(
"Content-Type" to "application/json",
"Accept" to "application/json",
"Method" to HTTP_METHOD,
"Host" to host,
"Date" to date,
"Digest" to digest,
"Authorization" to authorization
)
}
/**
* 统一错误码解释
*/
private fun interpretError(code: Int, message: String?): String {
return when (code) {
11200 -> "鉴权失败(11200): 请检查 APPID/APIKey/APISecret 是否正确,服务是否开通"
11201 -> "签名错误(11201): 请检查 Authorization/Date/Host/Digest 的签名串"
10105 -> "参数错误(10105): 请检查业务参数 from/to 与请求体"
else -> if (!message.isNullOrEmpty()) "HTTP错误: $code, $message" else "HTTP错误: $code"
}
}
/**
* 生成签名原文字段
*/
private fun buildSignString(host: String, date: String, digest: String): String {
return "host: $host\n" +
"date: $date\n" +
"$HTTP_METHOD $REQUEST_URI $HTTP_PROTO\n" +
"digest: $digest"
}
/**
* 生成 RFC 1123 GMT 时间字符串
*/
private fun httpDate(dt: Date): String {
val sdf = SimpleDateFormat("EEE, dd MMM yyyy HH:mm:ss 'GMT'", Locale.US)
sdf.timeZone = TimeZone.getTimeZone("GMT")
return sdf.format(dt)
}
}

271
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/IflytekTtsWs.kt

@ -0,0 +1,271 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import okhttp3.*
import okhttp3.HttpUrl.Companion.toHttpUrl
import okio.ByteString
import java.net.URL
import java.nio.charset.StandardCharsets
import java.text.SimpleDateFormat
import java.util.Base64
import java.util.Locale
import java.util.TimeZone
import javax.crypto.Mac
import javax.crypto.spec.SecretKeySpec
/**
* 讯飞在线TTS WebSocket客户端
*
* 负责生成鉴权URL并通过WebSocket进行流式语音合成,返回PCM原始数据
*/
class IflytekTtsWs(
private val hostUrl: String = "https://tts-api.xfyun.cn/v2/tts"
) {
private val tag = "IflytekTtsWs"
private val okHttp = OkHttpClient()
private var webSocket: WebSocket? = null
private var isClosed = false
/**
* 合成回调接口
*/
interface Callback {
fun onStarted()
fun onAudioChunk(data: ByteArray)
fun onCompleted(sid: String?)
fun onError(code: Int?, message: String)
}
/**
* 进行一次文本合成
*
* 参数说明:
* - appId: 讯飞应用 ID
* - apiKey: API Key
* - apiSecret: API Secret
* - text: 待合成文本(将按 `tte` 编码)
* - vcn: 发音人标识(如 `x4_yezi`),直接覆盖服务默认值
* - tte: 文本编码(`UTF8`/`UNICODE`),默认 `UTF8`
* - pitch: 音高 0–100,默认 50
* - speed: 语速 0–100,默认 50
* - aue: 音频编码,`raw` 表示 PCM 原始流
* - callback: 合成回调,流式返回音频分片并在完成时回调
*/
fun synthesize(
appId: String,
apiKey: String,
apiSecret: String,
text: String,
vcn: String = "x4_yezi",
tte: String = "UTF8",
pitch: Int = 50,
speed: Int = 50,
aue: String = "raw",
callback: Callback
) {
try {
val authUrl = buildAuthUrl(hostUrl, apiKey, apiSecret).replace("https://", "wss://")
val request = Request.Builder().url(authUrl).build()
isClosed = false
webSocket = okHttp.newWebSocket(request, object : WebSocketListener() {
override fun onOpen(ws: WebSocket, response: Response) {
callback.onStarted()
val payload = buildRequestJson(appId, text, vcn, tte, pitch, speed, aue)
ws.send(payload)
}
override fun onMessage(ws: WebSocket, textMsg: String) {
try {
val parse = JsonParse.fromJson(textMsg)
if (parse.code != 0) {
callback.onError(parse.code, "sid=${parse.sid}")
return
}
val data = parse.data
if (data != null) {
if (!data.audio.isNullOrEmpty()) {
val bytes = Base64.getDecoder().decode(data.audio)
callback.onAudioChunk(bytes)
}
if (data.status == 2) {
callback.onCompleted(parse.sid)
close()
}
}
} catch (e: Exception) {
callback.onError(null, "parse error: ${e.message}")
}
}
override fun onMessage(ws: WebSocket, bytes: ByteString) {
// iFlytek服务使用文本消息承载;此分支通常不会触发
}
override fun onFailure(ws: WebSocket, t: Throwable, response: Response?) {
callback.onError(null, t.message ?: "unknown error")
close()
}
override fun onClosing(ws: WebSocket, code: Int, reason: String) {
ws.close(code, reason)
}
override fun onClosed(ws: WebSocket, code: Int, reason: String) {
isClosed = true
}
})
} catch (e: Exception) {
callback.onError(null, e.message ?: "exception")
close()
}
}
/**
* 关闭当前连接
*/
fun close() {
try {
if (!isClosed) {
webSocket?.close(1000, "normal")
}
} catch (_: Exception) {
} finally {
webSocket = null
isClosed = true
}
}
/**
* 按语言进行文本合成
*
* 用法说明:
* - iFlytek TTS 的语种由发音人 `vcn` 决定
* - 可传入 `language`(如 `ja-JP`、`ko-KR`),并通过 `vcnOverride` 指定发音人
* - 若未提供覆盖值,将尝试按语言解析默认发音人;未配置则抛出错误
*/
fun synthesizeByLanguage(
appId: String,
apiKey: String,
apiSecret: String,
text: String,
language: String,
callback: Callback,
vcnOverride: String? = null,
tte: String = "UTF8",
pitch: Int = 50,
speed: Int = 50,
aue: String = "raw"
) {
val vcn = resolveVcn(language, vcnOverride)
synthesize(appId, apiKey, apiSecret, text, vcn, tte, pitch, speed, aue, callback)
}
/**
* 解析语言对应的默认发音人
*/
private fun resolveVcn(language: String, vcnOverride: String?): String {
if (!vcnOverride.isNullOrBlank()) return vcnOverride
return when (language.lowercase(Locale.ROOT)) {
"zh", "zh-cn" -> "x4_yezi"
// 以下语种需在控制台开启对应小语种发音人,并传入具体发音人ID
"ja", "ja-jp" -> throw IllegalArgumentException("未配置日语发音人vcn,请传入vcnOverride")
"ko", "ko-kr" -> throw IllegalArgumentException("未配置韩语发音人vcn,请传入vcnOverride")
else -> throw IllegalArgumentException("未配置语种(${language})发音人vcn,请传入vcnOverride")
}
}
/**
* 构造业务请求JSON
*/
private fun buildRequestJson(
appId: String,
text: String,
vcn: String,
tte: String,
pitch: Int,
speed: Int,
aue: String
): String {
val txt = when (tte.uppercase(Locale.ROOT)) {
"UNICODE" -> Base64.getEncoder().encodeToString(text.toByteArray(Charsets.UTF_16LE))
else -> Base64.getEncoder().encodeToString(text.toByteArray(StandardCharsets.UTF_8))
}
return """
{
"common": { "app_id": "$appId" },
"business": {
"aue": "$aue",
"tte": "$tte",
"ent": "intp65",
"vcn": "$vcn",
"pitch": $pitch,
"speed": $speed
},
"data": {
"status": 2,
"text": "$txt"
}
}
""".trimIndent()
}
/**
* 生成带鉴权的URL
*/
private fun buildAuthUrl(hostUrl: String, apiKey: String, apiSecret: String): String {
val url = URL(hostUrl)
val fmt = SimpleDateFormat("EEE, dd MMM yyyy HH:mm:ss z", Locale.US).apply {
timeZone = TimeZone.getTimeZone("GMT")
}
val date = fmt.format(java.util.Date())
val preStr = "host: ${url.host}\n" +
"date: $date\n" +
"GET ${url.path} HTTP/1.1"
val mac = Mac.getInstance("hmacsha256")
val spec = SecretKeySpec(apiSecret.toByteArray(StandardCharsets.UTF_8), "hmacsha256")
mac.init(spec)
val hex = mac.doFinal(preStr.toByteArray(StandardCharsets.UTF_8))
val sha = Base64.getEncoder().encodeToString(hex)
val authorization = String.format(
"api_key=\"%s\", algorithm=\"%s\", headers=\"%s\", signature=\"%s\"",
apiKey, "hmac-sha256", "host date request-line", sha
)
val httpUrl = "https://${url.host}${url.path}".toHttpUrl().newBuilder()
.addQueryParameter("authorization", Base64.getEncoder().encodeToString(authorization.toByteArray(StandardCharsets.UTF_8)))
.addQueryParameter("date", date)
.addQueryParameter("host", url.host)
.build()
return httpUrl.toString()
}
/**
* 解析服务端JSON
*/
private data class JsonParse(val code: Int, val sid: String?, val data: Data?) {
data class Data(val status: Int, val audio: String?)
companion object {
fun fromJson(s: String): JsonParse {
// 简单解析以减少依赖:仅提取必须字段
// 格式可能为 {"code":0, "sid":"...", "data":{"status":1,"audio":"..."}}
var code = -1
var sid: String? = null
var status = -1
var audio: String? = null
fun extract(re: Regex): String? = re.find(s)?.groupValues?.getOrNull(1)
extract(Regex("\"code\"\\s*:\\s*(\\d+)"))?.let { code = it.toIntOrNull() ?: -1 }
sid = extract(Regex("\"sid\"\\s*:\\s*\"([^\"]+)\""))
extract(Regex("\"status\"\\s*:\\s*(\\d+)"))?.let { status = it.toIntOrNull() ?: -1 }
audio = extract(Regex("\"audio\"\\s*:\\s*\"([^\"]*)\""))
return JsonParse(code, sid, if (status >= 0) Data(status, audio) else null)
}
}
}
}

250
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftAsrServiceImpl.kt

@ -0,0 +1,250 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import com.microsoft.cognitiveservices.speech.*
import com.microsoft.cognitiveservices.speech.audio.AudioConfig
import java.util.concurrent.atomic.AtomicBoolean
/**
* Microsoft Azure ASR (Speech-to-Text) 服务实现
*/
class MicrosoftAsrServiceImpl {
companion object {
private const val TAG = "MicrosoftAsrService"
}
private var speechConfig: SpeechConfig? = null
private var recognizer: SpeechRecognizer? = null
private var audioConfig: AudioConfig? = null
private var audioProcessor: AudioProcessor? = null
private var isRecognizing = AtomicBoolean(false)
private var callback: AsrCallback? = null
/**
* ASR 回调接口
*/
interface AsrCallback {
fun onRecognizing(text: String, confidence: Float)
fun onRecognized(text: String, confidence: Float)
fun onSessionStarted()
fun onSessionStopped()
fun onCanceled(error: String)
fun onError(message: String)
}
/**
* ASR 配置参数
*/
data class AsrConfig(
val subscriptionKey: String,
val region: String,
val language: String,
val enableAutoLanguageDetection: Boolean = false
)
/**
* 初始化 ASR 服务
*/
fun initialize(config: AsrConfig, callback: AsrCallback): Boolean {
this.callback = callback
return try {
Log.d(TAG, "初始化 Azure ASR 服务: region=${config.region}, lang=${config.language}")
// 1. 初始化 SpeechConfig
speechConfig = SpeechConfig.fromSubscription(config.subscriptionKey, config.region).apply {
speechRecognitionLanguage = config.language
if (config.enableAutoLanguageDetection) {
setProperty("SpeechServiceConnection_LanguageIdMode", "Continuous")
}
}
// 2. 初始化 AudioProcessor (录音处理)
initializeAudioProcessor()
// 3. 初始化 SpeechRecognizer
setupSpeechRecognizer(config.language)
Log.d(TAG, "Azure ASR 服务初始化成功")
true
} catch (e: Exception) {
Log.e(TAG, "Azure ASR 服务初始化失败", e)
callback.onError("初始化失败: ${e.message}")
false
}
}
private fun initializeAudioProcessor() {
audioProcessor = AudioProcessor().apply {
initialize()
}
// 使用 AudioProcessor 的流作为输入
audioConfig = AudioConfig.fromStreamInput(audioProcessor?.getAudioStream())
}
private fun setupSpeechRecognizer(language: String) {
recognizer?.close()
recognizer = SpeechRecognizer(speechConfig, audioConfig).apply {
// 识别中
recognizing.addEventListener { _, event ->
if (event.result.text.isNotEmpty()) {
val confidence = extractConfidence(event.result)
callback?.onRecognizing(event.result.text, confidence)
}
}
// 识别完成
recognized.addEventListener { _, event ->
if (event.result.reason == ResultReason.RecognizedSpeech) {
if (event.result.text.isNotEmpty()) {
val confidence = extractConfidence(event.result)
callback?.onRecognized(event.result.text, confidence)
}
} else if (event.result.reason == ResultReason.NoMatch) {
Log.d(TAG, "未识别到语音")
}
}
// 会话开始
sessionStarted.addEventListener { _, _ ->
Log.d(TAG, "会话开始")
isRecognizing.set(true)
callback?.onSessionStarted()
}
// 会话停止
sessionStopped.addEventListener { _, _ ->
Log.d(TAG, "会话停止")
isRecognizing.set(false)
callback?.onSessionStopped()
}
// 取消
canceled.addEventListener { _, event ->
Log.d(TAG, "识别取消: ${event.errorDetails}")
isRecognizing.set(false)
callback?.onCanceled(event.errorDetails ?: "未知错误")
}
}
}
/**
* 启动连续识别
*/
fun startContinuousRecognition(): Boolean {
if (isRecognizing.get()) {
Log.w(TAG, "识别已在进行中")
return true
}
return try {
audioProcessor?.startRecording()
recognizer?.startContinuousRecognitionAsync()
Log.d(TAG, "启动连续识别成功")
true
} catch (e: Exception) {
Log.e(TAG, "启动连续识别失败", e)
callback?.onError("启动识别失败: ${e.message}")
false
}
}
/**
* 停止连续识别
*/
fun stopContinuousRecognition() {
Log.d(TAG, "停止连续识别")
try {
recognizer?.stopContinuousRecognitionAsync()
audioProcessor?.stopRecording()
isRecognizing.set(false)
} catch (e: Exception) {
Log.e(TAG, "停止识别失败", e)
}
}
/**
* 推送音频数据
*/
fun pushAudioData(data: ByteArray) {
audioProcessor?.pushAudioData(data)
}
/**
* 释放资源
*/
fun dispose() {
try {
stopContinuousRecognition()
recognizer?.close()
speechConfig?.close()
audioConfig?.close() // AudioConfig close 可能会抛异常如果流已经关闭,视 SDK 版本而定,加 try-catch
recognizer = null
speechConfig = null
audioConfig = null
audioProcessor = null
callback = null
} catch (e: Exception) {
Log.e(TAG, "释放资源出错", e)
}
}
/**
* 提取置信度 (占位实现,与原逻辑一致)
*/
private fun extractConfidence(result: SpeechRecognitionResult): Float {
return try {
// 这里可以添加具体的提取逻辑
0.8f
} catch (e: Exception) {
0.5f
}
}
/**
* 开启录音文件保存 (如果需要直接在这里控制)
* 原逻辑中 IntegratedSpeechTranslationService 控制了 RecordFile,
* 这里可以提供接口让外部获取 AudioProcessor 或直接传入 RecordFile,
* 但为了简化,暂时保持 AudioProcessor 内部管理或外部通过其他方式处理。
* 根据原代码,RecordFile 似乎是独立的。
*/
}
/**
* 音频处理器
* 封装 Azure PushAudioInputStream,用于接收外部音频数据并提供给 ASR 服务
*/
class AudioProcessor {
private var pushStream: com.microsoft.cognitiveservices.speech.audio.PushAudioInputStream? = null
fun initialize() {
// 使用默认的 PCM 格式: 16kHz, 16bit, 单声道
val format = com.microsoft.cognitiveservices.speech.audio.AudioStreamFormat.getWaveFormatPCM(16000L, 16, 1)
pushStream = com.microsoft.cognitiveservices.speech.audio.AudioInputStream.createPushStream(format)
}
fun getAudioStream(): com.microsoft.cognitiveservices.speech.audio.AudioInputStream? {
return pushStream
}
fun startRecording() {
if (pushStream == null) {
initialize()
}
}
fun stopRecording() {
// 可以在这里做一些清理
}
fun pushAudioData(data: ByteArray) {
pushStream?.write(data)
}
fun close() {
pushStream?.close()
pushStream = null
}
}

266
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTTSServiceImpl.kt

@ -0,0 +1,266 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import com.microsoft.cognitiveservices.speech.*
import com.microsoft.cognitiveservices.speech.audio.AudioConfig
import com.microsoft.cognitiveservices.speech.audio.AudioOutputStream
import com.microsoft.cognitiveservices.speech.audio.PushAudioOutputStream
import com.microsoft.cognitiveservices.speech.audio.PushAudioOutputStreamCallback
import java.util.concurrent.atomic.AtomicBoolean
/**
* Microsoft Azure TTS (Text-to-Speech) 服务实现
*/
class MicrosoftTTSServiceImpl {
companion object {
private const val TAG = "MicrosoftTTSService"
}
private var speechConfig: SpeechConfig? = null
private var synthesizer: SpeechSynthesizer? = null
private var synthOutputStream: PushAudioOutputStream? = null
private var isSynthesizing = AtomicBoolean(false)
private var callback: TtsCallback? = null
// 内部音频缓冲,用于分片发送
private var ttsDataBuffer: ByteArray = ByteArray(0)
/**
* TTS 回调接口
*/
interface TtsCallback {
fun onSynthesisStarted(utteranceId: String, text: String)
fun onSynthesisProgress(utteranceId: String, text: String, progress: Float)
// 返回处理后的音频分片
fun onAudioChunkGenerated(utteranceId: String, text: String, audioChunk: ByteArray)
fun onSynthesisCompleted(utteranceId: String, text: String, fullAudio: ByteArray?)
fun onSynthesisFailed(utteranceId: String, text: String, error: String)
fun onError(message: String)
}
/**
* TTS 配置参数
*/
data class TtsConfig(
val subscriptionKey: String,
val region: String,
val targetLanguage: String,
val voiceName: String,
val speechRate: String = "0%",
val speechPitch: String = "0%",
val speechVolume: String = "100%",
val enableAudioPlayback: Boolean = false
)
private var currentConfig: TtsConfig? = null
/**
* 初始化 TTS 服务
*/
fun initialize(config: TtsConfig, callback: TtsCallback): Boolean {
this.callback = callback
this.currentConfig = config
return try {
Log.d(TAG, "初始化 Azure TTS 服务: region=${config.region}, voice=${config.voiceName}")
speechConfig = SpeechConfig.fromSubscription(config.subscriptionKey, config.region).apply {
speechSynthesisLanguage = config.targetLanguage
speechSynthesisVoiceName = config.voiceName.ifEmpty { config.targetLanguage }
setSpeechSynthesisOutputFormat(SpeechSynthesisOutputFormat.Riff16Khz16BitMonoPcm)
}
setupSpeechSynthesizer(config)
Log.d(TAG, "Azure TTS 服务初始化成功")
true
} catch (e: Exception) {
Log.e(TAG, "Azure TTS 服务初始化失败", e)
callback.onError("初始化失败: ${e.message}")
false
}
}
private fun setupSpeechSynthesizer(config: TtsConfig) {
synthesizer?.close()
// 根据配置决定音频输出方式
val audioConfig = if (config.enableAudioPlayback) {
AudioConfig.fromDefaultSpeakerOutput()
} else {
// 创建空消费的推流,避免使用 null AudioConfig
synthOutputStream?.close()
synthOutputStream = AudioOutputStream.createPushStream(object : PushAudioOutputStreamCallback() {
override fun write(dataBuffer: ByteArray): Int {
// 丢弃数据但保持流有效 (因为我们会通过事件获取数据)
return dataBuffer.size
}
override fun close() { /* no-op */ }
})
AudioConfig.fromStreamOutput(synthOutputStream)
}
synthesizer = SpeechSynthesizer(speechConfig, audioConfig).apply {
// 合成开始
SynthesisStarted.addEventListener { _, event ->
Log.d(TAG, "合成开始")
ttsDataBuffer = ByteArray(0) // 重置缓冲
isSynthesizing.set(true)
// 这里我们无法直接获取 utteranceId 和 text,因为事件对象里没有
// 它们需要由调用 synthesizeText 的地方管理,或者通过某种方式传递
// 为了简化,我们假设 Service 层不持有 currentUtteranceId,
// 而是由 orchestrator 传递,但事件监听器这里是异步的...
// *注意*: 原代码是使用了类成员 currentSynthesisId 来追踪。
// 我们这里也需要类似机制,或者简化回调只通知状态,不带 ID(但这不符合需求)。
// 只能使用成员变量来暂存当前的 ID。
}
// 合成进行中
Synthesizing.addEventListener { _, event ->
val incoming = event.result.audioData ?: ByteArray(0)
// 这里的 currentUtteranceId 依赖于 synthesizeText 调用时的设置
// 由于 SpeechSynthesizer 是单实例串行工作的,这通常是安全的
processSynthesisAudioChunk(incoming)
// 进度回调 (模拟)
callback?.onSynthesisProgress(currentUtteranceId, currentText, 0.5f)
}
// 合成完成
SynthesisCompleted.addEventListener { _, event ->
Log.d(TAG, "合成完成")
isSynthesizing.set(false)
callback?.onSynthesisCompleted(currentUtteranceId, currentText, event.result.audioData)
}
// 合成取消
SynthesisCanceled.addEventListener { _, event ->
Log.d(TAG, "合成取消: ${event.result.reason}")
isSynthesizing.set(false)
callback?.onSynthesisFailed(currentUtteranceId, currentText, "取消: ${event.result.reason}")
}
}
}
// 用于追踪当前合成任务的状态
private var currentUtteranceId: String = ""
private var currentText: String = ""
/**
* 执行语音合成
*/
fun synthesize(utteranceId: String, text: String): ByteArray? {
if (isSynthesizing.get()) {
Log.w(TAG, "正在合成中,忽略请求: $utteranceId")
return null
}
currentUtteranceId = utteranceId
currentText = text
callback?.onSynthesisStarted(utteranceId, text)
try {
isSynthesizing.set(true)
val ssml = generateOptimizedSsml(text)
// 使用 SpeakSsmlAsync 并等待结果
val result = synthesizer?.SpeakSsmlAsync(ssml)?.get()
if (result?.reason == ResultReason.SynthesizingAudioCompleted) {
// 发送剩余缓冲数据
flushAudioBuffer()
return result.audioData
} else {
return null
}
} catch (e: Exception) {
Log.e(TAG, "合成异常", e)
callback?.onSynthesisFailed(utteranceId, text, e.message ?: "未知异常")
return null
} finally {
isSynthesizing.set(false)
}
}
/**
* 停止/取消合成
*/
fun stop() {
try {
synthesizer?.StopSpeakingAsync()
} catch (e: Exception) {
Log.e(TAG, "停止合成失败", e)
}
}
fun dispose() {
try {
stop()
synthesizer?.close()
speechConfig?.close()
synthOutputStream?.close()
synthesizer = null
speechConfig = null
synthOutputStream = null
callback = null
} catch (e: Exception) {
Log.e(TAG, "释放资源失败", e)
}
}
private fun processSynthesisAudioChunk(incoming: ByteArray, multiple: Int = 1280) {
if (incoming.isNotEmpty()) {
ttsDataBuffer += incoming
}
val sendLen = (ttsDataBuffer.size / multiple) * multiple
if (sendLen > 0) {
val chunk = ttsDataBuffer.copyOfRange(0, sendLen)
callback?.onAudioChunkGenerated(currentUtteranceId, currentText, chunk)
ttsDataBuffer = ttsDataBuffer.copyOfRange(sendLen, ttsDataBuffer.size)
}
}
private fun flushAudioBuffer() {
if (ttsDataBuffer.isNotEmpty()) {
callback?.onAudioChunkGenerated(currentUtteranceId, currentText, ttsDataBuffer)
ttsDataBuffer = ByteArray(0)
}
}
private fun generateOptimizedSsml(text: String): String {
val config = currentConfig ?: return ""
val cleanText = cleanTextForSynthesis(text)
val escapedText = escapeXmlText(cleanText)
return """
<speak version="1.0" xmlns="http://www.w3.org/2001/10/synthesis" xml:lang="${config.targetLanguage}">
<voice name="${config.voiceName}">
<prosody rate="${config.speechRate}" pitch="${config.speechPitch}" volume="${config.speechVolume}">
$escapedText
</prosody>
</voice>
</speak>
""".trimIndent()
}
private fun cleanTextForSynthesis(text: String): String {
return text
.replace(Regex("https?://[^\\s]+"), "") // 移除URL
.replace(Regex("[\\p{So}\\p{Sk}]"), "") // 移除表情符号
.replace(Regex("\\s+"), " ") // 合并多个空格
.trim()
}
private fun escapeXmlText(text: String): String {
return text
.replace("&", "&amp;")
.replace("<", "&lt;")
.replace(">", "&gt;")
.replace("\"", "&quot;")
.replace("'", "&apos;")
}
}

189
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTranslationAndTtsService.kt

@ -0,0 +1,189 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.withContext
import java.io.File
import java.io.FileOutputStream
import java.util.UUID
/**
* 微软翻译与TTS综合服务
*
* 功能:
* 1. 接收输入文本、源语言代码、目标语言代码
* 2. 调用微软翻译服务将文本翻译为目标语言
* 3. 调用微软TTS服务将翻译后的文本合成为语音
* 4. 将合成的音频保存为 WAV 文件,文件名为 "{targetLanguage}.wav"
*/
class MicrosoftTranslationAndTtsService {
companion object {
private const val TAG = "MsTransAndTtsService"
}
private val translationService = MicrosoftTranslationServiceImpl()
private val ttsService = MicrosoftTTSServiceImpl()
private var serviceConfig: ServiceConfig? = null
/**
* 服务配置
* @param speechSubscriptionKey Azure 语音服务订阅密钥(用于 TTS)
* @param speechRegion Azure 语音服务区域 (例如 "eastus")
* @param translationSubscriptionKey Azure 翻译服务订阅密钥(用于文本翻译)
* @param translationRegion Azure 翻译服务区域
*/
data class ServiceConfig(
val speechSubscriptionKey: String,
val speechRegion: String,
val translationSubscriptionKey: String = speechSubscriptionKey,
val translationRegion: String = speechRegion
)
/**
* 初始化服务
* @param config 服务配置
*/
suspend fun initialize(config: ServiceConfig): Boolean {
this.serviceConfig = config
// 初始化翻译服务
// 翻译服务支持的配置键: subscriptionKey, region
val transConfig = mapOf(
"subscriptionKey" to config.translationSubscriptionKey,
"region" to config.translationRegion
)
Log.d(TAG, "正在初始化翻译服务...")
val transInitSuccess = translationService.initialize(transConfig)
if (!transInitSuccess) {
Log.e(TAG, "翻译服务初始化失败")
return false
}
Log.d(TAG, "服务初始化成功")
return true
}
/**
* 执行 翻译 -> TTS -> 保存文件 流程
*
* @param text 待翻译和合成的文本
* @param sourceLanguage 源语言代码 (例如 "zh-CN")
* @param targetLanguage 翻译目标语言代码 (例如 "en-US")
* @param targetTTSLanguage 实际用于TTS的语言代码(可与翻译目标不同)
* @param outputDirectory 音频文件保存目录
* @return 生成的音频文件绝对路径,如果失败则返回 null
*/
suspend fun processAndSaveAudio(
text: String,
sourceLanguage: String,
targetLanguage: String,
targetTTSLanguage: String,
outputDirectory: File
): String? = withContext(Dispatchers.IO) {
val config = serviceConfig
if (config == null) {
Log.e(TAG, "服务尚未初始化,请先调用 initialize()")
return@withContext null
}
// --- 步骤 1: 文本翻译 ---
Log.d(TAG, "开始翻译: text=$text, $sourceLanguage -> $targetLanguage")
val translationResult = translationService.translateText(text, sourceLanguage, targetLanguage)
if (!translationResult.success || translationResult.translatedText.isNullOrEmpty()) {
Log.e(TAG, "翻译失败: ${translationResult.error}")
return@withContext null
}
val translatedText = translationResult.translatedText!!
Log.i(TAG, "翻译完成: $translatedText")
// --- 步骤 2: TTS 合成 ---
Log.d(TAG, "开始TTS合成...")
// 根据上传的 TTS 语言代码优先配置 TTS,其次回退到翻译目标语言
val ttsConfig = resolveTtsConfig(targetTTSLanguage, config)
// 创建一个空的 Callback,因为我们主要依赖 synthesize 方法的返回值
val ttsCallback = object : MicrosoftTTSServiceImpl.TtsCallback {
override fun onSynthesisStarted(utteranceId: String, text: String) {}
override fun onSynthesisProgress(utteranceId: String, text: String, progress: Float) {}
override fun onAudioChunkGenerated(utteranceId: String, text: String, audioChunk: ByteArray) {}
override fun onSynthesisCompleted(utteranceId: String, text: String, fullAudio: ByteArray?) {}
override fun onSynthesisFailed(utteranceId: String, text: String, error: String) {
Log.w(TAG, "TTS Callback 报告失败: $error")
}
override fun onError(message: String) {
Log.w(TAG, "TTS Callback 报告错误: $message")
}
}
val ttsInitSuccess = ttsService.initialize(ttsConfig, ttsCallback)
if (!ttsInitSuccess) {
Log.e(TAG, "TTS 服务初始化失败")
return@withContext null
}
val utteranceId = UUID.randomUUID().toString()
val audioData = ttsService.synthesize(utteranceId, translatedText)
if (audioData == null || audioData.isEmpty()) {
Log.e(TAG, "TTS 合成返回空数据")
return@withContext null
}
Log.i(TAG, "TTS 合成成功,数据大小: ${audioData.size} bytes")
// --- 步骤 3: 保存文件 ---
try {
if (!outputDirectory.exists()) {
outputDirectory.mkdirs()
}
// 文件名为目标语言代码.wav
val fileName = "$targetLanguage.wav"
val outputFile = File(outputDirectory, fileName)
FileOutputStream(outputFile).use { fos ->
fos.write(audioData)
fos.flush()
}
Log.i(TAG, "音频文件已保存: ${outputFile.absolutePath}")
return@withContext outputFile.absolutePath
} catch (e: Exception) {
Log.e(TAG, "保存音频文件失败", e)
return@withContext null
}
}
/**
* 根据目标语言构建 TTS 配置
* @param targetTTSLanguage 翻译目标语言代码(如 en、en-US、zh-Hans)
* @param config 服务配置(包含语音服务 key/region)
*/
private fun resolveTtsConfig(
targetTTSLanguage: String,
config: ServiceConfig
): MicrosoftTTSServiceImpl.TtsConfig {
return MicrosoftTTSServiceImpl.TtsConfig(
subscriptionKey = config.speechSubscriptionKey,
region = config.speechRegion,
targetLanguage = targetTTSLanguage,
voiceName = targetTTSLanguage,
enableAudioPlayback = false
)
}
/**
* 释放资源
*/
fun dispose() {
translationService.dispose()
ttsService.dispose()
}
}

268
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/MicrosoftTranslationServiceImpl.kt

@ -0,0 +1,268 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.delay
import kotlinx.coroutines.withContext
import okhttp3.MediaType.Companion.toMediaType
import okhttp3.OkHttpClient
import okhttp3.Request
import okhttp3.RequestBody
import org.json.JSONArray
import org.json.JSONObject
import okhttp3.HttpUrl.Companion.toHttpUrl
import java.util.UUID
import java.util.concurrent.TimeUnit
/**
* 微软翻译服务实现
* 实现 `IntegratedSpeechTranslationService.TranslationServiceInterface` 接口
*/
class MicrosoftTranslationServiceImpl :
IntegratedSpeechTranslationService.TranslationServiceInterface {
companion object {
private const val TAG = "MicrosoftTranslationService"
private const val DEFAULT_ENDPOINT = "https://api.cognitive.microsofttranslator.com"
private const val API_PATH = "/translate"
private const val API_VERSION = "3.0"
}
private var isInitialized = false
private var subscriptionKey: String = ""
private var region: String = ""
private var endpoint: String = DEFAULT_ENDPOINT
private var maxRetryAttempts: Int = 3
private var timeoutMs: Long = 10000L
private val httpClient = OkHttpClient.Builder()
.connectTimeout(10, TimeUnit.SECONDS)
.readTimeout(30, TimeUnit.SECONDS)
.writeTimeout(30, TimeUnit.SECONDS)
.build()
/**
* 初始化翻译服务
* @param config 配置映射,支持 `subscriptionKey`/`accessKey`、`region`/`location`、`endpoint`、`maxRetryAttempts`、`timeout`
* @return 是否初始化成功
*/
override suspend fun initialize(config: Map<String, String>): Boolean {
return try {
subscriptionKey = (config["subscriptionKey"] ?: config["accessKey"] ?: "").trim()
region = (config["region"] ?: config["location"] ?: "").trim()
endpoint = (config["endpoint"] ?: DEFAULT_ENDPOINT).trim()
maxRetryAttempts = config["maxRetryAttempts"]?.toIntOrNull() ?: 3
timeoutMs = config["timeout"]?.toLongOrNull() ?: 10000L
Log.d(TAG, "初始化参数: endpoint=$endpoint, region=$region, maxRetryAttempts=$maxRetryAttempts, timeoutMs=$timeoutMs, key=${maskKey(subscriptionKey)}")
if (subscriptionKey.isEmpty()) {
Log.e(TAG, "缺少必要的订阅密钥(subscriptionKey)")
false
} else {
isInitialized = true
Log.d(TAG, "微软翻译服务初始化成功: endpoint=$endpoint, region=$region")
true
}
} catch (e: Exception) {
Log.e(TAG, "初始化失败", e)
false
}
}
/**
* 文本翻译
* @param text 待翻译文本
* @param sourceLanguage 源语言代码(如 `zh-CN`)
* @param targetLanguage 目标语言代码(如 `en-US`)
* @return 翻译结果
*/
override suspend fun translateText(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
Log.d(TAG, "翻译请求: 文本=${text.take(20)}..., 源=$sourceLanguage, 目标=$targetLanguage")
if (!isInitialized) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "翻译服务未初始化"
)
}
return withContext(Dispatchers.IO) {
Log.d(TAG, "翻译请求: 文本=${text.take(20)}..., 源=$sourceLanguage, 目标=$targetLanguage")
performTranslationWithRetry(text, sourceLanguage, targetLanguage)
}
}
/**
* 带重试的翻译执行
*/
private suspend fun performTranslationWithRetry(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
var lastError: String? = null
repeat(maxRetryAttempts) { attempt ->
Log.d(TAG, "翻译重试: 尝试=${attempt + 1}/$maxRetryAttempts, 文本长度=${text.length}, 源=$sourceLanguage, 目标=$targetLanguage")
val result = performTranslation(text, sourceLanguage, targetLanguage)
if (result.success) return result
lastError = result.error
if (attempt < maxRetryAttempts - 1) delay(500L * (attempt + 1))
}
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = lastError ?: "翻译失败"
)
}
/**
* 实际翻译执行
*/
private suspend fun performTranslation(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
try {
Log.d(TAG, "语言映射: sourceLanguage=$sourceLanguage, targetLanguage=$targetLanguage")
if (sourceLanguage.isEmpty() || targetLanguage.isEmpty()) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "不支持的语言代码: $sourceLanguage -> $targetLanguage"
)
}
val urlBuilder = endpoint.toHttpUrl().newBuilder()
.addEncodedPathSegments(API_PATH.trimStart('/'))
.addQueryParameter("api-version", API_VERSION)
.addQueryParameter("from", sourceLanguage)
.addQueryParameter("to", targetLanguage)
val url = urlBuilder.build()
Log.d(TAG, "请求URL: $url")
val bodyArray = JSONArray().put(JSONObject().put("text", text))
Log.d(TAG, "请求体长度: ${bodyArray.toString().length}")
val requestBody = RequestBody.create(
"application/json; charset=utf-8".toMediaType(),
bodyArray.toString()
)
val headers = mapOf(
"Ocp-Apim-Subscription-Key" to subscriptionKey,
"Ocp-Apim-Subscription-Region" to region,
"Content-Type" to "application/json",
"X-ClientTraceId" to UUID.randomUUID().toString()
)
Log.d(TAG, "请求头: key=${maskKey(subscriptionKey)}, region=$region")
val request = Request.Builder()
.url(url)
.post(requestBody)
.apply { headers.forEach { (k, v) -> addHeader(k, v) } }
.build()
val startNs = System.nanoTime()
val response = httpClient.newCall(request).execute()
val durationMs = TimeUnit.NANOSECONDS.toMillis(System.nanoTime() - startNs)
val responseBody = response.body?.string()
Log.d(TAG, "HTTP响应: code=${response.code}, bodyLen=${responseBody?.length ?: -1}, 耗时=${durationMs}ms")
if (!response.isSuccessful || responseBody == null) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "HTTP错误: ${response.code}"
)
}
val translation = extractTranslation(responseBody)
if (translation != null) {
Log.d(TAG, "解析翻译成功: 片段=${translation.take(64)}")
return IntegratedSpeechTranslationService.TranslationResult(
success = true,
translatedText = translation,
confidence = 0.9f
)
}
val error = extractError(responseBody)
Log.w(TAG, "解析翻译失败: ${error ?: "未知错误"}")
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = error ?: "翻译结果为空"
)
} catch (e: Exception) {
Log.e(TAG, "翻译请求异常", e)
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "请求异常: ${e.message}"
)
}
}
/**
* 解析翻译结果
*/
private fun extractTranslation(responseBody: String): String? {
return try {
val root = JSONArray(responseBody)
if (root.length() == 0) return null
val item = root.getJSONObject(0)
if (item.has("translations")) {
val translations = item.getJSONArray("translations")
Log.d(TAG, "translations 数量: ${translations.length()}")
if (translations.length() > 0) {
val first = translations.getJSONObject(0)
return first.optString("text", null)
}
}
null
} catch (e: Exception) {
Log.e(TAG, "提取翻译结果失败", e)
null
}
}
/**
* 解析错误信息
*/
private fun extractError(responseBody: String): String? {
return try {
// 错误响应通常为对象:{"error": {"code": ..., "message": ...}}
val obj = JSONObject(responseBody)
if (obj.has("error")) {
val err = obj.getJSONObject("error")
Log.w(TAG, "错误对象: code=${err.optString("code")}, message=${err.optString("message")}")
return err.optString("message", err.optString("code", "未知错误"))
}
null
} catch (_: Exception) {
null
}
}
/**
* 释放资源
*/
override fun dispose() {
isInitialized = false
try {
Log.d(TAG, "释放HTTP资源")
httpClient.dispatcher.executorService.shutdown()
httpClient.connectionPool.evictAll()
} catch (e: Exception) {
Log.e(TAG, "清理资源失败", e)
}
}
private fun maskKey(key: String): String {
if (key.isEmpty()) return "(empty)"
val shown = key.take(6)
return "$shown..." + "*".repeat(8)
}
}

357
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/VolcanoTranslationServiceImpl.kt

@ -0,0 +1,357 @@
package com.yunqiinnovation.azure_speech
import android.util.Log
import kotlinx.coroutines.Dispatchers
import kotlinx.coroutines.delay
import kotlinx.coroutines.withContext
import okhttp3.MediaType.Companion.toMediaType
import okhttp3.OkHttpClient
import okhttp3.Request
import okhttp3.RequestBody
import org.json.JSONArray
import org.json.JSONObject
import okhttp3.HttpUrl.Companion.toHttpUrl
import java.net.URLEncoder
import java.security.MessageDigest
import java.text.SimpleDateFormat
import java.util.Date
import java.util.Locale
import java.util.TimeZone
import java.util.concurrent.TimeUnit
import javax.crypto.Mac
import javax.crypto.spec.SecretKeySpec
/**
* 火山翻译服务实现
* 实现 `IntegratedSpeechTranslationService.TranslationServiceInterface` 接口
*/
class VolcanoTranslationServiceImpl :
IntegratedSpeechTranslationService.TranslationServiceInterface {
companion object {
private const val TAG = "VolcanoTranslationService"
private const val BASE_URL = "https://translate.volcengineapi.com"
private const val ENDPOINT = "/"
private const val SERVICE = "translate"
private const val VERSION = "2020-06-01"
private const val ACTION = "TranslateText"
private const val ALGORITHM = "HMAC-SHA256"
}
private var isInitialized = false
private var accessKey: String = ""
private var secretKey: String = ""
private var region: String = "cn-north-1"
private var maxRetryAttempts: Int = 3
private var timeout: Long = 10000L
// HTTP客户端
private val httpClient = OkHttpClient.Builder()
.connectTimeout(10, TimeUnit.SECONDS)
.readTimeout(30, TimeUnit.SECONDS)
.writeTimeout(30, TimeUnit.SECONDS)
.build()
/**
* 初始化翻译服务
* @param config 配置映射,包含 `accessKey`、`secretKey`、`region` 等
* @return 是否初始化成功
*/
override suspend fun initialize(config: Map<String, String>): Boolean {
return try {
accessKey = config["accessKey"] ?: ""
secretKey = config["secretKey"] ?: ""
region = config["region"] ?: "cn-north-1"
maxRetryAttempts = config["maxRetryAttempts"]?.toIntOrNull() ?: 3
timeout = config["timeout"]?.toLongOrNull() ?: 10000L
if (accessKey.isEmpty() || secretKey.isEmpty()) {
Log.e(TAG, "缺少必要的API密钥")
false
} else {
isInitialized = true
Log.d(TAG, "火山翻译服务初始化成功")
true
}
} catch (e: Exception) {
Log.e(TAG, "初始化失败", e)
false
}
}
/**
* 文本翻译
* @param text 待翻译文本
* @param sourceLanguage 源语言代码
* @param targetLanguage 目标语言代码
* @return 翻译结果
*/
override suspend fun translateText(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
if (!isInitialized) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "翻译服务未初始化"
)
}
Log.d(TAG, "火山翻译服务请求参数: text=$text, sourceLanguage=$sourceLanguage, targetLanguage=$targetLanguage")
return withContext(Dispatchers.IO) {
performTranslationWithRetry(text, sourceLanguage, targetLanguage)
}
}
/**
* 带重试的翻译执行
*/
private suspend fun performTranslationWithRetry(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
var lastException: Exception? = null
repeat(maxRetryAttempts) { attempt ->
try {
val result = performTranslation(text, sourceLanguage, targetLanguage)
if (result.success) return result
lastException = Exception(result.error)
} catch (e: Exception) {
lastException = e
Log.w(TAG, "翻译请求失败,尝试 ${attempt + 1}/$maxRetryAttempts", e)
if (attempt < maxRetryAttempts - 1) delay(500L * (attempt + 1))
}
}
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "翻译失败: ${lastException?.message ?: "未知错误"}"
)
}
/**
* 实际翻译执行
*/
private suspend fun performTranslation(
text: String,
sourceLanguage: String,
targetLanguage: String
): IntegratedSpeechTranslationService.TranslationResult {
try {
if (sourceLanguage.isEmpty() || targetLanguage.isEmpty()) {
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "不支持的语言代码: $sourceLanguage -> $targetLanguage"
)
}
val requestBody = JSONObject().apply {
put("SourceLanguage", sourceLanguage)
put("TargetLanguage", targetLanguage)
put("TextList", JSONArray().put(text))
}
val queryParams = mapOf(
"Action" to ACTION,
"Version" to VERSION,
"Region" to region,
"Service" to SERVICE
)
val headers = generateSignature("POST", requestBody, queryParams)
val urlBuilder = BASE_URL.toHttpUrl().newBuilder()
queryParams.forEach { (key, value) -> urlBuilder.addQueryParameter(key, value) }
val url = urlBuilder.build()
val requestBodyObj = RequestBody.create(
"application/json; charset=utf-8".toMediaType(),
requestBody.toString()
)
val request = Request.Builder()
.url(url)
.post(requestBodyObj)
.apply { headers.forEach { (k, v) -> addHeader(k, v) } }
.build()
val response = httpClient.newCall(request).execute()
if (response.isSuccessful) {
val responseBody = response.body?.string()
if (responseBody != null) {
val jsonResponse = JSONObject(responseBody)
val translation = extractTranslation(jsonResponse)
if (translation != null) {
return IntegratedSpeechTranslationService.TranslationResult(
success = true,
translatedText = translation,
confidence = 0.9f
)
} else {
val error = extractError(jsonResponse)
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = error ?: "翻译结果为空"
)
}
}
}
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "HTTP错误: ${response.code}"
)
} catch (e: Exception) {
Log.e(TAG, "翻译请求异常", e)
return IntegratedSpeechTranslationService.TranslationResult(
success = false,
error = "请求异常: ${e.message}"
)
}
}
/**
* 解析翻译结果
*/
private fun extractTranslation(jsonResponse: JSONObject): String? {
try {
if (jsonResponse.has("TranslationList")) {
val list = jsonResponse.getJSONArray("TranslationList")
if (list.length() > 0) {
val item = list.getJSONObject(0)
if (item.has("Translation")) return item.getString("Translation")
}
}
if (jsonResponse.has("Translation")) return jsonResponse.getString("Translation")
if (jsonResponse.has("Result")) {
val result = jsonResponse.getJSONObject("Result")
if (result.has("Translation")) return result.getString("Translation")
}
if (jsonResponse.has("Data")) {
val data = jsonResponse.getJSONObject("Data")
if (data.has("Translation")) return data.getString("Translation")
if (data.has("TranslationList")) {
val list = data.getJSONArray("TranslationList")
if (list.length() > 0) {
val item = list.getJSONObject(0)
if (item.has("Translation")) return item.getString("Translation")
}
}
}
} catch (e: Exception) {
Log.e(TAG, "提取翻译结果失败", e)
}
return null
}
/**
* 解析错误信息
*/
private fun extractError(jsonResponse: JSONObject): String? {
try {
if (jsonResponse.has("ResponseMetadata")) {
val metadata = jsonResponse.getJSONObject("ResponseMetadata")
if (metadata.has("Error")) {
val error = metadata.getJSONObject("Error")
return error.optString("Message", error.optString("Code", "未知错误"))
}
}
if (jsonResponse.has("Error")) {
val error = jsonResponse.getJSONObject("Error")
return error.optString("Message", error.optString("Code", "未知错误"))
}
} catch (e: Exception) {
Log.e(TAG, "提取错误信息失败", e)
}
return null
}
/**
* 生成签名头
*/
private fun generateSignature(
method: String,
requestBody: JSONObject,
queryParams: Map<String, String>
): Map<String, String> {
val now = Date()
val dateFormat = SimpleDateFormat("yyyyMMdd", Locale.US).apply { timeZone = TimeZone.getTimeZone("UTC") }
val timestampFormat = SimpleDateFormat("yyyyMMdd'T'HHmmss'Z'", Locale.US).apply { timeZone = TimeZone.getTimeZone("UTC") }
val date = dateFormat.format(now)
val timestamp = timestampFormat.format(now)
val canonicalQueryString = queryParams.toSortedMap().map { (k, v) ->
"${URLEncoder.encode(k, "UTF-8")}=${URLEncoder.encode(v, "UTF-8")}"
}.joinToString("&")
val contentType = "application/json"
val payloadHash = sha256(requestBody.toString())
val host = "translate.volcengineapi.com"
val canonicalHeaders = "host:$host\nx-date:$timestamp\n"
val signedHeaders = "host;x-date"
val canonicalRequest = "$method\n$ENDPOINT\n$canonicalQueryString\n$canonicalHeaders\n$signedHeaders\n$payloadHash"
val credentialScope = "$date/$region/$SERVICE/request"
val stringToSign = "$ALGORITHM\n$timestamp\n$credentialScope\n${sha256(canonicalRequest)}"
val kSecret = secretKey.toByteArray(Charsets.UTF_8)
val kDate = hmacSha256(kSecret, date)
val kRegion = hmacSha256(kDate, region)
val kService = hmacSha256(kRegion, SERVICE)
val kSigning = hmacSha256(kService, "request")
val signature = hmacSha256(kSigning, stringToSign).joinToString("") { "%02x".format(it) }
val authorization = "$ALGORITHM Credential=$accessKey/$credentialScope, SignedHeaders=$signedHeaders, Signature=$signature"
return mapOf(
"Content-Type" to contentType,
"X-Date" to timestamp,
"Authorization" to authorization,
"Host" to host
)
}
/**
* SHA-256 摘要
*/
private fun sha256(input: String): String {
val digest = MessageDigest.getInstance("SHA-256")
val hash = digest.digest(input.toByteArray(Charsets.UTF_8))
return hash.joinToString("") { "%02x".format(it) }
}
/**
* HMAC-SHA256 计算
*/
private fun hmacSha256(key: ByteArray, data: String): ByteArray {
val mac = Mac.getInstance("HmacSHA256")
val secretKeySpec = SecretKeySpec(key, "HmacSHA256")
mac.init(secretKeySpec)
return mac.doFinal(data.toByteArray(Charsets.UTF_8))
}
/**
* 释放资源
*/
override fun dispose() {
isInitialized = false
try {
val cleanup = Runnable {
try {
httpClient.dispatcher.executorService.shutdown()
httpClient.connectionPool.evictAll()
} catch (e: Exception) {
Log.e(TAG, "清理资源失败", e)
}
}
if (android.os.Looper.getMainLooper() == android.os.Looper.myLooper()) {
Thread(cleanup, "VolcanoDisposeThread").start()
} else {
cleanup.run()
}
} catch (e: Exception) {
Log.e(TAG, "清理资源失败", e)
}
}
}

235
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/ScreenCaptureForegroundService.kt

@ -0,0 +1,235 @@
package com.yunqiinnovation.azure_speech.tools
import android.app.Notification
import android.app.NotificationChannel
import android.app.NotificationManager
import android.app.PendingIntent
import android.app.Service
import android.content.Context
import android.content.Intent
import android.os.Build
import android.os.IBinder
import androidx.core.app.NotificationCompat
import android.util.Log
import android.media.projection.MediaProjectionManager
/**
* 屏幕捕获前台服务
* 以前台服务的形式执行 MediaProjection 屏幕捕获,满足 Android 14 的安全要求
*/
class ScreenCaptureForegroundService : Service() {
companion object {
private const val TAG = "ScreenCaptureService"
private const val NOTIFICATION_ID = 2001
private const val CHANNEL_ID = "screen_capture_channel"
private const val CHANNEL_NAME = "屏幕捕获服务"
const val ACTION_START = "com.yunqiinnovation.azure_speech.tools.SCREEN_CAPTURE_START"
const val ACTION_STOP = "com.yunqiinnovation.azure_speech.tools.SCREEN_CAPTURE_STOP"
const val EXTRA_RESULT_CODE = "result_code"
const val EXTRA_RESULT_DATA = "result_data"
const val EXTRA_WIDTH = "width"
const val EXTRA_HEIGHT = "height"
const val EXTRA_DPI = "dpi"
const val EXTRA_QUALITY = "quality"
// 全局音频数据监听器,供 Plugin 设置
var audioDataListener: ((ByteArray) -> Unit)? = null
/**
* 启动屏幕捕获前台服务
*/
fun startService(
context: Context,
resultCode: Int,
data: Intent,
width: Int?,
height: Int?,
dpi: Int?,
quality: Int
) {
val intent = Intent(context, ScreenCaptureForegroundService::class.java).apply {
action = ACTION_START
putExtra(EXTRA_RESULT_CODE, resultCode)
putExtra(EXTRA_RESULT_DATA, data)
if (width != null) putExtra(EXTRA_WIDTH, width)
if (height != null) putExtra(EXTRA_HEIGHT, height)
if (dpi != null) putExtra(EXTRA_DPI, dpi)
putExtra(EXTRA_QUALITY, quality)
}
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.O) {
context.startForegroundService(intent)
} else {
context.startService(intent)
}
Log.d(TAG, "屏幕捕获前台服务启动请求已发送")
}
/**
* 停止屏幕捕获前台服务
*/
fun stopService(context: Context) {
val intent = Intent(context, ScreenCaptureForegroundService::class.java).apply {
action = ACTION_STOP
}
context.startService(intent)
Log.d(TAG, "发送停止屏幕捕获服务指令")
}
}
private var screenCaptureManager: ScreenCaptureManager? = null
override fun onCreate() {
super.onCreate()
createNotificationChannel()
}
override fun onStartCommand(intent: Intent?, flags: Int, startId: Int): Int {
val action = intent?.action
Log.d(TAG, "onStartCommand action=$action")
// 如果是停止指令,直接停止服务,不需要 startForeground
if (action == ACTION_STOP) {
stopCapture()
stopForeground(true)
stopSelf()
return START_NOT_STICKY
}
// 创建前台通知,必须尽快调用
val notification = createNotification()
try {
// 检查权限,虽然 Manifest 声明了,但为了健壮性
startForeground(NOTIFICATION_ID, notification)
} catch (e: Exception) {
Log.e(TAG, "Start foreground failed: ${e.message}")
stopSelf()
return START_NOT_STICKY
}
when (action) {
ACTION_START -> {
try {
val resultCode = intent.getIntExtra(EXTRA_RESULT_CODE, 0)
val data = intent.getParcelableExtra<Intent>(EXTRA_RESULT_DATA)
val width = if (intent.hasExtra(EXTRA_WIDTH)) intent.getIntExtra(EXTRA_WIDTH, 0) else null
val height = if (intent.hasExtra(EXTRA_HEIGHT)) intent.getIntExtra(EXTRA_HEIGHT, 0) else null
val dpi = if (intent.hasExtra(EXTRA_DPI)) intent.getIntExtra(EXTRA_DPI, 0) else null
val quality = intent.getIntExtra(EXTRA_QUALITY, 80)
if (data == null) {
Log.e(TAG, "缺少授权数据,无法启动屏幕捕获")
stopSelf()
return START_NOT_STICKY
}
screenCaptureManager = ScreenCaptureManager(this)
// 设置音频回调,将数据转发给全局监听器
screenCaptureManager?.setAudioCallback(object : ScreenCaptureManager.AudioDataCallback {
override fun onAudio(data: ByteArray) {
// Log.d(TAG, "Service received audio data: ${data.size} bytes, listener: $audioDataListener")
if (audioDataListener != null) {
audioDataListener?.invoke(data)
} else {
Log.w(TAG, "audioDataListener is null")
}
}
})
screenCaptureManager?.startCapture(
resultCode = resultCode,
data = data,
targetWidth = width,
targetHeight = height,
targetDpi = dpi,
jpegQuality = quality
)
isCapturing = true
Log.i(TAG, "屏幕捕获已启动")
} catch (e: Exception) {
Log.e(TAG, "启动屏幕捕获失败: ${e.message}")
stopSelf()
return START_NOT_STICKY
}
}
ACTION_STOP -> {
// 已在顶部处理
return START_NOT_STICKY
}
else -> {
// 无操作
}
}
return START_STICKY
}
override fun onBind(intent: Intent?): IBinder? = null
override fun onDestroy() {
super.onDestroy()
stopCapture()
stopForeground(true)
}
/**
* 停止屏幕捕获并释放相关资源
*/
private fun stopCapture() {
try {
screenCaptureManager?.stopCapture()
screenCaptureManager = null
isCapturing = false
Log.i(TAG, "屏幕捕获已停止")
} catch (_: Exception) { }
}
/**
* 创建通知渠道
*/
private fun createNotificationChannel() {
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.O) {
val channel = NotificationChannel(
CHANNEL_ID,
CHANNEL_NAME,
NotificationManager.IMPORTANCE_LOW
).apply {
description = "用于屏幕捕获时保持应用运行"
setShowBadge(false)
setSound(null, null)
enableVibration(false)
}
val nm = getSystemService(Context.NOTIFICATION_SERVICE) as NotificationManager
nm.createNotificationChannel(channel)
}
}
/**
* 创建前台服务通知
*/
private fun createNotification(): Notification {
val intent = packageManager.getLaunchIntentForPackage(packageName)
val pendingIntent = PendingIntent.getActivity(
this,
0,
intent,
PendingIntent.FLAG_UPDATE_CURRENT or PendingIntent.FLAG_IMMUTABLE
)
return NotificationCompat.Builder(this, CHANNEL_ID)
.setContentTitle("语音助手正在运行")
.setContentText("正在进行屏幕捕获,点击返回应用")
.setSmallIcon(android.R.drawable.ic_menu_camera)
.setContentIntent(pendingIntent)
.setOngoing(true)
.setPriority(NotificationCompat.PRIORITY_LOW)
.setCategory(NotificationCompat.CATEGORY_SERVICE)
.setShowWhen(false)
.build()
}
}
@Volatile
var isCapturing: Boolean = false

318
local_plugins/azure_speech/android/src/main/kotlin/com/yunqiinnovation/azure_speech/tools/ScreenCaptureManager.kt

@ -0,0 +1,318 @@
package com.yunqiinnovation.azure_speech.tools
import android.content.Context
import android.content.Intent
import android.graphics.Bitmap
import android.graphics.PixelFormat
import android.hardware.display.DisplayManager
import android.hardware.display.VirtualDisplay
import android.media.ImageReader
import android.media.AudioRecord
import android.media.AudioFormat
import android.media.AudioAttributes
import android.media.AudioPlaybackCaptureConfiguration
import android.os.Build
import android.os.Environment
import android.media.projection.MediaProjection
import android.media.projection.MediaProjectionManager
import android.os.Handler
import android.os.HandlerThread
import android.util.DisplayMetrics
import android.util.Log
import android.view.WindowManager
import java.io.ByteArrayOutputStream
class ScreenCaptureManager(private val context: Context) {
companion object {
private const val TAG = "ScreenCaptureManager"
}
private var projectionManager: MediaProjectionManager? = null
private var mediaProjection: MediaProjection? = null
private var virtualDisplay: VirtualDisplay? = null
private var imageReader: ImageReader? = null
private var handlerThread: HandlerThread? = null
private var handler: Handler? = null
private var jpegQuality: Int = 80
// 音频捕获
private var audioRecord: AudioRecord? = null
private var audioThread: Thread? = null
private var isAudioCapturing: java.util.concurrent.atomic.AtomicBoolean =
java.util.concurrent.atomic.AtomicBoolean(false)
//private var audioRaf: java.io.RandomAccessFile? = null
//private var audioFile: java.io.File? = null
private var totalPcmBytes: Long = 0
private val audioSampleRate = 16000
private val audioChannels = 1
private val audioBitsPerSample = 16
private var audioCallback: AudioDataCallback? = null
/**
* 音频数据回调接口
*/
interface AudioDataCallback {
fun onAudio(data: ByteArray)
}
/**
* 设置音频数据回调
*/
fun setAudioCallback(callback: AudioDataCallback?) {
this.audioCallback = callback
}
/**
* 开始屏幕捕获并将帧通过提供的传输器发送
* @param resultCode 媒体投射授权结果码
* @param data 媒体投射授权返回的Intent
* @param targetWidth 目标宽度(可选,不传则使用屏幕真实宽度)
* @param targetHeight 目标高度(可选,不传则使用屏幕真实高度)
* @param targetDpi 目标DPI(可选,不传则使用屏幕DPI)
* @param jpegQuality JPEG压缩质量(0-100),默认80
*/
fun startCapture(
resultCode: Int,
data: Intent,
targetWidth: Int? = null,
targetHeight: Int? = null,
targetDpi: Int? = null,
jpegQuality: Int = 80
) {
if (mediaProjection != null) return
this.jpegQuality = jpegQuality.coerceIn(0, 100)
projectionManager = context.getSystemService(Context.MEDIA_PROJECTION_SERVICE) as MediaProjectionManager
mediaProjection = projectionManager?.getMediaProjection(resultCode, data)
if (mediaProjection == null) {
Log.e(TAG, "MediaProjection 创建失败")
return
}
val metrics = getRealDisplayMetrics()
val width = targetWidth ?: metrics.widthPixels
val height = targetHeight ?: metrics.heightPixels
val dpi = targetDpi ?: metrics.densityDpi
handlerThread = HandlerThread("ScreenCaptureThread").also { it.start() }
handler = Handler(handlerThread!!.looper)
// Android 14 要求在开始捕获前注册回调以管理资源
mediaProjection?.registerCallback(object : MediaProjection.Callback() {
override fun onStop() {
Log.i(TAG, "MediaProjection 回调: 捕获停止,清理资源")
try {
stopCapture()
} catch (e: Exception) {
Log.e(TAG, "回调清理资源失败: ${e.message}")
}
}
}, handler)
imageReader = ImageReader.newInstance(width, height, PixelFormat.RGBA_8888, 2)
imageReader?.setOnImageAvailableListener({ reader ->
val image = reader.acquireLatestImage() ?: return@setOnImageAvailableListener
try {
val w = image.width
val h = image.height
val plane = image.planes[0]
val buffer = plane.buffer
val pixelStride = plane.pixelStride
val rowStride = plane.rowStride
val rowPadding = rowStride - pixelStride * w
val bitmapWidth = w + rowPadding / pixelStride
val tempBitmap = Bitmap.createBitmap(bitmapWidth, h, Bitmap.Config.ARGB_8888)
tempBitmap.copyPixelsFromBuffer(buffer)
val finalBitmap = if (bitmapWidth != w) {
Bitmap.createBitmap(tempBitmap, 0, 0, w, h)
} else tempBitmap
val baos = ByteArrayOutputStream()
finalBitmap.compress(Bitmap.CompressFormat.JPEG, this.jpegQuality, baos)
val bytes = baos.toByteArray()
baos.reset()
//屏幕帧
//transmitter.sendFrame(bytes)
if (finalBitmap !== tempBitmap) finalBitmap.recycle()
tempBitmap.recycle()
} catch (e: Exception) {
Log.e(TAG, "处理屏幕帧失败: ${e.message}")
} finally {
image.close()
}
}, handler)
virtualDisplay = mediaProjection?.createVirtualDisplay(
"ScreenCapture",
width,
height,
dpi,
DisplayManager.VIRTUAL_DISPLAY_FLAG_AUTO_MIRROR,
imageReader!!.surface,
null,
handler
)
// 启动系统播放音频捕获(Android 10+)
if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.Q) {
startAudioCapture()
} else {
Log.w(TAG, "当前系统版本不支持AudioPlaybackCapture(需要Android 10+)")
}
Log.i(TAG, "屏幕捕获已启动 ${width}x${height}@${dpi}dpi")
}
/**
* 启动系统播放音频捕获(MediaProjection + AudioPlaybackCaptureConfiguration)
*/
private fun startAudioCapture() {
try {
val mp = mediaProjection ?: run {
Log.e(TAG, "MediaProjection 不可用,无法捕获系统音频")
return
}
// 构建播放音频捕获配置:捕获媒体/游戏等播放流
val config = AudioPlaybackCaptureConfiguration.Builder(mp)
.addMatchingUsage(AudioAttributes.USAGE_MEDIA)
.addMatchingUsage(AudioAttributes.USAGE_GAME)
.addMatchingUsage(AudioAttributes.USAGE_ASSISTANCE_SONIFICATION)
.build()
val sampleRate = audioSampleRate
val channelMask = AudioFormat.CHANNEL_IN_MONO
val encoding = AudioFormat.ENCODING_PCM_16BIT
val audioFormat = AudioFormat.Builder()
.setSampleRate(sampleRate)
.setChannelMask(channelMask)
.setEncoding(encoding)
.build()
audioRecord = AudioRecord.Builder()
.setAudioPlaybackCaptureConfig(config)
.setAudioFormat(audioFormat)
.build()
val minBuf = AudioRecord.getMinBufferSize(sampleRate, channelMask, encoding)
if (minBuf <= 0) {
Log.e(TAG, "无法获取有效的缓冲区大小: $minBuf")
return
}
// 准备WAV文件输出
// prepareWavFile()
audioRecord?.startRecording()
isAudioCapturing.set(true)
audioThread = Thread {
Log.i(TAG, "系统播放音频捕获线程已启动")
val buffer = ByteArray(minBuf)
var totalBytes: Long = 0
try {
while (isAudioCapturing.get()) {
val read = audioRecord?.read(buffer, 0, buffer.size) ?: 0
// Log.d(TAG, "AudioRecord read result: $read")
if (read > 0) {
totalBytes += read
totalPcmBytes += read
// 写入PCM到WAV数据区
// audioRaf?.write(buffer, 0, read)
// 回调音频数据
if (read == buffer.size) {
audioCallback?.onAudio(buffer.clone())
} else {
val validData = ByteArray(read)
System.arraycopy(buffer, 0, validData, 0, read)
audioCallback?.onAudio(validData)
}
// 调试输出:打印捕获到的字节数,避免过度日志
if (totalBytes % (minBuf * 50L) == 0L) {
Log.i(TAG, "累计捕获系统音频字节: $totalBytes")
}
}
}
} catch (e: Exception) {
Log.e(TAG, "系统音频捕获异常: ${e.message}")
} finally {
Log.i(TAG, "系统播放音频捕获线程结束,总字节: $totalBytes")
}
}.apply {
name = "PlaybackAudioCapture"
start()
}
} catch (e: Exception) {
Log.e(TAG, "启动系统音频捕获失败: ${e.message}")
stopAudioCapture()
}
}
/**
* 停止系统播放音频捕获并清理资源
*/
private fun stopAudioCapture() {
try {
isAudioCapturing.set(false)
audioRecord?.stop()
audioRecord?.release()
audioRecord = null
try { audioThread?.join(300) } catch (_: Exception) {}
audioThread = null
// finalizeWavFile()
} catch (e: Exception) {
Log.e(TAG, "停止系统音频捕获失败: ${e.message}")
}
}
/**
* 停止屏幕捕获并清理资源
*/
fun stopCapture() {
try {
virtualDisplay?.release()
virtualDisplay = null
imageReader?.close()
imageReader = null
mediaProjection?.stop()
mediaProjection = null
// 停止系统播放音频捕获
stopAudioCapture()
handlerThread?.quitSafely()
handlerThread = null
handler = null
} catch (e: Exception) {
Log.e(TAG, "停止捕获失败: ${e.message}")
}
}
/**
* 是否处于捕获中
* @return true 表示正在捕获,false 表示未捕获
*/
fun isCapturing(): Boolean {
return mediaProjection != null && virtualDisplay != null
}
private fun getRealDisplayMetrics(): DisplayMetrics {
val metrics = DisplayMetrics()
val wm = context.getSystemService(Context.WINDOW_SERVICE) as WindowManager
@Suppress("DEPRECATION")
wm.defaultDisplay.getRealMetrics(metrics)
return metrics
}
}

10
local_plugins/azure_speech/android/src/main/protos/HOWTO.md

@ -0,0 +1,10 @@
# gRPC-Protobuf文件构建指引
## Go语言
1. 安装[protoc](https://github.com/protocolbuffers/protobuf/releases/download/v21.12/protoc-21.12-linux-x86_64.zip)和[proto-gen-go](https://github.com/protocolbuffers/protobuf-go/releases/download/v1.32.0/protoc-gen-go.v1.32.0.linux.amd64.tar.gz)命令行工具
1. 假定你的Go项目代码仓库为`github.com/foo/bar`,并且已经初始化(若未初始化,请执行命令`go mod init github.com/foo/bar`命令完成初始化)
1. 将本目录(protos目录)复制到你的Go项目代码仓库中(务必放置在该仓库的根目录中,否则你需要自行修改`protos/build_go.sh`脚本)
3. 在仓库根目录中执行命令:`sh protos/build_go.sh`
4. 命令执行结束后,代码会生成到仓库根目录的`protogen`子目录中
5. 在Go代码中引用相关package,例:`import "../kotlin/com/yunqiinnovation/azure_speech/protos/github.com/foo/bar/protogen/products/understanding/ast"`

38
local_plugins/azure_speech/android/src/main/protos/build_go.sh

@ -0,0 +1,38 @@
#!/bin/bash
# You must have `protoc` CMD tool installed. You can download compiled binary for Linux here:
# https://github.com/protocolbuffers/protobuf/releases/download/v21.12/protoc-21.12-linux-x86_64.zip
# You must also have `proto-gen-go` installed, which you can find the compiled binary here:
# https://github.com/protocolbuffers/protobuf-go/releases/download/v1.32.0/protoc-gen-go.v1.32.0.linux.amd64.tar.gz
set -ex
proto_dir=$(dirname $0)
cd ${proto_dir}/..
import_prefix=$(go list -m)/protogen
protoc \
--proto_path=${proto_dir} \
--go_out=. \
--go_opt=Mcommon/events.proto=${import_prefix}/common/event \
--go_opt=Mcommon/rpcmeta.proto=${import_prefix}/common/rpcmeta \
--go_opt=Mproducts/understanding/base/au_base.proto=${import_prefix}/products/understanding/base \
--go_opt=Mproducts/understanding/ast/ast_service.proto=${import_prefix}/products/understanding/ast \
--go-grpc_out=. \
--go-grpc_opt=Mcommon/events.proto=${import_prefix}/common/event \
--go-grpc_opt=Mcommon/rpcmeta.proto=${import_prefix}/common/rpcmeta \
--go-grpc_opt=Mproducts/understanding/base/au_base.proto=${import_prefix}/products/understanding/base \
--go-grpc_opt=Mproducts/understanding/ast/ast_service.proto=${import_prefix}/products/understanding/ast \
${proto_dir}/common/events.proto \
${proto_dir}/common/rpcmeta.proto \
${proto_dir}/products/understanding/base/au_base.proto \
${proto_dir}/products/understanding/ast/ast_service.proto
# Copy generated files to the right place and remove redundant directories.
dest_dir=$(basename ${import_prefix})
redundant_dir=${import_prefix%%/*}
rsync -a --remove-source-files ${import_prefix}/ ${dest_dir}/
rm -rf ${redundant_dir}

133
local_plugins/azure_speech/android/src/main/protos/common/events.proto

@ -0,0 +1,133 @@
syntax = "proto3";
package data.speech.event;
option java_multiple_files = true;
option java_package = "data.speech.event";
enum Type {
option allow_alias = true;
/*
* All following event names do not conform to CONSTANT_CASE naming style
* because we intentionally make them consistant with the existing SAMI
* string-typed events so that these enum values can be easily converted to
* event strings by simply call their `String()` method (eg,
* `event.Type_StartTTS.String() == "StartTTS"`).
*/
// 默认事件,适用于不使用事件的方案或不需要传递事件的情况,
// 或者对于使用事件的方案,可以通过非0值来校验事件的合法性
None = 0;
/*
* 1 ~ 300 为与业务无关的通用事件
*/
/*** 1 ~ 99 为Connection相关事件 ***/
// 1 ~ 49 为上行Connection事件
StartConnection = 1;
StartTask = 1; // Alias of "StartConnection"
FinishConnection = 2;
FinishTask = 2; // Alias of "FinishConnection"
// 50 ~ 99 为下行Connection事件
// 成功建连
ConnectionStarted = 50;
TaskStarted = 50; // Alias of "ConnectionStarted"
// 建连失败(可能是无法通过权限认证)
ConnectionFailed = 51;
TaskFailed = 51; // Alias of "ConnectionFailed"
// 连接结束
ConnectionFinished = 52;
TaskFinished = 52; // Alias of "ConnectionFinished"
/*** 100 ~ 199 为Session、用量等类型事件 ***/
// 100 ~ 149 为上行Session事件
StartSession = 100;
CancelSession = 101;
FinishSession = 102;
// 150 ~ 199 为下行Session事件
SessionStarted = 150;
SessionCanceled = 151;
SessionFinished = 152;
SessionFailed = 153;
// 用量事件
UsageResponse = 154;
ChargeData = 154; // Alias of "UsageResponse"
/*** 200 ~ 299 为不跟特定方案绑定的通用事件,会被解决方案及下游消费 ***/
// 200 ~ 249 为上行通用事件
TaskRequest = 200;
UpdateConfig = 201;
ImageRequest = 202;
// 250 ~ 299 为下行通用事件
AudioMuted = 250;
/*
* 300及以上为业务相关的事件
*/
/*** 300 ~ 399 为TTS相关事件 ***/
// 300 ~ 349 为上行TTS事件
SayHello = 300;
// 350 ~ 399 为下行TTS事件
TTSSentenceStart = 350;
TTSSentenceEnd = 351;
TTSResponse = 352;
TTSEnded = 359;
PodcastRoundStart = 360;
PodcastRoundResponse = 361;
PodcastRoundEnd = 362;
/*** 400 ~ 499 为ASR相关事件 ***/
// 400 ~ 449 为上行ASR事件
// 450 ~ 499 为下行ASR事件
ASRInfo = 450;
ASRResponse = 451;
ASREnded = 459;
/*** 500 ~ 599 为对话相关事件 ***/
// 500 ~ 549 为上行对话事件
// (Ground-Truth-Alignment) text for speech synthesis
ChatTTSText = 500;
ChatTextQuery = 501;
// 550 ~ 599 为下行对话事件
ChatResponse = 550;
ChimeInStart = 551;
ChimeInEnd = 552;
ChatEnded = 559;
ThinkStart = 560;
ThinkResponse = 561;
ThinkEnd = 562;
ToolOutput = 563;
FCResponseStart = 564;
FCResponse = 565;
FCResponseEnd = 566;
/*** 600 ~ 699 为Machine Translation相关事件 ***/
// 600 ~ 649 为上行对话事件
// 650 ~ 699 为下行对话事件
// Events for source (original) language subtitle.
SourceSubtitleStart = 650;
SourceSubtitleResponse = 651;
SourceSubtitleEnd = 652;
// Events for target (translation) language subtitle.
TranslationSubtitleStart = 653;
TranslationSubtitleResponse = 654;
TranslationSubtitleEnd = 655;
}

100
local_plugins/azure_speech/android/src/main/protos/common/rpcmeta.proto

@ -0,0 +1,100 @@
syntax = "proto3";
package data.speech.common;
option java_multiple_files = true;
option java_package = "data.speech.common";
// NOTE(lucas): We define the fields of RequestMeta and ResponseMeta using
// PascalCase, which does not comform to normal practice, so that they are
// mostly compatible with the original definition declared as Go structs in:
// code.byted.org/lab-speech/gocommon/v2/srvutil/rpcmeta
message RequestMeta {
// Required.
// Backend endpoint name. Gateway will also use it to infer HTTP/WebSocket URL
// path.
string Endpoint = 1 [ json_name = "endpoint" ];
// Required.
// For historical reasons, this can be either a volcanic appid or a sail
// platform appKey. If both are transmitted at the same time, you need to put
// the volcengine appid in the AppID
string AppKey = 2 [ json_name = "app_key" ];
// Optional.
// appID in volcengine
string AppID = 3 [ json_name = "app_id" ];
// - Required by gateway, you must provide a ResourceID on calling gateway.
// - Optional for backend services, gateway doesn't need to pass this to
// backend.
string ResourceID = 4 [ json_name = "resource_id" ];
// Required for Websocket connection (SDK -> Gateway), optional otherwise.
// In websocket communication, a connection can be reused for multiple
// simultaneous sessions. Servers use this ID mainly for debugging purposes.
string ConnectionID = 5 [ json_name = "connection_id" ];
// Required. Session is the minimum unit that performs a specific task. The
// request data in a session can be devided into a number data packets
// transmitted (but not necessarily processed by the server) in a sequential
// manner. Each data packet can be denoted by a "Sequence" number (although
// populating the "Sequence" field is not mandatory). In the simplest case,
// all request data in a session is sent as a lump (ie, only one packet).
string SessionID = 6 [ json_name = "session_id" ];
// Optional, if not passed, default value 0 will be assumed. For streaming
// RPC, this field can be omitted (because the order of sending is preserved
// at the receiving side) unless the client needs the acknowledgement of
// packet receipt from the server side.
int32 Sequence = 7 [ json_name = "sequence" ];
}
message BillingItem {
// Required.
// Billing unit, eg:
// - minute (billed by the number of minutes used)
// - word (billed by the number of words submitted or generated)
// - call (billed by the number of RPC/HTTP/Websocket calls)
string Unit = 1 [ json_name = "unit" ];
// Optional.
// The amount that the consumer consumed counted by `BillingItem.Unit`. We use
// a float number because sometimes we want precision better than whole
// numbers.
float Quantity = 2 [ json_name = "quantity" ];
}
message Billing {
// Optional.
// For new billing items, use this field. This is a list because there may be
// more than one billing items in one request/session.
repeated BillingItem Items = 1 [ json_name = "items" ];
/*
* The following are reserved for compatibility. For new billing items, use
* field `Items`.
* We do NOT recommend using `Items` and the fields below at the same time.
*/
// Optional.
// For commodities that are priced w.r.t (typically audio/video) duration in
// milliseconds.
int64 DurationMsec = 2 [ json_name = "duration_msec" ];
// Optional.
// For commodities priced w.r.t text length (number of words/tokens).
int64 WordCount = 3 [ json_name = "word_count" ];
}
message ResponseMeta {
// Required.
// The same SessionID in RequestMeta.
string SessionID = 1 [ json_name = "session_id" ];
// Optional.
// The same sequence in RequestMeta (except that server *may* turn a positive
// sequence to its negative counterpart).
int32 Sequence = 2 [ json_name = "sequence" ];
// Optional.
// Response status code.
int32 StatusCode = 3 [ json_name = "status_code" ];
// Optional.
// Detailed status information.
string Message = 4 [ json_name = "message" ];
// Billing information in case only backend servers can collect this
// information.
optional Billing Billing = 5 [ json_name = "billing" ];
}

46
local_plugins/azure_speech/android/src/main/protos/products/understanding/ast/ast_service.proto

@ -0,0 +1,46 @@
syntax = "proto3";
import "common/events.proto";
import "common/rpcmeta.proto";
import "products/understanding/base/au_base.proto";
package data.speech.ast;
option java_multiple_files = true;
option java_package = "data.speech.ast";
message ReqParams {
string mode = 1; // 可能是s2t , s2s 选一个, 控制是否需要语音
string source_language = 2; // 源语言
string target_language = 3; // 目标语言
string speaker_id = 4;
data.speech.understanding.Corpus corpus = 100;
}
message TranslateRequest {
optional data.speech.common.RequestMeta request_meta = 1;
data.speech.event.Type event = 2;
data.speech.understanding.User user = 3;
// bytes data = 2; // request binary data
data.speech.understanding.Audio source_audio = 4; // 源音频信息
// 目标音频信息,只需要传format(pcm/ogg)、rate、bits、channel这些
data.speech.understanding.Audio target_audio = 5;
ReqParams request = 6; // 请求参数
optional bool denoise = 7; // 是否开启降噪
}
message TranslateResponse {
optional data.speech.common.ResponseMeta response_meta = 1;
data.speech.event.Type event = 2;
bytes data = 3; // response binary data
string text = 4; // 原文或者译文
int32 start_time = 5;
int32 end_time = 6;
bool spk_chg = 7;
int32 muted_duration_ms = 8;
}
service ASTService {
rpc Translate(stream TranslateRequest) returns (stream TranslateResponse);
}

195
local_plugins/azure_speech/android/src/main/protos/products/understanding/base/au_base.proto

@ -0,0 +1,195 @@
syntax = "proto3";
package data.speech.understanding;
option java_multiple_files = true;
option java_package = "data.speech.understanding";
enum Code {
ERROR_UNSPECIFIED = 0;
SUCCESS = 21000;
// request related
INVALID_REQUEST = 11100;
PERMISSION_DENIED = 11200;
LIMIT_QPS = 11301;
LIMIT_COUNT = 11302;
SERVER_BUSY = 11303;
INTERRUPTED = 21300;
ERROR_PARAMS = 11500;
// audio related
LONG_AUDIO = 11101;
LARGE_PACKET = 11102;
INVALID_FORMAT = 11103;
SILENT_AUDIO = 11104;
EMPTY_AUDIO = 11105;
AUDIO_DOWNLOAD_FAIL = 21701;
// recognition related
TIMEOUT_WAITING = 21200;
TIMEOUT_PROCESSING = 21201;
ERROR_PROCESSING = 21100;
// others
ERROR_UNKNOWN = 29900;
}
message Word {
string text = 1;
int32 start_time = 2;
int32 end_time = 3;
int32 blank_duration = 4;
string pronounce = 5;
double confidence = 6;
}
message UtteranceAddition {}
message ResultAddition {}
message Utterance {
string text = 1;
int32 start_time = 2;
int32 end_time = 3;
optional bool definite = 4;
repeated Word words = 5;
string language = 6;
double confidence = 7;
int32 speaker = 8;
int32 channel_id = 9;
map<string, string> additions = 100;
}
message LanguageDetail {
// language probability
double prob = 1;
}
message Result {
string text = 1;
repeated Utterance utterances = 2;
double confidence = 3;
double global_confidence = 4;
string language = 5;
map<string, LanguageDetail> language_details = 6;
optional bool termination = 7;
double volume = 8;
double speech_rate = 9;
string result_type = 10;
string ori_text = 11;
optional bool prefetch = 12;
map<string, string> additions = 100;
}
message AudioInfo {
int64 duration = 1;
double speech_rate = 2;
}
message App { string work_flow_name = 1; }
message User {
string uid = 1;
string did = 2;
// android or ios info
string platform = 3;
string sdk_version = 4;
string app_version = 5;
}
message Audio {
string data = 1;
string url = 2;
// url; vid...
string url_type = 3;
string format = 4;
string codec = 5;
string language = 6;
int32 rate = 7;
int32 bits = 8;
int32 channel = 9;
string tos_bucket = 10;
string tos_access_key = 11;
string audio_tos_object = 12;
string role_trn = 13;
bytes binary_data = 14;
}
enum AsrNotificationCode {
NOTIFICATION_UNSPECIFIED = 0;
DISCONNECT = 1000;
}
message AsrNotification {
AsrNotificationCode code = 1;
string message = 2;
}
message ReqParams {
string model_name = 1; // 用于映射集群
string config_name = 2; // 用于映射tcc配置
optional bool enable_vad = 3;
optional bool enable_punc = 4;
optional bool enable_itn = 5;
optional bool enable_ddc = 6;
optional bool enable_resample = 7;
optional bool vad_signal = 8;
optional bool enable_translate = 9;
optional bool disable_end_punc = 10;
optional bool enable_speaker_info = 11;
optional bool enable_channel_split = 12;
string app_lang = 13;
string country = 14;
string audio_type = 15;
optional bool enable_lid = 16;
int32 reco2act_num_spk = 17;
int32 reco2max_num_spk = 18;
optional bool enable_interrupt = 19;
optional bool vad_segment = 20;
optional bool enable_ori_text = 21;
optional bool enable_nonstream = 22;
optional bool show_utterances = 31;
optional bool show_words = 32;
optional bool show_duration = 33;
optional bool show_language = 34;
optional bool show_volume = 35;
optional bool show_volume_v = 362;
optional bool show_speech_rate = 37;
optional bool show_prefetch = 38;
string result_type = 39;
int32 start_silence_time = 61;
int32 end_silence_time = 62;
int32 vad_silence_time = 63;
int32 min_silence_time = 64;
string vad_mode = 65;
int32 vad_segment_duration = 66;
int32 end_window_size = 67;
int32 force_to_speech_time = 68;
Corpus corpus = 100;
AsrNotification notification = 101;
string sensitive_words_filter = 102;
}
message Corpus {
// json格式,包含热词,冷词,篇章,纠错词表等
string context = 1;
string boosting_table_id = 2;
string boosting_table_name = 3;
string boosting_id = 4;
string nnlm_id = 5;
string correct_table_id = 6;
string correct_table_name = 7;
string punc_hot_words = 8;
repeated string hot_words_list = 9;
map<string, string> glossary_list = 10;
optional bool show_match_hot_words = 61;
optional bool show_all_match_hot_words = 62;
optional bool show_effective_hot_words = 63;
}

427
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/AliyunBailianE2EHelper.swift

@ -0,0 +1,427 @@
import Foundation
import os.log
/**
* 阿里云百炼(Bailian)实时音视频翻译助手 (qwen3-livetranslate-flash-realtime)
* iOS 实现,使用 URLSessionWebSocketTask 和 JSON 协议。
*/
class AliyunBailianE2EHelper: NSObject, URLSessionWebSocketDelegate {
/**
* 回调接口
*/
protocol Callback: AnyObject {
func onSessionStarted(sessionId: String)
func onPartialText(sessionId: String, text: String)
func onPartialSourceText(sessionId: String, text: String)
func onFinalSourceText(sessionId: String, finalText: String)
func onFinalTranslatedText(sessionId: String, finalText: String)
func onPartialAudio(sessionId: String, data: Data)
func onSessionFinished(sessionId: String, finalText: String, finalAudio: Data)
func onSessionError(sessionId: String, code: Int, message: String)
}
/**
* 配置结构
*/
struct Config {
var wsUrl: String = "wss://dashscope.aliyuncs.com/api-ws/v1/realtime"
var apiKey: String = ""
var appId: String = "qwen3-livetranslate-flash-realtime"
var sampleRate: Int = 16000
var sourceLanguage: String = "zh"
var targetLanguage: String = "en"
var voice: String = ""
}
private let tag = "AliyunBailianE2EHelper"
private let log = OSLog(subsystem: "com.azure.speech", category: "AliyunBailian")
private var conf = Config()
private weak var callback: Callback?
private var urlSession: URLSession?
private var webSocket: URLSessionWebSocketTask?
private var sessionId: String = ""
private var fullTextBuffer = ""
private var fullAudioBuffer = Data()
private var recvTextBuffer = ""
private var audioChunkBuffer = Data()
private var isStarted = false
deinit {
os_log("AliyunBailianE2EHelper deinit", log: log, type: .info)
}
/**
* 初始化助手(异步)
* - 参数 config: 会话配置
* - 参数 cb: 回调实现
* - 返回: 是否成功触发连接逻辑
*/
func initialize(config: Config, cb: Callback) async -> Bool {
os_log("initialize: wsUrl=%{public}@", log: log, type: .info, config.wsUrl)
urlSession?.invalidateAndCancel()
urlSession = nil
conf = config
callback = cb
let configuration = URLSessionConfiguration.default
configuration.timeoutIntervalForRequest = 0
configuration.timeoutIntervalForResource = 0
urlSession = URLSession(configuration: configuration, delegate: self, delegateQueue: OperationQueue.main)
return startContinuousConversation()
}
/**
* 启动会话并建立 WebSocket 连接
*/
func startContinuousConversation() -> Bool {
guard !isStarted else {
os_log("Already started, ignore start request", log: log, type: .info)
return true
}
guard let session = urlSession else {
os_log("URLSession is nil, cannot start", log: log, type: .error)
return false
}
let urlWithModel: String
if conf.wsUrl.contains("?") {
urlWithModel = conf.wsUrl
} else {
urlWithModel = "\(conf.wsUrl)?model=qwen3-livetranslate-flash-realtime"
}
guard let url = URL(string: urlWithModel) else {
os_log("Invalid wsUrl: %{public}@", log: log, type: .error, urlWithModel)
return false
}
sessionId = UUID().uuidString
os_log("Start session: %{public}@", log: log, type: .info, sessionId)
recvTextBuffer = ""
fullTextBuffer = ""
fullAudioBuffer.removeAll()
audioChunkBuffer.removeAll()
var request = URLRequest(url: url)
request.timeoutInterval = 60
request.addValue("Bearer \(conf.apiKey)", forHTTPHeaderField: "Authorization")
webSocket = session.webSocketTask(with: request)
webSocket?.resume()
isStarted = true
return true
}
/**
* 发送 session.update 配置会话
*/
private func sendSessionUpdate() {
guard let ws = webSocket else { return }
do {
var session: [String: Any] = [:]
var transcription: [String: Any] = [:]
transcription["model"] = "qwen3-asr-flash-realtime"
transcription["language"] = conf.sourceLanguage
session["input_audio_transcription"] = transcription
var translation: [String: Any] = [:]
translation["language"] = conf.targetLanguage
session["translation"] = translation
if !conf.voice.isEmpty {
session["voice"] = conf.voice
} else {
if conf.targetLanguage != "yue" {
session["voice"] = "Cherry"
} else {
session["voice"] = "Kiki"
}
}
session["input_audio_format"] = "pcm16"
session["output_audio_format"] = "pcm24"
var modalities: [String] = []
modalities.append("text")
modalities.append("audio")
session["modalities"] = modalities
var root: [String: Any] = [:]
root["type"] = "session.update"
root["session"] = session
let data = try JSONSerialization.data(withJSONObject: root, options: [])
if let jsonString = String(data: data, encoding: .utf8) {
ws.send(.string(jsonString)) { [weak self] error in
if let e = error {
os_log("Error sending session.update: %{public}@", log: self?.log ?? .default, type: .error, e.localizedDescription)
} else {
os_log("Sent session.update: %{public}@", log: self?.log ?? .default, type: .info, jsonString)
}
}
}
} catch {
os_log("Error building session.update: %{public}@", log: log, type: .error, error.localizedDescription)
}
}
/**
* 将 24kHz PCM 重采样为 16kHz
*/
private func resample24kTo16k(_ input: Data) -> Data {
if input.count < 4 { return input }
let sampleCount = input.count / 2
var inputSamples = [Int16](repeating: 0, count: sampleCount)
_ = input.withUnsafeBytes { raw -> Void in
guard let base = raw.bindMemory(to: Int16.self).baseAddress else { return }
for i in 0..<sampleCount {
inputSamples[i] = Int16(littleEndian: base[i])
}
}
let outputCount = sampleCount * 2 / 3
if outputCount == 0 { return Data() }
var outputSamples = [Int16](repeating: 0, count: outputCount)
for i in 0..<outputCount {
let inputIndexFloat = Float(i) * 1.5
let index = Int(inputIndexFloat)
let frac = inputIndexFloat - Float(index)
if index + 1 < sampleCount {
let v1 = Float(inputSamples[index])
let v2 = Float(inputSamples[index + 1])
let v = v1 * (1.0 - frac) + v2 * frac
outputSamples[i] = Int16(max(min(Int(v), Int(Int16.max)), Int(Int16.min)))
} else if index < sampleCount {
outputSamples[i] = inputSamples[index]
}
}
var outData = Data(count: outputSamples.count * 2)
outData.withUnsafeMutableBytes { raw in
let ptr = raw.bindMemory(to: Int16.self).baseAddress
ptr?.assign(from: outputSamples, count: outputSamples.count)
}
return outData
}
/**
* 处理服务端 JSON 文本消息
*/
private func handleJsonMessage(_ text: String) {
guard let data = text.data(using: .utf8) else { return }
guard let obj = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { return }
let type = obj["type"] as? String ?? ""
os_log("handleJsonMessage: type=%{public}@", log: log, type: .info, type)
switch type {
case "error":
if let errorObj = obj["error"] as? [String: Any] {
let msg = errorObj["message"] as? String ?? "Unknown error"
let codeStr = errorObj["code"] as? String ?? ""
os_log("Error from server: %{public}@ - %{public}@", log: log, type: .error, codeStr, msg)
callback?.onSessionError(sessionId: sessionId, code: 1012, message: msg)
}
case "session.created":
os_log("Session created", log: log, type: .info)
case "session.updated":
os_log("Session updated", log: log, type: .info)
case "conversation.item.input_audio_transcription.text":
if let txt = obj["text"] as? String, !txt.isEmpty {
os_log("onPartialSourceText: %{public}@", log: log, type: .info, txt)
callback?.onPartialSourceText(sessionId: sessionId, text: txt)
}
case "conversation.item.input_audio_transcription.completed":
var finalTxt = ""
if let item = obj["item"] as? [String: Any],
let content = item["content"] as? [[String: Any]] {
for part in content {
if let t = part["type"] as? String, t == "input_audio" {
finalTxt = part["transcript"] as? String ?? ""
if !finalTxt.isEmpty { break }
}
}
}
if finalTxt.isEmpty {
finalTxt = obj["transcript"] as? String ?? ""
}
if !finalTxt.isEmpty {
os_log("onFinalSourceText: %{public}@", log: log, type: .info, finalTxt)
callback?.onFinalSourceText(sessionId: sessionId, finalText: finalTxt)
}
case "response.audio_transcript.delta":
if let delta = obj["delta"] as? String, !delta.isEmpty {
recvTextBuffer.append(delta)
os_log("onPartialText: %{public}@", log: log, type: .info, delta)
callback?.onPartialText(sessionId: sessionId, text: delta)
}
case "response.audio_transcript.done":
let transcript = obj["transcript"] as? String ?? ""
let finalText = transcript.isEmpty ? recvTextBuffer : transcript
os_log("onFinalTranslatedText: %{public}@", log: log, type: .info, finalText)
callback?.onFinalTranslatedText(sessionId: sessionId, finalText: finalText)
if !fullTextBuffer.isEmpty {
fullTextBuffer.append(" ")
}
fullTextBuffer.append(finalText)
recvTextBuffer = ""
case "response.audio.delta":
if let base64Audio = obj["delta"] as? String, !base64Audio.isEmpty {
if let audioBytes = Data(base64Encoded: base64Audio) {
let resampled = resample24kTo16k(audioBytes)
if !resampled.isEmpty {
fullAudioBuffer.append(resampled)
os_log("onPartialAudio: rawSize=%{public}d resampledSize=%{public}d", log: log, type: .info, audioBytes.count, resampled.count)
processAudioChunk(incoming: resampled, multiple: 1280)
}
}
}
case "response.done":
os_log("Response done", log: log, type: .info)
default:
break
}
}
/**
* 音频分片累积并按 multiple 字节回调
*/
private func processAudioChunk(incoming: Data, multiple: Int = 1280) {
if !incoming.isEmpty {
audioChunkBuffer.append(incoming)
}
let sendLen = (audioChunkBuffer.count / multiple) * multiple
if sendLen > 0 {
let chunk = audioChunkBuffer.prefix(sendLen)
callback?.onPartialAudio(sessionId: sessionId, data: Data(chunk))
audioChunkBuffer = Data(audioChunkBuffer.dropFirst(sendLen))
}
}
/**
* 推送音频 PCM 数据(Base64 编码)
*/
func pushAudioData(_ data: Data) -> Bool {
guard isStarted, let ws = webSocket else {
os_log("pushAudioData ignored: isStarted=%{public}d, ws is nil", log: log, type: .info, isStarted)
return false
}
os_log("pushAudioData: %{public}d bytes", log: log, type: .info, data.count)
do {
let base64Audio = data.base64EncodedString()
var root: [String: Any] = [:]
root["type"] = "input_audio_buffer.append"
root["audio"] = base64Audio
let jsonData = try JSONSerialization.data(withJSONObject: root, options: [])
if let jsonString = String(data: jsonData, encoding: .utf8) {
ws.send(.string(jsonString)) { [weak self] error in
if let e = error {
os_log("pushAudioData error: %{public}@", log: self?.log ?? .default, type: .error, e.localizedDescription)
}
}
}
return true
} catch {
os_log("pushAudioData serialization error: %{public}@", log: log, type: .error, error.localizedDescription)
return false
}
}
/**
* 停止会话
*/
func stopContinuousConversation() -> Bool {
os_log("stopContinuousConversation", log: log, type: .info)
if let ws = webSocket {
ws.cancel(with: .normalClosure, reason: "User stopped".data(using: .utf8))
}
isStarted = false
return true
}
/**
* 释放资源
*/
func dispose() {
os_log("dispose", log: log, type: .info)
do {
try webSocket?.cancel(with: .normalClosure, reason: "dispose".data(using: .utf8))
} catch {
}
webSocket = nil
isStarted = false
urlSession?.invalidateAndCancel()
urlSession = nil
}
/**
* WebSocket 打开回调
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didOpenWithProtocol protocol: String?) {
os_log("WebSocket didOpen", log: log, type: .info)
sendSessionUpdate()
callback?.onSessionStarted(sessionId: sessionId)
receiveLoop()
}
/**
* WebSocket 接收循环
*/
private func receiveLoop() {
webSocket?.receive { [weak self] result in
guard let self = self else { return }
switch result {
case .failure(let error):
os_log("WebSocket receive error: %{public}@", log: self.log, type: .error, error.localizedDescription)
self.callback?.onSessionError(sessionId: self.sessionId, code: 1011, message: error.localizedDescription)
case .success(let message):
switch message {
case .string(let text):
self.handleJsonMessage(text)
case .data(let data):
if let text = String(data: data, encoding: .utf8) {
self.handleJsonMessage(text)
}
@unknown default:
break
}
self.receiveLoop()
}
}
}
/**
* WebSocket 关闭回调
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didCloseWith closeCode: URLSessionWebSocketTask.CloseCode, reason: Data?) {
let reasonStr = String(data: reason ?? Data(), encoding: .utf8) ?? ""
os_log("WebSocket didClose code=%{public}d reason=%{public}@", log: log, type: .info, closeCode.rawValue, reasonStr)
isStarted = false
callback?.onSessionFinished(sessionId: sessionId, finalText: fullTextBuffer, finalAudio: fullAudioBuffer)
}
/**
* 任务完成回调(错误处理)
*/
func urlSession(_ session: URLSession, task: URLSessionTask, didCompleteWithError error: Error?) {
if let e = error {
os_log("WebSocket task error: %{public}@", log: log, type: .error, e.localizedDescription)
callback?.onSessionError(sessionId: sessionId, code: 1012, message: e.localizedDescription)
}
isStarted = false
}
}

650
local_plugins/azure_speech/ios/azure_speech/Sources/azure_speech/DoubaoE2ETranslateHelper.swift

@ -0,0 +1,650 @@
import Foundation
import os.log
/**
* Doubao AST 流式翻译助手(iOS)
* 提供与 Android `DoubaoE2ETranslateHelper` 类似的接口:
* - 初始化并建立 WebSocket 连接
* - 推送音频分片到服务端
* - 接收增量文本与音频分片
* - 在会话结束时回传完整文本与合成音频
*/
class DoubaoE2ETranslateHelper: NSObject, URLSessionWebSocketDelegate {
/**
* 回调接口
*/
protocol Callback: AnyObject {
func onSessionStarted(sessionId: String)
func onPartialText(sessionId: String, text: String)
func onPartialSourceText(sessionId: String, text: String)
func onFinalSourceText(sessionId: String, finalText: String)
func onFinalTranslatedText(sessionId: String, finalText: String)
func onPartialAudio(sessionId: String, data: Data)
func onSessionFinished(sessionId: String, finalText: String, finalAudio: Data)
func onSessionError(sessionId: String, code: Int, message: String)
}
/**
* 客户端配置
*/
struct Config {
var wsUrl: String = "wss://openspeech.bytedance.com/api/v4/ast/v2/translate"
var appKey: String = ""
var accessKey: String = ""
var resourceId: String = "volc.service_type.10053"
var sourceLanguage: String = "zh"
var targetLanguage: String = "en"
var clientWaitMs: Int64 = 60_000
}
private let tag = "DoubaoE2ETranslateHelper"
private let log = OSLog(subsystem: "com.azure.speech", category: "DoubaoE2E")
private var conf = Config()
private weak var callback: Callback?
private var urlSession: URLSession?
private var webSocket: URLSessionWebSocketTask?
private var sessionId: String = ""
private var recvAudio = Data()
private var recvText: [String] = []
private var recvSourceText: [String] = []
private var audioChunkBuffer = Data()
private var isStarted = false
private let lock = NSLock()
private var isWebSocketConnected = false
private var pendingAudioData: [Data] = []
private var keepAliveTimer: Timer?
private var lastAudioSentAt = Date.distantPast
private let keepAliveIntervalSec: TimeInterval = 1.0
private let keepAliveIdleThresholdSec: TimeInterval = 2.5
private let keepAliveSilentChunk = Data(repeating: 0, count: 640)
private var hasSentInitialWarmupPacket = false
/**
* 初始化助手(异步)
* 返回是否成功触发了连接逻辑
*/
func initialize(config: Config, cb: Callback) async -> Bool {
// 清理旧的 Session
if let session = urlSession {
session.invalidateAndCancel()
}
os_log("Initialize with wsUrl: %{public}@", log: log, type: .info, config.wsUrl)
conf = config
// 去除可能的空白字符
conf.appKey = conf.appKey.trimmingCharacters(in: .whitespacesAndNewlines)
conf.accessKey = conf.accessKey.trimmingCharacters(in: .whitespacesAndNewlines)
conf.resourceId = conf.resourceId.trimmingCharacters(in: .whitespacesAndNewlines)
callback = cb
let configuration = URLSessionConfiguration.default
configuration.timeoutIntervalForRequest = 0 // 恢复为0(无超时),因为 WebSocket 是长连接
configuration.timeoutIntervalForResource = 0
// 使用主队列,方便调试且避免并发问题
urlSession = URLSession(configuration: configuration, delegate: self, delegateQueue: OperationQueue.main)
// 等待连接建立(简单的延迟模拟,或者通过回调通知)
// 这里仅启动连接,实际连接结果是异步的
return startContinuousTranslation()
}
/**
* 重启会话(用于超时自动重连)
*/
private func restartSession() {
os_log("Restarting session due to timeout...", log: log, type: .info)
stopKeepAliveTimer()
hasSentInitialWarmupPacket = false
webSocket?.cancel(with: .normalClosure, reason: "restarting".data(using: .utf8))
webSocket = nil
isStarted = false
lock.lock()
isWebSocketConnected = false
lock.unlock()
DispatchQueue.main.asyncAfter(deadline: .now() + 0.2) { [weak self] in
guard let self = self else { return }
if !self.isStarted {
_ = self.startContinuousTranslation()
}
}
}
/**
* 启动会话并建立 WebSocket 连接
*/
func startContinuousTranslation() -> Bool {
guard !isStarted else {
os_log("Already started, ignoring start request", log: log, type: .info)
return true
}
guard let url = URL(string: conf.wsUrl) else {
os_log("Invalid WebSocket URL: %{public}@", log: log, type: .error, conf.wsUrl)
return false
}
sessionId = UUID().uuidString
os_log("Starting new session: %{public}@", log: log, type: .info, sessionId)
recvAudio.removeAll()
recvText.removeAll()
recvSourceText.removeAll()
hasSentInitialWarmupPacket = false
lock.lock()
isWebSocketConnected = false
pendingAudioData.removeAll()
lock.unlock()
var request = URLRequest(url: url)
request.timeoutInterval = 60 // 握手超时可以设置,但连接成功后不应受此限制
// 打印 Header 信息(部分脱敏)
let appKeyMasked = conf.appKey.count > 4 ? String(conf.appKey.prefix(4)) + "***" : conf.appKey
let accessKeyMasked = conf.accessKey.count > 4 ? String(conf.accessKey.prefix(4)) + "***" : "N/A"
os_log("Request Headers: AppKey=%{public}@, ResourceId=%{public}@", log: log, type: .info, appKeyMasked, conf.resourceId)
request.addValue(conf.appKey, forHTTPHeaderField: "X-Api-App-Key")
request.addValue(conf.accessKey, forHTTPHeaderField: "X-Api-Access-Key")
request.addValue(conf.resourceId, forHTTPHeaderField: "X-Api-Resource-Id")
request.addValue(UUID().uuidString, forHTTPHeaderField: "X-Api-Connect-Id")
// 尝试添加 Protocol 头
// request.addValue("ast-v2", forHTTPHeaderField: "Sec-WebSocket-Protocol")
if let session = urlSession {
webSocket = session.webSocketTask(with: request)
} else {
os_log("Failed to create WebSocket task: urlSession is nil", log: log, type: .error)
return false
}
webSocket?.resume()
isStarted = true
return true
}
/**
* 推送音频数据(PCM/WAV)
*/
func pushAudioData(_ data: Data) -> Bool {
// 先检查 isStarted,避免未启动就调用
guard isStarted else {
os_log("Push audio failed: not started", log: log, type: .error)
return false
}
// 尝试获取 webSocket,如果为空,说明还在连接中,应该进入缓冲逻辑
// 如果已经连接成功但 webSocket 还是 nil,那就是异常状态
lock.lock()
// 1. 如果已连接且有 ws 实例 -> 直接发
if isWebSocketConnected, let ws = webSocket {
lock.unlock()
return sendAudioData(ws, data: data)
}
// 2. 否则(连接中、ws 为空、或者刚断开)-> 缓冲
else {
// os_log("Buffering audio data, size: %d", log: log, type: .debug, data.count)
pendingAudioData.append(data)
lock.unlock()
return true
}
}
/**
* 发送音频分片(Protobuf)
*/
private func sendAudioData(_ ws: URLSessionWebSocketTask, data: Data) -> Bool {
if data.isEmpty { return true }
let req = makeChunkRequest(sessionId: sessionId, chunk: data)
guard let msg = try? req.serializedData() else { return false }
ws.send(.data(msg)) { [weak self] error in
if let e = error {
os_log("发送音频失败: %{public}@", log: self?.log ?? .default, type: .error, e.localizedDescription)
} else {
self?.markAudioSentNow()
}
}
return true
}
/**
* 停止会话并发送结束请求(Protobuf)
*/
func stopContinuousTranslation() -> Bool {
os_log("Stopping continuous translation", log: log, type: .info)
guard let ws = webSocket else { return false }
let req = makeFinishRequest(sessionId: sessionId)
guard let msg = try? req.serializedData() else { return false }
ws.send(.data(msg)) { _ in }
return true
}
/**
* 释放资源
*/
func dispose() {
os_log("Disposing resources", log: log, type: .info)
stopKeepAliveTimer()
do { try webSocket?.cancel(with: .normalClosure, reason: "dispose".data(using: .utf8)) } catch { }
webSocket = nil
isStarted = false
lock.lock()
isWebSocketConnected = false
pendingAudioData.removeAll()
lock.unlock()
urlSession?.invalidateAndCancel()
urlSession = nil
}
/**
* WebSocket打开回调(发送 Protobuf StartSession)
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didOpenWithProtocol protocol: String?) {
os_log("WebSocket didOpen", log: log, type: .info)
markAudioSentNow()
startKeepAliveTimer()
let startReq = makeStartRequest(sessionId: sessionId)
if let data = try? startReq.serializedData() {
os_log("Sending StartSession (protobuf), size=%{public}d", log: log, type: .info, data.count)
let message = URLSessionWebSocketTask.Message.data(data)
webSocketTask.send(message) { [weak self] error in
if let e = error {
os_log("StartSession 发送失败: %{public}@", log: self?.log ?? .default, type: .error, e.localizedDescription)
} else {
os_log("StartSession sent successfully", log: self?.log ?? .default, type: .info)
self?.sendInitialWarmupPacketIfNeeded(ws: webSocketTask)
}
}
}
lock.lock()
isWebSocketConnected = true
let pending = pendingAudioData
pendingAudioData.removeAll()
lock.unlock()
if !pending.isEmpty {
os_log("Flushing %d buffered audio chunks", log: log, type: .info, pending.count)
}
for data in pending {
_ = sendAudioData(webSocketTask, data: data)
}
callback?.onSessionStarted(sessionId: sessionId)
receiveLoop()
}
/**
* WebSocket关闭回调
*/
func urlSession(_ session: URLSession, webSocketTask: URLSessionWebSocketTask, didCloseWith closeCode: URLSessionWebSocketTask.CloseCode, reason: Data?) {
let reasonStr = String(data: reason ?? Data(), encoding: .utf8) ?? ""
os_log("WebSocket didClose, code: %d, reason: %{public}@", log: log, type: .info, closeCode.rawValue, reasonStr)
stopKeepAliveTimer()
lock.lock()
isWebSocketConnected = false
lock.unlock()
isStarted = false
}
/**
* 任务完成回调(处理错误)
*/
func urlSession(_ session: URLSession, task: URLSessionTask, didCompleteWithError error: Error?) {
if let e = error {
os_log("WebSocket task failed: %{public}@", log: log, type: .error, e.localizedDescription)
callback?.onSessionError(sessionId: sessionId, code: 1012, message: e.localizedDescription)
stopKeepAliveTimer()
lock.lock()
isWebSocketConnected = false
lock.unlock()
isStarted = false
}
}
private func startKeepAliveTimer() {
stopKeepAliveTimer()
keepAliveTimer = Timer.scheduledTimer(withTimeInterval: keepAliveIntervalSec, repeats: true) { [weak self] _ in
self?.sendKeepAliveIfNeeded()
}
if let timer = keepAliveTimer {
RunLoop.main.add(timer, forMode: .common)
}
}
private func stopKeepAliveTimer() {
keepAliveTimer?.invalidate()
keepAliveTimer = nil
}
private func markAudioSentNow() {
lastAudioSentAt = Date()
}
private func sendKeepAliveIfNeeded() {
lock.lock()
let canSend = isStarted && isWebSocketConnected
let ws = webSocket
lock.unlock()
guard canSend, let ws = ws else { return }
let idle = Date().timeIntervalSince(lastAudioSentAt)
if idle < keepAliveIdleThresholdSec { return }
os_log("Sending keepalive silent chunk, idle=%.2fs", log: log, type: .debug, idle)
_ = sendAudioData(ws, data: keepAliveSilentChunk)
}
private func sendInitialWarmupPacketIfNeeded(ws: URLSessionWebSocketTask) {
guard !hasSentInitialWarmupPacket else { return }
hasSentInitialWarmupPacket = true
os_log("Sending initial warmup packet to avoid first-packet timeout", log: log, type: .info)
_ = sendAudioData(ws, data: keepAliveSilentChunk)
}
/**
* 接收消息循环
*/
private func receiveLoop() {
webSocket?.receive { [weak self] result in
guard let self = self else { return }
switch result {
case .failure(let error):
self.callback?.onSessionError(sessionId: self.sessionId, code: 1011, message: error.localizedDescription)
case .success(let message):
self.handleMessage(message)
self.receiveLoop()
}
}
}
/**
* 处理服务端消息(优先按 Protobuf 解码)
*/
private func handleMessage(_ message: URLSessionWebSocketTask.Message) {
switch message {
case .data(let data):
handleProtoMessage(data)
case .string(let text):
handleTextMessage(text)
@unknown default:
break
}
}
/**
* 处理二进制消息(Protobuf TranslateResponse)
*/
private func handleProtoMessage(_ data: Data) {
do {
let resp = try Data_Speech_Ast_TranslateResponse(serializedData: data)
handleProtoResponse(resp)
} catch {
os_log("Proto parse failed, len=%{public}d error=%{public}@", log: log, type: .error, data.count, error.localizedDescription)
}
}
/**
* 按事件类型处理 Protobuf TranslateResponse
*/
private func handleProtoResponse(_ resp: Data_Speech_Ast_TranslateResponse) {
let event = resp.event
let text = resp.text
let audioData = resp.data
if event != .usageResponse {
os_log("Handling proto event: %{public}d", log: log, type: .info, event.rawValue)
os_log("Proto payload: textLen=%{public}d audioLen=%{public}d",
log: log, type: .info, text.count, audioData.count)
}
if event == .usageResponse {
return
}
if event == .sessionFailed || event == .sessionCanceled {
let message = resp.hasResponseMeta ? resp.responseMeta.message : ""
os_log("Session error event: %{public}d, message: %{public}@", log: log, type: .error, event.rawValue, message)
if message.contains("Timeout waiting next packet") {
os_log("Detected timeout error, triggering auto-restart", log: log, type: .info)
restartSession()
return
}
callback?.onSessionError(sessionId: sessionId, code: 1011, message: message)
webSocket?.cancel(with: .goingAway, reason: message.data(using: .utf8))
return
}
if event == .sessionFinished {
let finalText = recvText.joined(separator: " ")
let finalAudio = recvAudio
os_log("SessionFinished: finalTextLen=%{public}d finalAudioLen=%{public}d",
log: log, type: .info, finalText.count, finalAudio.count)
callback?.onSessionFinished(sessionId: sessionId, finalText: finalText, finalAudio: finalAudio)
webSocket?.cancel(with: .normalClosure, reason: "Completed".data(using: .utf8))
return
}
if !audioData.isEmpty {
recvAudio.append(audioData)
os_log("Append audio chunk: chunkSize=%{public}d totalSize=%{public}d",
log: log, type: .info, audioData.count, recvAudio.count)
processAudioChunk(incoming: audioData, multiple: 1280)
}
if !text.isEmpty {
switch event {
case .sourceSubtitleStart:
os_log("SourceSubtitleStart", log: log, type: .info)
recvSourceText.removeAll()
case .sourceSubtitleResponse:
os_log("SourceSubtitleResponse text=%{public}@", log: log, type: .info, text)
recvSourceText.append(text)
callback?.onPartialSourceText(sessionId: sessionId, text: text)
case .sourceSubtitleEnd:
os_log("SourceSubtitleEnd, finalSrcLen=%{public}d",
log: log, type: .info, recvSourceText.joined(separator: " ").count)
callback?.onFinalSourceText(sessionId: sessionId, finalText: recvSourceText.joined(separator: " "))
case .translationSubtitleStart:
os_log("TranslationSubtitleStart", log: log, type: .info)
recvText.removeAll()
case .translationSubtitleResponse:
os_log("TranslationSubtitleResponse text=%{public}@", log: log, type: .info, text)
recvText.append(text)
callback?.onPartialText(sessionId: sessionId, text: text)
case .translationSubtitleEnd:
os_log("TranslationSubtitleEnd, finalTgtLen=%{public}d",
log: log, type: .info, recvText.joined(separator: " ").count)
callback?.onFinalTranslatedText(sessionId: sessionId, finalText: recvText.joined(separator: " "))
startNewSubSession()
default:
os_log("Other text event=%{public}d text=%{public}@", log: log, type: .info, event.rawValue, text)
// recvText.append(text)
// callback?.onPartialText(sessionId: sessionId, text: text)
}
}
}
/**
* 处理文本消息(保留 JSON 兜底逻辑)
*/
private func handleTextMessage(_ text: String) {
guard let data = text.data(using: .utf8),
let obj = try? JSONSerialization.jsonObject(with: data) as? [String: Any] else { return }
let event = obj["event"] as? String ?? ""
let msgText = obj["text"] as? String
if event != "UsageResponse" {
os_log("Handling text event: %{public}@", log: log, type: .info, event)
}
if event == "UsageResponse" { return }
if event == "SessionFailed" || event == "SessionCanceled" {
let meta = obj["responseMeta"] as? [String: Any]
let message = meta?["message"] as? String ?? ""
os_log("Session error event(text): %{public}@, message: %{public}@", log: log, type: .error, event, message)
if message.contains("Timeout waiting next packet") {
os_log("Detected timeout error (text), triggering auto-restart", log: log, type: .info)
restartSession()
return
}
callback?.onSessionError(sessionId: sessionId, code: 1011, message: message)
webSocket?.cancel(with: .goingAway, reason: message.data(using: .utf8))
return
}
if event == "SessionFinished" {
let finalText = recvText.joined(separator: " ")
let finalAudio = recvAudio
callback?.onSessionFinished(sessionId: sessionId, finalText: finalText, finalAudio: finalAudio)
webSocket?.cancel(with: .normalClosure, reason: "Completed".data(using: .utf8))
return
}
if let audioB64 = (obj["data"] as? String), let audioData = Data(base64Encoded: audioB64) {
recvAudio.append(audioData)
processAudioChunk(incoming: audioData, multiple: 1280)
}
if let t = msgText, !t.isEmpty {
switch event {
case "SourceSubtitleStart":
recvSourceText.removeAll()
case "SourceSubtitleResponse":
recvSourceText.append(t)
callback?.onPartialSourceText(sessionId: sessionId, text: t)
case "SourceSubtitleEnd":
callback?.onFinalSourceText(sessionId: sessionId, finalText: recvSourceText.joined(separator: " "))
case "TranslationSubtitleStart":
recvText.removeAll()
case "TranslationSubtitleResponse":
recvText.append(t)
callback?.onPartialText(sessionId: sessionId, text: t)
case "TranslationSubtitleEnd":
callback?.onFinalTranslatedText(sessionId: sessionId, finalText: recvText.joined(separator: " "))
startNewSubSession()
default:
break
// recvText.append(t)
// callback?.onPartialText(sessionId: sessionId, text: t)
}
}
}
/**
* 新开子会话(句末切分)
*/
private func startNewSubSession() {
sessionId = UUID().uuidString
recvText.removeAll()
recvSourceText.removeAll()
}
/**
* 分片处理音频并按 1280 字节回调
*/
private func processAudioChunk(incoming: Data, multiple: Int) {
if !incoming.isEmpty { audioChunkBuffer.append(incoming) }
let sendLen = (audioChunkBuffer.count / multiple) * multiple
if sendLen > 0 {
let chunk = audioChunkBuffer.prefix(sendLen)
callback?.onPartialAudio(sessionId: sessionId, data: Data(chunk))
audioChunkBuffer = Data(audioChunkBuffer.dropFirst(sendLen))
}
}
/**
* 构造 StartSession 请求(Protobuf)
*/
private func makeStartRequest(sessionId: String) -> Data_Speech_Ast_TranslateRequest {
var user = Data_Speech_Understanding_User()
user.uid = "ast_ios_client"
user.did = "ast_ios_client"
user.platform = "ios"
var sourceAudio = Data_Speech_Understanding_Audio()
sourceAudio.format = "wav"
sourceAudio.rate = 16000
sourceAudio.bits = 16
sourceAudio.channel = 1
var targetAudio = Data_Speech_Understanding_Audio()
targetAudio.format = "pcm"
targetAudio.rate = 16000
var params = Data_Speech_Ast_ReqParams()
params.mode = "s2s"
params.sourceLanguage = conf.sourceLanguage
params.targetLanguage = conf.targetLanguage
var meta = Data_Speech_Common_RequestMeta()
meta.sessionID = sessionId
var req = Data_Speech_Ast_TranslateRequest()
req.requestMeta = meta
req.event = .startSession
req.user = user
req.sourceAudio = sourceAudio
req.targetAudio = targetAudio
req.request = params
return req
}
/**
* 构造音频分片请求(Protobuf)
*/
private func makeChunkRequest(sessionId: String, chunk: Data) -> Data_Speech_Ast_TranslateRequest {
var sourceAudio = Data_Speech_Understanding_Audio()
sourceAudio.format = "wav"
sourceAudio.rate = 16000
sourceAudio.bits = 16
sourceAudio.channel = 1
sourceAudio.binaryData = chunk
var meta = Data_Speech_Common_RequestMeta()
meta.sessionID = sessionId
var req = Data_Speech_Ast_TranslateRequest()
req.requestMeta = meta
req.event = .taskRequest
var user = Data_Speech_Understanding_User()
user.uid = "ast_ios_client"
user.did = "ast_ios_client"
req.user = user
req.sourceAudio = sourceAudio
return req
}
/**
* 构造 FinishSession 请求(Protobuf)
*/
private func makeFinishRequest(sessionId: String) -> Data_Speech_Ast_TranslateRequest {
var meta = Data_Speech_Common_RequestMeta()
meta.sessionID = sessionId
var user = Data_Speech_Understanding_User()
user.uid = "ast_ios_client"
user.did = "ast_ios_client"
var sourceAudio = Data_Speech_Understanding_Audio()
var req = Data_Speech_Ast_TranslateRequest()
req.requestMeta = meta
req.event = .finishSession
req.user = user
req.sourceAudio = sourceAudio
return req
}
}

301
local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/common/events.pb.swift

@ -0,0 +1,301 @@
// DO NOT EDIT.
// swift-format-ignore-file
// swiftlint:disable all
//
// Generated by the Swift generator plugin for the protocol buffer compiler.
// Source: common/events.proto
//
// For information on using the generated types, please see the documentation:
// https://github.com/apple/swift-protobuf/
import SwiftProtobuf
// If the compiler emits an error on this type, it is because this file
// was generated by a version of the `protoc` Swift plug-in that is
// incompatible with the version of SwiftProtobuf to which you are linking.
// Please ensure that you are building against the same version of the API
// that was used to generate this file.
fileprivate struct _GeneratedWithProtocGenSwiftVersion: SwiftProtobuf.ProtobufAPIVersionCheck {
struct _2: SwiftProtobuf.ProtobufAPIVersion_2 {}
typealias Version = _2
}
enum Data_Speech_Event_Type: SwiftProtobuf.Enum, Swift.CaseIterable {
typealias RawValue = Int
/// 默认事件,适用于不使用事件的方案或不需要传递事件的情况,
/// 或者对于使用事件的方案,可以通过非0值来校验事件的合法性
case none // = 0
/// 1 ~ 49 为上行Connection事件
case startConnection // = 1
/// Alias of "StartConnection"
static let startTask = startConnection
case finishConnection // = 2
/// Alias of "FinishConnection"
static let finishTask = finishConnection
/// 50 ~ 99 为下行Connection事件
/// 成功建连
case connectionStarted // = 50
/// Alias of "ConnectionStarted"
static let taskStarted = connectionStarted
/// 建连失败(可能是无法通过权限认证)
case connectionFailed // = 51
/// Alias of "ConnectionFailed"
static let taskFailed = connectionFailed
/// 连接结束
case connectionFinished // = 52
/// Alias of "ConnectionFinished"
static let taskFinished = connectionFinished
/// 100 ~ 149 为上行Session事件
case startSession // = 100
case cancelSession // = 101
case finishSession // = 102
/// 150 ~ 199 为下行Session事件
case sessionStarted // = 150
case sessionCanceled // = 151
case sessionFinished // = 152
case sessionFailed // = 153
/// 用量事件
case usageResponse // = 154
/// Alias of "UsageResponse"
static let chargeData = usageResponse
/// 200 ~ 249 为上行通用事件
case taskRequest // = 200
case updateConfig // = 201
case imageRequest // = 202
/// 250 ~ 299 为下行通用事件
case audioMuted // = 250
/// 300 ~ 349 为上行TTS事件
case sayHello // = 300
/// 350 ~ 399 为下行TTS事件
case ttssentenceStart // = 350
case ttssentenceEnd // = 351
case ttsresponse // = 352
case ttsended // = 359
case podcastRoundStart // = 360
case podcastRoundResponse // = 361
case podcastRoundEnd // = 362
/// 450 ~ 499 为下行ASR事件
case asrinfo // = 450
case asrresponse // = 451
case asrended // = 459
/// 500 ~ 549 为上行对话事件
/// (Ground-Truth-Alignment) text for speech synthesis
case chatTtstext // = 500
case chatTextQuery // = 501
/// 550 ~ 599 为下行对话事件
case chatResponse // = 550
case chimeInStart // = 551
case chimeInEnd // = 552
case chatEnded // = 559
case thinkStart // = 560
case thinkResponse // = 561
case thinkEnd // = 562
case toolOutput // = 563
case fcresponseStart // = 564
case fcresponse // = 565
case fcresponseEnd // = 566
/// 650 ~ 699 为下行对话事件
/// Events for source (original) language subtitle.
case sourceSubtitleStart // = 650
case sourceSubtitleResponse // = 651
case sourceSubtitleEnd // = 652
/// Events for target (translation) language subtitle.
case translationSubtitleStart // = 653
case translationSubtitleResponse // = 654
case translationSubtitleEnd // = 655
case UNRECOGNIZED(Int)
init() {
self = .none
}
init?(rawValue: Int) {
switch rawValue {
case 0: self = .none
case 1: self = .startConnection
case 2: self = .finishConnection
case 50: self = .connectionStarted
case 51: self = .connectionFailed
case 52: self = .connectionFinished
case 100: self = .startSession
case 101: self = .cancelSession
case 102: self = .finishSession
case 150: self = .sessionStarted
case 151: self = .sessionCanceled
case 152: self = .sessionFinished
case 153: self = .sessionFailed
case 154: self = .usageResponse
case 200: self = .taskRequest
case 201: self = .updateConfig
case 202: self = .imageRequest
case 250: self = .audioMuted
case 300: self = .sayHello
case 350: self = .ttssentenceStart
case 351: self = .ttssentenceEnd
case 352: self = .ttsresponse
case 359: self = .ttsended
case 360: self = .podcastRoundStart
case 361: self = .podcastRoundResponse
case 362: self = .podcastRoundEnd
case 450: self = .asrinfo
case 451: self = .asrresponse
case 459: self = .asrended
case 500: self = .chatTtstext
case 501: self = .chatTextQuery
case 550: self = .chatResponse
case 551: self = .chimeInStart
case 552: self = .chimeInEnd
case 559: self = .chatEnded
case 560: self = .thinkStart
case 561: self = .thinkResponse
case 562: self = .thinkEnd
case 563: self = .toolOutput
case 564: self = .fcresponseStart
case 565: self = .fcresponse
case 566: self = .fcresponseEnd
case 650: self = .sourceSubtitleStart
case 651: self = .sourceSubtitleResponse
case 652: self = .sourceSubtitleEnd
case 653: self = .translationSubtitleStart
case 654: self = .translationSubtitleResponse
case 655: self = .translationSubtitleEnd
default: self = .UNRECOGNIZED(rawValue)
}
}
var rawValue: Int {
switch self {
case .none: return 0
case .startConnection: return 1
case .finishConnection: return 2
case .connectionStarted: return 50
case .connectionFailed: return 51
case .connectionFinished: return 52
case .startSession: return 100
case .cancelSession: return 101
case .finishSession: return 102
case .sessionStarted: return 150
case .sessionCanceled: return 151
case .sessionFinished: return 152
case .sessionFailed: return 153
case .usageResponse: return 154
case .taskRequest: return 200
case .updateConfig: return 201
case .imageRequest: return 202
case .audioMuted: return 250
case .sayHello: return 300
case .ttssentenceStart: return 350
case .ttssentenceEnd: return 351
case .ttsresponse: return 352
case .ttsended: return 359
case .podcastRoundStart: return 360
case .podcastRoundResponse: return 361
case .podcastRoundEnd: return 362
case .asrinfo: return 450
case .asrresponse: return 451
case .asrended: return 459
case .chatTtstext: return 500
case .chatTextQuery: return 501
case .chatResponse: return 550
case .chimeInStart: return 551
case .chimeInEnd: return 552
case .chatEnded: return 559
case .thinkStart: return 560
case .thinkResponse: return 561
case .thinkEnd: return 562
case .toolOutput: return 563
case .fcresponseStart: return 564
case .fcresponse: return 565
case .fcresponseEnd: return 566
case .sourceSubtitleStart: return 650
case .sourceSubtitleResponse: return 651
case .sourceSubtitleEnd: return 652
case .translationSubtitleStart: return 653
case .translationSubtitleResponse: return 654
case .translationSubtitleEnd: return 655
case .UNRECOGNIZED(let i): return i
}
}
// The compiler won't synthesize support with the UNRECOGNIZED case.
static let allCases: [Data_Speech_Event_Type] = [
.none,
.startConnection,
.finishConnection,
.connectionStarted,
.connectionFailed,
.connectionFinished,
.startSession,
.cancelSession,
.finishSession,
.sessionStarted,
.sessionCanceled,
.sessionFinished,
.sessionFailed,
.usageResponse,
.taskRequest,
.updateConfig,
.imageRequest,
.audioMuted,
.sayHello,
.ttssentenceStart,
.ttssentenceEnd,
.ttsresponse,
.ttsended,
.podcastRoundStart,
.podcastRoundResponse,
.podcastRoundEnd,
.asrinfo,
.asrresponse,
.asrended,
.chatTtstext,
.chatTextQuery,
.chatResponse,
.chimeInStart,
.chimeInEnd,
.chatEnded,
.thinkStart,
.thinkResponse,
.thinkEnd,
.toolOutput,
.fcresponseStart,
.fcresponse,
.fcresponseEnd,
.sourceSubtitleStart,
.sourceSubtitleResponse,
.sourceSubtitleEnd,
.translationSubtitleStart,
.translationSubtitleResponse,
.translationSubtitleEnd,
]
}
// MARK: - Code below here is support for the SwiftProtobuf runtime.
extension Data_Speech_Event_Type: SwiftProtobuf._ProtoNameProviding {
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{2}\0None\0\u{9}StartConnection\0\u{1}StartTask\0\u{9}FinishConnection\0\u{1}FinishTask\0\u{a}0ConnectionStarted\0\u{1}TaskStarted\0\u{9}ConnectionFailed\0\u{1}TaskFailed\0\u{9}ConnectionFinished\0\u{1}TaskFinished\0\u{2}0StartSession\0\u{1}CancelSession\0\u{1}FinishSession\0\u{2}0SessionStarted\0\u{1}SessionCanceled\0\u{1}SessionFinished\0\u{1}SessionFailed\0\u{9}UsageResponse\0\u{1}ChargeData\0\u{2}.TaskRequest\0\u{1}UpdateConfig\0\u{1}ImageRequest\0\u{2}0AudioMuted\0\u{2}2SayHello\0\u{2}2TTSSentenceStart\0\u{1}TTSSentenceEnd\0\u{1}TTSResponse\0\u{2}\u{7}TTSEnded\0\u{1}PodcastRoundStart\0\u{1}PodcastRoundResponse\0\u{1}PodcastRoundEnd\0\u{2}X\u{1}ASRInfo\0\u{1}ASRResponse\0\u{2}\u{8}ASREnded\0\u{2})ChatTTSText\0\u{1}ChatTextQuery\0\u{2}1ChatResponse\0\u{1}ChimeInStart\0\u{1}ChimeInEnd\0\u{2}\u{7}ChatEnded\0\u{1}ThinkStart\0\u{1}ThinkResponse\0\u{1}ThinkEnd\0\u{1}ToolOutput\0\u{1}FCResponseStart\0\u{1}FCResponse\0\u{1}FCResponseEnd\0\u{2}T\u{1}SourceSubtitleStart\0\u{1}SourceSubtitleResponse\0\u{1}SourceSubtitleEnd\0\u{1}TranslationSubtitleStart\0\u{1}TranslationSubtitleResponse\0\u{1}TranslationSubtitleEnd\0")
}

350
local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/common/rpcmeta.pb.swift

@ -0,0 +1,350 @@
// DO NOT EDIT.
// swift-format-ignore-file
// swiftlint:disable all
//
// Generated by the Swift generator plugin for the protocol buffer compiler.
// Source: common/rpcmeta.proto
//
// For information on using the generated types, please see the documentation:
// https://github.com/apple/swift-protobuf/
import SwiftProtobuf
// If the compiler emits an error on this type, it is because this file
// was generated by a version of the `protoc` Swift plug-in that is
// incompatible with the version of SwiftProtobuf to which you are linking.
// Please ensure that you are building against the same version of the API
// that was used to generate this file.
fileprivate struct _GeneratedWithProtocGenSwiftVersion: SwiftProtobuf.ProtobufAPIVersionCheck {
struct _2: SwiftProtobuf.ProtobufAPIVersion_2 {}
typealias Version = _2
}
struct Data_Speech_Common_RequestMeta: Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
/// Required.
/// Backend endpoint name. Gateway will also use it to infer HTTP/WebSocket URL
/// path.
var endpoint: String = String()
/// Required.
/// For historical reasons, this can be either a volcanic appid or a sail
/// platform appKey. If both are transmitted at the same time, you need to put
/// the volcengine appid in the AppID
var appKey: String = String()
/// Optional.
/// appID in volcengine
var appID: String = String()
/// - Required by gateway, you must provide a ResourceID on calling gateway.
/// - Optional for backend services, gateway doesn't need to pass this to
/// backend.
var resourceID: String = String()
/// Required for Websocket connection (SDK -> Gateway), optional otherwise.
/// In websocket communication, a connection can be reused for multiple
/// simultaneous sessions. Servers use this ID mainly for debugging purposes.
var connectionID: String = String()
/// Required. Session is the minimum unit that performs a specific task. The
/// request data in a session can be devided into a number data packets
/// transmitted (but not necessarily processed by the server) in a sequential
/// manner. Each data packet can be denoted by a "Sequence" number (although
/// populating the "Sequence" field is not mandatory). In the simplest case,
/// all request data in a session is sent as a lump (ie, only one packet).
var sessionID: String = String()
/// Optional, if not passed, default value 0 will be assumed. For streaming
/// RPC, this field can be omitted (because the order of sending is preserved
/// at the receiving side) unless the client needs the acknowledgement of
/// packet receipt from the server side.
var sequence: Int32 = 0
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
}
struct Data_Speech_Common_BillingItem: Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
/// Required.
/// Billing unit, eg:
/// - minute (billed by the number of minutes used)
/// - word (billed by the number of words submitted or generated)
/// - call (billed by the number of RPC/HTTP/Websocket calls)
var unit: String = String()
/// Optional.
/// The amount that the consumer consumed counted by `BillingItem.Unit`. We use
/// a float number because sometimes we want precision better than whole
/// numbers.
var quantity: Float = 0
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
}
struct Data_Speech_Common_Billing: Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
/// Optional.
/// For new billing items, use this field. This is a list because there may be
/// more than one billing items in one request/session.
var items: [Data_Speech_Common_BillingItem] = []
/// Optional.
/// For commodities that are priced w.r.t (typically audio/video) duration in
/// milliseconds.
var durationMsec: Int64 = 0
/// Optional.
/// For commodities priced w.r.t text length (number of words/tokens).
var wordCount: Int64 = 0
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
}
struct Data_Speech_Common_ResponseMeta: Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
/// Required.
/// The same SessionID in RequestMeta.
var sessionID: String = String()
/// Optional.
/// The same sequence in RequestMeta (except that server *may* turn a positive
/// sequence to its negative counterpart).
var sequence: Int32 = 0
/// Optional.
/// Response status code.
var statusCode: Int32 = 0
/// Optional.
/// Detailed status information.
var message: String = String()
/// Billing information in case only backend servers can collect this
/// information.
var billing: Data_Speech_Common_Billing {
get {return _billing ?? Data_Speech_Common_Billing()}
set {_billing = newValue}
}
/// Returns true if `billing` has been explicitly set.
var hasBilling: Bool {return self._billing != nil}
/// Clears the value of `billing`. Subsequent reads from it will return its default value.
mutating func clearBilling() {self._billing = nil}
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
fileprivate var _billing: Data_Speech_Common_Billing? = nil
}
// MARK: - Code below here is support for the SwiftProtobuf runtime.
fileprivate let _protobuf_package = "data.speech.common"
extension Data_Speech_Common_RequestMeta: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".RequestMeta"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{5}Endpoint\0endpoint\0\u{5}AppKey\0app_key\0\u{5}AppID\0app_id\0\u{5}ResourceID\0resource_id\0\u{5}ConnectionID\0connection_id\0\u{5}SessionID\0session_id\0\u{5}Sequence\0sequence\0")
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularStringField(value: &self.endpoint) }()
case 2: try { try decoder.decodeSingularStringField(value: &self.appKey) }()
case 3: try { try decoder.decodeSingularStringField(value: &self.appID) }()
case 4: try { try decoder.decodeSingularStringField(value: &self.resourceID) }()
case 5: try { try decoder.decodeSingularStringField(value: &self.connectionID) }()
case 6: try { try decoder.decodeSingularStringField(value: &self.sessionID) }()
case 7: try { try decoder.decodeSingularInt32Field(value: &self.sequence) }()
default: break
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
if !self.endpoint.isEmpty {
try visitor.visitSingularStringField(value: self.endpoint, fieldNumber: 1)
}
if !self.appKey.isEmpty {
try visitor.visitSingularStringField(value: self.appKey, fieldNumber: 2)
}
if !self.appID.isEmpty {
try visitor.visitSingularStringField(value: self.appID, fieldNumber: 3)
}
if !self.resourceID.isEmpty {
try visitor.visitSingularStringField(value: self.resourceID, fieldNumber: 4)
}
if !self.connectionID.isEmpty {
try visitor.visitSingularStringField(value: self.connectionID, fieldNumber: 5)
}
if !self.sessionID.isEmpty {
try visitor.visitSingularStringField(value: self.sessionID, fieldNumber: 6)
}
if self.sequence != 0 {
try visitor.visitSingularInt32Field(value: self.sequence, fieldNumber: 7)
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Common_RequestMeta, rhs: Data_Speech_Common_RequestMeta) -> Bool {
if lhs.endpoint != rhs.endpoint {return false}
if lhs.appKey != rhs.appKey {return false}
if lhs.appID != rhs.appID {return false}
if lhs.resourceID != rhs.resourceID {return false}
if lhs.connectionID != rhs.connectionID {return false}
if lhs.sessionID != rhs.sessionID {return false}
if lhs.sequence != rhs.sequence {return false}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}
extension Data_Speech_Common_BillingItem: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".BillingItem"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{5}Unit\0unit\0\u{5}Quantity\0quantity\0")
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularStringField(value: &self.unit) }()
case 2: try { try decoder.decodeSingularFloatField(value: &self.quantity) }()
default: break
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
if !self.unit.isEmpty {
try visitor.visitSingularStringField(value: self.unit, fieldNumber: 1)
}
if self.quantity.bitPattern != 0 {
try visitor.visitSingularFloatField(value: self.quantity, fieldNumber: 2)
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Common_BillingItem, rhs: Data_Speech_Common_BillingItem) -> Bool {
if lhs.unit != rhs.unit {return false}
if lhs.quantity != rhs.quantity {return false}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}
extension Data_Speech_Common_Billing: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".Billing"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{5}Items\0items\0\u{5}DurationMsec\0duration_msec\0\u{5}WordCount\0word_count\0")
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeRepeatedMessageField(value: &self.items) }()
case 2: try { try decoder.decodeSingularInt64Field(value: &self.durationMsec) }()
case 3: try { try decoder.decodeSingularInt64Field(value: &self.wordCount) }()
default: break
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
if !self.items.isEmpty {
try visitor.visitRepeatedMessageField(value: self.items, fieldNumber: 1)
}
if self.durationMsec != 0 {
try visitor.visitSingularInt64Field(value: self.durationMsec, fieldNumber: 2)
}
if self.wordCount != 0 {
try visitor.visitSingularInt64Field(value: self.wordCount, fieldNumber: 3)
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Common_Billing, rhs: Data_Speech_Common_Billing) -> Bool {
if lhs.items != rhs.items {return false}
if lhs.durationMsec != rhs.durationMsec {return false}
if lhs.wordCount != rhs.wordCount {return false}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}
extension Data_Speech_Common_ResponseMeta: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".ResponseMeta"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{5}SessionID\0session_id\0\u{5}Sequence\0sequence\0\u{5}StatusCode\0status_code\0\u{5}Message\0message\0\u{5}Billing\0billing\0")
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularStringField(value: &self.sessionID) }()
case 2: try { try decoder.decodeSingularInt32Field(value: &self.sequence) }()
case 3: try { try decoder.decodeSingularInt32Field(value: &self.statusCode) }()
case 4: try { try decoder.decodeSingularStringField(value: &self.message) }()
case 5: try { try decoder.decodeSingularMessageField(value: &self._billing) }()
default: break
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every if/case branch local when no optimizations
// are enabled. https://github.com/apple/swift-protobuf/issues/1034 and
// https://github.com/apple/swift-protobuf/issues/1182
if !self.sessionID.isEmpty {
try visitor.visitSingularStringField(value: self.sessionID, fieldNumber: 1)
}
if self.sequence != 0 {
try visitor.visitSingularInt32Field(value: self.sequence, fieldNumber: 2)
}
if self.statusCode != 0 {
try visitor.visitSingularInt32Field(value: self.statusCode, fieldNumber: 3)
}
if !self.message.isEmpty {
try visitor.visitSingularStringField(value: self.message, fieldNumber: 4)
}
try { if let v = self._billing {
try visitor.visitSingularMessageField(value: v, fieldNumber: 5)
} }()
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Common_ResponseMeta, rhs: Data_Speech_Common_ResponseMeta) -> Bool {
if lhs.sessionID != rhs.sessionID {return false}
if lhs.sequence != rhs.sequence {return false}
if lhs.statusCode != rhs.statusCode {return false}
if lhs.message != rhs.message {return false}
if lhs._billing != rhs._billing {return false}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}

461
local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/products/understanding/ast/ast_service.pb.swift

@ -0,0 +1,461 @@
// DO NOT EDIT.
// swift-format-ignore-file
// swiftlint:disable all
//
// Generated by the Swift generator plugin for the protocol buffer compiler.
// Source: products/understanding/ast/ast_service.proto
//
// For information on using the generated types, please see the documentation:
// https://github.com/apple/swift-protobuf/
import Foundation
import SwiftProtobuf
// If the compiler emits an error on this type, it is because this file
// was generated by a version of the `protoc` Swift plug-in that is
// incompatible with the version of SwiftProtobuf to which you are linking.
// Please ensure that you are building against the same version of the API
// that was used to generate this file.
fileprivate struct _GeneratedWithProtocGenSwiftVersion: SwiftProtobuf.ProtobufAPIVersionCheck {
struct _2: SwiftProtobuf.ProtobufAPIVersion_2 {}
typealias Version = _2
}
struct Data_Speech_Ast_ReqParams: @unchecked Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
/// 可能是s2t , s2s 选一个, 控制是否需要语音
var mode: String {
get {return _storage._mode}
set {_uniqueStorage()._mode = newValue}
}
/// 源语言
var sourceLanguage: String {
get {return _storage._sourceLanguage}
set {_uniqueStorage()._sourceLanguage = newValue}
}
/// 目标语言
var targetLanguage: String {
get {return _storage._targetLanguage}
set {_uniqueStorage()._targetLanguage = newValue}
}
var speakerID: String {
get {return _storage._speakerID}
set {_uniqueStorage()._speakerID = newValue}
}
var corpus: Data_Speech_Understanding_Corpus {
get {return _storage._corpus ?? Data_Speech_Understanding_Corpus()}
set {_uniqueStorage()._corpus = newValue}
}
/// Returns true if `corpus` has been explicitly set.
var hasCorpus: Bool {return _storage._corpus != nil}
/// Clears the value of `corpus`. Subsequent reads from it will return its default value.
mutating func clearCorpus() {_uniqueStorage()._corpus = nil}
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
fileprivate var _storage = _StorageClass.defaultInstance
}
struct Data_Speech_Ast_TranslateRequest: @unchecked Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
var requestMeta: Data_Speech_Common_RequestMeta {
get {return _storage._requestMeta ?? Data_Speech_Common_RequestMeta()}
set {_uniqueStorage()._requestMeta = newValue}
}
/// Returns true if `requestMeta` has been explicitly set.
var hasRequestMeta: Bool {return _storage._requestMeta != nil}
/// Clears the value of `requestMeta`. Subsequent reads from it will return its default value.
mutating func clearRequestMeta() {_uniqueStorage()._requestMeta = nil}
var event: Data_Speech_Event_Type {
get {return _storage._event}
set {_uniqueStorage()._event = newValue}
}
var user: Data_Speech_Understanding_User {
get {return _storage._user ?? Data_Speech_Understanding_User()}
set {_uniqueStorage()._user = newValue}
}
/// Returns true if `user` has been explicitly set.
var hasUser: Bool {return _storage._user != nil}
/// Clears the value of `user`. Subsequent reads from it will return its default value.
mutating func clearUser() {_uniqueStorage()._user = nil}
/// bytes data = 2; // request binary data
var sourceAudio: Data_Speech_Understanding_Audio {
get {return _storage._sourceAudio ?? Data_Speech_Understanding_Audio()}
set {_uniqueStorage()._sourceAudio = newValue}
}
/// Returns true if `sourceAudio` has been explicitly set.
var hasSourceAudio: Bool {return _storage._sourceAudio != nil}
/// Clears the value of `sourceAudio`. Subsequent reads from it will return its default value.
mutating func clearSourceAudio() {_uniqueStorage()._sourceAudio = nil}
/// 目标音频信息,只需要传format(pcm/ogg)、rate、bits、channel这些
var targetAudio: Data_Speech_Understanding_Audio {
get {return _storage._targetAudio ?? Data_Speech_Understanding_Audio()}
set {_uniqueStorage()._targetAudio = newValue}
}
/// Returns true if `targetAudio` has been explicitly set.
var hasTargetAudio: Bool {return _storage._targetAudio != nil}
/// Clears the value of `targetAudio`. Subsequent reads from it will return its default value.
mutating func clearTargetAudio() {_uniqueStorage()._targetAudio = nil}
/// 请求参数
var request: Data_Speech_Ast_ReqParams {
get {return _storage._request ?? Data_Speech_Ast_ReqParams()}
set {_uniqueStorage()._request = newValue}
}
/// Returns true if `request` has been explicitly set.
var hasRequest: Bool {return _storage._request != nil}
/// Clears the value of `request`. Subsequent reads from it will return its default value.
mutating func clearRequest() {_uniqueStorage()._request = nil}
/// 是否开启降噪
var denoise: Bool {
get {return _storage._denoise ?? false}
set {_uniqueStorage()._denoise = newValue}
}
/// Returns true if `denoise` has been explicitly set.
var hasDenoise: Bool {return _storage._denoise != nil}
/// Clears the value of `denoise`. Subsequent reads from it will return its default value.
mutating func clearDenoise() {_uniqueStorage()._denoise = nil}
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
fileprivate var _storage = _StorageClass.defaultInstance
}
struct Data_Speech_Ast_TranslateResponse: Sendable {
// SwiftProtobuf.Message conformance is added in an extension below. See the
// `Message` and `Message+*Additions` files in the SwiftProtobuf library for
// methods supported on all messages.
var responseMeta: Data_Speech_Common_ResponseMeta {
get {return _responseMeta ?? Data_Speech_Common_ResponseMeta()}
set {_responseMeta = newValue}
}
/// Returns true if `responseMeta` has been explicitly set.
var hasResponseMeta: Bool {return self._responseMeta != nil}
/// Clears the value of `responseMeta`. Subsequent reads from it will return its default value.
mutating func clearResponseMeta() {self._responseMeta = nil}
var event: Data_Speech_Event_Type = .none
/// response binary data
var data: Data = Data()
/// 原文或者译文
var text: String = String()
var startTime: Int32 = 0
var endTime: Int32 = 0
var spkChg: Bool = false
var mutedDurationMs: Int32 = 0
var unknownFields = SwiftProtobuf.UnknownStorage()
init() {}
fileprivate var _responseMeta: Data_Speech_Common_ResponseMeta? = nil
}
// MARK: - Code below here is support for the SwiftProtobuf runtime.
fileprivate let _protobuf_package = "data.speech.ast"
extension Data_Speech_Ast_ReqParams: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".ReqParams"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{1}mode\0\u{3}source_language\0\u{3}target_language\0\u{3}speaker_id\0\u{2}`\u{1}corpus\0")
fileprivate class _StorageClass {
var _mode: String = String()
var _sourceLanguage: String = String()
var _targetLanguage: String = String()
var _speakerID: String = String()
var _corpus: Data_Speech_Understanding_Corpus? = nil
// This property is used as the initial default value for new instances of the type.
// The type itself is protecting the reference to its storage via CoW semantics.
// This will force a copy to be made of this reference when the first mutation occurs;
// hence, it is safe to mark this as `nonisolated(unsafe)`.
static nonisolated(unsafe) let defaultInstance = _StorageClass()
private init() {}
init(copying source: _StorageClass) {
_mode = source._mode
_sourceLanguage = source._sourceLanguage
_targetLanguage = source._targetLanguage
_speakerID = source._speakerID
_corpus = source._corpus
}
}
fileprivate mutating func _uniqueStorage() -> _StorageClass {
if !isKnownUniquelyReferenced(&_storage) {
_storage = _StorageClass(copying: _storage)
}
return _storage
}
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
_ = _uniqueStorage()
try withExtendedLifetime(_storage) { (_storage: _StorageClass) in
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularStringField(value: &_storage._mode) }()
case 2: try { try decoder.decodeSingularStringField(value: &_storage._sourceLanguage) }()
case 3: try { try decoder.decodeSingularStringField(value: &_storage._targetLanguage) }()
case 4: try { try decoder.decodeSingularStringField(value: &_storage._speakerID) }()
case 100: try { try decoder.decodeSingularMessageField(value: &_storage._corpus) }()
default: break
}
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
try withExtendedLifetime(_storage) { (_storage: _StorageClass) in
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every if/case branch local when no optimizations
// are enabled. https://github.com/apple/swift-protobuf/issues/1034 and
// https://github.com/apple/swift-protobuf/issues/1182
if !_storage._mode.isEmpty {
try visitor.visitSingularStringField(value: _storage._mode, fieldNumber: 1)
}
if !_storage._sourceLanguage.isEmpty {
try visitor.visitSingularStringField(value: _storage._sourceLanguage, fieldNumber: 2)
}
if !_storage._targetLanguage.isEmpty {
try visitor.visitSingularStringField(value: _storage._targetLanguage, fieldNumber: 3)
}
if !_storage._speakerID.isEmpty {
try visitor.visitSingularStringField(value: _storage._speakerID, fieldNumber: 4)
}
try { if let v = _storage._corpus {
try visitor.visitSingularMessageField(value: v, fieldNumber: 100)
} }()
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Ast_ReqParams, rhs: Data_Speech_Ast_ReqParams) -> Bool {
if lhs._storage !== rhs._storage {
let storagesAreEqual: Bool = withExtendedLifetime((lhs._storage, rhs._storage)) { (_args: (_StorageClass, _StorageClass)) in
let _storage = _args.0
let rhs_storage = _args.1
if _storage._mode != rhs_storage._mode {return false}
if _storage._sourceLanguage != rhs_storage._sourceLanguage {return false}
if _storage._targetLanguage != rhs_storage._targetLanguage {return false}
if _storage._speakerID != rhs_storage._speakerID {return false}
if _storage._corpus != rhs_storage._corpus {return false}
return true
}
if !storagesAreEqual {return false}
}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}
extension Data_Speech_Ast_TranslateRequest: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".TranslateRequest"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{3}request_meta\0\u{1}event\0\u{1}user\0\u{3}source_audio\0\u{3}target_audio\0\u{1}request\0\u{1}denoise\0")
fileprivate class _StorageClass {
var _requestMeta: Data_Speech_Common_RequestMeta? = nil
var _event: Data_Speech_Event_Type = .none
var _user: Data_Speech_Understanding_User? = nil
var _sourceAudio: Data_Speech_Understanding_Audio? = nil
var _targetAudio: Data_Speech_Understanding_Audio? = nil
var _request: Data_Speech_Ast_ReqParams? = nil
var _denoise: Bool? = nil
// This property is used as the initial default value for new instances of the type.
// The type itself is protecting the reference to its storage via CoW semantics.
// This will force a copy to be made of this reference when the first mutation occurs;
// hence, it is safe to mark this as `nonisolated(unsafe)`.
static nonisolated(unsafe) let defaultInstance = _StorageClass()
private init() {}
init(copying source: _StorageClass) {
_requestMeta = source._requestMeta
_event = source._event
_user = source._user
_sourceAudio = source._sourceAudio
_targetAudio = source._targetAudio
_request = source._request
_denoise = source._denoise
}
}
fileprivate mutating func _uniqueStorage() -> _StorageClass {
if !isKnownUniquelyReferenced(&_storage) {
_storage = _StorageClass(copying: _storage)
}
return _storage
}
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
_ = _uniqueStorage()
try withExtendedLifetime(_storage) { (_storage: _StorageClass) in
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularMessageField(value: &_storage._requestMeta) }()
case 2: try { try decoder.decodeSingularEnumField(value: &_storage._event) }()
case 3: try { try decoder.decodeSingularMessageField(value: &_storage._user) }()
case 4: try { try decoder.decodeSingularMessageField(value: &_storage._sourceAudio) }()
case 5: try { try decoder.decodeSingularMessageField(value: &_storage._targetAudio) }()
case 6: try { try decoder.decodeSingularMessageField(value: &_storage._request) }()
case 7: try { try decoder.decodeSingularBoolField(value: &_storage._denoise) }()
default: break
}
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
try withExtendedLifetime(_storage) { (_storage: _StorageClass) in
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every if/case branch local when no optimizations
// are enabled. https://github.com/apple/swift-protobuf/issues/1034 and
// https://github.com/apple/swift-protobuf/issues/1182
try { if let v = _storage._requestMeta {
try visitor.visitSingularMessageField(value: v, fieldNumber: 1)
} }()
if _storage._event != .none {
try visitor.visitSingularEnumField(value: _storage._event, fieldNumber: 2)
}
try { if let v = _storage._user {
try visitor.visitSingularMessageField(value: v, fieldNumber: 3)
} }()
try { if let v = _storage._sourceAudio {
try visitor.visitSingularMessageField(value: v, fieldNumber: 4)
} }()
try { if let v = _storage._targetAudio {
try visitor.visitSingularMessageField(value: v, fieldNumber: 5)
} }()
try { if let v = _storage._request {
try visitor.visitSingularMessageField(value: v, fieldNumber: 6)
} }()
try { if let v = _storage._denoise {
try visitor.visitSingularBoolField(value: v, fieldNumber: 7)
} }()
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Ast_TranslateRequest, rhs: Data_Speech_Ast_TranslateRequest) -> Bool {
if lhs._storage !== rhs._storage {
let storagesAreEqual: Bool = withExtendedLifetime((lhs._storage, rhs._storage)) { (_args: (_StorageClass, _StorageClass)) in
let _storage = _args.0
let rhs_storage = _args.1
if _storage._requestMeta != rhs_storage._requestMeta {return false}
if _storage._event != rhs_storage._event {return false}
if _storage._user != rhs_storage._user {return false}
if _storage._sourceAudio != rhs_storage._sourceAudio {return false}
if _storage._targetAudio != rhs_storage._targetAudio {return false}
if _storage._request != rhs_storage._request {return false}
if _storage._denoise != rhs_storage._denoise {return false}
return true
}
if !storagesAreEqual {return false}
}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}
extension Data_Speech_Ast_TranslateResponse: SwiftProtobuf.Message, SwiftProtobuf._MessageImplementationBase, SwiftProtobuf._ProtoNameProviding {
static let protoMessageName: String = _protobuf_package + ".TranslateResponse"
static let _protobuf_nameMap = SwiftProtobuf._NameMap(bytecode: "\0\u{3}response_meta\0\u{1}event\0\u{1}data\0\u{1}text\0\u{3}start_time\0\u{3}end_time\0\u{3}spk_chg\0\u{3}muted_duration_ms\0")
mutating func decodeMessage<D: SwiftProtobuf.Decoder>(decoder: inout D) throws {
while let fieldNumber = try decoder.nextFieldNumber() {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every case branch when no optimizations are
// enabled. https://github.com/apple/swift-protobuf/issues/1034
switch fieldNumber {
case 1: try { try decoder.decodeSingularMessageField(value: &self._responseMeta) }()
case 2: try { try decoder.decodeSingularEnumField(value: &self.event) }()
case 3: try { try decoder.decodeSingularBytesField(value: &self.data) }()
case 4: try { try decoder.decodeSingularStringField(value: &self.text) }()
case 5: try { try decoder.decodeSingularInt32Field(value: &self.startTime) }()
case 6: try { try decoder.decodeSingularInt32Field(value: &self.endTime) }()
case 7: try { try decoder.decodeSingularBoolField(value: &self.spkChg) }()
case 8: try { try decoder.decodeSingularInt32Field(value: &self.mutedDurationMs) }()
default: break
}
}
}
func traverse<V: SwiftProtobuf.Visitor>(visitor: inout V) throws {
// The use of inline closures is to circumvent an issue where the compiler
// allocates stack space for every if/case branch local when no optimizations
// are enabled. https://github.com/apple/swift-protobuf/issues/1034 and
// https://github.com/apple/swift-protobuf/issues/1182
try { if let v = self._responseMeta {
try visitor.visitSingularMessageField(value: v, fieldNumber: 1)
} }()
if self.event != .none {
try visitor.visitSingularEnumField(value: self.event, fieldNumber: 2)
}
if !self.data.isEmpty {
try visitor.visitSingularBytesField(value: self.data, fieldNumber: 3)
}
if !self.text.isEmpty {
try visitor.visitSingularStringField(value: self.text, fieldNumber: 4)
}
if self.startTime != 0 {
try visitor.visitSingularInt32Field(value: self.startTime, fieldNumber: 5)
}
if self.endTime != 0 {
try visitor.visitSingularInt32Field(value: self.endTime, fieldNumber: 6)
}
if self.spkChg != false {
try visitor.visitSingularBoolField(value: self.spkChg, fieldNumber: 7)
}
if self.mutedDurationMs != 0 {
try visitor.visitSingularInt32Field(value: self.mutedDurationMs, fieldNumber: 8)
}
try unknownFields.traverse(visitor: &visitor)
}
static func ==(lhs: Data_Speech_Ast_TranslateResponse, rhs: Data_Speech_Ast_TranslateResponse) -> Bool {
if lhs._responseMeta != rhs._responseMeta {return false}
if lhs.event != rhs.event {return false}
if lhs.data != rhs.data {return false}
if lhs.text != rhs.text {return false}
if lhs.startTime != rhs.startTime {return false}
if lhs.endTime != rhs.endTime {return false}
if lhs.spkChg != rhs.spkChg {return false}
if lhs.mutedDurationMs != rhs.mutedDurationMs {return false}
if lhs.unknownFields != rhs.unknownFields {return false}
return true
}
}

1803
local_plugins/azure_speech/ios/azure_speech/Sources/protos_swift/products/understanding/base/au_base.pb.swift

File diff suppressed because it is too large

20
local_plugins/azure_speech/lib/main.dart

@ -0,0 +1,20 @@
import 'package:flutter/material.dart';
void main() {
runApp(const MainApp());
}
class MainApp extends StatelessWidget {
const MainApp({super.key});
@override
Widget build(BuildContext context) {
return const MaterialApp(
home: Scaffold(
body: Center(
child: Text('Hello World!'),
),
),
);
}
}
Loading…
Cancel
Save