tts_voices.py 6.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243
  1. # Copyright (C) 2025 AIDC-AI
  2. #
  3. # Licensed under the Apache License, Version 2.0 (the "License");
  4. # you may not use this file except in compliance with the License.
  5. # You may obtain a copy of the License at
  6. # http://www.apache.org/licenses/LICENSE-2.0
  7. # Unless required by applicable law or agreed to in writing, software
  8. # distributed under the License is distributed on an "AS IS" BASIS,
  9. # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
  10. # See the License for the specific language governing permissions and
  11. # limitations under the License.
  12. """
  13. TTS Voice Configuration
  14. Defines available voices for local Edge TTS inference.
  15. """
  16. from typing import List, Dict, Any
  17. # Edge TTS voice presets for local inference
  18. EDGE_TTS_VOICES: List[Dict[str, Any]] = [
  19. # Chinese voices
  20. {
  21. "id": "zh-CN-XiaoxiaoNeural",
  22. "label_key": "tts.voice.zh_CN_XiaoxiaoNeural",
  23. "locale": "zh-CN",
  24. "gender": "female"
  25. },
  26. {
  27. "id": "zh-CN-XiaoyiNeural",
  28. "label_key": "tts.voice.zh_CN_XiaoyiNeural",
  29. "locale": "zh-CN",
  30. "gender": "female"
  31. },
  32. {
  33. "id": "zh-CN-YunjianNeural",
  34. "label_key": "tts.voice.zh_CN_YunjianNeural",
  35. "locale": "zh-CN",
  36. "gender": "male"
  37. },
  38. {
  39. "id": "zh-CN-YunxiNeural",
  40. "label_key": "tts.voice.zh_CN_YunxiNeural",
  41. "locale": "zh-CN",
  42. "gender": "male"
  43. },
  44. {
  45. "id": "zh-CN-YunyangNeural",
  46. "label_key": "tts.voice.zh_CN_YunyangNeural",
  47. "locale": "zh-CN",
  48. "gender": "male"
  49. },
  50. {
  51. "id": "zh-CN-YunyeNeural",
  52. "label_key": "tts.voice.zh_CN_YunyeNeural",
  53. "locale": "zh-CN",
  54. "gender": "male"
  55. },
  56. {
  57. "id": "zh-CN-YunfengNeural",
  58. "label_key": "tts.voice.zh_CN_YunfengNeural",
  59. "locale": "zh-CN",
  60. "gender": "male"
  61. },
  62. {
  63. "id": "zh-CN-liaoning-XiaobeiNeural",
  64. "label_key": "tts.voice.zh_CN_liaoning_XiaobeiNeural",
  65. "locale": "zh-CN",
  66. "gender": "female"
  67. },
  68. {
  69. "id": "en-US-AriaNeural",
  70. "label_key": "tts.voice.en_US_AriaNeural",
  71. "locale": "en-US",
  72. "gender": "female"
  73. },
  74. {
  75. "id": "en-US-JennyNeural",
  76. "label_key": "tts.voice.en_US_JennyNeural",
  77. "locale": "en-US",
  78. "gender": "female"
  79. },
  80. {
  81. "id": "en-US-GuyNeural",
  82. "label_key": "tts.voice.en_US_GuyNeural",
  83. "locale": "en-US",
  84. "gender": "male"
  85. },
  86. {
  87. "id": "en-US-DavisNeural",
  88. "label_key": "tts.voice.en_US_DavisNeural",
  89. "locale": "en-US",
  90. "gender": "male"
  91. },
  92. {
  93. "id": "en-GB-SoniaNeural",
  94. "label_key": "tts.voice.en_GB_SoniaNeural",
  95. "locale": "en-GB",
  96. "gender": "female"
  97. },
  98. {
  99. "id": "en-GB-RyanNeural",
  100. "label_key": "tts.voice.en_GB_RyanNeural",
  101. "locale": "en-GB",
  102. "gender": "male"
  103. },
  104. {
  105. "id": "ko-KR-InJoonNeural",
  106. "label_key": "tts.voice.ko-KR-InJoonNeural",
  107. "locale": "ko-KR",
  108. "gender": "male"
  109. },
  110. {
  111. "id": "ko-KR-SunHiNeural",
  112. "label_key": "tts.voice.ko-KR-SunHiNeural",
  113. "locale": "ko-KR",
  114. "gender": "female"
  115. },
  116. {
  117. "id": "fr-FR-EloiseNeural",
  118. "label_key": "tts.voice.fr-FR-EloiseNeural",
  119. "locale": "fr-FR",
  120. "gender": "female"
  121. },
  122. {
  123. "id": "fr-FR-HenriNeural",
  124. "label_key": "tts.voice.fr-FR-HenriNeural",
  125. "locale": "fr-FR",
  126. "gender": "male"
  127. },
  128. {
  129. "id": "pt-PT-DuarteNeural",
  130. "label_key": "tts.voice.pt-PT-DuarteNeural",
  131. "locale": "pt-PT",
  132. "gender": "male"
  133. },
  134. {
  135. "id": "pt-PT-RaquelNeural",
  136. "label_key": "tts.voice.pt-PT-RaquelNeural",
  137. "locale": "pt-PT",
  138. "gender": "female"
  139. },
  140. {
  141. "id": "de-DE-AmalaNeural",
  142. "label_key": "tts.voice.de-DE-AmalaNeural",
  143. "locale": "de-DE",
  144. "gender": "female"
  145. },
  146. {
  147. "id": "de-DE-ConradNeural",
  148. "label_key": "tts.voice.de-DE-ConradNeural",
  149. "locale": "de-DE",
  150. "gender": "male"
  151. },
  152. # English voices
  153. {
  154. "id": "ru-RU-DmitryNeural",
  155. "label_key": "tts.voice.ru-RU-DmitryNeural",
  156. "locale": "ru-RU",
  157. "gender": "male"
  158. },
  159. {
  160. "id": "ru-RU-SvetlanaNeural",
  161. "label_key": "tts.voice.ru-RU-SvetlanaNeural",
  162. "locale": "ru-RU",
  163. "gender": "female"
  164. },
  165. {
  166. "id": "tr-TR-AhmetNeural",
  167. "label_key": "tts.voice.tr-TR-AhmetNeural",
  168. "locale": "tr-TR",
  169. "gender": "male"
  170. },
  171. {
  172. "id": "tr-TR-EmelNeural",
  173. "label_key": "tts.voice.tr-TR-EmelNeural",
  174. "locale": "tr-TR",
  175. "gender": "female"
  176. },
  177. {
  178. "id": "es-ES-AlvaroNeural",
  179. "label_key": "tts.voice.es-ES-AlvaroNeural",
  180. "locale": "es-ES",
  181. "gender": "male"
  182. },
  183. {
  184. "id": "es-ES-ElviraNeural",
  185. "label_key": "tts.voice.es-ES-ElviraNeural",
  186. "locale": "es-ES",
  187. "gender": "female"
  188. },
  189. ]
  190. def get_voice_display_name(voice_id: str, tr_func=None, locale: str = "zh_CN") -> str:
  191. """
  192. Get display name for voice
  193. Args:
  194. voice_id: Voice ID (e.g., "zh-CN-YunjianNeural")
  195. tr_func: Translation function (optional)
  196. locale: Current locale (default: "zh_CN")
  197. Returns:
  198. Display name (translated label if in Chinese, otherwise voice ID)
  199. """
  200. # Find voice config
  201. voice_config = next((v for v in EDGE_TTS_VOICES if v["id"] == voice_id), None)
  202. if not voice_config:
  203. return voice_id
  204. # If Chinese locale and translation function available, use translated label
  205. if locale == "zh_CN" and tr_func:
  206. label_key = voice_config["label_key"]
  207. return tr_func(label_key)
  208. # For other locales, return voice ID
  209. return voice_id
  210. def speed_to_rate(speed: float) -> str:
  211. """
  212. Convert speed multiplier to Edge TTS rate parameter
  213. Args:
  214. speed: Speed multiplier (1.0 = normal, 1.2 = 120%)
  215. Returns:
  216. Rate string (e.g., "+20%", "-10%")
  217. Examples:
  218. 1.0 → "+0%"
  219. 1.2 → "+20%"
  220. 0.8 → "-20%"
  221. """
  222. percentage = int((speed - 1.0) * 100)
  223. sign = "+" if percentage >= 0 else ""
  224. return f"{sign}{percentage}%"