-
Notifications
You must be signed in to change notification settings - Fork 2
Expand file tree
/
Copy pathTtsEngine.cs
More file actions
289 lines (265 loc) · 12.8 KB
/
Copy pathTtsEngine.cs
File metadata and controls
289 lines (265 loc) · 12.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
using KokoroSharp;
using KokoroSharp.Core;
using KokoroSharp.Processing;
using KokoroSharp.Utilities;
// ═══════════════════════════════════════════════════════════════════════
// TtsEngine — in-process Kokoro neural TTS for POST /v1/audio/speech
//
// Cross-platform (ONNX runtime): needs kokoro.onnx + voices/ next to
// the executable. The csproj brings voices/ + voices-zh/ from the KokoroSharp
// package content and provides kokoro.onnx (copy from the sibling VoiceAgent
// build output, else curl download) — the endpoint reports itself unavailable
// (501) until the assets are present, so clients only activate TTS when the
// platform actually supports it.
//
// The engine is initialized in two steps so that merely ASKING whether TTS works
// never costs what RUNNING it costs. Loading the model is heavy: measured on this
// repo's own assets, one availability check used to take the process from 125 MB to
// 628 MB of resident memory and add ~8 threads, spending about 0.7 s of CPU on
// every cold call (issue #25) — and the status / capabilities endpoints are polled
// by the TUI and the web clients, so a user who never speaks still paid for the
// whole model. EnsureVoices() answers availability from the assets and the voice
// catalogue alone (the same call now ends at ~226 MB with no ONNX session loaded);
// the session is created by EnsureSynth() only when audio is really requested, and
// is reused after that. A static lock serializes synthesis because the underlying
// synthesizer queues jobs internally.
// ═══════════════════════════════════════════════════════════════════════
/// <summary>In-process Kokoro neural TTS engine backing POST /v1/audio/speech.</summary>
public static class TtsEngine
{
private static readonly object Sync = new();
private static KokoroWavSynthesizer? _synth;
private static List<string> _voices = new();
private static string? _unavailableReason;
private static bool _voicesChecked;
/// <summary>True when TTS can run on this machine (model + voices present).
/// Cheap: it checks the assets and the voice catalogue WITHOUT loading the model,
/// so the polled status/capabilities endpoints never pay for the ONNX session.
/// </summary>
public static bool IsAvailable
{
get { EnsureVoices(); return _unavailableReason == null; }
}
/// <summary>Human-readable reason when <see cref="IsAvailable"/> is false.</summary>
public static string UnavailableReason
{
get { EnsureVoices(); return _unavailableReason ?? "TTS not initialized"; }
}
/// <summary>Kokoro voice ids currently loaded from voices/, sorted. Cheap — see
/// <see cref="IsAvailable"/> for why the catalogue is read without the model.</summary>
public static IReadOnlyList<string> Voices
{
get { EnsureVoices(); return _voices; }
}
/// <summary>
/// Synthesizes text to WAV bytes with the requested voice. The voice name accepts an
/// OpenAI voice name ("alloy", "echo", ...) or a raw Kokoro voice id ("if_sara",
/// "af_heart", ...); unknown names fall back to the first loaded voice.
/// The optional <paramref name="lang"/> (two-letter ISO code) picks a voice of that
/// language: Kokoro voices are per-language (af_*/am_* = English, if_*/im_* = Italian,
/// ef_* = Spanish, ff_* = French, jf_* = Japanese, ...). When no voice is given, the
/// machine's language (<see cref="SystemLang.Get"/>) selects the default — every machine
/// speaks its own language, no hardcoded default.
/// </summary>
public static byte[] Synthesize(string text, string? voice, double? speed, string? lang = null)
{
// The heavy step happens here, on a request that actually wants audio — never
// while a client is only asking whether TTS exists (see the class header).
var synth = EnsureSynth();
// The single TTS normalization (canonical apostrophes — the typographic ’ breaks the
// Italian elision — plus markdown/emoji removal, punctuation spacing and the English
// "AI" pronunciation for Italian), shared with every TTS consumer.
text = AIOrchestrator.VoiceConversation.NormalizeForTts(text, lang);
var voiceId = ResolveVoiceId(voice, lang);
var kokoroVoice = _voices.Contains(voiceId, StringComparer.OrdinalIgnoreCase)
? KokoroVoiceManager.GetVoice(voiceId)
: null;
if (kokoroVoice == null && _voices.Count > 0)
kokoroVoice = KokoroVoiceManager.GetVoice(_voices[0]);
if (kokoroVoice == null)
throw new InvalidOperationException("No Kokoro voice available.");
var config = new KokoroTTSPipelineConfig(
new DefaultSegmentationConfig { MaxFirstSegmentLength = 510 })
{
Speed = (float)Math.Clamp(speed ?? 1.0, 0.25, 4.0)
};
byte[] pcm;
lock (Sync)
{
pcm = synth.Synthesize(text, kokoroVoice, config);
}
return WrapWav(pcm);
}
// KokoroSharp's KokoroWavSynthesizer returns RAW 16-bit PCM (no container); the WAV
// header below (RIFF/fmt/data chunks) makes the response a playable audio/wav file.
private static byte[] WrapWav(byte[] pcm)
{
var fmt = KokoroPlayback.waveFormat;
var sampleRate = fmt.SampleRate;
var channels = fmt.Channels;
var bits = fmt.BitsPerSample;
var blockAlign = (short)(channels * bits / 8);
var byteRate = sampleRate * blockAlign;
using var ms = new MemoryStream(44 + pcm.Length);
using var w = new BinaryWriter(ms);
w.Write("RIFF"u8);
w.Write(36 + pcm.Length);
w.Write("WAVE"u8);
w.Write("fmt "u8);
w.Write(16);
w.Write((short)1); // PCM
w.Write((short)channels);
w.Write(sampleRate);
w.Write(byteRate);
w.Write(blockAlign);
w.Write((short)bits);
w.Write("data"u8);
w.Write(pcm.Length);
w.Write(pcm);
w.Flush();
return ms.ToArray();
}
// INVARIANT (project rule): voice selection follows the MACHINE'S LANGUAGE
// (SystemLang.Get() / lang parameter) — the code contains no per-language
// setting. The map below is only the "ISO language → Kokoro voice prefix of
// that language" translation; if the machine's language is not in the map a
// neutral default is used, without assuming the user.
// OpenAI voice names → Kokoro voice ids (best-effort; the mapping only picks a
// pleasant default, the full Kokoro catalogue stays reachable by raw id).
private static readonly Dictionary<string, string> OpenAiToKokoro = new(StringComparer.OrdinalIgnoreCase)
{
["alloy"] = "af_heart",
["echo"] = "am_michael",
["fable"] = "bf_alice",
["onyx"] = "am_onyx",
["nova"] = "af_bella",
["shimmer"] = "ef_dora",
["coral"] = "bf_emma",
["sage"] = "bm_george",
["ash"] = "pm_alex",
["ballad"] = "ff_siwis",
["verse"] = "jf_alpha",
};
// Two-letter ISO language → Kokoro voice prefix (the part before the underscore).
private static readonly Dictionary<string, string> LangPrefix = new(StringComparer.OrdinalIgnoreCase)
{
["it"] = "if_", // Italian
["en"] = "af_", // American English (af_/am_ both exist)
["es"] = "ef_", // Spanish
["fr"] = "ff_", // French
["ja"] = "jf_", // Japanese
["zh"] = "cm_", // Mandarin
["ko"] = "kf_", // Korean
["ar"] = "am_", // Arabic (fallback to a male English id is wrong, but Kokoro's
// Arabic voices use am_/af_ prefixes in some builds — keep simple)
};
private static string ResolveVoiceId(string? voice, string? lang)
{
// Raw Kokoro voice ("if_sara") → use it as-is.
if (!string.IsNullOrWhiteSpace(voice) && !OpenAiToKokoro.ContainsKey(voice))
return voice;
var id = string.IsNullOrWhiteSpace(voice) ? "" : OpenAiToKokoro[voice];
var targetLang = string.IsNullOrWhiteSpace(lang) ? SystemLang.Get() : lang!;
// No voice requested: the language (explicit or system) picks the default.
if (string.IsNullOrWhiteSpace(voice))
{
var prefix = LangPrefix.TryGetValue(targetLang, out var p) ? p : null;
if (prefix != null)
{
var match = _voices.FirstOrDefault(v => v.StartsWith(prefix, StringComparison.OrdinalIgnoreCase));
if (match != null) return match;
}
return id.Length > 0 ? id : "af_heart";
}
// Named OpenAI voice but of a different language than requested → prefer the
// language (e.g. "alloy" + lang "it" → an if_* voice).
var mappedPrefix = LangPrefix.TryGetValue(targetLang, out var mp) ? mp : null;
if (mappedPrefix != null && !id.StartsWith(mappedPrefix, StringComparison.OrdinalIgnoreCase))
{
var match = _voices.FirstOrDefault(v => v.StartsWith(mappedPrefix, StringComparison.OrdinalIgnoreCase));
if (match != null) return match;
}
return id;
}
/// <summary>Resolves the model and voice paths the engine uses (the app base
/// directory, where the release archive puts them).</summary>
private static (string ModelPath, string VoicesDir) AssetPaths()
{
var baseDir = AppDomain.CurrentDomain.BaseDirectory;
return (Path.Combine(baseDir, "kokoro.onnx"), Path.Combine(baseDir, "voices"));
}
/// <summary>
/// Answers whether the TTS assets are usable, WITHOUT loading the model: checks
/// kokoro.onnx and voices/ are there and reads the voice catalogue. This is what
/// <see cref="IsAvailable"/>, <see cref="Voices"/> and the status/capabilities
/// endpoints use, so polling them costs a directory read instead of the whole
/// ONNX session. The result is cached: the assets only change with an update, and
/// an update restarts the process.
/// </summary>
private static void EnsureVoices()
{
if (_voicesChecked || _unavailableReason != null) return;
lock (Sync)
{
if (_voicesChecked || _unavailableReason != null) return;
var (modelPath, voicesDir) = AssetPaths();
if (!File.Exists(modelPath))
{
_unavailableReason = $"kokoro.onnx not found at '{modelPath}'. Build the server (the DownloadKokoroModel target copies or downloads it) to enable TTS.";
return;
}
if (!Directory.Exists(voicesDir))
{
_unavailableReason = $"voices/ directory not found at '{voicesDir}'. TTS unavailable.";
return;
}
try
{
KokoroVoiceManager.LoadVoicesFromPath(voicesDir);
_voices = KokoroVoiceManager.Voices
.Select(v => v.Name)
.OrderBy(n => n, StringComparer.OrdinalIgnoreCase)
.ToList();
if (_voices.Count == 0)
{
_unavailableReason = "voices/ contains no Kokoro voices.";
return;
}
_voicesChecked = true;
}
catch (Exception ex)
{
_unavailableReason = $"TTS initialization failed: {ex.Message}";
}
}
}
/// <summary>
/// Creates the Kokoro synthesizer: loads the ONNX model into memory and starts the
/// runtime thread pool. This is the expensive step (hundreds of MB, dozens of
/// threads, seconds) and must be reached only from a request that really wants
/// audio — see the class header for why it is not part of availability.
/// </summary>
private static KokoroWavSynthesizer EnsureSynth()
{
var existing = _synth;
if (existing != null) return existing;
lock (Sync)
{
if (_synth != null) return _synth;
EnsureVoices();
if (_unavailableReason != null)
throw new InvalidOperationException(_unavailableReason);
var (modelPath, _) = AssetPaths();
try
{
_synth = new KokoroWavSynthesizer(modelPath);
return _synth;
}
catch (Exception ex)
{
_unavailableReason = $"TTS initialization failed: {ex.Message}";
throw new InvalidOperationException(_unavailableReason, ex);
}
}
}
}