From 1e2e8e4a01960913b2c58d3bc568cccd22dbc88f Mon Sep 17 00:00:00 2001 From: Chris Thompson Date: Tue, 15 Sep 2026 14:35:26 -0600 Subject: [PATCH 1/2] kokoro_tts: skip phonemes the vocab has no id for, as KModel does eSpeak-ng glottalises /t/ before a syllabic nasal, so the built-in G2P produces a U+0329 syllabic mark for ordinary words -- "button", "kitten", "written", "forgotten" -- and encode_input_ids_and_count then threw on the very symbol it had just produced, losing the whole request. The reference implementation does not: hexgrad/Kokoro's KModel tokenizes with filter(None, map(vocab.get, phonemes)), dropping any phoneme the 114-entry vocab has no id for. Dropping the mark gives b'V?n for "button", a correct reading; throwing gives no audio at all. Malformed UTF-8 above still throws. An unknown but well-formed phoneme does not. --- src/models/kokoro_tts/frontend.cpp | 20 ++++++++++++++++++-- 1 file changed, 18 insertions(+), 2 deletions(-) diff --git a/src/models/kokoro_tts/frontend.cpp b/src/models/kokoro_tts/frontend.cpp index 98f183e62..6bdfa82ac 100644 --- a/src/models/kokoro_tts/frontend.cpp +++ b/src/models/kokoro_tts/frontend.cpp @@ -2,6 +2,8 @@ #include "engine/models/kokoro_tts/g2p_multilingual.h" +#include "engine/framework/debug/trace.h" + #include #include #include @@ -144,9 +146,23 @@ EncodedInputIds encode_input_ids_and_count( throw std::runtime_error("invalid UTF-8 continuation byte in Kokoro phoneme string"); } } - const auto it = assets.vocab.find(phonemes.substr(i, width)); + const std::string symbol = phonemes.substr(i, width); + const auto it = assets.vocab.find(symbol); if (it == assets.vocab.end()) { - throw std::runtime_error("Kokoro vocab is missing phoneme symbol: " + phonemes.substr(i, width)); + // Skipped, not fatal, matching the reference implementation: hexgrad/Kokoro's KModel + // tokenizes with `filter(None, map(vocab.get, phonemes))`, which drops any phoneme the + // 114-entry vocab has no id for. + // + // This matters because our OWN G2P produces such symbols for ordinary words: eSpeak-ng + // glottalises /t/ before a syllabic nasal, so "button" is `b'V?n` with a U+0329 + // syllabic mark the vocab does not carry. Throwing there loses the whole request; + // dropping the mark gives a correct reading of the word. + // + // Malformed UTF-8 above still throws — that is a real error. An unknown but + // well-formed phoneme is not. + engine::debug::trace_log_scalar("kokoro.skipped_phoneme", std::string_view(symbol)); + i += width; + continue; } encoded.ids.push_back(it->second); ++encoded.phoneme_count; From 66dceed7aa51fee2e1b31ccda0b119a550d9b41f Mon Sep 17 00:00:00 2001 From: Chris Thompson Date: Tue, 15 Sep 2026 14:39:06 -0600 Subject: [PATCH 2/2] kokoro_tts: accept a supplied phoneme stream, bypassing the built-in G2P MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds a `phonemes` request option. When set, the string is synthesized as-is instead of running eSpeak-ng over the text. The built-in G2P is one opinion about pronunciation, and a caller may have a better one for its material: a lexicon the engine does not carry, a language it does not cover, a domain vocabulary, or a pronunciation the application has already displayed to its user and must now speak the same way. Today there is no way to express any of that -- phonemize_text() is unconditional, and the option validator rejects anything undeclared, so the door is bolted rather than merely shut. Text is still required and its language must still agree with the voice; only the phonemization is replaced. Three details worth review: - Not chunked. Chunking splits the TEXT, and nothing here knows where the matching cut points in a caller's phoneme stream are -- only their G2P does. So a supplied stream runs whole and the existing 510-symbol guard tells the caller to split it, with a message that distinguishes the two cases rather than reporting a framework limitation to someone who can simply send less. - The run cache is keyed on the supplied stream. Without that, two requests with the same text and different phonemes hit the same entry and the second is spoken as the first. - prepare() sizes the graph on the unchunked request when phonemes are supplied, since that is what run() will feed it in a single pass. ⚠ The declared option set is embedded in the GGUF at conversion time, so existing model files need --model-spec-override (or re-conversion) before they will accept it. --- include/engine/models/kokoro_tts/frontend.h | 11 ++++++-- model_specs/kokoro_tts.json | 6 +++++ src/models/kokoro_tts/frontend.cpp | 23 +++++++++++++---- src/models/kokoro_tts/session.cpp | 28 +++++++++++++++++---- 4 files changed, 56 insertions(+), 12 deletions(-) diff --git a/include/engine/models/kokoro_tts/frontend.h b/include/engine/models/kokoro_tts/frontend.h index 7b66789bb..f1ed03533 100644 --- a/include/engine/models/kokoro_tts/frontend.h +++ b/include/engine/models/kokoro_tts/frontend.h @@ -31,14 +31,21 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state( const std::optional & voice, const KokoroAssets & assets); +// `phoneme_override`, when non-empty, is synthesized as-is INSTEAD of running the built-in +// G2P over `text`. It lets a caller with its own grapheme-to-phoneme stage — a lexicon the +// engine does not carry, a language it does not cover, a pronunciation the application has +// already shown its user — drive the model directly. `text` is still required and its +// language must still agree with the voice; only the phonemization is replaced. KokoroSynthesisInput build_kokoro_synthesis_input( const runtime::Transcript & text, const KokoroFrontendSessionState & state, - const KokoroAssets & assets); + const KokoroAssets & assets, + const std::string & phoneme_override = std::string()); int64_t estimate_kokoro_request_tokens( const runtime::SessionPreparationRequest & request, const KokoroFrontendSessionState & state, - const KokoroAssets & assets); + const KokoroAssets & assets, + const std::string & phoneme_override = std::string()); } // namespace engine::models::kokoro_tts diff --git a/model_specs/kokoro_tts.json b/model_specs/kokoro_tts.json index a4b4faff3..4443129c8 100644 --- a/model_specs/kokoro_tts.json +++ b/model_specs/kokoro_tts.json @@ -29,6 +29,12 @@ "required": false, "min": 0 }, + { + "name": "phonemes", + "type": "string", + "description": "Kokoro-vocabulary phoneme stream to synthesize directly, bypassing the built-in eSpeak-ng G2P. Text is still required and its language must still match the voice. Not chunked: supply at most 510 phonemes per request.", + "required": false + }, { "name": "text_chunk_size", "type": "int", diff --git a/src/models/kokoro_tts/frontend.cpp b/src/models/kokoro_tts/frontend.cpp index 6bdfa82ac..a4ef9ffeb 100644 --- a/src/models/kokoro_tts/frontend.cpp +++ b/src/models/kokoro_tts/frontend.cpp @@ -231,15 +231,27 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state( KokoroSynthesisInput build_kokoro_synthesis_input( const runtime::Transcript & text, const KokoroFrontendSessionState & state, - const KokoroAssets & assets) { + const KokoroAssets & assets, + const std::string & phoneme_override) { if (state.voice_pack == nullptr) { throw std::runtime_error("Kokoro frontend session voice pack was not prepared"); } - const std::string phonemes = phonemize_text(text, state.language_code, assets); + const bool supplied = !phoneme_override.empty(); + const std::string phonemes = + supplied ? phoneme_override : phonemize_text(text, state.language_code, assets); const EncodedInputIds encoded = encode_input_ids_and_count(phonemes, assets); if (encoded.phoneme_count > 510) { + // Two different failures wearing one message helps nobody: the caller who supplied the + // phonemes can fix this by sending less, and is told so; the caller who supplied text + // is hitting an engine limitation and is told that instead. throw std::runtime_error( - "Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet"); + supplied + ? "Kokoro phoneme string exceeds 510 symbols; supplied phonemes are not chunked, " + "so split them across requests" + : "Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet"); + } + if (supplied) { + engine::debug::trace_log_scalar("kokoro.supplied_phoneme_count", static_cast(encoded.phoneme_count)); } KokoroSynthesisInput input; input.voice_id = state.voice_id; @@ -257,11 +269,12 @@ KokoroSynthesisInput build_kokoro_synthesis_input( int64_t estimate_kokoro_request_tokens( const runtime::SessionPreparationRequest & request, const KokoroFrontendSessionState & state, - const KokoroAssets & assets) { + const KokoroAssets & assets, + const std::string & phoneme_override) { if (!request.text.has_value()) { return 0; } - const auto input = build_kokoro_synthesis_input(*request.text, state, assets); + const auto input = build_kokoro_synthesis_input(*request.text, state, assets, phoneme_override); return static_cast(input.input_ids.size()); } diff --git a/src/models/kokoro_tts/session.cpp b/src/models/kokoro_tts/session.cpp index 0547d50c7..63e9bb39f 100644 --- a/src/models/kokoro_tts/session.cpp +++ b/src/models/kokoro_tts/session.cpp @@ -248,10 +248,16 @@ void KokoroTTSSession::prepare(const runtime::SessionPreparationRequest & reques } auto adapter = make_graph_capacity_adapter(); int64_t request_size = 0; + const std::string prepare_phonemes = + runtime::find_option(request.options, {"phonemes"}).value_or(std::string()); if (request.text.has_value()) { const int64_t text_chunk_size = engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize); - const auto text_chunks = engine::text::split_text_chunks(request.text->text, text_chunk_size); + // Supplied phonemes are one stream for the whole request, so chunking the TEXT would + // size the graph for a fraction of what run() then feeds it in a single pass. + const auto text_chunks = prepare_phonemes.empty() + ? engine::text::split_text_chunks(request.text->text, text_chunk_size) + : std::vector{request.text->text}; for (const auto & chunk : text_chunks) { runtime::SessionPreparationRequest chunk_request = request; chunk_request.text = runtime::Transcript{chunk, request.text->language}; @@ -259,7 +265,7 @@ void KokoroTTSSession::prepare(const runtime::SessionPreparationRequest & reques resolve_kokoro_frontend_session_state(chunk_request.text, chunk_request.voice, *assets_); request_size = std::max( request_size, - estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_)); + estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_, prepare_phonemes)); } } graph_capacity_controller_.ensure_prepared(adapter, request_size); @@ -275,7 +281,15 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request) const int64_t text_chunk_size = engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize); - const auto chunk_requests = runtime::chunk_text_request(request, text_chunk_size); + // ⚠ A SUPPLIED PHONEME STREAM IS NOT CHUNKED. Chunking splits the TEXT, and there is no + // general way to cut a phoneme stream at the matching points — the caller's G2P is the only + // thing that knows where they are. So the request runs whole, and the 510-symbol guard in + // the frontend tells a caller who sent too much to split it themselves. + const std::string supplied_phonemes = + runtime::find_option(request.options, {"phonemes"}).value_or(std::string()); + const auto chunk_requests = supplied_phonemes.empty() + ? runtime::chunk_text_request(request, text_chunk_size) + : std::vector{request}; engine::debug::trace_log_scalar("kokoro.text_chunk_size", text_chunk_size); engine::debug::trace_log_scalar("kokoro.text_chunk_count", static_cast(chunk_requests.size())); double frontend_ms = 0.0; @@ -291,14 +305,18 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request) frontend_state.language_code + ":" + std::to_string(frontend_state.speaking_rate) + ":" + std::to_string(chunk_request.text_input->text.size()) + ":" + - chunk_request.text_input->text; + chunk_request.text_input->text + ":" + + // Without this, two requests with the same text and different supplied phonemes + // would hit the same cache entry and the second would be spoken as the first. + supplied_phonemes; KokoroSynthesisInput input; frontend_ms += measure_ms([&]() { if (!cache_key.empty() && cached_input_ && cache_key == cached_request_key_) { input = *cached_input_; return; } - input = build_kokoro_synthesis_input(*chunk_request.text_input, frontend_state, *assets_); + input = build_kokoro_synthesis_input( + *chunk_request.text_input, frontend_state, *assets_, supplied_phonemes); if (!cache_key.empty()) { cached_request_key_ = cache_key; cached_input_ = std::make_unique(input);