diff --git a/include/engine/models/kokoro_tts/frontend.h b/include/engine/models/kokoro_tts/frontend.h index 7b66789bb..f1ed03533 100644 --- a/include/engine/models/kokoro_tts/frontend.h +++ b/include/engine/models/kokoro_tts/frontend.h @@ -31,14 +31,21 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state( const std::optional & voice, const KokoroAssets & assets); +// `phoneme_override`, when non-empty, is synthesized as-is INSTEAD of running the built-in +// G2P over `text`. It lets a caller with its own grapheme-to-phoneme stage — a lexicon the +// engine does not carry, a language it does not cover, a pronunciation the application has +// already shown its user — drive the model directly. `text` is still required and its +// language must still agree with the voice; only the phonemization is replaced. KokoroSynthesisInput build_kokoro_synthesis_input( const runtime::Transcript & text, const KokoroFrontendSessionState & state, - const KokoroAssets & assets); + const KokoroAssets & assets, + const std::string & phoneme_override = std::string()); int64_t estimate_kokoro_request_tokens( const runtime::SessionPreparationRequest & request, const KokoroFrontendSessionState & state, - const KokoroAssets & assets); + const KokoroAssets & assets, + const std::string & phoneme_override = std::string()); } // namespace engine::models::kokoro_tts diff --git a/model_specs/kokoro_tts.json b/model_specs/kokoro_tts.json index a4b4faff3..4443129c8 100644 --- a/model_specs/kokoro_tts.json +++ b/model_specs/kokoro_tts.json @@ -29,6 +29,12 @@ "required": false, "min": 0 }, + { + "name": "phonemes", + "type": "string", + "description": "Kokoro-vocabulary phoneme stream to synthesize directly, bypassing the built-in eSpeak-ng G2P. Text is still required and its language must still match the voice. Not chunked: supply at most 510 phonemes per request.", + "required": false + }, { "name": "text_chunk_size", "type": "int", diff --git a/src/models/kokoro_tts/frontend.cpp b/src/models/kokoro_tts/frontend.cpp index 98f183e62..a4ef9ffeb 100644 --- a/src/models/kokoro_tts/frontend.cpp +++ b/src/models/kokoro_tts/frontend.cpp @@ -2,6 +2,8 @@ #include "engine/models/kokoro_tts/g2p_multilingual.h" +#include "engine/framework/debug/trace.h" + #include #include #include @@ -144,9 +146,23 @@ EncodedInputIds encode_input_ids_and_count( throw std::runtime_error("invalid UTF-8 continuation byte in Kokoro phoneme string"); } } - const auto it = assets.vocab.find(phonemes.substr(i, width)); + const std::string symbol = phonemes.substr(i, width); + const auto it = assets.vocab.find(symbol); if (it == assets.vocab.end()) { - throw std::runtime_error("Kokoro vocab is missing phoneme symbol: " + phonemes.substr(i, width)); + // Skipped, not fatal, matching the reference implementation: hexgrad/Kokoro's KModel + // tokenizes with `filter(None, map(vocab.get, phonemes))`, which drops any phoneme the + // 114-entry vocab has no id for. + // + // This matters because our OWN G2P produces such symbols for ordinary words: eSpeak-ng + // glottalises /t/ before a syllabic nasal, so "button" is `b'V?n` with a U+0329 + // syllabic mark the vocab does not carry. Throwing there loses the whole request; + // dropping the mark gives a correct reading of the word. + // + // Malformed UTF-8 above still throws — that is a real error. An unknown but + // well-formed phoneme is not. + engine::debug::trace_log_scalar("kokoro.skipped_phoneme", std::string_view(symbol)); + i += width; + continue; } encoded.ids.push_back(it->second); ++encoded.phoneme_count; @@ -215,15 +231,27 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state( KokoroSynthesisInput build_kokoro_synthesis_input( const runtime::Transcript & text, const KokoroFrontendSessionState & state, - const KokoroAssets & assets) { + const KokoroAssets & assets, + const std::string & phoneme_override) { if (state.voice_pack == nullptr) { throw std::runtime_error("Kokoro frontend session voice pack was not prepared"); } - const std::string phonemes = phonemize_text(text, state.language_code, assets); + const bool supplied = !phoneme_override.empty(); + const std::string phonemes = + supplied ? phoneme_override : phonemize_text(text, state.language_code, assets); const EncodedInputIds encoded = encode_input_ids_and_count(phonemes, assets); if (encoded.phoneme_count > 510) { + // Two different failures wearing one message helps nobody: the caller who supplied the + // phonemes can fix this by sending less, and is told so; the caller who supplied text + // is hitting an engine limitation and is told that instead. throw std::runtime_error( - "Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet"); + supplied + ? "Kokoro phoneme string exceeds 510 symbols; supplied phonemes are not chunked, " + "so split them across requests" + : "Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet"); + } + if (supplied) { + engine::debug::trace_log_scalar("kokoro.supplied_phoneme_count", static_cast(encoded.phoneme_count)); } KokoroSynthesisInput input; input.voice_id = state.voice_id; @@ -241,11 +269,12 @@ KokoroSynthesisInput build_kokoro_synthesis_input( int64_t estimate_kokoro_request_tokens( const runtime::SessionPreparationRequest & request, const KokoroFrontendSessionState & state, - const KokoroAssets & assets) { + const KokoroAssets & assets, + const std::string & phoneme_override) { if (!request.text.has_value()) { return 0; } - const auto input = build_kokoro_synthesis_input(*request.text, state, assets); + const auto input = build_kokoro_synthesis_input(*request.text, state, assets, phoneme_override); return static_cast(input.input_ids.size()); } diff --git a/src/models/kokoro_tts/session.cpp b/src/models/kokoro_tts/session.cpp index 0547d50c7..63e9bb39f 100644 --- a/src/models/kokoro_tts/session.cpp +++ b/src/models/kokoro_tts/session.cpp @@ -248,10 +248,16 @@ void KokoroTTSSession::prepare(const runtime::SessionPreparationRequest & reques } auto adapter = make_graph_capacity_adapter(); int64_t request_size = 0; + const std::string prepare_phonemes = + runtime::find_option(request.options, {"phonemes"}).value_or(std::string()); if (request.text.has_value()) { const int64_t text_chunk_size = engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize); - const auto text_chunks = engine::text::split_text_chunks(request.text->text, text_chunk_size); + // Supplied phonemes are one stream for the whole request, so chunking the TEXT would + // size the graph for a fraction of what run() then feeds it in a single pass. + const auto text_chunks = prepare_phonemes.empty() + ? engine::text::split_text_chunks(request.text->text, text_chunk_size) + : std::vector{request.text->text}; for (const auto & chunk : text_chunks) { runtime::SessionPreparationRequest chunk_request = request; chunk_request.text = runtime::Transcript{chunk, request.text->language}; @@ -259,7 +265,7 @@ void KokoroTTSSession::prepare(const runtime::SessionPreparationRequest & reques resolve_kokoro_frontend_session_state(chunk_request.text, chunk_request.voice, *assets_); request_size = std::max( request_size, - estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_)); + estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_, prepare_phonemes)); } } graph_capacity_controller_.ensure_prepared(adapter, request_size); @@ -275,7 +281,15 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request) const int64_t text_chunk_size = engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize); - const auto chunk_requests = runtime::chunk_text_request(request, text_chunk_size); + // ⚠ A SUPPLIED PHONEME STREAM IS NOT CHUNKED. Chunking splits the TEXT, and there is no + // general way to cut a phoneme stream at the matching points — the caller's G2P is the only + // thing that knows where they are. So the request runs whole, and the 510-symbol guard in + // the frontend tells a caller who sent too much to split it themselves. + const std::string supplied_phonemes = + runtime::find_option(request.options, {"phonemes"}).value_or(std::string()); + const auto chunk_requests = supplied_phonemes.empty() + ? runtime::chunk_text_request(request, text_chunk_size) + : std::vector{request}; engine::debug::trace_log_scalar("kokoro.text_chunk_size", text_chunk_size); engine::debug::trace_log_scalar("kokoro.text_chunk_count", static_cast(chunk_requests.size())); double frontend_ms = 0.0; @@ -291,14 +305,18 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request) frontend_state.language_code + ":" + std::to_string(frontend_state.speaking_rate) + ":" + std::to_string(chunk_request.text_input->text.size()) + ":" + - chunk_request.text_input->text; + chunk_request.text_input->text + ":" + + // Without this, two requests with the same text and different supplied phonemes + // would hit the same cache entry and the second would be spoken as the first. + supplied_phonemes; KokoroSynthesisInput input; frontend_ms += measure_ms([&]() { if (!cache_key.empty() && cached_input_ && cache_key == cached_request_key_) { input = *cached_input_; return; } - input = build_kokoro_synthesis_input(*chunk_request.text_input, frontend_state, *assets_); + input = build_kokoro_synthesis_input( + *chunk_request.text_input, frontend_state, *assets_, supplied_phonemes); if (!cache_key.empty()) { cached_request_key_ = cache_key; cached_input_ = std::make_unique(input);