Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 9 additions & 2 deletions include/engine/models/kokoro_tts/frontend.h
Original file line number Diff line number Diff line change
Expand Up @@ -31,14 +31,21 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state(
const std::optional<runtime::VoiceCondition> & voice,
const KokoroAssets & assets);

// `phoneme_override`, when non-empty, is synthesized as-is INSTEAD of running the built-in
// G2P over `text`. It lets a caller with its own grapheme-to-phoneme stage — a lexicon the
// engine does not carry, a language it does not cover, a pronunciation the application has
// already shown its user — drive the model directly. `text` is still required and its
// language must still agree with the voice; only the phonemization is replaced.
KokoroSynthesisInput build_kokoro_synthesis_input(
const runtime::Transcript & text,
const KokoroFrontendSessionState & state,
const KokoroAssets & assets);
const KokoroAssets & assets,
const std::string & phoneme_override = std::string());

int64_t estimate_kokoro_request_tokens(
const runtime::SessionPreparationRequest & request,
const KokoroFrontendSessionState & state,
const KokoroAssets & assets);
const KokoroAssets & assets,
const std::string & phoneme_override = std::string());

} // namespace engine::models::kokoro_tts
6 changes: 6 additions & 0 deletions model_specs/kokoro_tts.json
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,12 @@
"required": false,
"min": 0
},
{
"name": "phonemes",
"type": "string",
"description": "Kokoro-vocabulary phoneme stream to synthesize directly, bypassing the built-in eSpeak-ng G2P. Text is still required and its language must still match the voice. Not chunked: supply at most 510 phonemes per request.",
"required": false
},
{
"name": "text_chunk_size",
"type": "int",
Expand Down
43 changes: 36 additions & 7 deletions src/models/kokoro_tts/frontend.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@

#include "engine/models/kokoro_tts/g2p_multilingual.h"

#include "engine/framework/debug/trace.h"

#include <algorithm>
#include <cctype>
#include <cstring>
Expand Down Expand Up @@ -144,9 +146,23 @@ EncodedInputIds encode_input_ids_and_count(
throw std::runtime_error("invalid UTF-8 continuation byte in Kokoro phoneme string");
}
}
const auto it = assets.vocab.find(phonemes.substr(i, width));
const std::string symbol = phonemes.substr(i, width);
const auto it = assets.vocab.find(symbol);
if (it == assets.vocab.end()) {
throw std::runtime_error("Kokoro vocab is missing phoneme symbol: " + phonemes.substr(i, width));
// Skipped, not fatal, matching the reference implementation: hexgrad/Kokoro's KModel
// tokenizes with `filter(None, map(vocab.get, phonemes))`, which drops any phoneme the
// 114-entry vocab has no id for.
//
// This matters because our OWN G2P produces such symbols for ordinary words: eSpeak-ng
// glottalises /t/ before a syllabic nasal, so "button" is `b'V?n` with a U+0329
// syllabic mark the vocab does not carry. Throwing there loses the whole request;
// dropping the mark gives a correct reading of the word.
//
// Malformed UTF-8 above still throws — that is a real error. An unknown but
// well-formed phoneme is not.
engine::debug::trace_log_scalar("kokoro.skipped_phoneme", std::string_view(symbol));
i += width;
continue;
}
encoded.ids.push_back(it->second);
++encoded.phoneme_count;
Expand Down Expand Up @@ -215,15 +231,27 @@ KokoroFrontendSessionState resolve_kokoro_frontend_session_state(
KokoroSynthesisInput build_kokoro_synthesis_input(
const runtime::Transcript & text,
const KokoroFrontendSessionState & state,
const KokoroAssets & assets) {
const KokoroAssets & assets,
const std::string & phoneme_override) {
if (state.voice_pack == nullptr) {
throw std::runtime_error("Kokoro frontend session voice pack was not prepared");
}
const std::string phonemes = phonemize_text(text, state.language_code, assets);
const bool supplied = !phoneme_override.empty();
const std::string phonemes =
supplied ? phoneme_override : phonemize_text(text, state.language_code, assets);
const EncodedInputIds encoded = encode_input_ids_and_count(phonemes, assets);
if (encoded.phoneme_count > 510) {
// Two different failures wearing one message helps nobody: the caller who supplied the
// phonemes can fix this by sending less, and is told so; the caller who supplied text
// is hitting an engine limitation and is told that instead.
throw std::runtime_error(
"Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet");
supplied
? "Kokoro phoneme string exceeds 510 symbols; supplied phonemes are not chunked, "
"so split them across requests"
: "Kokoro phoneme string exceeds 510 symbols; segmenting is not implemented in the framework path yet");
}
if (supplied) {
engine::debug::trace_log_scalar("kokoro.supplied_phoneme_count", static_cast<int64_t>(encoded.phoneme_count));
}
KokoroSynthesisInput input;
input.voice_id = state.voice_id;
Expand All @@ -241,11 +269,12 @@ KokoroSynthesisInput build_kokoro_synthesis_input(
int64_t estimate_kokoro_request_tokens(
const runtime::SessionPreparationRequest & request,
const KokoroFrontendSessionState & state,
const KokoroAssets & assets) {
const KokoroAssets & assets,
const std::string & phoneme_override) {
if (!request.text.has_value()) {
return 0;
}
const auto input = build_kokoro_synthesis_input(*request.text, state, assets);
const auto input = build_kokoro_synthesis_input(*request.text, state, assets, phoneme_override);
return static_cast<int64_t>(input.input_ids.size());
}

Expand Down
28 changes: 23 additions & 5 deletions src/models/kokoro_tts/session.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -248,18 +248,24 @@ void KokoroTTSSession::prepare(const runtime::SessionPreparationRequest & reques
}
auto adapter = make_graph_capacity_adapter();
int64_t request_size = 0;
const std::string prepare_phonemes =
runtime::find_option(request.options, {"phonemes"}).value_or(std::string());
if (request.text.has_value()) {
const int64_t text_chunk_size =
engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize);
const auto text_chunks = engine::text::split_text_chunks(request.text->text, text_chunk_size);
// Supplied phonemes are one stream for the whole request, so chunking the TEXT would
// size the graph for a fraction of what run() then feeds it in a single pass.
const auto text_chunks = prepare_phonemes.empty()
? engine::text::split_text_chunks(request.text->text, text_chunk_size)
: std::vector<std::string>{request.text->text};
for (const auto & chunk : text_chunks) {
runtime::SessionPreparationRequest chunk_request = request;
chunk_request.text = runtime::Transcript{chunk, request.text->language};
const auto frontend_state =
resolve_kokoro_frontend_session_state(chunk_request.text, chunk_request.voice, *assets_);
request_size = std::max(
request_size,
estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_));
estimate_kokoro_request_tokens(chunk_request, frontend_state, *assets_, prepare_phonemes));
}
}
graph_capacity_controller_.ensure_prepared(adapter, request_size);
Expand All @@ -275,7 +281,15 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request)

const int64_t text_chunk_size =
engine::text::parse_text_chunk_size_override(request.options).value_or(kDefaultTextChunkSize);
const auto chunk_requests = runtime::chunk_text_request(request, text_chunk_size);
// ⚠ A SUPPLIED PHONEME STREAM IS NOT CHUNKED. Chunking splits the TEXT, and there is no
// general way to cut a phoneme stream at the matching points — the caller's G2P is the only
// thing that knows where they are. So the request runs whole, and the 510-symbol guard in
// the frontend tells a caller who sent too much to split it themselves.
const std::string supplied_phonemes =
runtime::find_option(request.options, {"phonemes"}).value_or(std::string());
const auto chunk_requests = supplied_phonemes.empty()
? runtime::chunk_text_request(request, text_chunk_size)
: std::vector<runtime::TaskRequest>{request};
engine::debug::trace_log_scalar("kokoro.text_chunk_size", text_chunk_size);
engine::debug::trace_log_scalar("kokoro.text_chunk_count", static_cast<int64_t>(chunk_requests.size()));
double frontend_ms = 0.0;
Expand All @@ -291,14 +305,18 @@ runtime::TaskResult KokoroTTSSession::run(const runtime::TaskRequest & request)
frontend_state.language_code + ":" +
std::to_string(frontend_state.speaking_rate) + ":" +
std::to_string(chunk_request.text_input->text.size()) + ":" +
chunk_request.text_input->text;
chunk_request.text_input->text + ":" +
// Without this, two requests with the same text and different supplied phonemes
// would hit the same cache entry and the second would be spoken as the first.
supplied_phonemes;
KokoroSynthesisInput input;
frontend_ms += measure_ms([&]() {
if (!cache_key.empty() && cached_input_ && cache_key == cached_request_key_) {
input = *cached_input_;
return;
}
input = build_kokoro_synthesis_input(*chunk_request.text_input, frontend_state, *assets_);
input = build_kokoro_synthesis_input(
*chunk_request.text_input, frontend_state, *assets_, supplied_phonemes);
if (!cache_key.empty()) {
cached_request_key_ = cache_key;
cached_input_ = std::make_unique<KokoroSynthesisInput>(input);
Expand Down