From 27136246295da068968b5a2a5c96fddb83b48e38 Mon Sep 17 00:00:00 2001 From: Chris Thompson Date: Tue, 15 Sep 2026 17:30:47 -0600 Subject: [PATCH] kokoro_tts: complete the upstream eSpeak frontend port MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Upstream misaki has TWO eSpeak arms: EspeakFallback for English, with a rich mapping table, and EspeakG2P for everything else, with only the tie-bar digraphs. This port had the second one and used it for both, so English was phonemized with the generic table. Measured against upstream over 60 sentences of ordinary English, that agreed on NONE of them: this file emitted ː in 60 sentences, ɚ in 50, ɐ in 39, ɾ in 39 upstream emits none of those, and ᵊ in 31 Every one of those symbols IS in Kokoro's vocabulary, so nothing ever failed and no test ever went red. The model was simply handed tokens it had not been trained on, on every English sentence. With the English arm restored the two agree on 60/60. Two normalisations were missing outright, and each was losing whole words: · A SYLLABIC CONSONANT has no token. eSpeak writes "button" as bˈʌʔn̩, and U+0329 is not in the vocabulary, so the request threw. Upstream rewrites it as schwa + consonant and then maps the glottal stop to the /t/ it stands for, giving bˈʌtn. That is ordinary English, not exotica -- kitten, written, forgotten, and every -tten/-tton word behaves the same way. · GUILLEMETS are not in the vocabulary either, but the curly quotes upstream turns them into are. Spanish and French prose carries « » as ordinary quotation marks; passing them through as punctuation put an untokenizable symbol in front of the model for ~12% of Spanish and ~15% of French sentences of real corpus text. Verified with the ORIGINAL throw still in place, so these are fixed at the source rather than masked: button/kitten/written, Rustenburg, and Spanish and French quoted text all synthesize, and Spanish that already worked is unchanged. espeak_text() is split into espeak_raw() plus the two mapping arms, because both tables must see eSpeak's output with its tie characters intact -- ə^l is a unit, and the old function had already collapsed it. espeak_text() itself keeps its signature and behaviour for every existing caller. ⚠ The generic arm stays narrow on purpose. Applying the English table to another language is destructive rather than approximate: r→ɹ flattens the Spanish and Italian trill, x→k flattens the jota, and stripping the nasalisation tilde deletes the French nasal vowels. English can afford those because it has no trill and its ɾ really is an allophone of /t/. --- src/models/kokoro_tts/g2p_multilingual.cpp | 89 +++++++++++++++++++++- 1 file changed, 86 insertions(+), 3 deletions(-) diff --git a/src/models/kokoro_tts/g2p_multilingual.cpp b/src/models/kokoro_tts/g2p_multilingual.cpp index cc624a116..5e04c4e73 100644 --- a/src/models/kokoro_tts/g2p_multilingual.cpp +++ b/src/models/kokoro_tts/g2p_multilingual.cpp @@ -82,7 +82,10 @@ class Library { } }; -std::string espeak_text(const std::string & text, const std::string & language, const std::filesystem::path & root) { +/// Phonemizes through eSpeak and returns its output WITH the tie characters intact, so that a +/// caller can still see `a^ɪ` and `ə^l` as units. The two mapping arms below both need that: +/// upstream applies its tables to eSpeak's raw output, not to an already-collapsed string. +std::string espeak_raw(const std::string & text, const std::string & language, const std::filesystem::path & root) { const auto * library_override = std::getenv("AUDIOCPP_ESPEAK_LIBRARY"); const auto * data_override = std::getenv("AUDIOCPP_ESPEAK_DATA"); const auto library = library_override && *library_override @@ -95,6 +98,14 @@ std::string espeak_text(const std::string & text, const std::string & language, else if (std::filesystem::is_regular_file(root / "espeak-ng-data" / "phontab")) data = root / "espeak-ng-data"; #endif engine::audio::EspeakPhonemizer phonemizer(library, data, {language == "fr-fr" ? "fr" : language}); + // ⚠ GUILLEMETS ARE NOT IN KOKORO'S VOCABULARY, but the curly quotes upstream turns them into + // are. Spanish and French prose carries « » as ordinary quotation marks, and passing them + // through as punctuation — which is what happened — put an untokenizable symbol in front of + // the model. Measured over real corpus text, that was ~12% of Spanish and ~15% of French + // sentences. + std::string source = text; + replace(source, u8"«", u8"“"); + replace(source, u8"»", u8"”"); // Preserve punctuation ourselves: TextToPhonemes consumes clause punctuation. const U punctuation = U";:,.!?¡¿—…\"«»“”()"; std::string out, chunk; @@ -105,7 +116,7 @@ std::string espeak_text(const std::string & text, const std::string & language, out += ps; chunk.clear(); }; - for (char32_t c : decode(text)) { + for (char32_t c : decode(source)) { if (punctuation.find(c) != U::npos) { const bool spaced = !chunk.empty() && chunk.back() == ' '; flush(); if (spaced && !out.empty() && out.back() != ' ') out += ' '; @@ -114,7 +125,17 @@ std::string espeak_text(const std::string & text, const std::string & language, else chunk += encode(U(1, c)); } flush(); - out = std::regex_replace(out, std::regex("\\([a-z-]+\\)"), ""); + return std::regex_replace(out, std::regex("\\([a-z-]+\\)"), ""); +} + +/// The mapping every language except English gets: the tie-bar digraphs, and nothing else. +/// +/// ⚠ DELIBERATELY NARROW. Upstream keeps a second, much richer table for English alone, and +/// applying it here would be destructive rather than approximate: `r`→`ɹ` flattens the Spanish +/// and Italian trill, `x`→`k` flattens the jota, and stripping the nasalisation tilde deletes +/// the French nasal vowels. English can afford those because it has no trill and its `ɾ` really +/// is an allophone of /t/. +std::string generic_kokoro_mapping(std::string out) { for (const auto & pair : std::vector>{ {u8"a^ɪ", "I"}, {u8"a^ʊ", "W"}, {"d^z", u8"ʣ"}, {u8"d^ʒ", u8"ʤ"}, {u8"e^ɪ", "A"}, {u8"o^ʊ", "O"}, {u8"ə^ʊ", "Q"}, {"s^s", "S"}, @@ -123,6 +144,64 @@ std::string espeak_text(const std::string & text, const std::string & language, return spaces(out); } +/// Rewrites a syllabic consonant as schwa + consonant: `n̩` becomes `ᵊn`. +/// +/// ⚠ Kokoro's vocabulary has no id for the combining mark (U+0329) but does have ᵊ, so this is +/// the difference between a word being spoken and being lost. eSpeak writes "button" as +/// `bˈʌʔn̩` — the mark turns up in ordinary English, not in exotica. +std::string syllabic_to_schwa(std::string value) { + static const std::regex syllabic(u8"(\\S)\u0329"); + value = std::regex_replace(value, syllabic, u8"ᵊ$1"); + replace(value, u8"\u0329", ""); // anything the rule could not pair with a segment + return value; +} + +/// The English arm, which upstream keeps separate from the one above and which this port did not +/// have. +/// +/// ⚠ THE TWO ARMS ARE NOT INTERCHANGEABLE. Measured against upstream over 60 sentences of +/// ordinary English, the generic table agreed on NONE of them: it emitted `ː` in 60 sentences, +/// `ɚ` in 50, `ɐ` in 39 and `ɾ` in 39, where upstream emits none of those and emits `ᵊ` in 31. +/// Every one of those symbols IS in Kokoro's vocabulary, so nothing ever failed — the model was +/// simply handed tokens it had not been trained on, on every English sentence. +std::string english_kokoro_mapping(std::string ps, bool british) { + // Longest key first, as upstream sorts it: a diphthong must be claimed before its bare + // vowel, and the glottal+syllabic pair before syllabic_to_schwa sees it. + static const std::vector> kE2M = { + {u8"ʔˌn\u0329", u8"ʔn"}, {u8"ʔn\u0329", u8"ʔn"}, + {u8"a^ɪ", "I"}, {u8"a^ʊ", "W"}, {u8"d^ʒ", u8"ʤ"}, {u8"e^ɪ", "A"}, + {u8"t^ʃ", u8"ʧ"}, {u8"ɔ^ɪ", "Y"}, {u8"ə^l", u8"ᵊl"}, + {u8"ʲo", "jo"}, {u8"ʲə", u8"jə"}, + {"e", "A"}, {u8"ʲ", ""}, {u8"ɚ", u8"əɹ"}, {"r", u8"ɹ"}, + {"x", "k"}, {u8"ç", "k"}, {u8"ɐ", u8"ə"}, {u8"ɬ", "l"}, {u8"\u0303", ""}, + }; + for (const auto & entry : kE2M) replace(ps, entry.first, entry.second); + + ps = syllabic_to_schwa(std::move(ps)); + + if (british) { + replace(ps, u8"e^ə", u8"ɛː"); + replace(ps, u8"iə", u8"ɪə"); + replace(ps, u8"ə^ʊ", "Q"); + } else { + replace(ps, u8"o^ʊ", "O"); + replace(ps, u8"ɜːɹ", u8"ɜɹ"); + replace(ps, u8"ɜː", u8"ɜɹ"); + replace(ps, u8"ɪə", u8"iə"); + replace(ps, u8"ː", ""); // en-us drops length marks; en-gb keeps them + } + replace(ps, "o", u8"ɔ"); // upstream: eSpeak < 1.52 compatibility + replace(ps, u8"ɾ", "T"); // the flap is its own token, not a tap + replace(ps, u8"ʔ", "t"); // ...and the glottal stop is the /t/ it stands for + replace(ps, "^", ""); + replace(ps, "-", ""); + return spaces(ps); +} + +std::string espeak_text(const std::string & text, const std::string & language, const std::filesystem::path & root) { + return generic_kokoro_mapping(espeak_raw(text, language, root)); +} + std::vector split(const std::string & s, char delim) { std::vector out; std::istringstream in(s); std::string item; @@ -410,6 +489,10 @@ std::string MultilingualG2P::phonemize(const std::string & text, const std::stri {"e", "es"}, {"f", "fr-fr"}, {"h", "hi"}, {"i", "it"}, {"p", "pt-br"}}; auto it = langs.find(language); if (it == langs.end()) throw std::runtime_error("Unsupported Kokoro language: " + language); + // English has its own mapping upstream, and gets it here too. + if (language == "a" || language == "b") { + return english_kokoro_mapping(espeak_raw(text, it->second, impl_->root), language == "b"); + } return espeak_text(text, it->second, impl_->root); } }