diff --git a/README.md b/README.md index d76b214..bf8aab8 100644 --- a/README.md +++ b/README.md @@ -62,7 +62,7 @@ Agents and external tools should inspect a model declaration before constructing import { createGenerationClient } from "@neta-art/generation"; const discoveryClient = createGenerationClient(); -const declaration = discoveryClient.getModel("qwen-tts"); +const declaration = discoveryClient.getModel("cosyvoice-v3.5-plus"); if (!declaration) throw new Error("Model is unavailable"); console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" })); @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI: ```bash neta-generation models list -neta-generation models export qwen-tts --out ./qwen-tts.yaml +neta-generation models export cosyvoice-v3.5-plus --out ./cosyvoice-v3.5-plus.yaml neta-generation models export-all --out ./models ``` @@ -170,7 +170,8 @@ const client = createGenerationClient({ - `gpt-image-2` - `z-image-turbo` - `qwen-image-edit` -- `qwen-tts` +- `cosyvoice-v3.5-plus` +- `cosyvoice-v3.5-flash` - `qwen-audio-3.0-tts-plus` - `qwen-audio-3.0-tts-flash` - `higgs-tts` @@ -268,22 +269,38 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au | Requirement | Model choice | | --- | --- | -| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default | +| Create a voice from a text-only description, without reference audio | Use an explicitly requested cosyvoice variant; otherwise use `cosyvoice-v3.5-flash` as the deterministic default | | Maximize fidelity to one reference voice | `higgs-tts` | | Blend 2-16 weighted reference voices | `higgs-tts` | | Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` | -- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points. +- CosyVoice / Qwen-Audio-TTS: `voice_prompt` design OR one-reference clone; `cosyvoice-v3.5-plus`, `cosyvoice-v3.5-flash`, `qwen-audio-3.0-tts-plus`, and `qwen-audio-3.0-tts-flash` all require at least 15 Unicode code points (and at most 200 in design mode); `cosyvoice-v3.5-flash` is the deterministic default when no variant is requested. - Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend. - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. - Dependency: clone prior generated audio. -- Ranking: no declared Qwen quality, latency, or cost order. +- Ranking: no declared CosyVoice / Qwen quality, latency, or cost order. + +> **Migrating from `qwen-tts` (removed in 0.2.0):** DashScope retires `qwen-tts` on 2026-10-10; this SDK removed +> the declaration in the same release. Switch to `cosyvoice-v3.5-flash` (or `-plus`) — same request shape +> (`voice_prompt` design OR one-reference clone), but `qwen-tts` accepted text of **any length** while +> `cosyvoice-v3.5-*` requires **at least 15 Unicode code points** (and at most 200 in design mode, like the rest +> of this model family). A caller doing a plain model-name swap on short input will start seeing +> `GenerationValidationError` where it previously succeeded — pad short inputs or catch the error. + +> **Server-side wire contract:** this SDK talks to the router over HTTP; it does not call the DashScope-facing +> worker (`talesofai/background`'s `qwen_tts_actor`) directly, and as of 2026-09-20 nothing in the pipeline +> between them performs the translation yet. For whoever builds that glue, per the worker's own field names +> (see its module docstring — they do **not** match DashScope's or this SDK's field names one-for-one, which is +> the trap to avoid): the worker's `preview_text` (what's actually spoken, required on every request) is this +> request's primary `text` content block; the worker's `text` field (the voice STYLE description, design mode +> only — unrelated to this SDK's own `input`/text-content naming despite the shared word) is this request's +> `meta.voice_prompt`; `target_model` is this request's `model`, unchanged. ```ts await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "欢迎使用语音合成功能。" }], + model: "cosyvoice-v3.5-flash", + content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", }, diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 009b59f..740b64c 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-flash", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/cosyvoice-v3.5-flash.yaml b/models/cosyvoice-v3.5-flash.yaml new file mode 100644 index 0000000..e51de69 --- /dev/null +++ b/models/cosyvoice-v3.5-flash.yaml @@ -0,0 +1,46 @@ +schema: neta.generation.model.v1 +model: cosyvoice-v3.5-flash +title: CosyVoice v3.5 Flash +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." +adapter: + type: openai.audioSpeech +content: + input: + - type: text + required: true + min: 1 + max: 1 + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). + - type: audio + required: false + max: 1 + sources: + - url + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." +meta: + fields: + voice_prompt: + type: string + optional: true + description: "Design: custom voice text; no reference audio." +examples: + - title: Voice design + request: + model: cosyvoice-v3.5-flash + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + meta: + voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 + - title: Voice clone + request: + model: cosyvoice-v3.5-flash + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + - type: audio + source: + type: url + url: https://example.com/reference.mp3 diff --git a/models/cosyvoice-v3.5-plus.yaml b/models/cosyvoice-v3.5-plus.yaml new file mode 100644 index 0000000..158f0f0 --- /dev/null +++ b/models/cosyvoice-v3.5-plus.yaml @@ -0,0 +1,46 @@ +schema: neta.generation.model.v1 +model: cosyvoice-v3.5-plus +title: CosyVoice v3.5 Plus +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." +adapter: + type: openai.audioSpeech +content: + input: + - type: text + required: true + min: 1 + max: 1 + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). + - type: audio + required: false + max: 1 + sources: + - url + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." +meta: + fields: + voice_prompt: + type: string + optional: true + description: "Design: custom voice text; no reference audio." +examples: + - title: Voice design + request: + model: cosyvoice-v3.5-plus + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + meta: + voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 + - title: Voice clone + request: + model: cosyvoice-v3.5-plus + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + - type: audio + source: + type: url + url: https://example.com/reference.mp3 diff --git a/models/qwen-audio-3.0-tts-flash.yaml b/models/qwen-audio-3.0-tts-flash.yaml index 3d111b9..42ed56a 100644 --- a/models/qwen-audio-3.0-tts-flash.yaml +++ b/models/qwen-audio-3.0-tts-flash.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 model: qwen-audio-3.0-tts-flash title: Qwen Audio 3.0 TTS Flash -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -10,19 +11,20 @@ content: required: true min: 1 max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). - type: audio required: false max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: diff --git a/models/qwen-audio-3.0-tts-plus.yaml b/models/qwen-audio-3.0-tts-plus.yaml index d6bf7fa..872ae5a 100644 --- a/models/qwen-audio-3.0-tts-plus.yaml +++ b/models/qwen-audio-3.0-tts-plus.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 model: qwen-audio-3.0-tts-plus title: Qwen Audio 3.0 TTS Plus -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -10,19 +11,20 @@ content: required: true min: 1 max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). - type: audio required: false max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: diff --git a/models/qwen-tts.yaml b/models/qwen-tts.yaml deleted file mode 100644 index 1136fc7..0000000 --- a/models/qwen-tts.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-tts -title: Qwen TTS -description: 'Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/package.json b/package.json index 4895da2..2a7f219 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@neta-art/generation", - "version": "0.1.31", + "version": "0.2.0", "description": "A lightweight multimodal generation SDK with built-in model presets and adapter-based provider calls.", "keywords": [ "ai", diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index b174984..18d5164 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,8 +8,14 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); -const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const VOICE_ENROLLMENT_MODELS = new Set([ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", +]); +const VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS = 200; +const VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS = 500; const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -71,7 +77,11 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (audio.length > 1) { throw new GenerationValidationError(`${input.declaration.model} supports at most one reference audio`); } - // Keep accepting the retired preview_text key for compatibility, but never use or forward it. + // Keep accepting the retired preview_text key for compatibility, but never use or forward it: + // the text spoken in the preview/output clip is this request's primary `text` content block + // (validated below against the same 15-200 code point window the worker enforces on its own + // `preview_text` field downstream), not a separate meta value. `preview_text` was how older + // callers passed that text directly; treat any value here as a no-op alias, not live input. const requestMetaKeys = new Set(["voice_prompt", "preview_text"]); validateMetaKeys("request.metadata", input.request.metadata, requestMetaKeys); validateMetaKeys("request.meta", input.request.meta, requestMetaKeys); @@ -83,6 +93,19 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (voicePrompt !== undefined && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} meta.voice_prompt must be a non-empty string`); } + // Upper-bound checks below (here and on `text`) count the RAW string's code + // points, not the trimmed one: buildPayload sends the untrimmed original + // string (deliberate — see the wire-contract preservation tests), and the + // worker counts with plain len() on what it receives, so validating here + // against anything shorter than what's actually sent could let through a + // value the worker then rejects. Lower-bound/non-empty checks stay on the + // trimmed string — trimmed-length >= a minimum only strengthens the + // guarantee, since trimming can only shorten a string. + if (hasVoicePrompt && Array.from(voicePrompt).length > VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS) { + throw new GenerationValidationError( + `${input.declaration.model} meta.voice_prompt must be at most ${VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS} Unicode code points`, + ); + } if (audio.length === 0 && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} requires one reference audio or meta.voice_prompt`); } @@ -92,9 +115,19 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { + const codePoints = Array.from(text.text.trim()).length; + if (codePoints < 15) { throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); } + // Voice design speaks exactly this text as the preview clip, so the model's + // 15-200 character preview window applies; cloning only feeds the separate + // synthesis call, where long-form text is legitimate. Upper bound counts the + // raw (untrimmed) string sent by buildPayload — see the comment above. + if (hasVoicePrompt && Array.from(text.text).length > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { + throw new GenerationValidationError( + `${input.declaration.model} voice design requires input of at most ${VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS} Unicode code points`, + ); + } } function hasOwnWeight(block: AudioBlock): boolean { @@ -124,7 +157,7 @@ function validateHiggs(input: ResolvedGenerationRequest, text: TextBlock, audio: function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void { const { text, audio } = validateCommonContent(input); - if (QWEN_MODELS.has(input.declaration.model)) { + if (VOICE_ENROLLMENT_MODELS.has(input.declaration.model)) { validateQwen(input, text, audio); return; } @@ -148,7 +181,7 @@ function buildPayload(input: ResolvedGenerationRequest): Record input: text.text, }; - if (QWEN_MODELS.has(input.declaration.model)) { + if (VOICE_ENROLLMENT_MODELS.has(input.declaration.model)) { if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim(); else payload.metadata = { voice_prompt: input.meta.voice_prompt }; return payload; diff --git a/src/builtins.ts b/src/builtins.ts index 8ba25a5..2632627 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -708,15 +708,7 @@ function geminiImageModel( }; } -function qwenTtsModel( - model: string, - title: string, - description: string, - options: { minimumTextCodePoints?: number } = {}, -): GenerationModelDeclaration { - const text = options.minimumTextCodePoints - ? "这是一段长度足够并且表达清晰自然的语音合成测试文本。" - : "这是一次清晰自然的语音合成测试。"; +function voiceEnrollmentModel(model: string, title: string, description: string): GenerationModelDeclaration { return { schema: MODEL_SCHEMA, model, @@ -730,9 +722,8 @@ function qwenTtsModel( required: true, min: 1, max: 1, - description: options.minimumTextCodePoints - ? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.` - : "Exactly one non-empty text block to speak.", + description: + "Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in voice-design mode).", }, { type: "audio", @@ -757,7 +748,7 @@ function qwenTtsModel( title: "Voice design", request: { model, - content: [{ type: "text", text }], + content: [{ type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, @@ -766,7 +757,7 @@ function qwenTtsModel( request: { model, content: [ - { type: "text", text }, + { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, ], }, @@ -776,22 +767,25 @@ function qwenTtsModel( } const audioSpeechModels = [ - qwenTtsModel( - "qwen-tts", - "Qwen TTS", - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + voiceEnrollmentModel( + "cosyvoice-v3.5-plus", + "CosyVoice v3.5 Plus", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + ), + voiceEnrollmentModel( + "cosyvoice-v3.5-flash", + "CosyVoice v3.5 Flash", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), - qwenTtsModel( + voiceEnrollmentModel( "qwen-audio-3.0-tts-plus", "Qwen Audio 3.0 TTS Plus", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), - qwenTtsModel( + voiceEnrollmentModel( "qwen-audio-3.0-tts-flash", "Qwen Audio 3.0 TTS Flash", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), { schema: MODEL_SCHEMA, diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 236718b..e0916d6 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -39,10 +39,10 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC }; } -function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { +function designRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-tts", - content: [{ type: "text", text: "这是需要朗读的文本。" }], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是需要朗读的一段完整试听文本。" }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, }; @@ -65,15 +65,15 @@ function requestBody(call: { init: RequestInit } | undefined): Record { - it("sends the fixed wire contract and preserves Qwen input and voice prompt", async () => { + it("sends the fixed wire contract and preserves voice-enrollment input and voice prompt", async () => { const { client, calls } = recordingClient(() => routerSuccess({}, { headers: { "x-request-id": "request-primary", "x-oneapi-request-id": "request-fallback" } }), ); - const input = " 原样保留的朗读文本。\n"; + const input = " 原样保留的一段完整朗读试听文本。\n"; const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -85,7 +85,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", input, metadata: { voice_prompt: voicePrompt }, }); @@ -98,16 +98,16 @@ describe("openai.audioSpeech adapter requests", () => { ]); }); - it("maps Qwen reference audio and trims only its URL", async () => { + it("maps voice-enrollment reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段足够长的克隆试听文本。" }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "短句", + model: "cosyvoice-v3.5-plus", + input: "这是一段足够长的克隆试听文本。", ref_audio: REFERENCE_URL, }); }); @@ -115,14 +115,14 @@ describe("openai.audioSpeech adapter requests", () => { it("silently ignores deprecated preview_text without sending it", async () => { const { client, calls } = recordingClient(); await client.generate( - qwenDesignRequest({ + designRequest({ meta: { voice_prompt: "清晰女声", preview_text: "旧字段不应发送" }, }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "这是需要朗读的文本。", + model: "cosyvoice-v3.5-plus", + input: "这是需要朗读的一段完整试听文本。", metadata: { voice_prompt: "清晰女声" }, }); }); @@ -194,12 +194,12 @@ describe("openai.audioSpeech adapter validation", () => { expect(fetchMock).not.toHaveBeenCalled(); }); - it("accepts retired preview_text without treating it as a Qwen voice source", () => { + it("accepts retired preview_text without treating it as a voice source", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "有效语音文本" }], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段有效的语音试听文本。" }], meta: { preview_text: "已经失效的文本来源" }, }), ).toThrow("requires one reference audio or meta.voice_prompt"); @@ -208,41 +208,41 @@ describe("openai.audioSpeech adapter validation", () => { it.each([ { name: "neither voice source", - request: qwenDesignRequest({ meta: {} }), + request: designRequest({ meta: {} }), }, { name: "both voice sources", - request: qwenDesignRequest({ + request: designRequest({ content: [{ type: "text", text: "文本" }, audio()], }), }, { name: "blank voice prompt", - request: qwenDesignRequest({ meta: { voice_prompt: " \n " } }), + request: designRequest({ meta: { voice_prompt: " \n " } }), }, { name: "audio weight", - request: qwenDesignRequest({ + request: designRequest({ content: [{ type: "text", text: "文本" }, audio(REFERENCE_URL, { weight: 1 })], meta: {}, }), }, - ])("rejects Qwen $name", ({ request }) => { + ])("rejects voice-enrollment $name", ({ request }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate(request)).toThrow(GenerationValidationError); }); - it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-tts"); - if (!declaration) throw new Error("qwen-tts declaration is unavailable"); + it("enforces voice-enrollment reference limits inside the adapter hook when a declaration is overridden", () => { + const declaration = getBuiltinGenerationModel("cosyvoice-v3.5-plus"); + if (!declaration) throw new Error("cosyvoice-v3.5-plus declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-tts audio spec is unavailable"); + if (!audioSpec) throw new Error("cosyvoice-v3.5-plus audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); @@ -270,6 +270,8 @@ describe("openai.audioSpeech adapter validation", () => { }); it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(14) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(14) }, { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(14) }, { model: "qwen-audio-3.0-tts-flash", text: "😀".repeat(14) }, ])("rejects $model input below 15 Unicode code points", ({ model, text }) => { @@ -280,20 +282,93 @@ describe("openai.audioSpeech adapter validation", () => { }); it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(15) }, + { model: "cosyvoice-v3.5-flash", text: ` ${"😀".repeat(15)} ` }, { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - { model: "qwen-tts", text: "短" }, - { model: "qwen-tts", text: "a".repeat(40) }, - ])("accepts the $model input boundary", ({ model, text }) => { + ])("accepts the $model clone input boundary", ({ model, text }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); }); + it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(201) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(201) }, + ])("rejects $model voice-design input above 200 Unicode code points", ({ model, text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => client.validate({ model, content: [{ type: "text", text }], meta: { voice_prompt: "声音" } })).toThrow( + "voice design requires input of at most 200 Unicode code points", + ); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(201) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(201) }, + ])("accepts $model clone input above 200 Unicode code points", ({ model, text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", voice_prompt: "a".repeat(501) }, + { model: "cosyvoice-v3.5-flash", voice_prompt: "😀".repeat(501) }, + { model: "qwen-audio-3.0-tts-plus", voice_prompt: "a".repeat(501) }, + { model: "qwen-audio-3.0-tts-flash", voice_prompt: "😀".repeat(501) }, + ])("rejects $model meta.voice_prompt above 500 Unicode code points", ({ model, voice_prompt }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model, + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt }, + }), + ).toThrow("meta.voice_prompt must be at most 500 Unicode code points"); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", voice_prompt: "a".repeat(500) }, + { model: "cosyvoice-v3.5-flash", voice_prompt: "😀".repeat(500) }, + ])("accepts $model meta.voice_prompt at the 500 Unicode code point boundary", ({ model, voice_prompt }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model, + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt }, + }), + ).not.toThrow(); + }); + + it("rejects meta.voice_prompt at exactly 500 trimmed code points plus surrounding whitespace", () => { + // buildPayload sends the untrimmed string, and the worker counts raw length — + // trailing whitespace pushes the raw length past the cap even though the + // trimmed length is exactly at it. + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt: ` ${"a".repeat(500)} ` }, + }), + ).toThrow("meta.voice_prompt must be at most 500 Unicode code points"); + }); + + it("rejects voice-design input at exactly 200 trimmed code points plus surrounding whitespace", () => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: ` ${"a".repeat(200)} ` }], + meta: { voice_prompt: "声音" }, + }), + ).toThrow("voice design requires input of at most 200 Unicode code points"); + }); + it.each<{ label: string; request: GenerateRequest }>([ - { label: "request typo", request: qwenDesignRequest({ meta: { voice_promt: "拼错" } }) }, + { label: "request typo", request: designRequest({ meta: { voice_promt: "拼错" } }) }, { label: "raw ref_audio", - request: qwenDesignRequest({ meta: { voice_prompt: "声音", ref_audio: REFERENCE_URL } }), + request: designRequest({ meta: { voice_prompt: "声音", ref_audio: REFERENCE_URL } }), }, { label: "raw references", diff --git a/test/config.test.ts b/test/config.test.ts index 408c951..4956181 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,7 +69,8 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-tts", + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash", "suno_cover_chirp_v5", @@ -216,9 +217,14 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const voiceEnrollmentModels = [ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", + ]; - for (const model of qwenModels) { + for (const model of voiceEnrollmentModels) { const declaration = client.getModel(model); expect(declaration?.description).toContain("Modes: voice_prompt design OR one-reference clone"); expect(declaration?.description).not.toMatch(/Higgs|stronger|HTTP|URL/i); @@ -236,17 +242,19 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const qwen = client.getModel("qwen-tts"); - expect(qwen?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(qwen?.content.input.find((input) => input.type === "text")?.description).not.toContain( - "Unicode code points", - ); + for (const model of ["cosyvoice-v3.5-plus", "cosyvoice-v3.5-flash"]) { + const declaration = client.getModel(model); + expect(declaration?.description).toBe( + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + ); + expect(declaration?.content.input.find((input) => input.type === "text")?.description).toContain( + "at most 200 in voice-design mode", + ); + } const plus = client.getModel("qwen-audio-3.0-tts-plus"); expect(plus?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); expect(plus?.content.input.find((input) => input.type === "text")?.description).toContain( "at least 15 Unicode code points", @@ -254,7 +262,7 @@ describe("config", () => { const flash = client.getModel("qwen-audio-3.0-tts-flash"); expect(flash?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); expect(flash?.content.input.find((input) => input.type === "text")?.description).toContain( "at least 15 Unicode code points", @@ -287,12 +295,12 @@ describe("config", () => { expect(Object.keys(packageJson.exports ?? {})).toEqual([".", "./models"]); const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); expect(readme).not.toContain("@neta-art/generation/models/"); - expect(readme).toContain("Qwen: `voice_prompt` design OR one-reference clone"); + expect(readme).toContain("CosyVoice / Qwen-Audio-TTS: `voice_prompt` design OR one-reference clone"); expect(readme).toContain("Higgs: delegated default voice, high-fidelity one-reference clone"); expect(readme).toContain("Conflict: reference + redesign requires user choice before generation"); expect(readme).toContain("Blend: all references, full text, one request"); expect(readme).toContain("Dependency: clone prior generated audio"); - expect(readme).toContain("Ranking: no declared Qwen quality, latency, or cost order"); + expect(readme).toContain("Ranking: no declared CosyVoice / Qwen quality, latency, or cost order"); expect(readme).not.toMatch(/quality prioritized over latency|latency prioritized over maximum quality/); }); @@ -517,10 +525,22 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of [ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", + ]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } - expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); + // Prose may explain (for implementers) that preview_text is a deprecated, + // never-forwarded field — that's not the same as telling a caller to use + // it. What must never happen is a copy-pasteable code example showing + // preview_text as something to send. + const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); + const codeBlocks = readme.match(/```[a-z]*\r?\n[\s\S]*?\r?\n```/g) ?? []; + expect(codeBlocks.length).toBeGreaterThan(0); + for (const block of codeBlocks) expect(block).not.toContain("preview_text"); }); it("validates every built-in model example", () => { diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index 4878eb1..3774eac 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -47,18 +47,18 @@ liveDescribe("audio speech live router smoke", () => { const runId = `${Date.now()}`; const cases: Array<{ name: string; request: GenerateRequest }> = [ { - name: "qwen voice design", + name: "cosyvoice v3.5 plus voice design", request: { - model: "qwen-tts", - content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], + model: "cosyvoice-v3.5-plus", + content: [text(`这是CosyVoice设计音色端到端测试,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { - name: "qwen voice clone", + name: "cosyvoice v3.5 flash voice clone", request: { - model: "qwen-tts", - content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], + model: "cosyvoice-v3.5-flash", + content: [text(`这是CosyVoice克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, {