From e40e231899bdff5734d6a65729fea9068b078b7a Mon Sep 17 00:00:00 2001 From: LingXuanYin <95487306+LingXuanYin@users.noreply.github.com> Date: Fri, 11 Sep 2026 18:24:35 +0000 Subject: [PATCH 1/5] feat!: replace qwen-tts with cosyvoice-v3.5-plus/flash DashScope retires qwen-tts (qwen-voice-design/enrollment + qwen3-tts-vd/vc) on 2026-10-10 (notices 118331/118332/118434). The SDK now declares cosyvoice-v3.5-plus and cosyvoice-v3.5-flash, which share one request contract with the remaining qwen-audio-3.0-tts-* models. - builtins + exported YAMLs regenerated; `qwen-tts` removed (breaking) - adapter validation mirrors the service bounds: >=15 Unicode code points, <=200 in voice-design mode - README, text-to-speech example and live smoke updated; 0.2.0 --- README.md | 17 +++--- examples/text-to-speech.ts | 2 +- models/cosyvoice-v3.5-flash.yaml | 46 +++++++++++++++ models/cosyvoice-v3.5-plus.yaml | 46 +++++++++++++++ models/qwen-audio-3.0-tts-flash.yaml | 10 ++-- models/qwen-audio-3.0-tts-plus.yaml | 10 ++-- models/qwen-tts.yaml | 44 -------------- package.json | 2 +- src/adapters/audio-speech.ts | 24 ++++++-- src/builtins.ts | 42 ++++++-------- test/adapters/audio-speech.test.ts | 86 +++++++++++++++++----------- test/config.test.ts | 43 +++++++++----- test/live/audio-speech-live.test.ts | 12 ++-- 13 files changed, 239 insertions(+), 145 deletions(-) create mode 100644 models/cosyvoice-v3.5-flash.yaml create mode 100644 models/cosyvoice-v3.5-plus.yaml delete mode 100644 models/qwen-tts.yaml diff --git a/README.md b/README.md index d76b214..d677a9b 100644 --- a/README.md +++ b/README.md @@ -62,7 +62,7 @@ Agents and external tools should inspect a model declaration before constructing import { createGenerationClient } from "@neta-art/generation"; const discoveryClient = createGenerationClient(); -const declaration = discoveryClient.getModel("qwen-tts"); +const declaration = discoveryClient.getModel("cosyvoice-v3.5-plus"); if (!declaration) throw new Error("Model is unavailable"); console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" })); @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI: ```bash neta-generation models list -neta-generation models export qwen-tts --out ./qwen-tts.yaml +neta-generation models export cosyvoice-v3.5-plus --out ./cosyvoice-v3.5-plus.yaml neta-generation models export-all --out ./models ``` @@ -170,7 +170,8 @@ const client = createGenerationClient({ - `gpt-image-2` - `z-image-turbo` - `qwen-image-edit` -- `qwen-tts` +- `cosyvoice-v3.5-plus` +- `cosyvoice-v3.5-flash` - `qwen-audio-3.0-tts-plus` - `qwen-audio-3.0-tts-flash` - `higgs-tts` @@ -268,22 +269,22 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au | Requirement | Model choice | | --- | --- | -| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default | +| Create a voice from a text-only description, without reference audio | Use an explicitly requested cosyvoice variant; otherwise use `cosyvoice-v3.5-flash` as the deterministic default | | Maximize fidelity to one reference voice | `higgs-tts` | | Blend 2-16 weighted reference voices | `higgs-tts` | | Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` | -- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points. +- CosyVoice / Qwen-Audio-TTS: `voice_prompt` design OR one-reference clone; `cosyvoice-v3.5-plus`, `cosyvoice-v3.5-flash`, `qwen-audio-3.0-tts-plus`, and `qwen-audio-3.0-tts-flash` all require at least 15 Unicode code points (and at most 200 in design mode); `cosyvoice-v3.5-flash` is the deterministic default when no variant is requested. - Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend. - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. - Dependency: clone prior generated audio. -- Ranking: no declared Qwen quality, latency, or cost order. +- Ranking: no declared CosyVoice / Qwen quality, latency, or cost order. ```ts await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "欢迎使用语音合成功能。" }], + model: "cosyvoice-v3.5-flash", + content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", }, diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 009b59f..740b64c 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-flash", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/cosyvoice-v3.5-flash.yaml b/models/cosyvoice-v3.5-flash.yaml new file mode 100644 index 0000000..e51de69 --- /dev/null +++ b/models/cosyvoice-v3.5-flash.yaml @@ -0,0 +1,46 @@ +schema: neta.generation.model.v1 +model: cosyvoice-v3.5-flash +title: CosyVoice v3.5 Flash +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." +adapter: + type: openai.audioSpeech +content: + input: + - type: text + required: true + min: 1 + max: 1 + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). + - type: audio + required: false + max: 1 + sources: + - url + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." +meta: + fields: + voice_prompt: + type: string + optional: true + description: "Design: custom voice text; no reference audio." +examples: + - title: Voice design + request: + model: cosyvoice-v3.5-flash + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + meta: + voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 + - title: Voice clone + request: + model: cosyvoice-v3.5-flash + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + - type: audio + source: + type: url + url: https://example.com/reference.mp3 diff --git a/models/cosyvoice-v3.5-plus.yaml b/models/cosyvoice-v3.5-plus.yaml new file mode 100644 index 0000000..158f0f0 --- /dev/null +++ b/models/cosyvoice-v3.5-plus.yaml @@ -0,0 +1,46 @@ +schema: neta.generation.model.v1 +model: cosyvoice-v3.5-plus +title: CosyVoice v3.5 Plus +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." +adapter: + type: openai.audioSpeech +content: + input: + - type: text + required: true + min: 1 + max: 1 + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). + - type: audio + required: false + max: 1 + sources: + - url + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." +meta: + fields: + voice_prompt: + type: string + optional: true + description: "Design: custom voice text; no reference audio." +examples: + - title: Voice design + request: + model: cosyvoice-v3.5-plus + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + meta: + voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 + - title: Voice clone + request: + model: cosyvoice-v3.5-plus + content: + - type: text + text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 + - type: audio + source: + type: url + url: https://example.com/reference.mp3 diff --git a/models/qwen-audio-3.0-tts-flash.yaml b/models/qwen-audio-3.0-tts-flash.yaml index 3d111b9..42ed56a 100644 --- a/models/qwen-audio-3.0-tts-flash.yaml +++ b/models/qwen-audio-3.0-tts-flash.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 model: qwen-audio-3.0-tts-flash title: Qwen Audio 3.0 TTS Flash -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -10,19 +11,20 @@ content: required: true min: 1 max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). - type: audio required: false max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: diff --git a/models/qwen-audio-3.0-tts-plus.yaml b/models/qwen-audio-3.0-tts-plus.yaml index d6bf7fa..872ae5a 100644 --- a/models/qwen-audio-3.0-tts-plus.yaml +++ b/models/qwen-audio-3.0-tts-plus.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 model: qwen-audio-3.0-tts-plus title: Qwen Audio 3.0 TTS Plus -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). + Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -10,19 +11,20 @@ content: required: true min: 1 max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. + description: Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in + voice-design mode). - type: audio required: false max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: diff --git a/models/qwen-tts.yaml b/models/qwen-tts.yaml deleted file mode 100644 index 1136fc7..0000000 --- a/models/qwen-tts.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-tts -title: Qwen TTS -description: 'Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/package.json b/package.json index 4895da2..2a7f219 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@neta-art/generation", - "version": "0.1.31", + "version": "0.2.0", "description": "A lightweight multimodal generation SDK with built-in model presets and adapter-based provider calls.", "keywords": [ "ai", diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index b174984..cf57be4 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,8 +8,13 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); -const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const VOICE_ENROLLMENT_MODELS = new Set([ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", +]); +const VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS = 200; const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -92,9 +97,18 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { + const codePoints = Array.from(text.text.trim()).length; + if (codePoints < 15) { throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); } + // Voice design speaks exactly this text as the preview clip, so the model's + // 15-200 character preview window applies; cloning only feeds the separate + // synthesis call, where long-form text is legitimate. + if (audio.length === 0 && codePoints > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { + throw new GenerationValidationError( + `${input.declaration.model} voice design requires input of at most ${VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS} Unicode code points`, + ); + } } function hasOwnWeight(block: AudioBlock): boolean { @@ -124,7 +138,7 @@ function validateHiggs(input: ResolvedGenerationRequest, text: TextBlock, audio: function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void { const { text, audio } = validateCommonContent(input); - if (QWEN_MODELS.has(input.declaration.model)) { + if (VOICE_ENROLLMENT_MODELS.has(input.declaration.model)) { validateQwen(input, text, audio); return; } @@ -148,7 +162,7 @@ function buildPayload(input: ResolvedGenerationRequest): Record input: text.text, }; - if (QWEN_MODELS.has(input.declaration.model)) { + if (VOICE_ENROLLMENT_MODELS.has(input.declaration.model)) { if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim(); else payload.metadata = { voice_prompt: input.meta.voice_prompt }; return payload; diff --git a/src/builtins.ts b/src/builtins.ts index 8ba25a5..2632627 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -708,15 +708,7 @@ function geminiImageModel( }; } -function qwenTtsModel( - model: string, - title: string, - description: string, - options: { minimumTextCodePoints?: number } = {}, -): GenerationModelDeclaration { - const text = options.minimumTextCodePoints - ? "这是一段长度足够并且表达清晰自然的语音合成测试文本。" - : "这是一次清晰自然的语音合成测试。"; +function voiceEnrollmentModel(model: string, title: string, description: string): GenerationModelDeclaration { return { schema: MODEL_SCHEMA, model, @@ -730,9 +722,8 @@ function qwenTtsModel( required: true, min: 1, max: 1, - description: options.minimumTextCodePoints - ? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.` - : "Exactly one non-empty text block to speak.", + description: + "Exactly one non-empty text block to speak, with at least 15 Unicode code points (at most 200 in voice-design mode).", }, { type: "audio", @@ -757,7 +748,7 @@ function qwenTtsModel( title: "Voice design", request: { model, - content: [{ type: "text", text }], + content: [{ type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, @@ -766,7 +757,7 @@ function qwenTtsModel( request: { model, content: [ - { type: "text", text }, + { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, ], }, @@ -776,22 +767,25 @@ function qwenTtsModel( } const audioSpeechModels = [ - qwenTtsModel( - "qwen-tts", - "Qwen TTS", - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + voiceEnrollmentModel( + "cosyvoice-v3.5-plus", + "CosyVoice v3.5 Plus", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + ), + voiceEnrollmentModel( + "cosyvoice-v3.5-flash", + "CosyVoice v3.5 Flash", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), - qwenTtsModel( + voiceEnrollmentModel( "qwen-audio-3.0-tts-plus", "Qwen Audio 3.0 TTS Plus", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), - qwenTtsModel( + voiceEnrollmentModel( "qwen-audio-3.0-tts-flash", "Qwen Audio 3.0 TTS Flash", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ), { schema: MODEL_SCHEMA, diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 236718b..625cd0c 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -39,10 +39,10 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC }; } -function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { +function designRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-tts", - content: [{ type: "text", text: "这是需要朗读的文本。" }], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是需要朗读的一段完整试听文本。" }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, }; @@ -65,15 +65,15 @@ function requestBody(call: { init: RequestInit } | undefined): Record { - it("sends the fixed wire contract and preserves Qwen input and voice prompt", async () => { + it("sends the fixed wire contract and preserves voice-enrollment input and voice prompt", async () => { const { client, calls } = recordingClient(() => routerSuccess({}, { headers: { "x-request-id": "request-primary", "x-oneapi-request-id": "request-fallback" } }), ); - const input = " 原样保留的朗读文本。\n"; + const input = " 原样保留的一段完整朗读试听文本。\n"; const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -85,7 +85,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", input, metadata: { voice_prompt: voicePrompt }, }); @@ -98,16 +98,16 @@ describe("openai.audioSpeech adapter requests", () => { ]); }); - it("maps Qwen reference audio and trims only its URL", async () => { + it("maps voice-enrollment reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段足够长的克隆试听文本。" }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "短句", + model: "cosyvoice-v3.5-plus", + input: "这是一段足够长的克隆试听文本。", ref_audio: REFERENCE_URL, }); }); @@ -115,14 +115,14 @@ describe("openai.audioSpeech adapter requests", () => { it("silently ignores deprecated preview_text without sending it", async () => { const { client, calls } = recordingClient(); await client.generate( - qwenDesignRequest({ + designRequest({ meta: { voice_prompt: "清晰女声", preview_text: "旧字段不应发送" }, }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "这是需要朗读的文本。", + model: "cosyvoice-v3.5-plus", + input: "这是需要朗读的一段完整试听文本。", metadata: { voice_prompt: "清晰女声" }, }); }); @@ -194,12 +194,12 @@ describe("openai.audioSpeech adapter validation", () => { expect(fetchMock).not.toHaveBeenCalled(); }); - it("accepts retired preview_text without treating it as a Qwen voice source", () => { + it("accepts retired preview_text without treating it as a voice source", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "有效语音文本" }], + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段有效的语音试听文本。" }], meta: { preview_text: "已经失效的文本来源" }, }), ).toThrow("requires one reference audio or meta.voice_prompt"); @@ -208,41 +208,41 @@ describe("openai.audioSpeech adapter validation", () => { it.each([ { name: "neither voice source", - request: qwenDesignRequest({ meta: {} }), + request: designRequest({ meta: {} }), }, { name: "both voice sources", - request: qwenDesignRequest({ + request: designRequest({ content: [{ type: "text", text: "文本" }, audio()], }), }, { name: "blank voice prompt", - request: qwenDesignRequest({ meta: { voice_prompt: " \n " } }), + request: designRequest({ meta: { voice_prompt: " \n " } }), }, { name: "audio weight", - request: qwenDesignRequest({ + request: designRequest({ content: [{ type: "text", text: "文本" }, audio(REFERENCE_URL, { weight: 1 })], meta: {}, }), }, - ])("rejects Qwen $name", ({ request }) => { + ])("rejects voice-enrollment $name", ({ request }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate(request)).toThrow(GenerationValidationError); }); - it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-tts"); - if (!declaration) throw new Error("qwen-tts declaration is unavailable"); + it("enforces voice-enrollment reference limits inside the adapter hook when a declaration is overridden", () => { + const declaration = getBuiltinGenerationModel("cosyvoice-v3.5-plus"); + if (!declaration) throw new Error("cosyvoice-v3.5-plus declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-tts audio spec is unavailable"); + if (!audioSpec) throw new Error("cosyvoice-v3.5-plus audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", + model: "cosyvoice-v3.5-plus", content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); @@ -270,6 +270,8 @@ describe("openai.audioSpeech adapter validation", () => { }); it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(14) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(14) }, { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(14) }, { model: "qwen-audio-3.0-tts-flash", text: "😀".repeat(14) }, ])("rejects $model input below 15 Unicode code points", ({ model, text }) => { @@ -280,20 +282,38 @@ describe("openai.audioSpeech adapter validation", () => { }); it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(15) }, + { model: "cosyvoice-v3.5-flash", text: ` ${"😀".repeat(15)} ` }, { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - { model: "qwen-tts", text: "短" }, - { model: "qwen-tts", text: "a".repeat(40) }, - ])("accepts the $model input boundary", ({ model, text }) => { + ])("accepts the $model clone input boundary", ({ model, text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(201) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(201) }, + ])("rejects $model voice-design input above 200 Unicode code points", ({ model, text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => client.validate({ model, content: [{ type: "text", text }], meta: { voice_prompt: "声音" } })).toThrow( + "voice design requires input of at most 200 Unicode code points", + ); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", text: "a".repeat(201) }, + { model: "cosyvoice-v3.5-flash", text: "😀".repeat(201) }, + ])("accepts $model clone input above 200 Unicode code points", ({ model, text }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); }); it.each<{ label: string; request: GenerateRequest }>([ - { label: "request typo", request: qwenDesignRequest({ meta: { voice_promt: "拼错" } }) }, + { label: "request typo", request: designRequest({ meta: { voice_promt: "拼错" } }) }, { label: "raw ref_audio", - request: qwenDesignRequest({ meta: { voice_prompt: "声音", ref_audio: REFERENCE_URL } }), + request: designRequest({ meta: { voice_prompt: "声音", ref_audio: REFERENCE_URL } }), }, { label: "raw references", diff --git a/test/config.test.ts b/test/config.test.ts index 408c951..db03e96 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,7 +69,8 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-tts", + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash", "suno_cover_chirp_v5", @@ -216,9 +217,14 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const voiceEnrollmentModels = [ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", + ]; - for (const model of qwenModels) { + for (const model of voiceEnrollmentModels) { const declaration = client.getModel(model); expect(declaration?.description).toContain("Modes: voice_prompt design OR one-reference clone"); expect(declaration?.description).not.toMatch(/Higgs|stronger|HTTP|URL/i); @@ -236,17 +242,19 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const qwen = client.getModel("qwen-tts"); - expect(qwen?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(qwen?.content.input.find((input) => input.type === "text")?.description).not.toContain( - "Unicode code points", - ); + for (const model of ["cosyvoice-v3.5-plus", "cosyvoice-v3.5-flash"]) { + const declaration = client.getModel(model); + expect(declaration?.description).toBe( + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + ); + expect(declaration?.content.input.find((input) => input.type === "text")?.description).toContain( + "at most 200 in voice-design mode", + ); + } const plus = client.getModel("qwen-audio-3.0-tts-plus"); expect(plus?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); expect(plus?.content.input.find((input) => input.type === "text")?.description).toContain( "at least 15 Unicode code points", @@ -254,7 +262,7 @@ describe("config", () => { const flash = client.getModel("qwen-audio-3.0-tts-flash"); expect(flash?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points (design mode: <=200). Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); expect(flash?.content.input.find((input) => input.type === "text")?.description).toContain( "at least 15 Unicode code points", @@ -287,12 +295,12 @@ describe("config", () => { expect(Object.keys(packageJson.exports ?? {})).toEqual([".", "./models"]); const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); expect(readme).not.toContain("@neta-art/generation/models/"); - expect(readme).toContain("Qwen: `voice_prompt` design OR one-reference clone"); + expect(readme).toContain("CosyVoice / Qwen-Audio-TTS: `voice_prompt` design OR one-reference clone"); expect(readme).toContain("Higgs: delegated default voice, high-fidelity one-reference clone"); expect(readme).toContain("Conflict: reference + redesign requires user choice before generation"); expect(readme).toContain("Blend: all references, full text, one request"); expect(readme).toContain("Dependency: clone prior generated audio"); - expect(readme).toContain("Ranking: no declared Qwen quality, latency, or cost order"); + expect(readme).toContain("Ranking: no declared CosyVoice / Qwen quality, latency, or cost order"); expect(readme).not.toMatch(/quality prioritized over latency|latency prioritized over maximum quality/); }); @@ -517,7 +525,12 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of [ + "cosyvoice-v3.5-plus", + "cosyvoice-v3.5-flash", + "qwen-audio-3.0-tts-plus", + "qwen-audio-3.0-tts-flash", + ]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index 4878eb1..3774eac 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -47,18 +47,18 @@ liveDescribe("audio speech live router smoke", () => { const runId = `${Date.now()}`; const cases: Array<{ name: string; request: GenerateRequest }> = [ { - name: "qwen voice design", + name: "cosyvoice v3.5 plus voice design", request: { - model: "qwen-tts", - content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], + model: "cosyvoice-v3.5-plus", + content: [text(`这是CosyVoice设计音色端到端测试,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { - name: "qwen voice clone", + name: "cosyvoice v3.5 flash voice clone", request: { - model: "qwen-tts", - content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], + model: "cosyvoice-v3.5-flash", + content: [text(`这是CosyVoice克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, { From 0628896e3e9995cb7f6bc6d05baed3d2bcbd82c4 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Sun, 20 Sep 2026 16:38:53 +0800 Subject: [PATCH 2/5] fix: enforce voice_prompt length cap and document preview_text/qwen-tts migration MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - validateQwen now rejects meta.voice_prompt over 500 Unicode code points, matching the worker's _MAX_VOICE_PROMPT_LEN — previously the SDK let an oversized voice_prompt through and the worker only rejected it after the request had already been dispatched. - Clarify in-code why meta.preview_text is accepted-but-ignored: the spoken text is the request's primary text content block (already validated against the same 15-200 code point window), not a separate meta field. preview_text is a legacy alias, not live input. - Use hasVoicePrompt instead of audio.length === 0 to select the design-mode code point ceiling — same behavior (mutual exclusivity is already enforced above), clearer intent. - README: add a migration note for the qwen-tts removal — cosyvoice models require at least 15 Unicode code points where qwen-tts accepted any length, so a same-name-swap on short input will start throwing GenerationValidationError. --- README.md | 7 +++++++ src/adapters/audio-speech.ts | 14 ++++++++++++-- test/adapters/audio-speech.test.ts | 30 ++++++++++++++++++++++++++++++ 3 files changed, 49 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index d677a9b..75863ad 100644 --- a/README.md +++ b/README.md @@ -281,6 +281,13 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au - Dependency: clone prior generated audio. - Ranking: no declared CosyVoice / Qwen quality, latency, or cost order. +> **Migrating from `qwen-tts` (removed in 0.2.0):** DashScope retires `qwen-tts` on 2026-10-10; this SDK removed +> the declaration in the same release. Switch to `cosyvoice-v3.5-flash` (or `-plus`) — same request shape +> (`voice_prompt` design OR one-reference clone), but `qwen-tts` accepted text of **any length** while +> `cosyvoice-v3.5-*` requires **at least 15 Unicode code points** (and at most 200 in design mode, like the rest +> of this model family). A caller doing a plain model-name swap on short input will start seeing +> `GenerationValidationError` where it previously succeeded — pad short inputs or catch the error. + ```ts await client.generate({ model: "cosyvoice-v3.5-flash", diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index cf57be4..fb8d01d 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -15,6 +15,7 @@ const VOICE_ENROLLMENT_MODELS = new Set([ "qwen-audio-3.0-tts-flash", ]); const VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS = 200; +const VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS = 500; const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -76,7 +77,11 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (audio.length > 1) { throw new GenerationValidationError(`${input.declaration.model} supports at most one reference audio`); } - // Keep accepting the retired preview_text key for compatibility, but never use or forward it. + // Keep accepting the retired preview_text key for compatibility, but never use or forward it: + // the text spoken in the preview/output clip is this request's primary `text` content block + // (validated below against the same 15-200 code point window the worker enforces on its own + // `preview_text` field downstream), not a separate meta value. `preview_text` was how older + // callers passed that text directly; treat any value here as a no-op alias, not live input. const requestMetaKeys = new Set(["voice_prompt", "preview_text"]); validateMetaKeys("request.metadata", input.request.metadata, requestMetaKeys); validateMetaKeys("request.meta", input.request.meta, requestMetaKeys); @@ -88,6 +93,11 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (voicePrompt !== undefined && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} meta.voice_prompt must be a non-empty string`); } + if (hasVoicePrompt && Array.from(voicePrompt.trim()).length > VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS) { + throw new GenerationValidationError( + `${input.declaration.model} meta.voice_prompt must be at most ${VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS} Unicode code points`, + ); + } if (audio.length === 0 && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} requires one reference audio or meta.voice_prompt`); } @@ -104,7 +114,7 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: // Voice design speaks exactly this text as the preview clip, so the model's // 15-200 character preview window applies; cloning only feeds the separate // synthesis call, where long-form text is legitimate. - if (audio.length === 0 && codePoints > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { + if (hasVoicePrompt && codePoints > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { throw new GenerationValidationError( `${input.declaration.model} voice design requires input of at most ${VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS} Unicode code points`, ); diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 625cd0c..fe95fcc 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -309,6 +309,36 @@ describe("openai.audioSpeech adapter validation", () => { expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); }); + it.each([ + { model: "cosyvoice-v3.5-plus", voice_prompt: "a".repeat(501) }, + { model: "cosyvoice-v3.5-flash", voice_prompt: "😀".repeat(501) }, + { model: "qwen-audio-3.0-tts-plus", voice_prompt: "a".repeat(501) }, + { model: "qwen-audio-3.0-tts-flash", voice_prompt: "😀".repeat(501) }, + ])("rejects $model meta.voice_prompt above 500 Unicode code points", ({ model, voice_prompt }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model, + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt }, + }), + ).toThrow("meta.voice_prompt must be at most 500 Unicode code points"); + }); + + it.each([ + { model: "cosyvoice-v3.5-plus", voice_prompt: "a".repeat(500) }, + { model: "cosyvoice-v3.5-flash", voice_prompt: "😀".repeat(500) }, + ])("accepts $model meta.voice_prompt at the 500 Unicode code point boundary", ({ model, voice_prompt }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model, + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt }, + }), + ).not.toThrow(); + }); + it.each<{ label: string; request: GenerateRequest }>([ { label: "request typo", request: designRequest({ meta: { voice_promt: "拼错" } }) }, { From 7f1c30fb0e7c8f2c061a4e8f9fb89ebe05e1f7e1 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Sun, 20 Sep 2026 17:00:34 +0800 Subject: [PATCH 3/5] docs: write down the SDK-to-worker field mapping, note trim/raw length gap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - README: document that this SDK's `input`/`meta.voice_prompt` map to the background worker's `preview_text`/`voice_prompt` respectively — nothing in the pipeline between them performs that translation yet (talesofai backend has no live caller, new-api passes metadata through opaquely), so this was previously only documented in a commit message. Point whoever builds that glue at it instead of re-deriving the mapping. - Comment the known trim-vs-raw-length gap: validation counts trimmed code points (matching the worker), but buildPayload sends the untrimmed string (deliberate, existing wire-contract behavior) — a value at exactly a length cap plus padding can pass here and still get rejected downstream. Narrow, not fixed, just no longer silent. - Loosen the "does not publish the retired preview_text field" test to check code examples specifically instead of the whole README: the new mapping note legitimately explains, in prose, that preview_text is deprecated and never forwarded — that's not the same failure mode the test was written to catch (a copy-pasteable example telling callers to send it). --- README.md | 7 +++++++ src/adapters/audio-speech.ts | 5 +++++ test/config.test.ts | 8 +++++++- 3 files changed, 19 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 75863ad..727dc34 100644 --- a/README.md +++ b/README.md @@ -288,6 +288,13 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au > of this model family). A caller doing a plain model-name swap on short input will start seeing > `GenerationValidationError` where it previously succeeded — pad short inputs or catch the error. +> **Server-side wire contract:** this SDK talks to the router over HTTP; it does not call the DashScope-facing +> worker (`talesofai/background`'s `qwen_tts_actor`) directly, and as of 2026-09-20 nothing in the pipeline +> between them performs the translation yet. For whoever builds that glue: the worker's `preview_text` is this +> request's primary `text` content block (same value, same 15/200 bound — not the SDK's separate, deprecated, +> never-forwarded `meta.preview_text`), and the worker's `voice_prompt`/`target_model` are this request's +> `meta.voice_prompt`/`model` unchanged. + ```ts await client.generate({ model: "cosyvoice-v3.5-flash", diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index fb8d01d..19ed531 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -93,6 +93,11 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (voicePrompt !== undefined && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} meta.voice_prompt must be a non-empty string`); } + // Length checks below (here and on `text`) count trimmed code points, matching + // the worker's own bounds, but buildPayload sends the untrimmed original string + // (deliberate — see the wire-contract preservation tests). A value that's exactly + // at a cap plus surrounding whitespace can therefore pass here and still get + // rejected downstream: a narrow, known gap, not a correctness bug in either layer. if (hasVoicePrompt && Array.from(voicePrompt.trim()).length > VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS) { throw new GenerationValidationError( `${input.declaration.model} meta.voice_prompt must be at most ${VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS} Unicode code points`, diff --git a/test/config.test.ts b/test/config.test.ts index db03e96..cfbc4e6 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -533,7 +533,13 @@ describe("config", () => { ]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } - expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); + // Prose may explain (for implementers) that preview_text is a deprecated, + // never-forwarded field — that's not the same as telling a caller to use + // it. What must never happen is a copy-pasteable code example showing + // preview_text as something to send. + const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); + const codeBlocks = readme.match(/```ts\n[\s\S]*?\n```/g) ?? []; + for (const block of codeBlocks) expect(block).not.toContain("preview_text"); }); it("validates every built-in model example", () => { From e4e900dc3e685397dab0982069f3e46948ab37c2 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Sun, 20 Sep 2026 17:10:17 +0800 Subject: [PATCH 4/5] fix: correct the SDK-worker field mapping doc, widen and guard the fence test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - README's "server-side wire contract" note had the field mapping backwards for voice_prompt/text — see the paired background fix for the full correction and how it was verified against the actor's actual field reads. - The fence-block regex only matched ```ts, so a ```bash/```dotenv/bare ``` example could carry "preview_text" unchecked; widened to any fence tag. Also add codeBlocks.length > 0 so the test can't silently pass on zero matches if the fence format ever changes — this fired immediately: the widened regex's literal \n didn't match this Windows checkout's CRLF README, so it needed \r?\n to match anything at all. Without the length assertion this would have been a permanently-green dead test. --- README.md | 10 ++++++---- test/config.test.ts | 3 ++- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 727dc34..bf8aab8 100644 --- a/README.md +++ b/README.md @@ -290,10 +290,12 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au > **Server-side wire contract:** this SDK talks to the router over HTTP; it does not call the DashScope-facing > worker (`talesofai/background`'s `qwen_tts_actor`) directly, and as of 2026-09-20 nothing in the pipeline -> between them performs the translation yet. For whoever builds that glue: the worker's `preview_text` is this -> request's primary `text` content block (same value, same 15/200 bound — not the SDK's separate, deprecated, -> never-forwarded `meta.preview_text`), and the worker's `voice_prompt`/`target_model` are this request's -> `meta.voice_prompt`/`model` unchanged. +> between them performs the translation yet. For whoever builds that glue, per the worker's own field names +> (see its module docstring — they do **not** match DashScope's or this SDK's field names one-for-one, which is +> the trap to avoid): the worker's `preview_text` (what's actually spoken, required on every request) is this +> request's primary `text` content block; the worker's `text` field (the voice STYLE description, design mode +> only — unrelated to this SDK's own `input`/text-content naming despite the shared word) is this request's +> `meta.voice_prompt`; `target_model` is this request's `model`, unchanged. ```ts await client.generate({ diff --git a/test/config.test.ts b/test/config.test.ts index cfbc4e6..4956181 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -538,7 +538,8 @@ describe("config", () => { // it. What must never happen is a copy-pasteable code example showing // preview_text as something to send. const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); - const codeBlocks = readme.match(/```ts\n[\s\S]*?\n```/g) ?? []; + const codeBlocks = readme.match(/```[a-z]*\r?\n[\s\S]*?\r?\n```/g) ?? []; + expect(codeBlocks.length).toBeGreaterThan(0); for (const block of codeBlocks) expect(block).not.toContain("preview_text"); }); From 79097cf5ed487f20d8739c3c34b226d8580c35a1 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Wed, 23 Sep 2026 10:42:23 +0800 Subject: [PATCH 5/5] fix: validate upper length bounds against the raw string, not trimmed MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-2 review pointed out the round-1 fix only commented the gap instead of closing it. Validate the 200/500 code-point ceilings against the untrimmed string buildPayload actually sends (matching what the worker's plain len() sees), while keeping the >=15 lower-bound and non-empty checks on the trimmed string — trimming can only shorten a string, so a trimmed-length minimum still holds for the raw string too. Closes the gap where a value at exactly a cap plus surrounding whitespace passed the SDK and failed downstream. --- src/adapters/audio-speech.ts | 20 ++++++++++++-------- test/adapters/audio-speech.test.ts | 25 +++++++++++++++++++++++++ 2 files changed, 37 insertions(+), 8 deletions(-) diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index 19ed531..18d5164 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -93,12 +93,15 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: if (voicePrompt !== undefined && !hasVoicePrompt) { throw new GenerationValidationError(`${input.declaration.model} meta.voice_prompt must be a non-empty string`); } - // Length checks below (here and on `text`) count trimmed code points, matching - // the worker's own bounds, but buildPayload sends the untrimmed original string - // (deliberate — see the wire-contract preservation tests). A value that's exactly - // at a cap plus surrounding whitespace can therefore pass here and still get - // rejected downstream: a narrow, known gap, not a correctness bug in either layer. - if (hasVoicePrompt && Array.from(voicePrompt.trim()).length > VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS) { + // Upper-bound checks below (here and on `text`) count the RAW string's code + // points, not the trimmed one: buildPayload sends the untrimmed original + // string (deliberate — see the wire-contract preservation tests), and the + // worker counts with plain len() on what it receives, so validating here + // against anything shorter than what's actually sent could let through a + // value the worker then rejects. Lower-bound/non-empty checks stay on the + // trimmed string — trimmed-length >= a minimum only strengthens the + // guarantee, since trimming can only shorten a string. + if (hasVoicePrompt && Array.from(voicePrompt).length > VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS) { throw new GenerationValidationError( `${input.declaration.model} meta.voice_prompt must be at most ${VOICE_ENROLLMENT_VOICE_PROMPT_MAX_CODE_POINTS} Unicode code points`, ); @@ -118,8 +121,9 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: } // Voice design speaks exactly this text as the preview clip, so the model's // 15-200 character preview window applies; cloning only feeds the separate - // synthesis call, where long-form text is legitimate. - if (hasVoicePrompt && codePoints > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { + // synthesis call, where long-form text is legitimate. Upper bound counts the + // raw (untrimmed) string sent by buildPayload — see the comment above. + if (hasVoicePrompt && Array.from(text.text).length > VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS) { throw new GenerationValidationError( `${input.declaration.model} voice design requires input of at most ${VOICE_ENROLLMENT_DESIGN_MAX_CODE_POINTS} Unicode code points`, ); diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index fe95fcc..e0916d6 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -339,6 +339,31 @@ describe("openai.audioSpeech adapter validation", () => { ).not.toThrow(); }); + it("rejects meta.voice_prompt at exactly 500 trimmed code points plus surrounding whitespace", () => { + // buildPayload sends the untrimmed string, and the worker counts raw length — + // trailing whitespace pushes the raw length past the cap even though the + // trimmed length is exactly at it. + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: "这是一段长度足够的语音设计试听文本。" }], + meta: { voice_prompt: ` ${"a".repeat(500)} ` }, + }), + ).toThrow("meta.voice_prompt must be at most 500 Unicode code points"); + }); + + it("rejects voice-design input at exactly 200 trimmed code points plus surrounding whitespace", () => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ + model: "cosyvoice-v3.5-plus", + content: [{ type: "text", text: ` ${"a".repeat(200)} ` }], + meta: { voice_prompt: "声音" }, + }), + ).toThrow("voice design requires input of at most 200 Unicode code points"); + }); + it.each<{ label: string; request: GenerateRequest }>([ { label: "request typo", request: designRequest({ meta: { voice_promt: "拼错" } }) }, {