From 749927595e1c8a4ccd87d28d94fd4ef62a9b24cb Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Wed, 23 Sep 2026 13:49:24 +0800 Subject: [PATCH] feat!: replace qwen-tts with qwen3-tts-vc-2026-01-22 qwen-tts is retiring on DashScope (2026-10-10); qwen3-tts-vc-2026-01-22 is the long-term-supported replacement on the worker side (see background/tasks/qwen_tts_actor.py). Unlike qwen-tts, the new model is clone-only -- no voice-design mode exists for this DashScope model family -- so it gets its own adapter validate/buildPayload path instead of joining QWEN_MODELS (which still covers qwen-audio-3.0-tts-plus/flash, unaffected by this migration, design+clone dual mode retained as-is). cosyvoice-v3.5 is not included: it was never shipped (only ever existed on an abandoned PR), so there's nothing to keep exposed. The worker independently keeps serving it for any caller with a pinned target_model. --- README.md | 28 ++++++--- examples/text-to-speech.ts | 2 +- models/qwen-tts.yaml | 44 ------------- models/qwen3-tts-vc-2026-01-22.yaml | 32 ++++++++++ src/adapters/audio-speech.ts | 36 +++++++++-- src/builtins.ts | 68 ++++++++++++++------ test/adapters/audio-speech.test.ts | 98 +++++++++++++++++++++++------ test/config.test.ts | 38 ++++++++--- test/live/audio-speech-live.test.ts | 12 ++-- 9 files changed, 249 insertions(+), 109 deletions(-) delete mode 100644 models/qwen-tts.yaml create mode 100644 models/qwen3-tts-vc-2026-01-22.yaml diff --git a/README.md b/README.md index d76b214..3464b74 100644 --- a/README.md +++ b/README.md @@ -62,13 +62,13 @@ Agents and external tools should inspect a model declaration before constructing import { createGenerationClient } from "@neta-art/generation"; const discoveryClient = createGenerationClient(); -const declaration = discoveryClient.getModel("qwen-tts"); +const declaration = discoveryClient.getModel("qwen3-tts-vc-2026-01-22"); if (!declaration) throw new Error("Model is unavailable"); console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" })); -const request = declaration.examples?.find((example) => example.title === "Voice design")?.request; -if (!request) throw new Error("Voice-design example is unavailable"); +const request = declaration.examples?.find((example) => example.title === "Voice clone")?.request; +if (!request) throw new Error("Voice-clone example is unavailable"); // Discovery and validation do not require an API key or access the network. discoveryClient.validate(request); @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI: ```bash neta-generation models list -neta-generation models export qwen-tts --out ./qwen-tts.yaml +neta-generation models export qwen3-tts-vc-2026-01-22 --out ./qwen3-tts-vc-2026-01-22.yaml neta-generation models export-all --out ./models ``` @@ -170,7 +170,7 @@ const client = createGenerationClient({ - `gpt-image-2` - `z-image-turbo` - `qwen-image-edit` -- `qwen-tts` +- `qwen3-tts-vc-2026-01-22` - `qwen-audio-3.0-tts-plus` - `qwen-audio-3.0-tts-flash` - `higgs-tts` @@ -268,12 +268,14 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au | Requirement | Model choice | | --- | --- | -| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default | +| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-audio-3.0-tts-plus` as the deterministic default | +| Clone one reference voice on DashScope's long-term-supported model line | `qwen3-tts-vc-2026-01-22` | | Maximize fidelity to one reference voice | `higgs-tts` | | Blend 2-16 weighted reference voices | `higgs-tts` | | Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` | -- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points. +- Qwen (`qwen-audio-3.0-tts-plus`/`-flash`): `voice_prompt` design OR one-reference clone; require at least 15 Unicode code points. +- Qwen3-TTS (`qwen3-tts-vc-2026-01-22`): clone-only, no `voice_prompt` design mode; reference audio is required; accepts any text length. - Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend. - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. @@ -282,8 +284,8 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au ```ts await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "欢迎使用语音合成功能。" }], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: "欢迎使用长度足够的语音合成功能进行试听。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", }, @@ -296,6 +298,14 @@ await client.generate({ { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, ], }); + +await client.generate({ + model: "qwen3-tts-vc-2026-01-22", + content: [ + { type: "text", text: "欢迎使用语音合成功能。" }, + { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, + ], +}); ``` ```ts diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 009b59f..11be478 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/qwen-tts.yaml b/models/qwen-tts.yaml deleted file mode 100644 index 1136fc7..0000000 --- a/models/qwen-tts.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-tts -title: Qwen TTS -description: 'Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/models/qwen3-tts-vc-2026-01-22.yaml b/models/qwen3-tts-vc-2026-01-22.yaml new file mode 100644 index 0000000..de7a702 --- /dev/null +++ b/models/qwen3-tts-vc-2026-01-22.yaml @@ -0,0 +1,32 @@ +schema: neta.generation.model.v1 +model: qwen3-tts-vc-2026-01-22 +title: Qwen3 TTS Voice Clone +description: "Clone-only: requires exactly one reference audio; no voice-design mode. Text: any length. Conflict: N/A, + single mode. Dependency: clone prior generated audio." +adapter: + type: openai.audioSpeech +content: + input: + - type: text + required: true + min: 1 + max: 1 + description: Exactly one non-empty text block to speak. + - type: audio + required: true + min: 1 + max: 1 + sources: + - url + description: "Required reference audio URL to clone. Dependency: use prior generated audio." +examples: + - title: Voice clone + request: + model: qwen3-tts-vc-2026-01-22 + content: + - type: text + text: 这是一次清晰自然的语音合成测试。 + - type: audio + source: + type: url + url: https://example.com/reference.mp3 diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index b174984..2cf32d4 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,8 +8,14 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); -const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const QWEN_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const QWEN_MINIMUM_TEXT_CODE_POINTS = 15; +// qwen3-tts-vc is clone-only -- DashScope's voice-design call shape doesn't +// exist for this model family (see background/tasks/qwen_tts_actor.py's +// module docstring), unlike QWEN_MODELS above, which supports design OR +// clone through the same wire shape. It also has no minimum text length +// (background's worker-side implementation enforces only non-empty). +const QWEN3_TTS_CLONE_MODELS = new Set(["qwen3-tts-vc-2026-01-22"]); const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -92,8 +98,21 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { - throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); + if (Array.from(text.text.trim()).length < QWEN_MINIMUM_TEXT_CODE_POINTS) { + throw new GenerationValidationError( + `${input.declaration.model} requires input of at least ${QWEN_MINIMUM_TEXT_CODE_POINTS} Unicode code points`, + ); + } +} + +function validateQwen3TtsClone(input: ResolvedGenerationRequest, text: TextBlock, audio: AudioBlock[]): void { + validateMetaKeys("request.metadata", input.request.metadata, new Set()); + validateMetaKeys("request.meta", input.request.meta, new Set()); + validateMetaKeys("text content meta", text.meta, new Set()); + for (const block of audio) validateMetaKeys("audio content meta", block.meta, new Set()); + + if (audio.length !== 1) { + throw new GenerationValidationError(`${input.declaration.model} requires exactly one reference audio`); } } @@ -128,6 +147,10 @@ function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void { validateQwen(input, text, audio); return; } + if (QWEN3_TTS_CLONE_MODELS.has(input.declaration.model)) { + validateQwen3TtsClone(input, text, audio); + return; + } if (input.declaration.model === HIGGS_MODEL) { validateHiggs(input, text, audio); return; @@ -154,6 +177,11 @@ function buildPayload(input: ResolvedGenerationRequest): Record return payload; } + if (QWEN3_TTS_CLONE_MODELS.has(input.declaration.model)) { + if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim(); + return payload; + } + const firstAudio = audio[0]; if (audio.length === 1 && firstAudio && (!hasOwnWeight(firstAudio) || firstAudio.meta?.weight === undefined)) { if (firstAudio.source.type === "url") payload.ref_audio = firstAudio.source.url.trim(); diff --git a/src/builtins.ts b/src/builtins.ts index 8ba25a5..d5faf79 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -708,15 +708,10 @@ function geminiImageModel( }; } -function qwenTtsModel( - model: string, - title: string, - description: string, - options: { minimumTextCodePoints?: number } = {}, -): GenerationModelDeclaration { - const text = options.minimumTextCodePoints - ? "这是一段长度足够并且表达清晰自然的语音合成测试文本。" - : "这是一次清晰自然的语音合成测试。"; +const QWEN_AUDIO_3_MINIMUM_TEXT_CODE_POINTS = 15; + +function qwenTtsModel(model: string, title: string, description: string): GenerationModelDeclaration { + const text = "这是一段长度足够并且表达清晰自然的语音合成测试文本。"; return { schema: MODEL_SCHEMA, model, @@ -730,9 +725,7 @@ function qwenTtsModel( required: true, min: 1, max: 1, - description: options.minimumTextCodePoints - ? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.` - : "Exactly one non-empty text block to speak.", + description: `Exactly one non-empty text block to speak, with at least ${QWEN_AUDIO_3_MINIMUM_TEXT_CODE_POINTS} Unicode code points.`, }, { type: "audio", @@ -775,23 +768,62 @@ function qwenTtsModel( }; } +function qwen3TtsCloneModel(model: string, title: string, description: string): GenerationModelDeclaration { + return { + schema: MODEL_SCHEMA, + model, + title, + description, + adapter: { type: "openai.audioSpeech" }, + content: { + input: [ + { + type: "text", + required: true, + min: 1, + max: 1, + description: "Exactly one non-empty text block to speak.", + }, + { + type: "audio", + required: true, + min: 1, + max: 1, + sources: ["url"], + description: "Required reference audio URL to clone. Dependency: use prior generated audio.", + }, + ], + }, + examples: [ + { + title: "Voice clone", + request: { + model, + content: [ + { type: "text", text: "这是一次清晰自然的语音合成测试。" }, + { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, + ], + }, + }, + ], + }; +} + const audioSpeechModels = [ - qwenTtsModel( - "qwen-tts", - "Qwen TTS", - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + qwen3TtsCloneModel( + "qwen3-tts-vc-2026-01-22", + "Qwen3 TTS Voice Clone", + "Clone-only: requires exactly one reference audio; no voice-design mode. Text: any length. Conflict: N/A, single mode. Dependency: clone prior generated audio.", ), qwenTtsModel( "qwen-audio-3.0-tts-plus", "Qwen Audio 3.0 TTS Plus", "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, ), qwenTtsModel( "qwen-audio-3.0-tts-flash", "Qwen Audio 3.0 TTS Flash", "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, ), { schema: MODEL_SCHEMA, diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 236718b..4c53a42 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -11,6 +11,10 @@ import { const REFERENCE_URL = "https://example.com/reference.mp3"; const SECOND_REFERENCE_URL = "https://example.com/reference-2.mp3"; +// qwen-audio-3.0-tts-plus/flash require >=15 Unicode code points regardless +// of design or clone mode -- reused wherever a test needs valid Qwen input +// but doesn't care about its exact content. +const QWEN_VALID_TEXT = "这是一段长度足够并且清晰自然的语音试听文本。"; function routerSuccess( body: Record = {}, @@ -41,13 +45,21 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-tts", - content: [{ type: "text", text: "这是需要朗读的文本。" }], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_VALID_TEXT }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, }; } +function qwen3CloneRequest(overrides: Partial = {}): GenerateRequest { + return { + model: "qwen3-tts-vc-2026-01-22", + content: [{ type: "text", text: "短句也可以。" }, audio()], + ...overrides, + }; +} + function recordingClient(responseFactory: () => Response = () => routerSuccess()) { const calls: Array<{ url: string; init: RequestInit }> = []; const fetchMock = async (url: string | URL | Request, init?: RequestInit) => { @@ -69,11 +81,11 @@ describe("openai.audioSpeech adapter requests", () => { const { client, calls } = recordingClient(() => routerSuccess({}, { headers: { "x-request-id": "request-primary", "x-oneapi-request-id": "request-fallback" } }), ); - const input = " 原样保留的朗读文本。\n"; + const input = ` ${QWEN_VALID_TEXT}\n`; const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -85,7 +97,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", input, metadata: { voice_prompt: voicePrompt }, }); @@ -101,13 +113,13 @@ describe("openai.audioSpeech adapter requests", () => { it("maps Qwen reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_VALID_TEXT }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "短句", + model: "qwen-audio-3.0-tts-plus", + input: QWEN_VALID_TEXT, ref_audio: REFERENCE_URL, }); }); @@ -121,12 +133,27 @@ describe("openai.audioSpeech adapter requests", () => { ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "这是需要朗读的文本。", + model: "qwen-audio-3.0-tts-plus", + input: QWEN_VALID_TEXT, metadata: { voice_prompt: "清晰女声" }, }); }); + it("maps qwen3-tts-vc reference audio and trims only its URL", async () => { + const { client, calls } = recordingClient(); + await client.generate( + qwen3CloneRequest({ + content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + }), + ); + + expect(requestBody(calls[0])).toEqual({ + model: "qwen3-tts-vc-2026-01-22", + input: "短句", + ref_audio: REFERENCE_URL, + }); + }); + it("maps all Higgs reference modes", async () => { const { client, calls } = recordingClient(); @@ -198,8 +225,8 @@ describe("openai.audioSpeech adapter validation", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "有效语音文本" }], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_VALID_TEXT }], meta: { preview_text: "已经失效的文本来源" }, }), ).toThrow("requires one reference audio or meta.voice_prompt"); @@ -233,21 +260,37 @@ describe("openai.audioSpeech adapter validation", () => { }); it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-tts"); - if (!declaration) throw new Error("qwen-tts declaration is unavailable"); + const declaration = getBuiltinGenerationModel("qwen-audio-3.0-tts-plus"); + if (!declaration) throw new Error("qwen-audio-3.0-tts-plus declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-tts audio spec is unavailable"); + if (!audioSpec) throw new Error("qwen-audio-3.0-tts-plus audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); }); + it("enforces qwen3-tts-vc reference limits inside the adapter hook when a declaration is overridden", () => { + const declaration = getBuiltinGenerationModel("qwen3-tts-vc-2026-01-22"); + if (!declaration) throw new Error("qwen3-tts-vc-2026-01-22 declaration is unavailable"); + const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); + if (!audioSpec) throw new Error("qwen3-tts-vc-2026-01-22 audio spec is unavailable"); + audioSpec.max = 2; + const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); + + expect(() => + client.validate({ + model: "qwen3-tts-vc-2026-01-22", + content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], + }), + ).toThrow("requires exactly one reference audio"); + }); + it("enforces Higgs reference limits inside the adapter hook when a declaration is overridden", () => { const declaration = getBuiltinGenerationModel("higgs-tts"); if (!declaration) throw new Error("higgs-tts declaration is unavailable"); @@ -282,13 +325,30 @@ describe("openai.audioSpeech adapter validation", () => { it.each([ { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - { model: "qwen-tts", text: "短" }, - { model: "qwen-tts", text: "a".repeat(40) }, ])("accepts the $model input boundary", ({ model, text }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); }); + it.each(["短", "a".repeat(40)])("accepts qwen3-tts-vc input of any length: %s", (text) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => client.validate(qwen3CloneRequest({ content: [{ type: "text", text }, audio()] }))).not.toThrow(); + }); + + it("rejects qwen3-tts-vc voice_prompt entirely -- clone-only, no voice-design mode", () => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate(qwen3CloneRequest({ meta: { voice_prompt: "声音" } })), + ).toThrow("Unknown request.meta field: voice_prompt"); + }); + + it("rejects qwen3-tts-vc with no reference audio via the declared content schema", () => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate(qwen3CloneRequest({ content: [{ type: "text", text: "文本" }] })), + ).toThrow("Missing required audio content block"); + }); + it.each<{ label: string; request: GenerateRequest }>([ { label: "request typo", request: qwenDesignRequest({ meta: { voice_promt: "拼错" } }) }, { diff --git a/test/config.test.ts b/test/config.test.ts index 408c951..a5ecfb7 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,7 +69,7 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-tts", + "qwen3-tts-vc-2026-01-22", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash", "suno_cover_chirp_v5", @@ -207,6 +207,21 @@ describe("config", () => { expect(() => client.stringifyModelConfig("suno_music")).toThrow("Generation model is unavailable: suno_music"); }); + it("does not expose the retired qwen-tts model", () => { + // qwen-tts is DashScope's own retirement (2026-10-10); qwen3-tts-vc-2026-01-22 + // replaces it as the long-term-supported clone model. qwen-audio-3.0-tts-plus/flash + // are unaffected by this migration and stay published as-is. + const client = createGenerationClient({ apiKey: "test" }); + expect(client.getModel("qwen-tts")).toBeNull(); + expect(() => + client.validate({ + model: "qwen-tts", + content: [{ type: "text", text: "这是一段有效的语音试听文本。" }], + }), + ).toThrow("Generation model is unavailable: qwen-tts"); + expect(() => client.stringifyModelConfig("qwen-tts")).toThrow("Generation model is unavailable: qwen-tts"); + }); + it("keeps infrastructure names out of model descriptions", () => { const client = createGenerationClient({ apiKey: "test" }); for (const model of client.listModels()) { @@ -216,7 +231,7 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const qwenModels = ["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; for (const model of qwenModels) { const declaration = client.getModel(model); @@ -236,13 +251,19 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const qwen = client.getModel("qwen-tts"); - expect(qwen?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", + const qwen3Clone = client.getModel("qwen3-tts-vc-2026-01-22"); + expect(qwen3Clone?.description).toBe( + "Clone-only: requires exactly one reference audio; no voice-design mode. Text: any length. Conflict: N/A, single mode. Dependency: clone prior generated audio.", ); - expect(qwen?.content.input.find((input) => input.type === "text")?.description).not.toContain( + expect(qwen3Clone?.content.input.find((input) => input.type === "text")?.description).not.toContain( "Unicode code points", ); + expect(qwen3Clone?.content.input.find((input) => input.type === "audio")?.required).toBe(true); + expect(qwen3Clone?.meta?.fields?.voice_prompt).toBeUndefined(); + expect(qwen3Clone?.examples?.map((example) => example.title)).toEqual(["Voice clone"]); + expect(JSON.parse(client.stringifyModelConfig("qwen3-tts-vc-2026-01-22", { format: "json" }))).toEqual( + qwen3Clone, + ); const plus = client.getModel("qwen-audio-3.0-tts-plus"); expect(plus?.description).toBe( @@ -287,7 +308,8 @@ describe("config", () => { expect(Object.keys(packageJson.exports ?? {})).toEqual([".", "./models"]); const readme = await readFile(join(process.cwd(), "README.md"), "utf8"); expect(readme).not.toContain("@neta-art/generation/models/"); - expect(readme).toContain("Qwen: `voice_prompt` design OR one-reference clone"); + expect(readme).toContain("Qwen (`qwen-audio-3.0-tts-plus`/`-flash`): `voice_prompt` design OR one-reference clone"); + expect(readme).toContain("Qwen3-TTS (`qwen3-tts-vc-2026-01-22`): clone-only, no `voice_prompt` design mode"); expect(readme).toContain("Higgs: delegated default voice, high-fidelity one-reference clone"); expect(readme).toContain("Conflict: reference + redesign requires user choice before generation"); expect(readme).toContain("Blend: all references, full text, one request"); @@ -517,7 +539,7 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of ["qwen3-tts-vc-2026-01-22", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index 4878eb1..18044b3 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -47,18 +47,18 @@ liveDescribe("audio speech live router smoke", () => { const runId = `${Date.now()}`; const cases: Array<{ name: string; request: GenerateRequest }> = [ { - name: "qwen voice design", + name: "qwen audio 3 voice design", request: { - model: "qwen-tts", - content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], + model: "qwen-audio-3.0-tts-plus", + content: [text(`这是增强版本设计音色端到端测试文本,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { - name: "qwen voice clone", + name: "qwen3 tts voice clone", request: { - model: "qwen-tts", - content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], + model: "qwen3-tts-vc-2026-01-22", + content: [text(`这是长期支持模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, {