From 29355d328218e01c5a7e26c55487dc1cdbda69d7 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Mon, 28 Sep 2026 11:18:58 +0800 Subject: [PATCH 1/6] feat!: retire qwen-tts, no replacement id Alibaba retires qwen-tts on 2026-10-10 (https://www.alibabacloud.com/en/notice/model_studio_notice_of_retirement_for_selected_legacy_models_7d9). Pure removal: qwen-audio-3.0-tts-plus/flash already cover the same design+clone shape and are not retiring, so nothing new is added in its place. The actual DashScope-side compatibility mapping (qwen-family target models collapsing onto qwen-audio-3.1-tts-flash) lives in background's qwen_tts_actor.py, not in this SDK -- callers still bypassing the SDK with the old wire model name are covered there regardless. --- examples/text-to-speech.ts | 2 +- models/qwen-tts.yaml | 44 ----------------------------- src/adapters/audio-speech.ts | 5 ++-- src/builtins.ts | 5 ---- test/adapters/audio-speech.test.ts | 43 ++++++++++++++-------------- test/config.test.ts | 13 ++------- test/live/audio-speech-live.test.ts | 4 +-- 7 files changed, 28 insertions(+), 88 deletions(-) delete mode 100644 models/qwen-tts.yaml diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 009b59f..11be478 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/qwen-tts.yaml b/models/qwen-tts.yaml deleted file mode 100644 index 1136fc7..0000000 --- a/models/qwen-tts.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-tts -title: Qwen TTS -description: 'Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index b174984..8573bc8 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,8 +8,7 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); -const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const QWEN_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -92,7 +91,7 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { + if (QWEN_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); } } diff --git a/src/builtins.ts b/src/builtins.ts index 8ba25a5..bcc1ab0 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -776,11 +776,6 @@ function qwenTtsModel( } const audioSpeechModels = [ - qwenTtsModel( - "qwen-tts", - "Qwen TTS", - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ), qwenTtsModel( "qwen-audio-3.0-tts-plus", "Qwen Audio 3.0 TTS Plus", diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 236718b..18df021 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -11,6 +11,7 @@ import { const REFERENCE_URL = "https://example.com/reference.mp3"; const SECOND_REFERENCE_URL = "https://example.com/reference-2.mp3"; +const QWEN_TEXT = "这是用于语音合成测试的一整句中文示例文本。"; function routerSuccess( body: Record = {}, @@ -41,8 +42,8 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-tts", - content: [{ type: "text", text: "这是需要朗读的文本。" }], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_TEXT }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, }; @@ -69,11 +70,11 @@ describe("openai.audioSpeech adapter requests", () => { const { client, calls } = recordingClient(() => routerSuccess({}, { headers: { "x-request-id": "request-primary", "x-oneapi-request-id": "request-fallback" } }), ); - const input = " 原样保留的朗读文本。\n"; + const input = " 原样保留的朗读文本内容,不应被修改。\n"; const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -85,7 +86,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", input, metadata: { voice_prompt: voicePrompt }, }); @@ -101,13 +102,13 @@ describe("openai.audioSpeech adapter requests", () => { it("maps Qwen reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_TEXT }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "短句", + model: "qwen-audio-3.0-tts-plus", + input: QWEN_TEXT, ref_audio: REFERENCE_URL, }); }); @@ -121,8 +122,8 @@ describe("openai.audioSpeech adapter requests", () => { ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "这是需要朗读的文本。", + model: "qwen-audio-3.0-tts-plus", + input: QWEN_TEXT, metadata: { voice_prompt: "清晰女声" }, }); }); @@ -198,8 +199,8 @@ describe("openai.audioSpeech adapter validation", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "有效语音文本" }], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_TEXT }], meta: { preview_text: "已经失效的文本来源" }, }), ).toThrow("requires one reference audio or meta.voice_prompt"); @@ -213,7 +214,7 @@ describe("openai.audioSpeech adapter validation", () => { { name: "both voice sources", request: qwenDesignRequest({ - content: [{ type: "text", text: "文本" }, audio()], + content: [{ type: "text", text: QWEN_TEXT }, audio()], }), }, { @@ -223,7 +224,7 @@ describe("openai.audioSpeech adapter validation", () => { { name: "audio weight", request: qwenDesignRequest({ - content: [{ type: "text", text: "文本" }, audio(REFERENCE_URL, { weight: 1 })], + content: [{ type: "text", text: QWEN_TEXT }, audio(REFERENCE_URL, { weight: 1 })], meta: {}, }), }, @@ -233,17 +234,17 @@ describe("openai.audioSpeech adapter validation", () => { }); it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-tts"); - if (!declaration) throw new Error("qwen-tts declaration is unavailable"); + const declaration = getBuiltinGenerationModel("qwen-audio-3.0-tts-plus"); + if (!declaration) throw new Error("qwen-audio-3.0-tts-plus declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-tts audio spec is unavailable"); + if (!audioSpec) throw new Error("qwen-audio-3.0-tts-plus audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], + model: "qwen-audio-3.0-tts-plus", + content: [{ type: "text", text: QWEN_TEXT }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); }); @@ -282,8 +283,6 @@ describe("openai.audioSpeech adapter validation", () => { it.each([ { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - { model: "qwen-tts", text: "短" }, - { model: "qwen-tts", text: "a".repeat(40) }, ])("accepts the $model input boundary", ({ model, text }) => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); diff --git a/test/config.test.ts b/test/config.test.ts index 408c951..0c6d3ea 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,7 +69,6 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash", "suno_cover_chirp_v5", @@ -216,7 +215,7 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const qwenModels = ["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; for (const model of qwenModels) { const declaration = client.getModel(model); @@ -236,14 +235,6 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const qwen = client.getModel("qwen-tts"); - expect(qwen?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(qwen?.content.input.find((input) => input.type === "text")?.description).not.toContain( - "Unicode code points", - ); - const plus = client.getModel("qwen-audio-3.0-tts-plus"); expect(plus?.description).toBe( "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", @@ -517,7 +508,7 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of ["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index 4878eb1..d39e027 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -49,7 +49,7 @@ liveDescribe("audio speech live router smoke", () => { { name: "qwen voice design", request: { - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, @@ -57,7 +57,7 @@ liveDescribe("audio speech live router smoke", () => { { name: "qwen voice clone", request: { - model: "qwen-tts", + model: "qwen-audio-3.0-tts-plus", content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, From 82317bf2c17f4e8c47568b4e777ceba2d188b03e Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Mon, 28 Sep 2026 11:49:57 +0800 Subject: [PATCH 2/6] feat!: drop qwen-audio-3.0-tts-plus, move flash to qwen-audio-3.1-tts-flash Owner decision 2026-09-28: collapse the SDK's Qwen offering down to the one model background's compat map actually terminates on. There is no qwen-audio-3.1-tts-plus (Alibaba only ships a flash tier at 3.1), so plus has no replacement and is removed outright rather than staying on 3.0 as a orphaned tier. --- README.md | 18 +++--- examples/text-to-speech.ts | 2 +- models/qwen-audio-3.0-tts-plus.yaml | 44 --------------- ...ash.yaml => qwen-audio-3.1-tts-flash.yaml} | 15 ++--- src/adapters/audio-speech.ts | 2 +- src/builtins.ts | 10 +--- test/adapters/audio-speech.test.ts | 56 ++++++++++--------- test/config.test.ts | 17 ++---- test/live/audio-speech-live.test.ts | 22 ++------ 9 files changed, 57 insertions(+), 129 deletions(-) delete mode 100644 models/qwen-audio-3.0-tts-plus.yaml rename models/{qwen-audio-3.0-tts-flash.yaml => qwen-audio-3.1-tts-flash.yaml} (66%) diff --git a/README.md b/README.md index d76b214..689cb9d 100644 --- a/README.md +++ b/README.md @@ -62,7 +62,7 @@ Agents and external tools should inspect a model declaration before constructing import { createGenerationClient } from "@neta-art/generation"; const discoveryClient = createGenerationClient(); -const declaration = discoveryClient.getModel("qwen-tts"); +const declaration = discoveryClient.getModel("qwen-audio-3.1-tts-flash"); if (!declaration) throw new Error("Model is unavailable"); console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" })); @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI: ```bash neta-generation models list -neta-generation models export qwen-tts --out ./qwen-tts.yaml +neta-generation models export qwen-audio-3.1-tts-flash --out ./qwen-audio-3.1-tts-flash.yaml neta-generation models export-all --out ./models ``` @@ -170,9 +170,7 @@ const client = createGenerationClient({ - `gpt-image-2` - `z-image-turbo` - `qwen-image-edit` -- `qwen-tts` -- `qwen-audio-3.0-tts-plus` -- `qwen-audio-3.0-tts-flash` +- `qwen-audio-3.1-tts-flash` - `higgs-tts` - `gemini-3.1-flash-image-preview` - `kling-text-to-video` @@ -268,12 +266,12 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au | Requirement | Model choice | | --- | --- | -| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default | +| Create a voice from a text-only description, without reference audio | `qwen-audio-3.1-tts-flash` | | Maximize fidelity to one reference voice | `higgs-tts` | | Blend 2-16 weighted reference voices | `higgs-tts` | | Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` | -- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points. +- Qwen: `voice_prompt` design OR one-reference clone via `qwen-audio-3.1-tts-flash`; requires at least 15 Unicode code points. - Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend. - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. @@ -282,15 +280,15 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au ```ts await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "欢迎使用语音合成功能。" }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", }, }); await client.generate({ - model: "qwen-audio-3.0-tts-flash", + model: "qwen-audio-3.1-tts-flash", content: [ { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 11be478..db30b94 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/qwen-audio-3.0-tts-plus.yaml b/models/qwen-audio-3.0-tts-plus.yaml deleted file mode 100644 index d6bf7fa..0000000 --- a/models/qwen-audio-3.0-tts-plus.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-audio-3.0-tts-plus -title: Qwen Audio 3.0 TTS Plus -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-audio-3.0-tts-plus - content: - - type: text - text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-audio-3.0-tts-plus - content: - - type: text - text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/models/qwen-audio-3.0-tts-flash.yaml b/models/qwen-audio-3.1-tts-flash.yaml similarity index 66% rename from models/qwen-audio-3.0-tts-flash.yaml rename to models/qwen-audio-3.1-tts-flash.yaml index 3d111b9..10233cb 100644 --- a/models/qwen-audio-3.0-tts-flash.yaml +++ b/models/qwen-audio-3.1-tts-flash.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 -model: qwen-audio-3.0-tts-flash -title: Qwen Audio 3.0 TTS Flash -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +model: qwen-audio-3.1-tts-flash +title: Qwen Audio 3.1 TTS Flash +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; + never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -16,17 +17,17 @@ content: max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: - model: qwen-audio-3.0-tts-flash + model: qwen-audio-3.1-tts-flash content: - type: text text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 @@ -34,7 +35,7 @@ examples: voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - title: Voice clone request: - model: qwen-audio-3.0-tts-flash + model: qwen-audio-3.1-tts-flash content: - type: text text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index 8573bc8..a0ca70b 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,7 +8,7 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const QWEN_MODELS = new Set(["qwen-audio-3.1-tts-flash"]); const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; diff --git a/src/builtins.ts b/src/builtins.ts index bcc1ab0..bdbc08b 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -777,14 +777,8 @@ function qwenTtsModel( const audioSpeechModels = [ qwenTtsModel( - "qwen-audio-3.0-tts-plus", - "Qwen Audio 3.0 TTS Plus", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, - ), - qwenTtsModel( - "qwen-audio-3.0-tts-flash", - "Qwen Audio 3.0 TTS Flash", + "qwen-audio-3.1-tts-flash", + "Qwen Audio 3.1 TTS Flash", "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", { minimumTextCodePoints: 15 }, ), diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 18df021..bcaa6b2 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -42,7 +42,7 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: QWEN_TEXT }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, @@ -74,7 +74,7 @@ describe("openai.audioSpeech adapter requests", () => { const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -86,7 +86,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", input, metadata: { voice_prompt: voicePrompt }, }); @@ -102,12 +102,12 @@ describe("openai.audioSpeech adapter requests", () => { it("maps Qwen reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: QWEN_TEXT }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", input: QWEN_TEXT, ref_audio: REFERENCE_URL, }); @@ -122,7 +122,7 @@ describe("openai.audioSpeech adapter requests", () => { ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", input: QWEN_TEXT, metadata: { voice_prompt: "清晰女声" }, }); @@ -199,7 +199,7 @@ describe("openai.audioSpeech adapter validation", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: QWEN_TEXT }], meta: { preview_text: "已经失效的文本来源" }, }), @@ -234,16 +234,16 @@ describe("openai.audioSpeech adapter validation", () => { }); it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-audio-3.0-tts-plus"); - if (!declaration) throw new Error("qwen-audio-3.0-tts-plus declaration is unavailable"); + const declaration = getBuiltinGenerationModel("qwen-audio-3.1-tts-flash"); + if (!declaration) throw new Error("qwen-audio-3.1-tts-flash declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-audio-3.0-tts-plus audio spec is unavailable"); + if (!audioSpec) throw new Error("qwen-audio-3.1-tts-flash audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-audio-3.0-tts-plus", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: QWEN_TEXT }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); @@ -270,23 +270,25 @@ describe("openai.audioSpeech adapter validation", () => { ).toThrow("supports at most 16 references"); }); - it.each([ - { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(14) }, - { model: "qwen-audio-3.0-tts-flash", text: "😀".repeat(14) }, - ])("rejects $model input below 15 Unicode code points", ({ model, text }) => { - const client = createGenerationClient({ apiKey: "key" }); - expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).toThrow( - "requires input of at least 15 Unicode code points", - ); - }); + it.each([{ text: "a".repeat(14) }, { text: "😀".repeat(14) }])( + "rejects qwen-audio-3.1-tts-flash input below 15 Unicode code points ($text)", + ({ text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).toThrow("requires input of at least 15 Unicode code points"); + }, + ); - it.each([ - { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, - { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - ])("accepts the $model input boundary", ({ model, text }) => { - const client = createGenerationClient({ apiKey: "key" }); - expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); - }); + it.each([{ text: "a".repeat(15) }, { text: ` ${"😀".repeat(15)} ` }])( + "accepts the qwen-audio-3.1-tts-flash input boundary ($text)", + ({ text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).not.toThrow(); + }, + ); it.each<{ label: string; request: GenerateRequest }>([ { label: "request typo", request: qwenDesignRequest({ meta: { voice_promt: "拼错" } }) }, diff --git a/test/config.test.ts b/test/config.test.ts index 0c6d3ea..a3535c8 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,8 +69,7 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-audio-3.0-tts-plus", - "qwen-audio-3.0-tts-flash", + "qwen-audio-3.1-tts-flash", "suno_cover_chirp_v5", "suno_image_to_song_chirp_v5", "suno_infill_chirp_v5", @@ -215,7 +214,7 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const qwenModels = ["qwen-audio-3.1-tts-flash"]; for (const model of qwenModels) { const declaration = client.getModel(model); @@ -235,15 +234,7 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const plus = client.getModel("qwen-audio-3.0-tts-plus"); - expect(plus?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(plus?.content.input.find((input) => input.type === "text")?.description).toContain( - "at least 15 Unicode code points", - ); - - const flash = client.getModel("qwen-audio-3.0-tts-flash"); + const flash = client.getModel("qwen-audio-3.1-tts-flash"); expect(flash?.description).toBe( "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); @@ -508,7 +499,7 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of ["qwen-audio-3.1-tts-flash"]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index d39e027..bea40d3 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -49,30 +49,16 @@ liveDescribe("audio speech live router smoke", () => { { name: "qwen voice design", request: { - model: "qwen-audio-3.0-tts-plus", - content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], + model: "qwen-audio-3.1-tts-flash", + content: [text(`这是设计音色端到端测试,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { name: "qwen voice clone", request: { - model: "qwen-audio-3.0-tts-plus", - content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], - }, - }, - { - name: "qwen audio 3 plus", - request: { - model: "qwen-audio-3.0-tts-plus", - content: [text(`这是增强版本语音合成端到端测试文本,运行编号${runId}。`), audio(REFERENCE_A)], - }, - }, - { - name: "qwen audio 3 flash", - request: { - model: "qwen-audio-3.0-tts-flash", - content: [text(`这是快速版本语音合成端到端测试文本,运行编号${runId}。`), audio(REFERENCE_A)], + model: "qwen-audio-3.1-tts-flash", + content: [text(`这是克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, { From 1447adc7c686ac0ad883dd15843122f2d9359efb Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Mon, 28 Sep 2026 12:40:38 +0800 Subject: [PATCH 3/6] refactor: drop dead Qwen-tier scaffolding now that only one model remains Review findings (Linus-persona pass on this PR): - QWEN_MODELS was a one-element Set; the length check inside validateQwen re-checked membership in a set it could only be called from, making the condition a tautology. Collapsed to a single QWEN_MODEL string constant and dropped the redundant check. - qwenTtsModel(...) was a factory built for 2-3 Qwen tiers with only one caller left; inlined to a literal declaration object matching the neighboring higgs-tts entry's style. Re-exported YAML is byte-identical to the prior factory output (verified via CLI export diff). - README's "no declared Qwen quality/cost ranking" bullet only made sense when there were multiple tiers to rank; replaced with the actual capability loss this migration introduces (no voice-design path for sub-15-code-point text now that qwen-tts is gone). - Documented the cross-repo constraint inline: background's qwen_tts_actor collapses every qwen*-branded target_model onto this one id, so adding another Qwen tier here requires changing that rule first. --- README.md | 2 +- src/adapters/audio-speech.ts | 11 ++++++---- src/builtins.ts | 42 +++++++++++------------------------- test/config.test.ts | 2 +- 4 files changed, 21 insertions(+), 36 deletions(-) diff --git a/README.md b/README.md index 689cb9d..a458e0e 100644 --- a/README.md +++ b/README.md @@ -276,7 +276,7 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. - Dependency: clone prior generated audio. -- Ranking: no declared Qwen quality, latency, or cost order. +- Short text: input under 15 Unicode code points has no voice-design path on Qwen; ask the user to lengthen it, or use `higgs-tts` with a default/reference voice instead. ```ts await client.generate({ diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index a0ca70b..a3a42ad 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,7 +8,10 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-audio-3.1-tts-flash"]); +// background's qwen_tts_actor rewrites every qwen*-branded target_model to this +// one id -- adding another Qwen tier here requires changing that rule first, +// or it silently gets collapsed onto this model regardless of what's declared. +const QWEN_MODEL = "qwen-audio-3.1-tts-flash"; const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -91,7 +94,7 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { + if (Array.from(text.text.trim()).length < 15) { throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); } } @@ -123,7 +126,7 @@ function validateHiggs(input: ResolvedGenerationRequest, text: TextBlock, audio: function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void { const { text, audio } = validateCommonContent(input); - if (QWEN_MODELS.has(input.declaration.model)) { + if (input.declaration.model === QWEN_MODEL) { validateQwen(input, text, audio); return; } @@ -147,7 +150,7 @@ function buildPayload(input: ResolvedGenerationRequest): Record input: text.text, }; - if (QWEN_MODELS.has(input.declaration.model)) { + if (input.declaration.model === QWEN_MODEL) { if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim(); else payload.metadata = { voice_prompt: input.meta.voice_prompt }; return payload; diff --git a/src/builtins.ts b/src/builtins.ts index bdbc08b..0eec7f9 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -708,20 +708,13 @@ function geminiImageModel( }; } -function qwenTtsModel( - model: string, - title: string, - description: string, - options: { minimumTextCodePoints?: number } = {}, -): GenerationModelDeclaration { - const text = options.minimumTextCodePoints - ? "这是一段长度足够并且表达清晰自然的语音合成测试文本。" - : "这是一次清晰自然的语音合成测试。"; - return { +const audioSpeechModels = [ + { schema: MODEL_SCHEMA, - model, - title, - description, + model: "qwen-audio-3.1-tts-flash", + title: "Qwen Audio 3.1 TTS Flash", + description: + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", adapter: { type: "openai.audioSpeech" }, content: { input: [ @@ -730,9 +723,7 @@ function qwenTtsModel( required: true, min: 1, max: 1, - description: options.minimumTextCodePoints - ? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.` - : "Exactly one non-empty text block to speak.", + description: "Exactly one non-empty text block to speak, with at least 15 Unicode code points.", }, { type: "audio", @@ -756,32 +747,23 @@ function qwenTtsModel( { title: "Voice design", request: { - model, - content: [{ type: "text", text }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { title: "Voice clone", request: { - model, + model: "qwen-audio-3.1-tts-flash", content: [ - { type: "text", text }, + { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, ], }, }, ], - }; -} - -const audioSpeechModels = [ - qwenTtsModel( - "qwen-audio-3.1-tts-flash", - "Qwen Audio 3.1 TTS Flash", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, - ), + }, { schema: MODEL_SCHEMA, model: "higgs-tts", diff --git a/test/config.test.ts b/test/config.test.ts index a3535c8..7a1d2ad 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -274,7 +274,7 @@ describe("config", () => { expect(readme).toContain("Conflict: reference + redesign requires user choice before generation"); expect(readme).toContain("Blend: all references, full text, one request"); expect(readme).toContain("Dependency: clone prior generated audio"); - expect(readme).toContain("Ranking: no declared Qwen quality, latency, or cost order"); + expect(readme).toContain("Short text: input under 15 Unicode code points has no voice-design path on Qwen"); expect(readme).not.toMatch(/quality prioritized over latency|latency prioritized over maximum quality/); }); From e4a464629e95d944d8d9bdb78d86c2b400d68465 Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Mon, 28 Sep 2026 12:45:27 +0800 Subject: [PATCH 4/6] chore: bump version to 0.2.0 Two feat! commits landed on this branch (drop qwen-tts/qwen-audio-3.0-tts-plus, rename flash to qwen-audio-3.1-tts-flash) -- minor bump per this package's pre-1.0 convention for breaking changes. --- package.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/package.json b/package.json index 4895da2..2a7f219 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@neta-art/generation", - "version": "0.1.31", + "version": "0.2.0", "description": "A lightweight multimodal generation SDK with built-in model presets and adapter-based provider calls.", "keywords": [ "ai", From 817a662e83eaf7442847a1ecb37118af748f7c1e Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Mon, 28 Sep 2026 18:05:22 +0800 Subject: [PATCH 5/6] style: trim comment --- src/adapters/audio-speech.ts | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index a3a42ad..df8bd61 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,9 +8,7 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -// background's qwen_tts_actor rewrites every qwen*-branded target_model to this -// one id -- adding another Qwen tier here requires changing that rule first, -// or it silently gets collapsed onto this model regardless of what's declared. +// background's qwen_tts_actor rewrites every qwen* target_model onto this id. const QWEN_MODEL = "qwen-audio-3.1-tts-flash"; const HIGGS_MODEL = "higgs-tts"; From 7e29d3bc1275a19f853174f465819421c5e00b9d Mon Sep 17 00:00:00 2001 From: LingXuanYin <3546599908@qq.com> Date: Tue, 29 Sep 2026 11:24:12 +0800 Subject: [PATCH 6/6] style: format the boundary tests, drop misleading comment --- src/adapters/audio-speech.ts | 1 - test/adapters/audio-speech.test.ts | 36 +++++++++++++++--------------- 2 files changed, 18 insertions(+), 19 deletions(-) diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index df8bd61..4333733 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,7 +8,6 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -// background's qwen_tts_actor rewrites every qwen* target_model onto this id. const QWEN_MODEL = "qwen-audio-3.1-tts-flash"; const HIGGS_MODEL = "higgs-tts"; diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index bcaa6b2..cef1606 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -270,25 +270,25 @@ describe("openai.audioSpeech adapter validation", () => { ).toThrow("supports at most 16 references"); }); - it.each([{ text: "a".repeat(14) }, { text: "😀".repeat(14) }])( - "rejects qwen-audio-3.1-tts-flash input below 15 Unicode code points ($text)", - ({ text }) => { - const client = createGenerationClient({ apiKey: "key" }); - expect(() => - client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), - ).toThrow("requires input of at least 15 Unicode code points"); - }, - ); + it.each([ + { text: "a".repeat(14) }, + { text: "😀".repeat(14) }, + ])("rejects qwen-audio-3.1-tts-flash input below 15 Unicode code points ($text)", ({ text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).toThrow("requires input of at least 15 Unicode code points"); + }); - it.each([{ text: "a".repeat(15) }, { text: ` ${"😀".repeat(15)} ` }])( - "accepts the qwen-audio-3.1-tts-flash input boundary ($text)", - ({ text }) => { - const client = createGenerationClient({ apiKey: "key" }); - expect(() => - client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), - ).not.toThrow(); - }, - ); + it.each([ + { text: "a".repeat(15) }, + { text: ` ${"😀".repeat(15)} ` }, + ])("accepts the qwen-audio-3.1-tts-flash input boundary ($text)", ({ text }) => { + const client = createGenerationClient({ apiKey: "key" }); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).not.toThrow(); + }); it.each<{ label: string; request: GenerateRequest }>([ { label: "request typo", request: qwenDesignRequest({ meta: { voice_promt: "拼错" } }) },