diff --git a/README.md b/README.md index d76b214..a458e0e 100644 --- a/README.md +++ b/README.md @@ -62,7 +62,7 @@ Agents and external tools should inspect a model declaration before constructing import { createGenerationClient } from "@neta-art/generation"; const discoveryClient = createGenerationClient(); -const declaration = discoveryClient.getModel("qwen-tts"); +const declaration = discoveryClient.getModel("qwen-audio-3.1-tts-flash"); if (!declaration) throw new Error("Model is unavailable"); console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" })); @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI: ```bash neta-generation models list -neta-generation models export qwen-tts --out ./qwen-tts.yaml +neta-generation models export qwen-audio-3.1-tts-flash --out ./qwen-audio-3.1-tts-flash.yaml neta-generation models export-all --out ./models ``` @@ -170,9 +170,7 @@ const client = createGenerationClient({ - `gpt-image-2` - `z-image-turbo` - `qwen-image-edit` -- `qwen-tts` -- `qwen-audio-3.0-tts-plus` -- `qwen-audio-3.0-tts-flash` +- `qwen-audio-3.1-tts-flash` - `higgs-tts` - `gemini-3.1-flash-image-preview` - `kling-text-to-video` @@ -268,29 +266,29 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au | Requirement | Model choice | | --- | --- | -| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default | +| Create a voice from a text-only description, without reference audio | `qwen-audio-3.1-tts-flash` | | Maximize fidelity to one reference voice | `higgs-tts` | | Blend 2-16 weighted reference voices | `higgs-tts` | | Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` | -- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points. +- Qwen: `voice_prompt` design OR one-reference clone via `qwen-audio-3.1-tts-flash`; requires at least 15 Unicode code points. - Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend. - Conflict: reference + redesign requires user choice before generation. - Blend: all references, full text, one request. - Dependency: clone prior generated audio. -- Ranking: no declared Qwen quality, latency, or cost order. +- Short text: input under 15 Unicode code points has no voice-design path on Qwen; ask the user to lengthen it, or use `higgs-tts` with a default/reference voice instead. ```ts await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "欢迎使用语音合成功能。" }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", }, }); await client.generate({ - model: "qwen-audio-3.0-tts-flash", + model: "qwen-audio-3.1-tts-flash", content: [ { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, diff --git a/examples/text-to-speech.ts b/examples/text-to-speech.ts index 009b59f..db30b94 100644 --- a/examples/text-to-speech.ts +++ b/examples/text-to-speech.ts @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY"); const client = createGenerationClient({ apiKey }); const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }], meta: { voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中", diff --git a/models/qwen-audio-3.0-tts-plus.yaml b/models/qwen-audio-3.0-tts-plus.yaml deleted file mode 100644 index d6bf7fa..0000000 --- a/models/qwen-audio-3.0-tts-plus.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-audio-3.0-tts-plus -title: Qwen Audio 3.0 TTS Plus -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak, with at least 15 Unicode code points. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-audio-3.0-tts-plus - content: - - type: text - text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-audio-3.0-tts-plus - content: - - type: text - text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/models/qwen-audio-3.0-tts-flash.yaml b/models/qwen-audio-3.1-tts-flash.yaml similarity index 66% rename from models/qwen-audio-3.0-tts-flash.yaml rename to models/qwen-audio-3.1-tts-flash.yaml index 3d111b9..10233cb 100644 --- a/models/qwen-audio-3.0-tts-flash.yaml +++ b/models/qwen-audio-3.1-tts-flash.yaml @@ -1,7 +1,8 @@ schema: neta.generation.model.v1 -model: qwen-audio-3.0-tts-flash -title: Qwen Audio 3.0 TTS Flash -description: 'Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' +model: qwen-audio-3.1-tts-flash +title: Qwen Audio 3.1 TTS Flash +description: "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; + never combine/reinterpret. Dependency: clone prior generated audio." adapter: type: openai.audioSpeech content: @@ -16,17 +17,17 @@ content: max: 1 sources: - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' + description: "Clone: one URL; no voice_prompt. Dependency: use prior generated audio." meta: fields: voice_prompt: type: string optional: true - description: 'Design: custom voice text; no reference audio.' + description: "Design: custom voice text; no reference audio." examples: - title: Voice design request: - model: qwen-audio-3.0-tts-flash + model: qwen-audio-3.1-tts-flash content: - type: text text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 @@ -34,7 +35,7 @@ examples: voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - title: Voice clone request: - model: qwen-audio-3.0-tts-flash + model: qwen-audio-3.1-tts-flash content: - type: text text: 这是一段长度足够并且表达清晰自然的语音合成测试文本。 diff --git a/models/qwen-tts.yaml b/models/qwen-tts.yaml deleted file mode 100644 index 1136fc7..0000000 --- a/models/qwen-tts.yaml +++ /dev/null @@ -1,44 +0,0 @@ -schema: neta.generation.model.v1 -model: qwen-tts -title: Qwen TTS -description: 'Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.' -adapter: - type: openai.audioSpeech -content: - input: - - type: text - required: true - min: 1 - max: 1 - description: Exactly one non-empty text block to speak. - - type: audio - required: false - max: 1 - sources: - - url - description: 'Clone: one URL; no voice_prompt. Dependency: use prior generated audio.' -meta: - fields: - voice_prompt: - type: string - optional: true - description: 'Design: custom voice text; no reference audio.' -examples: - - title: Voice design - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - meta: - voice_prompt: 一位沉稳干练的男性播音员声音,吐字清晰有力 - - title: Voice clone - request: - model: qwen-tts - content: - - type: text - text: 这是一次清晰自然的语音合成测试。 - - type: audio - source: - type: url - url: https://example.com/reference.mp3 diff --git a/package.json b/package.json index 4895da2..2a7f219 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@neta-art/generation", - "version": "0.1.31", + "version": "0.2.0", "description": "A lightweight multimodal generation SDK with built-in model presets and adapter-based provider calls.", "keywords": [ "ai", diff --git a/src/adapters/audio-speech.ts b/src/adapters/audio-speech.ts index b174984..4333733 100644 --- a/src/adapters/audio-speech.ts +++ b/src/adapters/audio-speech.ts @@ -8,8 +8,7 @@ import type { } from "../types.js"; const REQUEST_TIMEOUT_MS = 210_000; -const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); -const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]); +const QWEN_MODEL = "qwen-audio-3.1-tts-flash"; const HIGGS_MODEL = "higgs-tts"; type TextBlock = Extract; @@ -92,7 +91,7 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio: ); } - if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) { + if (Array.from(text.text.trim()).length < 15) { throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`); } } @@ -124,7 +123,7 @@ function validateHiggs(input: ResolvedGenerationRequest, text: TextBlock, audio: function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void { const { text, audio } = validateCommonContent(input); - if (QWEN_MODELS.has(input.declaration.model)) { + if (input.declaration.model === QWEN_MODEL) { validateQwen(input, text, audio); return; } @@ -148,7 +147,7 @@ function buildPayload(input: ResolvedGenerationRequest): Record input: text.text, }; - if (QWEN_MODELS.has(input.declaration.model)) { + if (input.declaration.model === QWEN_MODEL) { if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim(); else payload.metadata = { voice_prompt: input.meta.voice_prompt }; return payload; diff --git a/src/builtins.ts b/src/builtins.ts index 8ba25a5..0eec7f9 100644 --- a/src/builtins.ts +++ b/src/builtins.ts @@ -708,20 +708,13 @@ function geminiImageModel( }; } -function qwenTtsModel( - model: string, - title: string, - description: string, - options: { minimumTextCodePoints?: number } = {}, -): GenerationModelDeclaration { - const text = options.minimumTextCodePoints - ? "这是一段长度足够并且表达清晰自然的语音合成测试文本。" - : "这是一次清晰自然的语音合成测试。"; - return { +const audioSpeechModels = [ + { schema: MODEL_SCHEMA, - model, - title, - description, + model: "qwen-audio-3.1-tts-flash", + title: "Qwen Audio 3.1 TTS Flash", + description: + "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", adapter: { type: "openai.audioSpeech" }, content: { input: [ @@ -730,9 +723,7 @@ function qwenTtsModel( required: true, min: 1, max: 1, - description: options.minimumTextCodePoints - ? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.` - : "Exactly one non-empty text block to speak.", + description: "Exactly one non-empty text block to speak, with at least 15 Unicode code points.", }, { type: "audio", @@ -756,43 +747,23 @@ function qwenTtsModel( { title: "Voice design", request: { - model, - content: [{ type: "text", text }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { title: "Voice clone", request: { - model, + model: "qwen-audio-3.1-tts-flash", content: [ - { type: "text", text }, + { type: "text", text: "这是一段长度足够并且表达清晰自然的语音合成测试文本。" }, { type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } }, ], }, }, ], - }; -} - -const audioSpeechModels = [ - qwenTtsModel( - "qwen-tts", - "Qwen TTS", - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ), - qwenTtsModel( - "qwen-audio-3.0-tts-plus", - "Qwen Audio 3.0 TTS Plus", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, - ), - qwenTtsModel( - "qwen-audio-3.0-tts-flash", - "Qwen Audio 3.0 TTS Flash", - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - { minimumTextCodePoints: 15 }, - ), + }, { schema: MODEL_SCHEMA, model: "higgs-tts", diff --git a/test/adapters/audio-speech.test.ts b/test/adapters/audio-speech.test.ts index 236718b..cef1606 100644 --- a/test/adapters/audio-speech.test.ts +++ b/test/adapters/audio-speech.test.ts @@ -11,6 +11,7 @@ import { const REFERENCE_URL = "https://example.com/reference.mp3"; const SECOND_REFERENCE_URL = "https://example.com/reference-2.mp3"; +const QWEN_TEXT = "这是用于语音合成测试的一整句中文示例文本。"; function routerSuccess( body: Record = {}, @@ -41,8 +42,8 @@ function audio(url = REFERENCE_URL, meta?: Record): GenerationC function qwenDesignRequest(overrides: Partial = {}): GenerateRequest { return { - model: "qwen-tts", - content: [{ type: "text", text: "这是需要朗读的文本。" }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: QWEN_TEXT }], meta: { voice_prompt: "沉稳清晰的男性播音员声音" }, ...overrides, }; @@ -69,11 +70,11 @@ describe("openai.audioSpeech adapter requests", () => { const { client, calls } = recordingClient(() => routerSuccess({}, { headers: { "x-request-id": "request-primary", "x-oneapi-request-id": "request-fallback" } }), ); - const input = " 原样保留的朗读文本。\n"; + const input = " 原样保留的朗读文本内容,不应被修改。\n"; const voicePrompt = " 沉稳清晰的声音。\n"; const output = await client.generate({ - model: "qwen-tts", + model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text: input }], meta: { voice_prompt: voicePrompt }, }); @@ -85,7 +86,7 @@ describe("openai.audioSpeech adapter requests", () => { new Headers({ Authorization: "Bearer secret-key", "Content-Type": "application/json" }), ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", + model: "qwen-audio-3.1-tts-flash", input, metadata: { voice_prompt: voicePrompt }, }); @@ -101,13 +102,13 @@ describe("openai.audioSpeech adapter requests", () => { it("maps Qwen reference audio and trims only its URL", async () => { const { client, calls } = recordingClient(); await client.generate({ - model: "qwen-tts", - content: [{ type: "text", text: "短句" }, audio(` ${REFERENCE_URL}\n`)], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: QWEN_TEXT }, audio(` ${REFERENCE_URL}\n`)], }); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "短句", + model: "qwen-audio-3.1-tts-flash", + input: QWEN_TEXT, ref_audio: REFERENCE_URL, }); }); @@ -121,8 +122,8 @@ describe("openai.audioSpeech adapter requests", () => { ); expect(requestBody(calls[0])).toEqual({ - model: "qwen-tts", - input: "这是需要朗读的文本。", + model: "qwen-audio-3.1-tts-flash", + input: QWEN_TEXT, metadata: { voice_prompt: "清晰女声" }, }); }); @@ -198,8 +199,8 @@ describe("openai.audioSpeech adapter validation", () => { const client = createGenerationClient({ apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "有效语音文本" }], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: QWEN_TEXT }], meta: { preview_text: "已经失效的文本来源" }, }), ).toThrow("requires one reference audio or meta.voice_prompt"); @@ -213,7 +214,7 @@ describe("openai.audioSpeech adapter validation", () => { { name: "both voice sources", request: qwenDesignRequest({ - content: [{ type: "text", text: "文本" }, audio()], + content: [{ type: "text", text: QWEN_TEXT }, audio()], }), }, { @@ -223,7 +224,7 @@ describe("openai.audioSpeech adapter validation", () => { { name: "audio weight", request: qwenDesignRequest({ - content: [{ type: "text", text: "文本" }, audio(REFERENCE_URL, { weight: 1 })], + content: [{ type: "text", text: QWEN_TEXT }, audio(REFERENCE_URL, { weight: 1 })], meta: {}, }), }, @@ -233,17 +234,17 @@ describe("openai.audioSpeech adapter validation", () => { }); it("enforces Qwen reference limits inside the adapter hook when a declaration is overridden", () => { - const declaration = getBuiltinGenerationModel("qwen-tts"); - if (!declaration) throw new Error("qwen-tts declaration is unavailable"); + const declaration = getBuiltinGenerationModel("qwen-audio-3.1-tts-flash"); + if (!declaration) throw new Error("qwen-audio-3.1-tts-flash declaration is unavailable"); const audioSpec = declaration.content.input.find((spec) => spec.type === "audio"); - if (!audioSpec) throw new Error("qwen-tts audio spec is unavailable"); + if (!audioSpec) throw new Error("qwen-audio-3.1-tts-flash audio spec is unavailable"); audioSpec.max = 2; const client = createGenerationClient({ models: [declaration], includeBuiltinModels: false, apiKey: "key" }); expect(() => client.validate({ - model: "qwen-tts", - content: [{ type: "text", text: "文本" }, audio(), audio(SECOND_REFERENCE_URL)], + model: "qwen-audio-3.1-tts-flash", + content: [{ type: "text", text: QWEN_TEXT }, audio(), audio(SECOND_REFERENCE_URL)], }), ).toThrow("supports at most one reference audio"); }); @@ -270,23 +271,23 @@ describe("openai.audioSpeech adapter validation", () => { }); it.each([ - { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(14) }, - { model: "qwen-audio-3.0-tts-flash", text: "😀".repeat(14) }, - ])("rejects $model input below 15 Unicode code points", ({ model, text }) => { + { text: "a".repeat(14) }, + { text: "😀".repeat(14) }, + ])("rejects qwen-audio-3.1-tts-flash input below 15 Unicode code points ($text)", ({ text }) => { const client = createGenerationClient({ apiKey: "key" }); - expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).toThrow( - "requires input of at least 15 Unicode code points", - ); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).toThrow("requires input of at least 15 Unicode code points"); }); it.each([ - { model: "qwen-audio-3.0-tts-plus", text: "a".repeat(15) }, - { model: "qwen-audio-3.0-tts-flash", text: ` ${"😀".repeat(15)} ` }, - { model: "qwen-tts", text: "短" }, - { model: "qwen-tts", text: "a".repeat(40) }, - ])("accepts the $model input boundary", ({ model, text }) => { + { text: "a".repeat(15) }, + { text: ` ${"😀".repeat(15)} ` }, + ])("accepts the qwen-audio-3.1-tts-flash input boundary ($text)", ({ text }) => { const client = createGenerationClient({ apiKey: "key" }); - expect(() => client.validate({ model, content: [{ type: "text", text }, audio()] })).not.toThrow(); + expect(() => + client.validate({ model: "qwen-audio-3.1-tts-flash", content: [{ type: "text", text }, audio()] }), + ).not.toThrow(); }); it.each<{ label: string; request: GenerateRequest }>([ diff --git a/test/config.test.ts b/test/config.test.ts index 408c951..7a1d2ad 100644 --- a/test/config.test.ts +++ b/test/config.test.ts @@ -69,9 +69,7 @@ describe("config", () => { expect(byCategory.audio.sort()).toEqual([...expected.audio]); for (const model of [ "higgs-tts", - "qwen-tts", - "qwen-audio-3.0-tts-plus", - "qwen-audio-3.0-tts-flash", + "qwen-audio-3.1-tts-flash", "suno_cover_chirp_v5", "suno_image_to_song_chirp_v5", "suno_infill_chirp_v5", @@ -216,7 +214,7 @@ describe("config", () => { it("publishes agent-discoverable audio speech declarations", () => { const client = createGenerationClient(); - const qwenModels = ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]; + const qwenModels = ["qwen-audio-3.1-tts-flash"]; for (const model of qwenModels) { const declaration = client.getModel(model); @@ -236,23 +234,7 @@ describe("config", () => { expect(JSON.parse(client.stringifyModelConfig(model, { format: "json" }))).toEqual(declaration); } - const qwen = client.getModel("qwen-tts"); - expect(qwen?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(qwen?.content.input.find((input) => input.type === "text")?.description).not.toContain( - "Unicode code points", - ); - - const plus = client.getModel("qwen-audio-3.0-tts-plus"); - expect(plus?.description).toBe( - "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", - ); - expect(plus?.content.input.find((input) => input.type === "text")?.description).toContain( - "at least 15 Unicode code points", - ); - - const flash = client.getModel("qwen-audio-3.0-tts-flash"); + const flash = client.getModel("qwen-audio-3.1-tts-flash"); expect(flash?.description).toBe( "Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.", ); @@ -292,7 +274,7 @@ describe("config", () => { expect(readme).toContain("Conflict: reference + redesign requires user choice before generation"); expect(readme).toContain("Blend: all references, full text, one request"); expect(readme).toContain("Dependency: clone prior generated audio"); - expect(readme).toContain("Ranking: no declared Qwen quality, latency, or cost order"); + expect(readme).toContain("Short text: input under 15 Unicode code points has no voice-design path on Qwen"); expect(readme).not.toMatch(/quality prioritized over latency|latency prioritized over maximum quality/); }); @@ -517,7 +499,7 @@ describe("config", () => { it("does not publish the retired Qwen preview field", async () => { const client = createGenerationClient({ apiKey: "test" }); - for (const model of ["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]) { + for (const model of ["qwen-audio-3.1-tts-flash"]) { expect(client.stringifyModelConfig(model)).not.toContain("preview_text"); } expect(await readFile(join(process.cwd(), "README.md"), "utf8")).not.toContain("preview_text"); diff --git a/test/live/audio-speech-live.test.ts b/test/live/audio-speech-live.test.ts index 4878eb1..bea40d3 100644 --- a/test/live/audio-speech-live.test.ts +++ b/test/live/audio-speech-live.test.ts @@ -49,30 +49,16 @@ liveDescribe("audio speech live router smoke", () => { { name: "qwen voice design", request: { - model: "qwen-tts", - content: [text(`这是基础模型设计音色端到端测试,运行编号${runId}。`)], + model: "qwen-audio-3.1-tts-flash", + content: [text(`这是设计音色端到端测试,运行编号${runId}。`)], meta: { voice_prompt: "一位沉稳干练的男性播音员声音,吐字清晰有力" }, }, }, { name: "qwen voice clone", request: { - model: "qwen-tts", - content: [text(`这是基础模型克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], - }, - }, - { - name: "qwen audio 3 plus", - request: { - model: "qwen-audio-3.0-tts-plus", - content: [text(`这是增强版本语音合成端到端测试文本,运行编号${runId}。`), audio(REFERENCE_A)], - }, - }, - { - name: "qwen audio 3 flash", - request: { - model: "qwen-audio-3.0-tts-flash", - content: [text(`这是快速版本语音合成端到端测试文本,运行编号${runId}。`), audio(REFERENCE_A)], + model: "qwen-audio-3.1-tts-flash", + content: [text(`这是克隆音色端到端测试,运行编号${runId}。`), audio(REFERENCE_A)], }, }, {