Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 19 additions & 9 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -62,13 +62,13 @@ Agents and external tools should inspect a model declaration before constructing
import { createGenerationClient } from "@neta-art/generation";

const discoveryClient = createGenerationClient();
const declaration = discoveryClient.getModel("qwen-tts");
const declaration = discoveryClient.getModel("qwen3-tts-vc-2026-01-22");
if (!declaration) throw new Error("Model is unavailable");

console.log(discoveryClient.stringifyModelConfig(declaration.model, { format: "json" }));

const request = declaration.examples?.find((example) => example.title === "Voice design")?.request;
if (!request) throw new Error("Voice-design example is unavailable");
const request = declaration.examples?.find((example) => example.title === "Voice clone")?.request;
if (!request) throw new Error("Voice-clone example is unavailable");

// Discovery and validation do not require an API key or access the network.
discoveryClient.validate(request);
Expand All @@ -84,7 +84,7 @@ The same declarations can be exported as YAML through the existing CLI:

```bash
neta-generation models list
neta-generation models export qwen-tts --out ./qwen-tts.yaml
neta-generation models export qwen3-tts-vc-2026-01-22 --out ./qwen3-tts-vc-2026-01-22.yaml
neta-generation models export-all --out ./models
```

Expand Down Expand Up @@ -170,7 +170,7 @@ const client = createGenerationClient({
- `gpt-image-2`
- `z-image-turbo`
- `qwen-image-edit`
- `qwen-tts`
- `qwen3-tts-vc-2026-01-22`
- `qwen-audio-3.0-tts-plus`
- `qwen-audio-3.0-tts-flash`
- `higgs-tts`
Expand Down Expand Up @@ -268,12 +268,14 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au

| Requirement | Model choice |
| --- | --- |
| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-tts` as the deterministic default |
| Create a voice from a text-only description, without reference audio | Use an explicitly requested Qwen variant; otherwise use `qwen-audio-3.0-tts-plus` as the deterministic default |
| Clone one reference voice on DashScope's long-term-supported model line | `qwen3-tts-vc-2026-01-22` |
| Maximize fidelity to one reference voice | `higgs-tts` |
| Blend 2-16 weighted reference voices | `higgs-tts` |
| Use a default voice, including a delegated choice expressed only as any, random, suitable, or natural | `higgs-tts` |

- Qwen: `voice_prompt` design OR one-reference clone; `qwen-tts` is the unspecified-design default and accepts any text length; Plus / Flash require at least 15 Unicode code points.
- Qwen (`qwen-audio-3.0-tts-plus`/`-flash`): `voice_prompt` design OR one-reference clone; require at least 15 Unicode code points.
- Qwen3-TTS (`qwen3-tts-vc-2026-01-22`): clone-only, no `voice_prompt` design mode; reference audio is required; accepts any text length.
- Higgs: delegated default voice, high-fidelity one-reference clone, or weighted 2-16-reference blend.
- Conflict: reference + redesign requires user choice before generation.
- Blend: all references, full text, one request.
Expand All @@ -282,8 +284,8 @@ Each TTS request accepts exactly one non-empty text block and returns one URL au

```ts
await client.generate({
model: "qwen-tts",
content: [{ type: "text", text: "欢迎使用语音合成功能。" }],
model: "qwen-audio-3.0-tts-plus",
content: [{ type: "text", text: "欢迎使用长度足够的语音合成功能进行试听。" }],
meta: {
voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中",
},
Expand All @@ -296,6 +298,14 @@ await client.generate({
{ type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } },
],
});

await client.generate({
model: "qwen3-tts-vc-2026-01-22",
content: [
{ type: "text", text: "欢迎使用语音合成功能。" },
{ type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } },
],
});
```

```ts
Expand Down
2 changes: 1 addition & 1 deletion examples/text-to-speech.ts
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@ if (!apiKey) throw new Error("Set NETA_ROUTER_API_KEY or NETA_API_KEY");

const client = createGenerationClient({ apiKey });
const output = await client.generate({
model: "qwen-tts",
model: "qwen-audio-3.0-tts-plus",
content: [{ type: "text", text: "欢迎使用语音合成功能,这是一段示例文本。" }],
meta: {
voice_prompt: "一位沉稳自然的中文播音员,吐字清晰,语速适中",
Expand Down
44 changes: 0 additions & 44 deletions models/qwen-tts.yaml

This file was deleted.

32 changes: 32 additions & 0 deletions models/qwen3-tts-vc-2026-01-22.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
schema: neta.generation.model.v1
model: qwen3-tts-vc-2026-01-22
title: Qwen3 TTS Voice Clone
description: "Clone-only: requires exactly one reference audio; no voice-design mode. Text: any length. Conflict: N/A,
single mode. Dependency: clone prior generated audio."
adapter:
type: openai.audioSpeech
content:
input:
- type: text
required: true
min: 1
max: 1
description: Exactly one non-empty text block to speak.
- type: audio
required: true
min: 1
max: 1
sources:
- url
description: "Required reference audio URL to clone. Dependency: use prior generated audio."
examples:
- title: Voice clone
request:
model: qwen3-tts-vc-2026-01-22
content:
- type: text
text: 这是一次清晰自然的语音合成测试。
- type: audio
source:
type: url
url: https://example.com/reference.mp3
36 changes: 32 additions & 4 deletions src/adapters/audio-speech.ts
Original file line number Diff line number Diff line change
Expand Up @@ -8,8 +8,14 @@ import type {
} from "../types.js";

const REQUEST_TIMEOUT_MS = 210_000;
const QWEN_MODELS = new Set(["qwen-tts", "qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]);
const QWEN_AUDIO_3_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]);
const QWEN_MODELS = new Set(["qwen-audio-3.0-tts-plus", "qwen-audio-3.0-tts-flash"]);
const QWEN_MINIMUM_TEXT_CODE_POINTS = 15;
// qwen3-tts-vc is clone-only -- DashScope's voice-design call shape doesn't
// exist for this model family (see background/tasks/qwen_tts_actor.py's
// module docstring), unlike QWEN_MODELS above, which supports design OR
// clone through the same wire shape. It also has no minimum text length
// (background's worker-side implementation enforces only non-empty).
const QWEN3_TTS_CLONE_MODELS = new Set(["qwen3-tts-vc-2026-01-22"]);
const HIGGS_MODEL = "higgs-tts";

type TextBlock = Extract<GenerationContentBlock, { type: "text" }>;
Expand Down Expand Up @@ -92,8 +98,21 @@ function validateQwen(input: ResolvedGenerationRequest, text: TextBlock, audio:
);
}

if (QWEN_AUDIO_3_MODELS.has(input.declaration.model) && Array.from(text.text.trim()).length < 15) {
throw new GenerationValidationError(`${input.declaration.model} requires input of at least 15 Unicode code points`);
if (Array.from(text.text.trim()).length < QWEN_MINIMUM_TEXT_CODE_POINTS) {
throw new GenerationValidationError(
`${input.declaration.model} requires input of at least ${QWEN_MINIMUM_TEXT_CODE_POINTS} Unicode code points`,
);
}
}

function validateQwen3TtsClone(input: ResolvedGenerationRequest, text: TextBlock, audio: AudioBlock[]): void {
validateMetaKeys("request.metadata", input.request.metadata, new Set());
validateMetaKeys("request.meta", input.request.meta, new Set());
validateMetaKeys("text content meta", text.meta, new Set());
for (const block of audio) validateMetaKeys("audio content meta", block.meta, new Set());

if (audio.length !== 1) {
throw new GenerationValidationError(`${input.declaration.model} requires exactly one reference audio`);
}
}

Expand Down Expand Up @@ -128,6 +147,10 @@ function validateAudioSpeechRequest(input: ResolvedGenerationRequest): void {
validateQwen(input, text, audio);
return;
}
if (QWEN3_TTS_CLONE_MODELS.has(input.declaration.model)) {
validateQwen3TtsClone(input, text, audio);
return;
}
if (input.declaration.model === HIGGS_MODEL) {
validateHiggs(input, text, audio);
return;
Expand All @@ -154,6 +177,11 @@ function buildPayload(input: ResolvedGenerationRequest): Record<string, unknown>
return payload;
}

if (QWEN3_TTS_CLONE_MODELS.has(input.declaration.model)) {
if (audio[0]?.source.type === "url") payload.ref_audio = audio[0].source.url.trim();
return payload;
}

const firstAudio = audio[0];
if (audio.length === 1 && firstAudio && (!hasOwnWeight(firstAudio) || firstAudio.meta?.weight === undefined)) {
if (firstAudio.source.type === "url") payload.ref_audio = firstAudio.source.url.trim();
Expand Down
68 changes: 50 additions & 18 deletions src/builtins.ts
Original file line number Diff line number Diff line change
Expand Up @@ -708,15 +708,10 @@ function geminiImageModel(
};
}

function qwenTtsModel(
model: string,
title: string,
description: string,
options: { minimumTextCodePoints?: number } = {},
): GenerationModelDeclaration {
const text = options.minimumTextCodePoints
? "这是一段长度足够并且表达清晰自然的语音合成测试文本。"
: "这是一次清晰自然的语音合成测试。";
const QWEN_AUDIO_3_MINIMUM_TEXT_CODE_POINTS = 15;

function qwenTtsModel(model: string, title: string, description: string): GenerationModelDeclaration {
const text = "这是一段长度足够并且表达清晰自然的语音合成测试文本。";
return {
schema: MODEL_SCHEMA,
model,
Expand All @@ -730,9 +725,7 @@ function qwenTtsModel(
required: true,
min: 1,
max: 1,
description: options.minimumTextCodePoints
? `Exactly one non-empty text block to speak, with at least ${options.minimumTextCodePoints} Unicode code points.`
: "Exactly one non-empty text block to speak.",
description: `Exactly one non-empty text block to speak, with at least ${QWEN_AUDIO_3_MINIMUM_TEXT_CODE_POINTS} Unicode code points.`,
},
{
type: "audio",
Expand Down Expand Up @@ -775,23 +768,62 @@ function qwenTtsModel(
};
}

function qwen3TtsCloneModel(model: string, title: string, description: string): GenerationModelDeclaration {
return {
schema: MODEL_SCHEMA,
model,
title,
description,
adapter: { type: "openai.audioSpeech" },
content: {
input: [
{
type: "text",
required: true,
min: 1,
max: 1,
description: "Exactly one non-empty text block to speak.",
},
{
type: "audio",
required: true,
min: 1,
max: 1,
sources: ["url"],
description: "Required reference audio URL to clone. Dependency: use prior generated audio.",
},
],
},
examples: [
{
title: "Voice clone",
request: {
model,
content: [
{ type: "text", text: "这是一次清晰自然的语音合成测试。" },
{ type: "audio", source: { type: "url", url: "https://example.com/reference.mp3" } },
],
},
},
],
};
}

const audioSpeechModels = [
qwenTtsModel(
"qwen-tts",
"Qwen TTS",
"Modes: voice_prompt design OR one-reference clone. Default: unspecified Qwen design. Text: any length. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.",
qwen3TtsCloneModel(
"qwen3-tts-vc-2026-01-22",
"Qwen3 TTS Voice Clone",
"Clone-only: requires exactly one reference audio; no voice-design mode. Text: any length. Conflict: N/A, single mode. Dependency: clone prior generated audio.",
),
qwenTtsModel(
"qwen-audio-3.0-tts-plus",
"Qwen Audio 3.0 TTS Plus",
"Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.",
{ minimumTextCodePoints: 15 },
),
qwenTtsModel(
"qwen-audio-3.0-tts-flash",
"Qwen Audio 3.0 TTS Flash",
"Modes: voice_prompt design OR one-reference clone. Text: >=15 Unicode code points. Conflict: ask user; never combine/reinterpret. Dependency: clone prior generated audio.",
{ minimumTextCodePoints: 15 },
),
{
schema: MODEL_SCHEMA,
Expand Down
Loading