From 6ed0da86c3356fe4d7dfbd1866e1f7ab1a048460 Mon Sep 17 00:00:00 2001 From: RainbowBird Date: Sat, 16 May 2026 17:22:13 +0800 Subject: [PATCH] feat(server/tts): upgrade dashscope-cosyvoice adapter to v2 two-step REST MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DashScope dropped cosyvoice-v1 from its REST-supported model list. v2 (and v3+) speak a different shape: voice / format / sample_rate live under `input`, not `parameters`; non-streaming responses return `output.audio.url` (signed OSS URL) instead of inline `output.audio.data` base64. The previous adapter sent v1-shaped bodies to a bare `https://dashscope-intl.aliyuncs.com/api/v1` baseURL and parsed `audio.data`, which 404'd before the migration and would 200-with-no- audio after — both invisible regressions for the gateway. Adapter changes: - Rewrite request body to v2 schema (voice/format under input). - Add follow-up GET against `output.audio.url`; stream into ArrayBuffer with a 25 MB hard cap and explicit drain-tracking finally, so a misbehaving URL cannot exhaust memory and a half-read body cannot hang a connection. - Re-document baseURL contract: adapters do NOT append path; ops must configure the FULL endpoint URL (root cause of the original 404 storm). DEFAULT_COSYVOICE_MODEL bumped to `cosyvoice-v2`, default voice to `longxiaochun_v2`. Voice catalog: regenerated with 19 representative cosyvoice-v2 voices (assistant / customer-service / child / en-US / en-GB / ja-JP / ko-KR) so the frontend voice picker is no longer a 2-entry stub. Full catalog (100+) remains on the Alibaba docs page — we'll sync on demand rather than scrape. Seed script: `--dashscope-region intl|cn` (default `intl`), `--dashscope-upstream-model cosyvoice-v2`, baseURL now resolves to `https:///api/v1/services/audio/tts/SpeechSynthesizer` so a mis-typed region or path cannot reintroduce the 404. Tests: new dashscope-cosyvoice.test.ts covers v2 body shape (asserts `parameters` absent — regression), audio.url follow-up fetch, 401 propagation with `.status`, empty-envelope falling back into the router's recoverable-error path, and catalog freshness (no leftover v1 ids). Verified locally against the staging DashScope key: 200 + playable mp3 end to end. --- apps/server/scripts/seed-router-config.ts | 19 +- .../tts-adapters/dashscope-cosyvoice.test.ts | 157 +++++++++++ .../tts-adapters/dashscope-cosyvoice.ts | 186 +++++++++---- .../voices/dashscope-cosyvoice.json | 263 +++++++++++++++++- 4 files changed, 559 insertions(+), 66 deletions(-) create mode 100644 apps/server/src/services/tts-adapters/dashscope-cosyvoice.test.ts diff --git a/apps/server/scripts/seed-router-config.ts b/apps/server/scripts/seed-router-config.ts index 77df8479f..09ecd2d11 100644 --- a/apps/server/scripts/seed-router-config.ts +++ b/apps/server/scripts/seed-router-config.ts @@ -49,6 +49,10 @@ interface Args { azureRegion: string azureTtsModel: string dashscopeTtsModel: string + /** `intl` → dashscope-intl.aliyuncs.com (Singapore); `cn` → dashscope.aliyuncs.com (Beijing). */ + dashscopeRegion: string + /** Concrete cosyvoice variant the adapter calls upstream. Independent from `dashscopeTtsModel` (the gateway-facing alias). */ + dashscopeUpstreamModel: string defaultTtsModel: string | undefined } @@ -76,7 +80,9 @@ function parseArgs(argv: string[]): Args { defaultChatModel: values['default-chat-model'] ?? 'chat-default', azureRegion: values['azure-region'] ?? 'eastasia', azureTtsModel: values['azure-tts-model'] ?? 'microsoft/v1', - dashscopeTtsModel: values['dashscope-tts-model'] ?? 'alibaba/cosyvoice-v1', + dashscopeTtsModel: values['dashscope-tts-model'] ?? 'alibaba/cosyvoice-v2', + dashscopeRegion: values['dashscope-region'] ?? 'intl', + dashscopeUpstreamModel: values['dashscope-upstream-model'] ?? 'cosyvoice-v2', defaultTtsModel: values['default-tts-model'], } } @@ -132,14 +138,21 @@ function buildDashscope(args: Args, plaintext: string, envelope: EnvelopeCrypto) modelName: args.dashscopeTtsModel, keyEntryId, }) + // dashscope-cosyvoice adapter expects the FULL non-streaming endpoint path — + // it does not append `/services/audio/tts/SpeechSynthesizer` itself. A bare + // `/api/v1` baseURL was the root cause of the 404 storm during the v1→v2 + // migration; do not regress here. + const host = args.dashscopeRegion === 'cn' + ? 'dashscope.aliyuncs.com' + : 'dashscope-intl.aliyuncs.com' return { ttsModelName: args.dashscopeTtsModel, ttsModel: { provider: 'dashscope-cosyvoice', upstreams: [{ - baseURL: 'https://dashscope-intl.aliyuncs.com/api/v1', + baseURL: `https://${host}/api/v1/services/audio/tts/SpeechSynthesizer`, keys: [{ id: keyEntryId, ciphertext }], - adapterParams: {}, + adapterParams: { model: args.dashscopeUpstreamModel }, }], }, } diff --git a/apps/server/src/services/tts-adapters/dashscope-cosyvoice.test.ts b/apps/server/src/services/tts-adapters/dashscope-cosyvoice.test.ts new file mode 100644 index 000000000..b74c119ea --- /dev/null +++ b/apps/server/src/services/tts-adapters/dashscope-cosyvoice.test.ts @@ -0,0 +1,157 @@ +import { Buffer } from 'node:buffer' + +import { describe, expect, it, vi } from 'vitest' + +import { dashscopeCosyvoiceAdapter } from './dashscope-cosyvoice' + +const FULL_ENDPOINT = 'https://dashscope-intl.aliyuncs.com/api/v1/services/audio/tts/SpeechSynthesizer' +const AUDIO_URL = 'https://dashscope-internal.aliyuncs.com/audio/abc.mp3' + +function jsonResponse(body: object, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' }, + }) +} + +function binaryResponse(bytes: Uint8Array, status = 200) { + return new Response(bytes, { + status, + headers: { 'content-type': 'audio/mpeg' }, + }) +} + +describe('dashscopeCosyvoiceAdapter', () => { + it('sends v2-shaped body: voice / format under input, not parameters (regression — old shape returned no audio.data)', async () => { + const fetchImpl = vi.fn() + .mockResolvedValueOnce(jsonResponse({ output: { audio: { url: AUDIO_URL } } })) + .mockResolvedValueOnce(binaryResponse(new Uint8Array([1, 2, 3, 4]))) + + await dashscopeCosyvoiceAdapter.send( + { text: 'hi there', voice: 'longxiaochun_v2', responseFormat: 'mp3' }, + { + keyPlaintext: Buffer.from('sk-test', 'utf8'), + baseURL: FULL_ENDPOINT, + adapterParams: { model: 'cosyvoice-v2' }, + fetchImpl: fetchImpl as unknown as typeof fetch, + }, + ) + + expect(fetchImpl).toHaveBeenCalledTimes(2) + const [synthesizeUrl, synthesizeInit] = fetchImpl.mock.calls[0] + expect(synthesizeUrl).toBe(FULL_ENDPOINT) + expect(synthesizeInit.method).toBe('POST') + + const body = JSON.parse(synthesizeInit.body as string) + expect(body).toEqual({ + model: 'cosyvoice-v2', + input: { + text: 'hi there', + voice: 'longxiaochun_v2', + format: 'mp3', + }, + }) + // Critical: voice / format must NOT leak into a top-level `parameters` block. + expect(body.parameters).toBeUndefined() + }) + + it('follows output.audio.url to fetch the actual audio bytes and returns them as ArrayBuffer (regression — v1 parsed base64 from output.audio.data, v2 returns a URL instead)', async () => { + const audioBytes = new Uint8Array([0x49, 0x44, 0x33, 0x04, 0x00, 0x00]) // ID3v2 mp3 header + const fetchImpl = vi.fn() + .mockResolvedValueOnce(jsonResponse({ output: { audio: { url: AUDIO_URL, data: '' } } })) + .mockResolvedValueOnce(binaryResponse(audioBytes)) + + const result = await dashscopeCosyvoiceAdapter.send( + { text: 'hi', responseFormat: 'mp3' }, + { + keyPlaintext: Buffer.from('sk-test', 'utf8'), + baseURL: FULL_ENDPOINT, + adapterParams: {}, + fetchImpl: fetchImpl as unknown as typeof fetch, + }, + ) + + const [audioFetchUrl, audioFetchInit] = fetchImpl.mock.calls[1] + expect(audioFetchUrl).toBe(AUDIO_URL) + expect(audioFetchInit.method).toBe('GET') + + expect(result.contentType).toBe('audio/mpeg') + expect(result.body).toBeInstanceOf(ArrayBuffer) + const out = new Uint8Array(result.body as ArrayBuffer) + expect(Array.from(out)).toEqual(Array.from(audioBytes)) + }) + + it('throws Error with .status when synthesis endpoint returns non-2xx (router maps to fallback chain)', async () => { + const fetchImpl = vi.fn() + .mockResolvedValueOnce(jsonResponse({ code: 'InvalidApiKey', message: 'bad key' }, 401)) + + await expect( + dashscopeCosyvoiceAdapter.send( + { text: 'hi' }, + { + keyPlaintext: Buffer.from('sk-test', 'utf8'), + baseURL: FULL_ENDPOINT, + adapterParams: {}, + fetchImpl: fetchImpl as unknown as typeof fetch, + }, + ), + ).rejects.toMatchObject({ status: 401, message: expect.stringContaining('401') }) + + expect(fetchImpl).toHaveBeenCalledTimes(1) + }) + + it('throws Error with .status when output envelope contains no audio.url (treat as recoverable upstream error)', async () => { + // ROOT CAUSE: + // + // CosyVoice v2 non-streaming returns audio.url; if upstream returns 200 + // with an empty envelope (rare — policy reject / region edge case), the + // v1 adapter silently returned "no audio data" while modeled status was + // 200. Router can't decide whether to fall back without a status. We + // attach the response status to the error so the router treats it as a + // recoverable upstream failure and walks to the next key. + const fetchImpl = vi.fn() + .mockResolvedValueOnce(jsonResponse({ output: {} })) + + await expect( + dashscopeCosyvoiceAdapter.send( + { text: 'hi' }, + { + keyPlaintext: Buffer.from('sk-test', 'utf8'), + baseURL: FULL_ENDPOINT, + adapterParams: {}, + fetchImpl: fetchImpl as unknown as typeof fetch, + }, + ), + ).rejects.toMatchObject({ status: 200, message: expect.stringContaining('no audio.url') }) + }) + + it('uses cosyvoice-v2 + longxiaochun_v2 as defaults when caller omits model / voice (regression — v1 defaults broke against the v2 endpoint)', async () => { + const fetchImpl = vi.fn() + .mockResolvedValueOnce(jsonResponse({ output: { audio: { url: AUDIO_URL } } })) + .mockResolvedValueOnce(binaryResponse(new Uint8Array([0]))) + + await dashscopeCosyvoiceAdapter.send( + { text: 'hi' }, + { + keyPlaintext: Buffer.from('sk-test', 'utf8'), + baseURL: FULL_ENDPOINT, + adapterParams: {}, + fetchImpl: fetchImpl as unknown as typeof fetch, + }, + ) + + const body = JSON.parse(fetchImpl.mock.calls[0][1].body as string) + expect(body.model).toBe('cosyvoice-v2') + expect(body.input.voice).toBe('longxiaochun_v2') + expect(body.input.format).toBe('mp3') + }) + + it('voice catalog contains v2-suffixed ids and is non-empty', () => { + const catalog = dashscopeCosyvoiceAdapter.getVoiceCatalog() + expect(catalog.length).toBeGreaterThan(0) + expect(catalog.some(v => v.id === 'longxiaochun_v2')).toBe(true) + // No bare v1 ids should survive the migration. + expect(catalog.find(v => v.id === 'longxiaochun')).toBeUndefined() + expect(catalog.find(v => v.id === 'longxiaobai')).toBeUndefined() + }) +}) diff --git a/apps/server/src/services/tts-adapters/dashscope-cosyvoice.ts b/apps/server/src/services/tts-adapters/dashscope-cosyvoice.ts index c7a75bb0c..cf7b47d64 100644 --- a/apps/server/src/services/tts-adapters/dashscope-cosyvoice.ts +++ b/apps/server/src/services/tts-adapters/dashscope-cosyvoice.ts @@ -11,13 +11,14 @@ import cosyvoiceVoices from './voices/dashscope-cosyvoice.json' with { type: 'js import { createInternalError } from '../../utils/error' /** - * Default DashScope cosyvoice voice id. `longxiaochun` is the most commonly - * referenced general-purpose Chinese voice in Alibaba's docs. + * Default DashScope cosyvoice voice id. v2 voice ids carry an explicit `_v2` + * suffix; `longxiaochun_v2` is the general-purpose Chinese assistant voice + * called out in Alibaba's voice list. * * NOTICE: * Hardcoded near use; promote when ops want per-tenant defaults. */ -const DEFAULT_COSYVOICE_VOICE = 'longxiaochun' +const DEFAULT_COSYVOICE_VOICE = 'longxiaochun_v2' /** * Default cosyvoice audio format. Mirrors the OpenAI `mp3` default expected by @@ -26,27 +27,46 @@ const DEFAULT_COSYVOICE_VOICE = 'longxiaochun' const DEFAULT_COSYVOICE_FORMAT = 'mp3' /** - * Default cosyvoice model id targeted by v1. Adapters can be retargeted via - * `adapterParams.model` if ops want to A/B between cosyvoice variants without - * deploying a code change. + * Default cosyvoice model id. v1 was dropped from the official "REST-supported + * models" list (the official list now starts at v2 and runs through v3.5); + * v2 is the most conservative current default and shares a request body shape + * with v3/v3.5 so ops can retarget via `adapterParams.model` without code. + * NOTICE: + * If you bump this past v2, verify the chosen `DEFAULT_COSYVOICE_VOICE` exists + * for that model — voice catalogs differ between v2 (`*_v2`) and v3 (`*_v3`). */ -const DEFAULT_COSYVOICE_MODEL = 'cosyvoice-v1' +const DEFAULT_COSYVOICE_MODEL = 'cosyvoice-v2' + +/** + * Hard cap on the audio bytes we will pull from the `output.audio.url` + * follow-up fetch. CosyVoice non-streaming responses for normal TTS prompts + * stay well under this; the limit exists so a misbehaving / hijacked URL + * cannot exhaust memory on the gateway instance. + */ +const MAX_AUDIO_BYTES = 25 * 1024 * 1024 /** * DashScope cosyvoice non-streaming REST adapter. * * Use when: - * - Routing a hosted TTS request to Alibaba DashScope's cosyvoice family of - * models (Chinese + English speech synthesis). + * - Routing a hosted TTS request to Alibaba DashScope's cosyvoice v2 / v3 + * family of models (Chinese + English + selected multilingual voices). * * Expects: - * - `ctx.baseURL` points at the DashScope multimodal-generation endpoint, e.g. - * `https://dashscope.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation`. + * - `ctx.baseURL` is the **full** non-streaming endpoint, e.g. + * `https://dashscope.aliyuncs.com/api/v1/services/audio/tts/SpeechSynthesizer` + * (or `dashscope-intl.aliyuncs.com` for the Singapore region). The adapter + * does not append a path — pointing at a bare `/api/v1` will 404. * - `ctx.keyPlaintext` is the DashScope API key (sent as `Bearer ...`). + * - `ctx.adapterParams.model` (optional) names the cosyvoice variant; defaults + * to {@link DEFAULT_COSYVOICE_MODEL}. * * Returns: - * - {@link TtsResult} with the audio bytes as an `ArrayBuffer`. Body is - * decoded from the upstream JSON's `output.audio.data` base64 payload. + * - {@link TtsResult} with the audio bytes as an `ArrayBuffer`. The non- + * streaming endpoint returns a JSON envelope whose `output.audio.url` is + * a short-lived signed URL; this adapter performs the follow-up GET and + * surfaces the final bytes so router callers get the same single-shot + * contract as the Azure / Volcengine paths. */ export const dashscopeCosyvoiceAdapter: TtsAdapter = { id: 'dashscope-cosyvoice', @@ -58,16 +78,17 @@ export const dashscopeCosyvoiceAdapter: TtsAdapter = { const voice = input.voice ?? DEFAULT_COSYVOICE_VOICE const format = input.responseFormat ?? DEFAULT_COSYVOICE_FORMAT - // DashScope multimodal-generation body: `input.text` for the prompt and - // `parameters` for synthesis options. cosyvoice v1 accepts a `rate` - // multiplier (defaults to 1.0 server-side when omitted). + // v2 / v3 request shape: voice / format / sample_rate live under `input` + // (NOT `parameters`, which was the v1 multimodal-generation schema). Speed + // is currently dropped on v2 non-streaming — there is no documented field + // for it on this endpoint; SSML rate is the supported substitute on + // SSML-enabled voices. const body: Record = { model, - input: { text: input.text }, - parameters: { + input: { + text: input.text, voice, format, - ...(input.speed != null ? { rate: input.speed } : {}), }, } @@ -96,9 +117,6 @@ export const dashscopeCosyvoiceAdapter: TtsAdapter = { throw err } - // cosyvoice v1 synthesize mode returns JSON. The audio payload sits under - // `output.audio.data` as a base64 string. We decode to ArrayBuffer here so - // the router/handler can re-stream identically to the Azure path. let payload: unknown try { payload = await response.json() @@ -107,19 +125,27 @@ export const dashscopeCosyvoiceAdapter: TtsAdapter = { throw createInternalError(`dashscope-cosyvoice tts response parse failed: ${errorMessageFrom(error) ?? 'unknown'}`) } - const audioData = extractCosyvoiceAudioBase64(payload) - if (!audioData) { - const err = new Error('dashscope-cosyvoice tts upstream returned no audio data') as Error & { status?: number } + const audioUrl = extractCosyvoiceAudioUrl(payload) + if (!audioUrl) { + // Could be a request-mode mismatch (SSE was enabled, response shape + // changed) or an upstream-policy reject that returned 200 with an empty + // envelope. Treat as a recoverable upstream error so the router can + // try the next key / upstream. + const err = new Error(`dashscope-cosyvoice tts upstream returned no audio.url (envelope: ${stringifyEnvelope(payload)})`) as Error & { status?: number } err.status = response.status throw err } - const buf = Buffer.from(audioData, 'base64') - // Slice avoids returning a view over the larger pooled Node buffer. - const arrayBuffer = buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength) - const contentType = formatToMime(format) + let audioBytes: ArrayBuffer + try { + audioBytes = await fetchAudioBytes(ctx.fetchImpl, audioUrl, ctx.abortSignal) + } + catch (error) { + throw createInternalError(`dashscope-cosyvoice tts audio download failed: ${errorMessageFrom(error) ?? 'unknown'}`) + } - return { contentType, body: arrayBuffer } + const contentType = formatToMime(format) + return { contentType, body: audioBytes } }, getVoiceCatalog() { @@ -128,33 +154,95 @@ export const dashscopeCosyvoiceAdapter: TtsAdapter = { } /** - * Pulls the base64 audio string out of a cosyvoice JSON response. Returns - * `null` if the response shape doesn't match (e.g. error envelope) so the - * caller can surface a clear upstream error. + * Pulls the `output.audio.url` short-lived signed URL out of a cosyvoice + * non-streaming JSON envelope. Returns `null` if the response shape doesn't + * match (e.g. error envelope, SSE leak, or v1-style `audio.data` payload), + * so the caller can surface a clear upstream error. */ -function extractCosyvoiceAudioBase64(payload: unknown): string | null { - if (!payload || typeof payload !== 'object') +function extractCosyvoiceAudioUrl(payload: unknown): string | null { + if (payload == null || typeof payload !== 'object') return null - const output = (payload as Record).output - if (!output || typeof output !== 'object') + const output = (payload as { output?: unknown }).output + if (output == null || typeof output !== 'object') return null - const audio = (output as Record).audio - if (!audio || typeof audio !== 'object') + const audio = (output as { audio?: unknown }).audio + if (audio == null || typeof audio !== 'object') return null - const data = (audio as Record).data - return typeof data === 'string' ? data : null + const url = (audio as { url?: unknown }).url + if (typeof url !== 'string' || url.length === 0) + return null + return url +} + +/** + * Stringify just enough of the upstream JSON to make the "no audio.url" error + * actionable, without leaking secrets like API keys. Keeps the snippet small + * so it fits inside the router's bodySnippet propagation path. + */ +function stringifyEnvelope(payload: unknown): string { + try { + return JSON.stringify(payload).slice(0, 256) + } + catch { + return '' + } +} + +/** + * Follow-up GET against the short-lived signed URL the cosyvoice endpoint + * returns. Streamed into an ArrayBuffer with a hard size cap so a misbehaving + * URL cannot exhaust memory on the gateway. + */ +async function fetchAudioBytes(fetchImpl: typeof fetch, url: string, abortSignal: AbortSignal | undefined): Promise { + const audioResp = await fetchImpl(url, { method: 'GET', signal: abortSignal }) + if (!audioResp.ok) { + const err = new Error(`audio.url responded ${audioResp.status}`) as Error & { status?: number } + err.status = audioResp.status + throw err + } + + if (audioResp.body == null) { + throw new Error('audio.url response had no body') + } + + const reader = audioResp.body.getReader() + const chunks: Uint8Array[] = [] + let total = 0 + let drained = false + try { + while (true) { + const { value, done } = await reader.read() + if (done) { + drained = true + break + } + total += value.length + if (total > MAX_AUDIO_BYTES) + throw new Error(`audio payload exceeded ${MAX_AUDIO_BYTES} bytes`) + chunks.push(value) + } + } + finally { + // Cancel only if we exited early; double-cancel after a clean drain is + // a no-op in spec but `reader.closed` is a Promise (always truthy), so + // we track the drain explicitly instead of testing `closed`. + if (!drained) + reader.cancel().catch(() => {}) + } + + const buf = Buffer.concat(chunks.map(c => Buffer.from(c))) + return buf.buffer.slice(buf.byteOffset, buf.byteOffset + buf.byteLength) } /** * Maps cosyvoice's `format` (`mp3` / `wav` / `pcm`) to a MIME type for the - * gateway response. + * client. Keeps the router contract symmetric with Azure / Volcengine. */ function formatToMime(format: string): string { - if (format === 'mp3') - return 'audio/mpeg' - if (format === 'wav') - return 'audio/wav' - if (format === 'pcm') - return 'audio/L16' - return 'application/octet-stream' + switch (format) { + case 'mp3': return 'audio/mpeg' + case 'wav': return 'audio/wav' + case 'pcm': return 'audio/L16' + default: return 'application/octet-stream' + } } diff --git a/apps/server/src/services/tts-adapters/voices/dashscope-cosyvoice.json b/apps/server/src/services/tts-adapters/voices/dashscope-cosyvoice.json index 074d8fa5e..de3055f21 100644 --- a/apps/server/src/services/tts-adapters/voices/dashscope-cosyvoice.json +++ b/apps/server/src/services/tts-adapters/voices/dashscope-cosyvoice.json @@ -1,30 +1,265 @@ [ { - "id": "longxiaochun", - "name": "Longxiaochun", - "description": "DashScope CosyVoice default female voice", - "compatible_models": ["cosyvoice-v1"], + "id": "longxiaochun_v2", + "name": "Longxiaochun (龙小淳)", + "description": "Knowing and energetic female assistant voice", + "compatible_models": ["cosyvoice-v2"], "formats": [ { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } ], - "labels": { "gender": "Female", "type": "neural" }, + "labels": { "gender": "Female", "type": "neural", "scenario": "assistant" }, "languages": [ - { "code": "zh-CN", "title": "Chinese (Simplified)" } + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } ], - "tags": ["chinese", "neural"] + "tags": ["assistant", "chinese", "english", "neural"] }, { - "id": "longxiaobai", - "name": "Longxiaobai", - "description": "DashScope CosyVoice male voice", - "compatible_models": ["cosyvoice-v1"], + "id": "longxiaoxia_v2", + "name": "Longxiaoxia (龙小夏)", + "description": "Calm authoritative female assistant voice", + "compatible_models": ["cosyvoice-v2"], "formats": [ { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } ], - "labels": { "gender": "Male", "type": "neural" }, + "labels": { "gender": "Female", "type": "neural", "scenario": "assistant" }, "languages": [ - { "code": "zh-CN", "title": "Chinese (Simplified)" } + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } ], - "tags": ["chinese", "neural"] + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longyumi_v2", + "name": "Yumi", + "description": "Composed young female assistant voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "assistant" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longanli", + "name": "Longanli (龙安莉)", + "description": "Crisp confident female assistant voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "assistant" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longanlang", + "name": "Longanlang (龙安朗)", + "description": "Cool and clean male assistant voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "assistant" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longanwen", + "name": "Longanwen (龙安温)", + "description": "Elegant and graceful female assistant voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "assistant" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longanyun", + "name": "Longanyun (龙安昂)", + "description": "Warm at-home male assistant voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "assistant" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["assistant", "chinese", "english", "neural"] + }, + { + "id": "longyingmu", + "name": "Longyingmu (龙应沐)", + "description": "Polished service representative female voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "customer_service" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["customer-service", "chinese", "english", "neural"] + }, + { + "id": "longyingtian", + "name": "Longyingtian (龙应甜)", + "description": "Warm sweet service representative female voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "customer_service" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["customer-service", "chinese", "english", "neural"] + }, + { + "id": "longhuhu", + "name": "Longhuhu (龙呼呼)", + "description": "Bright young girl voice for child-facing content", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "child" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["child", "chinese", "english", "neural"] + }, + { + "id": "longniuniu", + "name": "Longniuniu (龙牛牛)", + "description": "Sunny young boy voice for child-facing content", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "child" }, + "languages": [ + { "code": "zh-CN", "title": "Chinese (Simplified)" }, + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["child", "chinese", "english", "neural"] + }, + { + "id": "loongabby_v2", + "name": "Abby", + "description": "American English female voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["english", "us", "neural"] + }, + { + "id": "loongdavid_v2", + "name": "David", + "description": "American English male voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "en-US", "title": "English (United States)" } + ], + "tags": ["english", "us", "neural"] + }, + { + "id": "loongbrian_v2", + "name": "Brian", + "description": "British English male voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "en-GB", "title": "English (United Kingdom)" } + ], + "tags": ["english", "uk", "neural"] + }, + { + "id": "loongeva_v2", + "name": "Eva", + "description": "British English female voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "en-GB", "title": "English (United Kingdom)" } + ], + "tags": ["english", "uk", "neural"] + }, + { + "id": "loongyuuna_v2", + "name": "Yuuna", + "description": "Genki Japanese female voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Female", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "ja-JP", "title": "Japanese (Japan)" } + ], + "tags": ["japanese", "neural"] + }, + { + "id": "loongyuuma_v2", + "name": "Yuuma", + "description": "Steady Japanese male voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "ja-JP", "title": "Japanese (Japan)" } + ], + "tags": ["japanese", "neural"] + }, + { + "id": "loongjihun_v2", + "name": "Jihun", + "description": "Bright Korean male voice", + "compatible_models": ["cosyvoice-v2"], + "formats": [ + { "name": "mp3", "format_code": "mp3", "mime_type": "audio/mpeg", "extension": "mp3", "bitrate": 64, "sample_rate": 22050 } + ], + "labels": { "gender": "Male", "type": "neural", "scenario": "narration" }, + "languages": [ + { "code": "ko-KR", "title": "Korean (Korea)" } + ], + "tags": ["korean", "neural"] } ]