feat(server,stage-ui): bidirectional streaming TTS + audio path refactor
Why:
- Add a real bidirectional streaming TTS path: raw LLM tokens are
forwarded to the upstream model (Volcengine v3 via the unspeech ws
bridge) without client-side segmentation, so the model owns sentence
splitting and audio chunks play as they arrive.
- Move audio endpoints out of /api/v1/openai/. `/audio/voices`,
`/audio/models`, `/audio/voices/streaming` are not real OpenAI public
APIs, and the streaming TTS surface has nothing to do with OpenAI —
keeping them under /openai/ mislabelled the contract.
- Introduce `capabilities.speech.transport` on ProviderDefinition so
future streaming providers (ElevenLabs / Cartesia / OpenAI Realtime)
opt in without touching Stage.vue or the session factory.
- Unify Stage.vue's TTS path through a single StageTtsSession so the
chat-orchestrator hooks no longer branch on provider id.
What:
- apps/server: new ws proxy /api/v1/audio/speech/ws bridges client ↔
unspeech with auth, pre-flight flux check, billing from upstream
session.finished.usage, OTel spans.
- apps/server: audio routes moved from /api/v1/openai/audio/* to
/api/v1/audio/* (hard cutover; 404 sentinel tests added).
- apps/server: new /api/v1/audio/voices/streaming proxy reads voices
from unspeech /api/voices?provider=volcengine.
- apps/server: new STREAMING_TTS_UPSTREAM configKV entry +
scripts/seed-streaming-tts.ts.
- stage-ui: new libs/speech/streaming-pipeline.ts opens one ws per LLM
intent (appendText / finish / cancel + onSentence / onError / onDone).
- stage-ui: new libs/speech/tts-session.ts — StageTtsSession interface
with segmenter and streaming adapters; factory dispatches by
capabilities.speech.transport instead of hard-coded provider id.
- stage-ui: providerOfficialSpeechStreaming with capabilities.speech =
{ transport: 'bidirectional-ws' }; settings page with model/voice
picker + ws-based preview.
- stage-ui: Stage.vue chat hooks collapsed to a single currentSession;
hot-swap watcher cancels mid-session on provider/voice/model change;
unmount cancels and drains playback.
Tests:
- 9 streaming-pipeline tests (happy path / buffered / error / cancel /
truncation)
- 11 tts-session tests (factory branch coverage + adapter contracts)
- 4 audio-speech-ws route tests (forwarding / billing / pre-flight /
config-missing)
- 3 legacy-path 404 sentinels in v1 route tests
- Verification doc updated to reflect automated coverage.
This commit is contained in:
+12
-3
@@ -53,7 +53,7 @@ import { createCharacterRoutes } from './routes/characters'
|
||||
import { createChatWsHandlers } from './routes/chat-ws'
|
||||
import { createChatRoutes } from './routes/chats'
|
||||
import { createFluxRoutes } from './routes/flux'
|
||||
import { createV1CompletionsRoutes } from './routes/openai/v1'
|
||||
import { createV1Routes } from './routes/openai/v1'
|
||||
import { createProviderRoutes } from './routes/providers'
|
||||
import { createStripeRoutes } from './routes/stripe'
|
||||
import { createAdminFluxGrantsService } from './services/admin-flux-grants'
|
||||
@@ -204,6 +204,11 @@ export async function buildApp(deps: AppDeps) {
|
||||
logger: useLogger('config-sync').useGlobalConfig(),
|
||||
})
|
||||
|
||||
// Built once so the OpenAI-compat and audio routers share the same closure
|
||||
// (helpers like recordMetrics / recordRequestLog cross both surfaces) but
|
||||
// mount at different prefixes — see the `.route` calls below.
|
||||
const v1Routes = createV1Routes(deps.fluxService, deps.billingService, deps.configKV, deps.requestLogService, deps.ttsMeter, deps.llmRouter, deps.otel?.genAi, deps.otel?.revenue, deps.otel?.rateLimit)
|
||||
|
||||
const builtApp = app
|
||||
.use('*', sessionMiddleware(deps.auth, deps.env))
|
||||
.use('*', bodyLimit({ maxSize: 1024 * 1024 }))
|
||||
@@ -311,9 +316,13 @@ export async function buildApp(deps: AppDeps) {
|
||||
.route('/api/v1/chats', createChatRoutes(deps.chatService))
|
||||
|
||||
/**
|
||||
* V1 routes for official provider.
|
||||
* V1 OpenAI-compatible and audio routes. The factory returns two
|
||||
* sibling routers because the audio surface deliberately lives outside
|
||||
* `/openai/` — its `/voices`, `/voices/streaming`, and `/models`
|
||||
* extensions aren't OpenAI public APIs.
|
||||
*/
|
||||
.route('/api/v1/openai', createV1CompletionsRoutes(deps.fluxService, deps.billingService, deps.configKV, deps.requestLogService, deps.ttsMeter, deps.llmRouter, deps.otel?.genAi, deps.otel?.revenue, deps.otel?.rateLimit))
|
||||
.route('/api/v1/openai', v1Routes.openaiRoutes)
|
||||
.route('/api/v1/audio', v1Routes.audioRoutes)
|
||||
|
||||
/**
|
||||
* Flux routes.
|
||||
|
||||
Reference in New Issue
Block a user