mirror of
https://github.com/simstudioai/sim.git
synced 2026-09-24 15:45:35 +08:00
improvement(chat-voice): modernize ElevenLabs TTS to Flash v2.5 (#4943)
* improvement(chat-voice): modernize ElevenLabs TTS to Flash v2.5 - Switch default TTS model from eleven_turbo_v2_5 to eleven_flash_v2_5 (ElevenLabs recommends Flash over Turbo in all cases; ~75ms latency) - Drop deprecated optimize_streaming_latency knob plus legacy use_pvc_as_ivc / enable_ssml_parsing flags - Move output_format to the query string and raise it from mp3_22050_32 to mp3_44100_128 for higher audio quality - Switch apply_text_normalization from off to auto for correct number/date pronunciation * improvement(chat-voice): default to Jessica voice (Flash v2.5-optimized) Replace the legacy Sarah default (EXAVITQu4vr4xnSDxMaL), which has no high-quality eleven_flash_v2_5 base, with Jessica (cgSgspJ2msm6clMCkdW9) — a current premade conversational voice verified against the live account and optimized for Flash v2.5.
This commit is contained in:
@@ -92,7 +92,8 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
return new Response('ElevenLabs service not configured', { status: 503 })
|
||||
}
|
||||
|
||||
const endpoint = `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}/stream`
|
||||
const query = new URLSearchParams({ output_format: 'mp3_44100_128' })
|
||||
const endpoint = `https://api.elevenlabs.io/v1/text-to-speech/${voiceId}/stream?${query.toString()}`
|
||||
|
||||
const response = await fetch(endpoint, {
|
||||
method: 'POST',
|
||||
@@ -104,17 +105,13 @@ export const POST = withRouteHandler(async (request: NextRequest) => {
|
||||
body: JSON.stringify({
|
||||
text,
|
||||
model_id: modelId,
|
||||
optimize_streaming_latency: 4,
|
||||
output_format: 'mp3_22050_32', // Fastest format
|
||||
voice_settings: {
|
||||
stability: 0.5,
|
||||
similarity_boost: 0.8,
|
||||
style: 0.0,
|
||||
use_speaker_boost: false,
|
||||
},
|
||||
enable_ssml_parsing: false,
|
||||
apply_text_normalization: 'off',
|
||||
use_pvc_as_ivc: false,
|
||||
apply_text_normalization: 'auto',
|
||||
}),
|
||||
})
|
||||
|
||||
|
||||
@@ -44,7 +44,7 @@ interface ChatRequestPayload {
|
||||
}
|
||||
|
||||
const DEFAULT_VOICE_SETTINGS = {
|
||||
voiceId: 'EXAVITQu4vr4xnSDxMaL', // Default ElevenLabs voice (Bella)
|
||||
voiceId: 'cgSgspJ2msm6clMCkdW9', // Default ElevenLabs voice (Jessica) — Flash v2.5-optimized
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -79,7 +79,7 @@ export function useAudioStreaming(sharedAudioContextRef?: RefObject<AudioContext
|
||||
const { text, options } = item
|
||||
const {
|
||||
voiceId,
|
||||
modelId = 'eleven_turbo_v2_5',
|
||||
modelId = 'eleven_flash_v2_5',
|
||||
chatId,
|
||||
onAudioStart,
|
||||
onAudioEnd,
|
||||
|
||||
@@ -5,7 +5,7 @@ export const ttsStreamBodySchema = z
|
||||
.object({
|
||||
text: z.string().min(1),
|
||||
voiceId: z.string().min(1),
|
||||
modelId: z.string().optional().default('eleven_turbo_v2_5'),
|
||||
modelId: z.string().optional().default('eleven_flash_v2_5'),
|
||||
chatId: z.string().min(1),
|
||||
})
|
||||
.passthrough()
|
||||
|
||||
Reference in New Issue
Block a user