- Third-party
Gemini 3.8 Flash TTS is Google's flagship text-to-speech model for expressive, natural speech generation.
| Model Info | |
|---|---|
| Context Window ↗ | 8,192 tokens |
| More information | link ↗ |
| Pricing |
|
const response = await env.AI.run(
'google/gemini-3.8-flash-tts',
{ text: 'Hello, welcome to Cloudflare AI Gateway!' },
)
console.log(response)curl https://api.cloudflare.com/client/v4/accounts/$CLOUDFLARE_ACCOUNT_ID/ai/run \
--header "Authorization: Bearer $CLOUDFLARE_API_TOKEN" \
--header "Content-Type: application/json" \
--data '{
"model": "google/gemini-3.8-flash-tts",
"input": {
"text": "Hello, welcome to Cloudflare AI Gateway!"
}
}'{
"audio": "https://examples.aig.cloudflare.com/google/gemini-3.8-flash-tts/simple-text-to-speech.wav",
"gatewayMetadata": {
"keySource": "Unified"
}
}Expressive Pause — Generate expressive speech with an inline vocal pause
const response = await env.AI.run(
'google/gemini-3.8-flash-tts',
{
text: 'Welcome to the future of voice. <short pause> Let us build something remarkable together.',
voice: 'Puck',
},
)
console.log(response)curl https://api.cloudflare.com/client/v4/accounts/$CLOUDFLARE_ACCOUNT_ID/ai/run \
--header "Authorization: Bearer $CLOUDFLARE_API_TOKEN" \
--header "Content-Type: application/json" \
--data '{
"model": "google/gemini-3.8-flash-tts",
"input": {
"text": "Welcome to the future of voice. <short pause> Let us build something remarkable together.",
"voice": "Puck"
}
}'{
"audio": "https://examples.aig.cloudflare.com/google/gemini-3.8-flash-tts/expressive-pause.wav",
"gatewayMetadata": {
"keySource": "Unified"
}
}Narrated Passage — Generate a longer narrated passage with a different voice
const response = await env.AI.run(
'google/gemini-3.8-flash-tts',
{
text: 'At sunrise, the research team opened the observatory and watched the first light move across the valley. Every sensor came online in sequence, and the quiet room filled with the soft rhythm of discovery.',
voice: 'Charon',
},
)
console.log(response)curl https://api.cloudflare.com/client/v4/accounts/$CLOUDFLARE_ACCOUNT_ID/ai/run \
--header "Authorization: Bearer $CLOUDFLARE_API_TOKEN" \
--header "Content-Type: application/json" \
--data '{
"model": "google/gemini-3.8-flash-tts",
"input": {
"text": "At sunrise, the research team opened the observatory and watched the first light move across the valley. Every sensor came online in sequence, and the quiet room filled with the soft rhythm of discovery.",
"voice": "Charon"
}
}'{
"audio": "https://examples.aig.cloudflare.com/google/gemini-3.8-flash-tts/narrated-passage.wav",
"gatewayMetadata": {
"keySource": "Unified"
}
}French Greeting — Generate speech from a French greeting using a distinct voice
const response = await env.AI.run(
'google/gemini-3.8-flash-tts',
{
text: "Bonjour et bienvenue. Nous sommes ravis de vous accompagner aujourd'hui.",
voice: 'Kore',
},
)
console.log(response)curl https://api.cloudflare.com/client/v4/accounts/$CLOUDFLARE_ACCOUNT_ID/ai/run \
--header "Authorization: Bearer $CLOUDFLARE_API_TOKEN" \
--header "Content-Type: application/json" \
--data '{
"model": "google/gemini-3.8-flash-tts",
"input": {
"text": "Bonjour et bienvenue. Nous sommes ravis de vous accompagner aujourd'\''hui.",
"voice": "Kore"
}
}'{
"audio": "https://examples.aig.cloudflare.com/google/gemini-3.8-flash-tts/french-greeting.wav",
"gatewayMetadata": {
"keySource": "Unified"
}
}text
stringrequiredmaxLength: 10000The text to convert to speech. Maximum 10,000 characters.voice
stringenum: Zephyr, Puck, Charon, Kore, Fenrir, Leda, Orus, Aoede, Callirrhoe, Autonoe, Enceladus, Iapetus, Umbriel, Algieba, Despina, Erinome, Algenib, Rasalgethi, Laomedeia, Achernar, Alnilam, Schedar, Gacrux, Pulcherrima, Achird, Zubenelgenubi, Vindemiatrix, Sadachbia, Sadaltager, SulafatThe voice to use for speech synthesistemperature
numberminimum: 0maximum: 2Controls randomness in generation (0-2)topP
numberminimum: 0maximum: 1Nucleus sampling threshold (0-1). Tokens with cumulative probability up to topP are consideredtopK
integerexclusiveMinimum: 0maximum: 9007199254740991Only sample from the top K tokens. Smaller K = more focused, larger K = more diversemaxOutputTokens
integerexclusiveMinimum: 0maximum: 9007199254740991Maximum number of tokens to generate▶stopSequences[]
arraySequences where the model will stop generating further tokensaudio
stringBase64-encoded audio data (WAV format)