curl --request POST \
--url https://api.boson.ai/v1/audio/speech \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": "Hello, this is a test.",
"model": "higgs-tts-3",
"voice": "default",
"response_format": "mp3",
"stream": false,
"ref_audio": "<string>",
"ref_text": "<string>",
"enable_tn": true,
"tn_language": null,
"timestamps": false
}
'import requests
url = "https://api.boson.ai/v1/audio/speech"
payload = {
"input": "Hello, this is a test.",
"model": "higgs-tts-3",
"voice": "default",
"response_format": "mp3",
"stream": False,
"ref_audio": "<string>",
"ref_text": "<string>",
"enable_tn": True,
"tn_language": None,
"timestamps": False
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: 'Hello, this is a test.',
model: 'higgs-tts-3',
voice: 'default',
response_format: 'mp3',
stream: false,
ref_audio: '<string>',
ref_text: '<string>',
enable_tn: true,
tn_language: null,
timestamps: false
})
};
fetch('https://api.boson.ai/v1/audio/speech', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.boson.ai/v1/audio/speech",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => 'Hello, this is a test.',
'model' => 'higgs-tts-3',
'voice' => 'default',
'response_format' => 'mp3',
'stream' => false,
'ref_audio' => '<string>',
'ref_text' => '<string>',
'enable_tn' => true,
'tn_language' => null,
'timestamps' => false
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.boson.ai/v1/audio/speech"
payload := strings.NewReader("{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.boson.ai/v1/audio/speech")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.boson.ai/v1/audio/speech")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}"
response = http.request(request)
puts response.read_body"<string>"{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}Create a speech
Generate speech audio from text. Returns an audio file, or a stream of raw PCM chunks when stream is true. The body may be JSON or multipart/form-data — the latter lets you upload ref_audio as a raw file instead of base64-encoding it.
curl --request POST \
--url https://api.boson.ai/v1/audio/speech \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": "Hello, this is a test.",
"model": "higgs-tts-3",
"voice": "default",
"response_format": "mp3",
"stream": false,
"ref_audio": "<string>",
"ref_text": "<string>",
"enable_tn": true,
"tn_language": null,
"timestamps": false
}
'import requests
url = "https://api.boson.ai/v1/audio/speech"
payload = {
"input": "Hello, this is a test.",
"model": "higgs-tts-3",
"voice": "default",
"response_format": "mp3",
"stream": False,
"ref_audio": "<string>",
"ref_text": "<string>",
"enable_tn": True,
"tn_language": None,
"timestamps": False
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: 'Hello, this is a test.',
model: 'higgs-tts-3',
voice: 'default',
response_format: 'mp3',
stream: false,
ref_audio: '<string>',
ref_text: '<string>',
enable_tn: true,
tn_language: null,
timestamps: false
})
};
fetch('https://api.boson.ai/v1/audio/speech', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.boson.ai/v1/audio/speech",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => 'Hello, this is a test.',
'model' => 'higgs-tts-3',
'voice' => 'default',
'response_format' => 'mp3',
'stream' => false,
'ref_audio' => '<string>',
'ref_text' => '<string>',
'enable_tn' => true,
'tn_language' => null,
'timestamps' => false
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.boson.ai/v1/audio/speech"
payload := strings.NewReader("{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.boson.ai/v1/audio/speech")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.boson.ai/v1/audio/speech")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": \"Hello, this is a test.\",\n \"model\": \"higgs-tts-3\",\n \"voice\": \"default\",\n \"response_format\": \"mp3\",\n \"stream\": false,\n \"ref_audio\": \"<string>\",\n \"ref_text\": \"<string>\",\n \"enable_tn\": true,\n \"tn_language\": null,\n \"timestamps\": false\n}"
response = http.request(request)
puts response.read_body"<string>"{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}Authorizations
Your Boson API key, sent as Authorization: Bearer $BOSON_API_KEY.
Body
Text to convert to speech. May contain inline tags. Inputs longer than 5000 characters return a 400 input_too_long. This is a validation limit; we recommend keeping each request to roughly 300 characters, as longer text is not yet reliably supported.
1 - 5000"Hello, this is a test."
TTS model ID / public alias. Resolved to the served model server-side.
higgs-tts-3 Preset voice name or custom voice ID. Mutually exclusive with ref_audio / ref_text when explicitly provided.
Output audio format. Streaming requires pcm.
mp3, opus, pcm, wav, aac, flac If true, stream raw PCM chunks as they are decoded. Requires response_format to be pcm. Speed adjustment is not supported when streaming.
Inline reference audio for one-off cloning: an http(s) URL, data URI, or base64-encoded raw audio bytes. Supported formats: AAC, WAV, MP3, FLAC, OPUS. Inline (base64 / data-URI) payloads: max 10 MB.
Recommended transcript of ref_audio.
Expand non-standard words into spoken form before synthesis, including numbers, dates, times, currency, units, URLs, email addresses, phone numbers, abbreviations, and symbols. Enabled by default; set to false to synthesize the text exactly as written. Text with nothing to normalize is passed through unchanged. Inline tags are preserved.
Normalization language as an ISO 639-1 code. Defaults to null, which auto-detects Chinese, English, French, German, Hungarian, Italian, Japanese, Korean, Portuguese, Russian, Spanish, Swedish, Thai, and Vietnamese. Set it explicitly for any other language, and for short or mixed-language text where detection is least reliable. An unsupported code returns 400 invalid_field_value, including when normalization is disabled.
Attach word-level timestamps to the response. The response becomes a JSON envelope carrying the word list and the audio base64-encoded in the requested response_format, instead of raw audio bytes. Requires stream to be false; combining the two returns 400 timestamps_streaming_unsupported. Available for English, Chinese, and Spanish; other languages return a null word list. Overrides enable_tn, which does not run when timestamps are requested.
Response
Generated audio. The content type depends on response_format. When timestamps is true the response is application/json instead, carrying the audio base64-encoded alongside the word list.
The response is of type file.