SpyBara
Go Premium

Documentation 2026-08-21 18:57 UTC to 2026-08-25 22:58 UTC

11 files changed +217 −48. View all changes and history on the product overview
2026
Mon 31 20:58 Tue 25 22:58 Fri 21 18:57 Thu 20 15:58 Wed 19 18:02 Tue 18 04:01 Thu 13 22:00 Wed 12 23:59 Tue 11 20:57 Sat 8 23:00 Fri 7 17:57 Thu 6 20:01 Mon 3 23:00 Sat 1 01:59
Details

127# Image generation127# Image generation

128image_req = client.image.prepare(128image_req = client.image.prepare(

129 prompt="A sleek modern laptop on a minimalist desk",129 prompt="A sleek modern laptop on a minimalist desk",

130 model="grok-imagine-image",130 model="grok-imagine-image-2.0",

131 batch_request_id="img_001",131 batch_request_id="img_001",

132)132)

133batch_requests.append(image_req)133batch_requests.append(image_req)


135# Image edit135# Image edit

136image_edit_req = client.image.prepare(136image_edit_req = client.image.prepare(

137 prompt="Add a rainbow in the background",137 prompt="Add a rainbow in the background",

138 model="grok-imagine-image",138 model="grok-imagine-image-2.0",

139 image_url="https://picsum.photos/800",139 image_url="https://picsum.photos/800",

140 batch_request_id="img_edit_001",140 batch_request_id="img_edit_001",

141)141)


144# Video generation144# Video generation

145video_req = client.video.prepare(145video_req = client.video.prepare(

146 prompt="A product rotating on a turntable with dramatic lighting",146 prompt="A product rotating on a turntable with dramatic lighting",

147 model="grok-imagine-video",147 model="grok-imagine-video-1.5",

148 batch_request_id="vid_001",148 batch_request_id="vid_001",

149)149)

150batch_requests.append(video_req)150batch_requests.append(video_req)


240 batch_request: {240 batch_request: {

241 image_generation: {241 image_generation: {

242 prompt: "A sleek modern laptop on a minimalist desk",242 prompt: "A sleek modern laptop on a minimalist desk",

243 model: "grok-imagine-image",243 model: "grok-imagine-image-2.0",

244 },244 },

245 },245 },

246});246});


251 batch_request: {251 batch_request: {

252 image_edit: {252 image_edit: {

253 prompt: "Add a rainbow in the background",253 prompt: "Add a rainbow in the background",

254 model: "grok-imagine-image",254 model: "grok-imagine-image-2.0",

255 image: { url: "https://picsum.photos/800", type: "image_url" },255 image: { url: "https://picsum.photos/800", type: "image_url" },

256 },256 },

257 },257 },


263 batch_request: {263 batch_request: {

264 video_generation: {264 video_generation: {

265 prompt: "A product rotating on a turntable with dramatic lighting",265 prompt: "A product rotating on a turntable with dramatic lighting",

266 model: "grok-imagine-video",266 model: "grok-imagine-video-1.5",

267 },267 },

268 },268 },

269});269});


870{"custom_id": "chat-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "grok-4.3", "messages": [{"role": "user", "content": "Classify this as positive, negative, or neutral: The product exceeded my expectations!"}]}}870{"custom_id": "chat-1", "method": "POST", "url": "/v1/chat/completions", "body": {"model": "grok-4.3", "messages": [{"role": "user", "content": "Classify this as positive, negative, or neutral: The product exceeded my expectations!"}]}}

871{"custom_id": "search-1", "method": "POST", "url": "/v1/responses", "body": {"model": "grok-4.3", "tools": [{"type": "web_search"}, {"type": "x_search"}], "input": [{"role": "user", "content": "What are the latest SpaceX launches?"}]}}871{"custom_id": "search-1", "method": "POST", "url": "/v1/responses", "body": {"model": "grok-4.3", "tools": [{"type": "web_search"}, {"type": "x_search"}], "input": [{"role": "user", "content": "What are the latest SpaceX launches?"}]}}

872{"custom_id": "mcp-1", "method": "POST", "url": "/v1/responses", "body": {"model": "grok-4.3", "tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp"}], "input": [{"role": "user", "content": "What does the xai-sdk-python repo do?"}]}}872{"custom_id": "mcp-1", "method": "POST", "url": "/v1/responses", "body": {"model": "grok-4.3", "tools": [{"type": "mcp", "server_label": "deepwiki", "server_url": "https://mcp.deepwiki.com/mcp"}], "input": [{"role": "user", "content": "What does the xai-sdk-python repo do?"}]}}

873{"custom_id": "img-1", "method": "POST", "url": "/v1/images/generations", "body": {"model": "grok-imagine-image", "prompt": "A futuristic city skyline at sunset"}}873{"custom_id": "img-1", "method": "POST", "url": "/v1/images/generations", "body": {"model": "grok-imagine-image-2.0", "prompt": "A futuristic city skyline at sunset"}}

874{"custom_id": "img-edit-1", "method": "POST", "url": "/v1/images/edits", "body": {"model": "grok-imagine-image", "prompt": "Add a rainbow", "image": {"url": "https://picsum.photos/800"}}}874{"custom_id": "img-edit-1", "method": "POST", "url": "/v1/images/edits", "body": {"model": "grok-imagine-image-2.0", "prompt": "Add a rainbow", "image": {"url": "https://picsum.photos/800"}}}

875{"custom_id": "vid-1", "method": "POST", "url": "/v1/videos/generations", "body": {"model": "grok-imagine-video", "prompt": "A rocket launching from Mars", "duration": 8}}875{"custom_id": "vid-1", "method": "POST", "url": "/v1/videos/generations", "body": {"model": "grok-imagine-video-1.5", "prompt": "A rocket launching from Mars", "duration": 8}}

876{"custom_id": "vid-edit-1", "method": "POST", "url": "/v1/videos/edits", "body": {"model": "grok-imagine-video", "prompt": "Make it slow motion", "video": {"url": "https://lorem.video/cat_360p_3s"}}}876{"custom_id": "vid-edit-1", "method": "POST", "url": "/v1/videos/edits", "body": {"model": "grok-imagine-video", "prompt": "Make it slow motion", "video": {"url": "https://lorem.video/cat_360p_3s"}}}

877{"custom_id": "vid-ext-1", "method": "POST", "url": "/v1/videos/extensions", "body": {"model": "grok-imagine-video", "prompt": "The camera slowly pans to reveal a sunset", "video": {"url": "https://lorem.video/cat_360p_3s"}, "duration": 6}}877{"custom_id": "vid-ext-1", "method": "POST", "url": "/v1/videos/extensions", "body": {"model": "grok-imagine-video", "prompt": "The camera slowly pans to reveal a sunset", "video": {"url": "https://lorem.video/cat_360p_3s"}, "duration": 6}}

878```878```


891| `/v1/videos/edits` | [Video editing](/developers/model-capabilities/video/editing) |891| `/v1/videos/edits` | [Video editing](/developers/model-capabilities/video/editing) |

892| `/v1/videos/extensions` | [Video extension](/developers/model-capabilities/video/extension) |892| `/v1/videos/extensions` | [Video extension](/developers/model-capabilities/video/extension) |

893 893 

894Only batch-enabled models are accepted. Image and video requests currently support `grok-imagine-image` and `grok-imagine-video`; other Imagine models (including `grok-imagine-image-2.0` and `grok-imagine-video-1.5`) are rejected with "not supported for batch processing".894Only batch-enabled models are accepted. Refer to the relevant [model pages](/developers/models) for the most up-to-date information; models that are not batch-enabled are rejected with "not supported for batch processing".

895 895 

896Upload the file via the [Files API](/developers/files), then create a batch referencing it:896Upload the file via the [Files API](/developers/files), then create a batch referencing it:

897 897 

grok-4-6.md +1 −1

Details

75|----------|-------|75|----------|-------|

76| Model name | `grok-4.6` |76| Model name | `grok-4.6` |

77| Context window | 500,000 tokens |77| Context window | 500,000 tokens |

78| Knowledge cutoff | February 1, 2026 |78| Knowledge cutoff | January 2026 |

79| Modalities | Text and image input; text output |79| Modalities | Text and image input; text output |

80| Output limit | No text output limit |80| Output limit | No text output limit |

81| Input price | $2.00 / 1M tokens |81| Input price | $2.00 / 1M tokens |

Details

103 103 

104## Call control104## Call control

105 105 

106Use `refer` to transfer the caller to another PSTN or SIP destination:106Use `refer` to transfer the caller to another PSTN or SIP destination. The request blocks until the transfer resolves; the HTTP status reports whether the destination answered. See [FAQ](#faq) for status codes, failed-transfer session behavior, and conversation resumption.

107 107 

108```bash customLanguage="bash"108```bash customLanguage="bash"

109curl -X POST "https://api.x.ai/v1/realtime/calls/$CALL_ID/refer" \109curl -X POST "https://api.x.ai/v1/realtime/calls/$CALL_ID/refer" \


180 180 

1811. In your carrier, contact center, or PBX, create an outbound route or SIP trunk.1811. In your carrier, contact center, or PBX, create an outbound route or SIP trunk.

1822. Set the destination to `sip:{number}@sip.voice.x.ai;transport=tls`.1822. Set the destination to `sip:{number}@sip.voice.x.ai;transport=tls`.

183 

184## FAQ

185 

186These answers cover the Speech to Speech API SIP path: transfer success and failure with reason codes, what happens when a transfer fails, and how to continue a conversation on a later SIP call.

187 

188### How do I get transfer success or failure with a reason code?

189 

190There is no separate WebSocket transfer event. `POST /v1/realtime/calls/{call_id}/refer` returns the outcome synchronously.

191 

192A `200` with an empty JSON body (`{}`) means the transfer completed: the REFER succeeded and the destination answered, not merely that the REFER was accepted or that the destination started ringing.

193 

194Downstream SIP rejections return `502` with the carrier's SIP status in the body:

195 

196```json customLanguage="json"

197{

198 "error": "transfer rejected by downstream: SIP 403 Forbidden"

199}

200```

201 

202Other statuses are also possible:

203 

204| Status | Meaning |

205| --- | --- |

206| `400` | `target_uri` is not a `tel:` or `sip:` URI |

207| `404` | No SIP participant on this call |

208| `502` | Transfer rejected by downstream; the body includes the SIP code and reason when available |

209| `504` | Transfer timed out |

210| `500` | Internal error |

211 

212### Does the session stay usable if a transfer fails?

213 

214Yes. While the REFER is pending, the WebSocket stays open and the caller hears dialtone. After a `502`, that same realtime session stays connected. The failed transfer is a no-op on the xAI side: the caller remains on the call, and the agent can keep talking.

215 

216This path does not automatically start a new session or inject a failure-reason prompt. If the agent should tell the caller why the transfer failed, read the `refer` HTTP response and continue on this session, or resume later as below.

217 

218### Can I reattach a resumed conversation to a new SIP call?

219 

220[Session resumption](/developers/model-capabilities/audio/speech-to-speech#session-resumption) caches transcripts and tool results so a later SIP call can continue the same conversation. You must opt in on both the original session and the resuming session. History expires after 30 minutes of inactivity.

221 

2221. On the first call, open `wss://api.x.ai/v1/realtime?call_id={call_id}` and immediately send `session.update` with `resumption.enabled` set to `true`. Save that `call_id`.

2232. On the later call, open `wss://api.x.ai/v1/realtime?call_id={new_call_id}&conversation_id={saved_call_id}` and send the same `session.update` again. That restores the prior turns and keeps saving for future reconnects.

2243. There is no dedicated resumption-complete event. Restore happens as soon as that `session.update` is processed; replayed turns arrive as `conversation.item.created` events.

225 

226```json customLanguage="json"

227{

228 "type": "session.update",

229 "session": {

230 "resumption": { "enabled": true }

231 }

232}

233```

234 

235### Can I transfer to a Twilio SIP Domain?

236 

237Yes. A `sip:` `target_uri` is forwarded as the SIP `Refer-To` value, including URI parameters used to route into a Twilio Programmable Voice SIP Domain or conference. Custom SIP headers are not sent on the REFER.

Details

201 201 

202| Parameter | Type | Default | Description |202| Parameter | Type | Default | Description |

203|-----------|------|---------|-------------|203|-----------|------|---------|-------------|

204| `sample_rate` | integer | `16000` | Audio sample rate in Hz. |204| `sample_rate` | integer | `16000` | Audio sample rate in Hz. With `encoding=opus`: `8000`, `16000`, `24000`, or `48000` only. |

205| `encoding` | string | `pcm` | Audio encoding: `pcm`, `mulaw`, or `alaw`. |205| `encoding` | string | `pcm` | Audio encoding: `pcm`, `mulaw`, `alaw`, or `opus`. See [Opus Streaming](#opus-streaming). |

206| `interim_results` | boolean | `false` | When `true`, emit partial transcripts `is_final=false` every ~500 ms. |206| `interim_results` | boolean | `false` | When `true`, emit partial transcripts `is_final=false` every ~500 ms. |

207| `endpointing` | integer | `10` | Silence duration (ms) before utterance-final event. Range: 0–5000. `0` = fire on any VAD silence boundary. |207| `endpointing` | integer | `400` | Silence duration (ms) before utterance-final event. Range: 0–5000. `0` = fire on any VAD silence boundary. |

208| `language` | string | | Language code for text formatting. See [Supported Languages](#supported-languages). |208| `language` | string | | Language code for text formatting. See [Supported Languages](#supported-languages). |

209| `diarize` | boolean | | When `true`, enables speaker diarization. Words include a `speaker` field identifying the detected speaker. |209| `diarize` | boolean | | When `true`, enables speaker diarization. Words include a `speaker` field identifying the detected speaker. |

210| `filler_words` | boolean | `false` | When `true`, filler words (e.g. `uh`, `um`, `er`) are included in the transcript. When `false` (default), filler words are automatically removed. |210| `filler_words` | boolean | `false` | When `true`, filler words (e.g. `uh`, `um`, `er`) are included in the transcript. When `false` (default), filler words are automatically removed. |

211| `multichannel` | boolean | `false` | Per-channel transcription. Requires `channels` ≥ 2. |211| `multichannel` | boolean | `false` | Per-channel transcription. Requires `channels` ≥ 2. Not supported with `encoding=opus`. |

212| `channels` | integer | `1` | Number of interleaved audio channels (max 8). |212| `channels` | integer | `1` | Number of interleaved audio channels (max 8). |

213| `keyterm` | string | | A key term to bias transcription toward (e.g. product names, proper nouns). Repeat the parameter for multiple terms (e.g. `keyterm=Understand+The+Universe`). Max 100 terms, each up to 50 characters. |213| `keyterm` | string | | A key term to bias transcription toward (e.g. product names, proper nouns). Repeat the parameter for multiple terms (e.g. `keyterm=Understand+The+Universe`). Max 100 terms, each up to 50 characters. |

214| `smart_turn` | number | | End-of-turn detection threshold (0.0–1.0). When set, enables Smart Turn — an ML model predicts whether the speaker has finished their thought at each silence boundary. See [Smart Turn](#smart-turn). |214| `smart_turn` | number | | End-of-turn detection threshold (0.0–1.0). When set, enables Smart Turn — an ML model predicts whether the speaker has finished their thought at each silence boundary. See [Smart Turn](#smart-turn). |


222| `transcript.created` | Server ready — wait for this before sending audio. |222| `transcript.created` | Server ready — wait for this before sending audio. |

223| `transcript.partial` | Transcript result with `text`, `words`, `is_final`, `speech_final`, `start`, `duration`. Includes `channel_index` when `multichannel=true`. Includes `end_of_turn_confidence` when `smart_turn` is enabled. |223| `transcript.partial` | Transcript result with `text`, `words`, `is_final`, `speech_final`, `start`, `duration`. Includes `channel_index` when `multichannel=true`. Includes `end_of_turn_confidence` when `smart_turn` is enabled. |

224| `transcript.done` | Final transcript after `audio.done`. `duration` always present. Includes `channel_index` when `multichannel=true` — one event sent per channel. Connection closes after this. |224| `transcript.done` | Final transcript after `audio.done`. `duration` always present. Includes `channel_index` when `multichannel=true` — one event sent per channel. Connection closes after this. |

225| `error` | Error with `message` field. Connection stays open. |225| `error` | Error with `message` field. Most errors (including undecodable audio frames) close the connection; client message parse errors keep it open. |

226 226 

227The `transcript.partial` event uses `is_final` and `speech_final` to convey three states:227The `transcript.partial` event uses `is_final` and `speech_final` to convey three states:

228 228 


234 234 

235### Client Messages235### Client Messages

236 236 

237* **Binary frames** — raw audio in the specified encoding (streamed in real-time-paced chunks, e.g. 100 ms)237* **Binary frames** — raw audio in the specified encoding (streamed in real-time-paced chunks, e.g. 100 ms). With `encoding=opus`, each binary frame must contain exactly one Opus packet — see [Opus Streaming](#opus-streaming).

238* **`{"type": "finalize"}`** — force the current utterance to finalize as `speech_final` immediately for PTT238* **`{"type": "finalize"}`** — force the current utterance to finalize as `speech_final` immediately for PTT

239* **`{"type": "audio.done"}`** — signal end of audio, triggers `transcript.done`239* **`{"type": "audio.done"}`** — signal end of audio, triggers `transcript.done`

240 240 


252{"type": "Finalize", "channel": 0}252{"type": "Finalize", "channel": 0}

253```253```

254 254 

255### Opus Streaming

256 

257Set `encoding=opus` to stream compressed audio instead of raw PCM — roughly 4 KB/s at 24 kHz versus 48 KB/s for PCM16, with no client-side resampling.

258 

259**How it works:**

260 

261* Send **exactly one raw Opus packet per binary WebSocket frame**. Opus packets don't mark their own boundaries — the WebSocket framing does — so never concatenate packets or split one across frames.

262* Encode mono audio at `8000`, `16000`, `24000`, or `48000` Hz. Other sample rates are rejected with `400`.

263* `multichannel` is not supported with Opus — mono only.

264* Send raw packets, not containers. To transcribe an Ogg-Opus or WebM file, use the [batch endpoint](#supported-audio-formats) instead — it auto-detects containers.

265* If a frame can't be decoded, the server sends an `error` event and closes the session. Misframed packets often decode as noise rather than an error, so double-check that your encoder emits one whole packet per frame.

266 

267**Example URL:**

268 

269```

270wss://api.x.ai/v1/stt?sample_rate=24000&encoding=opus&interim_results=true

271```

272 

273**Typical use case:** Dictation and live transcription from mobile or other bandwidth-constrained clients, where streaming raw PCM is wasteful. Most platform audio APIs and WebRTC stacks produce Opus packets natively.

274 

255### Multichannel Streaming275### Multichannel Streaming

256 276 

257When `multichannel=true` and `channels` ≥ 2, the server transcribes each audio channel independently. Send interleaved multichannel PCM (e.g. L,R,L,R,… for stereo) as binary frames, and the server de-interleaves and processes each channel in parallel.277When `multichannel=true` and `channels` ≥ 2, the server transcribes each audio channel independently. Send interleaved multichannel PCM (e.g. L,R,L,R,… for stereo) as binary frames, and the server de-interleaves and processes each channel in parallel.


421* **Enable `interim_results`** for responsive UX — show transcription as the user speaks441* **Enable `interim_results`** for responsive UX — show transcription as the user speaks

422* **Use `language=en`** to enable text formatting — numbers and currencies are written in their standard form442* **Use `language=en`** to enable text formatting — numbers and currencies are written in their standard form

423* **Send 100 ms audio chunks** (3,200 bytes at 16 kHz PCM16) for a good balance of latency and efficiency443* **Send 100 ms audio chunks** (3,200 bytes at 16 kHz PCM16) for a good balance of latency and efficiency

444* **Use `encoding=opus` on bandwidth-constrained clients** — ~4 KB/s at 24 kHz versus 48 KB/s for raw PCM. See [Opus Streaming](#opus-streaming)

424* **Wait for `transcript.created`** before sending audio — the server needs to initialize its ASR backend445* **Wait for `transcript.created`** before sending audio — the server needs to initialize its ASR backend

425 446 

426## Error Handling447## Error Handling

Details

81 81 

82### Multiple Images82### Multiple Images

83 83 

84Generate multiple images in a single request using the `sample_batch()` method and the `n` parameter. This returns a list of `ImageResponse` objects.84Generate multiple images in a single request with the `n` parameter (`1`–`10`). On the REST API and OpenAI-compatible SDKs, `n` is optional and defaults to `1`. The xAI Python SDK uses `sample()` for a single image and `sample_batch(n=...)` for more than one — `n` is required on `sample_batch()`.

85 85 

86```python customLanguage="pythonXAI"86```python customLanguage="pythonXAI"

87import xai_sdk87import xai_sdk


165 165 

166### Aspect Ratio166### Aspect Ratio

167 167 

168Control image dimensions with the `aspect_ratio` parameter.168Control image dimensions with the `aspect_ratio` parameter. When omitted, the default is `auto`, which lets the model pick the best ratio for the prompt.

169 169 

170| Ratio | Use case |170| Ratio | Use case |

171|-------|----------|171|-------|----------|


174| `4:3` / `3:4` | Presentations, portraits |174| `4:3` / `3:4` | Presentations, portraits |

175| `3:2` / `2:3` | Photography |175| `3:2` / `2:3` | Photography |

176| `2:1` / `1:2` | Banners, headers |176| `2:1` / `1:2` | Banners, headers |

177| `19.5:9` / `9:19.5` | Modern smartphone displays |177| `19.5:9` / `9:19.5` | Modern smartphone displays (iPhone) |

178| `20:9` / `9:20` | Ultra-wide displays |178| `20:9` / `9:20` | Modern smartphone displays (Android) |

179| `21:9` | Cinematic widescreen |

180| `5:2` | Wide banners |

179| `auto` | Model auto-selects the best ratio for the prompt |181| `auto` | Model auto-selects the best ratio for the prompt |

180 182 

181```python customLanguage="pythonXAI"183```python customLanguage="pythonXAI"


253 255 

254### Resolution256### Resolution

255 257 

256You can specify different resolutions of the output image. Currently supported image resolutions are:258You can specify different resolutions of the output image with the `resolution` parameter. Currently supported image resolutions are:

257 259 

258* 1k260* `1k` (default when omitted)

259* 2k261* `2k`

260 262 

261```python customLanguage="pythonXAI"263```python customLanguage="pythonXAI"

262import xai_sdk264import xai_sdk


337 339 

338Control generation quality with the optional `quality` parameter. Allowed values are `low` and `medium`. When omitted, the default is `medium`. The parameter is only supported for `grok-imagine-image-2.0`.340Control generation quality with the optional `quality` parameter. Allowed values are `low` and `medium`. When omitted, the default is `medium`. The parameter is only supported for `grok-imagine-image-2.0`.

339 341 

342```python customLanguage="pythonXAI"

343import xai_sdk

344 

345client = xai_sdk.Client()

346 

347response = client.image.sample(

348 prompt="A watercolor painting of a lighthouse at dawn",

349 model="grok-imagine-image-2.0",

350 quality="low",

351)

352 

353print(response.url)

354```

355 

356```bash

357curl -X POST https://api.x.ai/v1/images/generations \

358 -H "Content-Type: application/json" \

359 -H "Authorization: Bearer $XAI_API_KEY" \

360 -d '{

361 "model": "grok-imagine-image-2.0",

362 "prompt": "A watercolor painting of a lighthouse at dawn",

363 "quality": "low"

364 }'

365```

366 

340### Base64 Output367### Base64 Output

341 368 

342For embedding images directly without downloading, request base64:369Control the output format with the `response_format` parameter. When omitted, the default is `url`, which returns temporary hosted URLs. For embedding images directly without downloading, request base64:

343 370 

344```python customLanguage="pythonXAI"371```python customLanguage="pythonXAI"

345import xai_sdk372import xai_sdk

Details

244| Limit | Max **3** voices per request |244| Limit | Max **3** voices per request |

245| Prompt | Reference voices by index: `<AUDIO_0>`, `<AUDIO_1>`, `<AUDIO_2>` |245| Prompt | Reference voices by index: `<AUDIO_0>`, `<AUDIO_1>`, `<AUDIO_2>` |

246 246 

247Generated videos include an audio track by default.247Generated videos include an audio track by default. Pass `generate_audio=False` to request a silent video:

248 

249```python customLanguage="pythonXAI"

250import os

251import xai_sdk

252 

253client = xai_sdk.Client(api_key=os.getenv("XAI_API_KEY"))

254 

255response = client.video.generate(

256 prompt="A paper boat drifting down a rain-soaked street",

257 model="grok-imagine-video-1.5",

258 generate_audio=False,

259)

260 

261print(response.url)

262```

263 

264```bash

265curl -X POST https://api.x.ai/v1/videos/generations \

266 -H "Content-Type: application/json" \

267 -H "Authorization: Bearer $XAI_API_KEY" \

268 -d '{

269 "model": "grok-imagine-video-1.5",

270 "prompt": "A paper boat drifting down a rain-soaked street",

271 "generate_audio": false

272 }'

273```

248 274 

249### Example275### Example

250 276 

Details

147 147 

148`reference_audios` accepts preset voices; voice references with your own audio files are available to trusted partners [on request](https://x.ai/contact-sales?interest=imagine). Use a voice alongside reference images or on its own, and tag voices in the prompt as `<AUDIO_0>`, `<AUDIO_1>`, and `<AUDIO_2>` (with `<IMAGE_0>`… when you also pass images).148`reference_audios` accepts preset voices; voice references with your own audio files are available to trusted partners [on request](https://x.ai/contact-sales?interest=imagine). Use a voice alongside reference images or on its own, and tag voices in the prompt as `<AUDIO_0>`, `<AUDIO_1>`, and `<AUDIO_2>` (with `<IMAGE_0>`… when you also pass images).

149 149 

150```python customLanguage="pythonXAI"

151import os

152import xai_sdk

153 

154client = xai_sdk.Client(api_key=os.getenv("XAI_API_KEY"))

155 

156response = client.video.generate(

157 prompt="The person from <IMAGE_1> presents the product from <IMAGE_2> on the set from <IMAGE_3>, speaking with the voice from <AUDIO_0>. A second speaker with the voice from <AUDIO_1> replies.",

158 model="grok-imagine-video-1.5",

159 reference_image_urls=[

160 "<IMAGE_URL_1>",

161 "<IMAGE_URL_2>",

162 "<IMAGE_URL_3>",

163 ],

164 reference_audios=[

165 {"voice_id": "eve"},

166 {"voice_id": "leo"},

167 ],

168 duration=8,

169 aspect_ratio="9:16",

170 resolution="720p",

171)

172 

173print(response.url)

174```

175 

150```python customLanguage="pythonRequests"176```python customLanguage="pythonRequests"

151import os177import os

152import time178import time


162 headers=headers,188 headers=headers,

163 json={189 json={

164 "model": "grok-imagine-video-1.5",190 "model": "grok-imagine-video-1.5",

165 "prompt": "The person from <IMAGE_1> speaks to camera with the voice from <AUDIO_0>.",191 "prompt": "The person from <IMAGE_1> presents the product from <IMAGE_2> on the set from <IMAGE_3>, speaking with the voice from <AUDIO_0>. A second speaker with the voice from <AUDIO_1> replies.",

166 "reference_images": [{"url": "<IMAGE_URL_1>"}],192 "reference_images": [

167 "reference_audios": [{"voice_id": "eve"}],193 {"url": "<IMAGE_URL_1>"},

194 {"url": "<IMAGE_URL_2>"},

195 {"url": "<IMAGE_URL_3>"},

196 ],

197 "reference_audios": [

198 {"voice_id": "eve"},

199 {"voice_id": "leo"},

200 ],

168 "duration": 8,201 "duration": 8,

169 "aspect_ratio": "9:16",202 "aspect_ratio": "9:16",

170 "resolution": "720p",203 "resolution": "720p",


194 -H "Authorization: Bearer $XAI_API_KEY" \227 -H "Authorization: Bearer $XAI_API_KEY" \

195 -d '{228 -d '{

196 "model": "grok-imagine-video-1.5",229 "model": "grok-imagine-video-1.5",

197 "prompt": "The person from <IMAGE_1> speaks to camera with the voice from <AUDIO_0>.",230 "prompt": "The person from <IMAGE_1> presents the product from <IMAGE_2> on the set from <IMAGE_3>, speaking with the voice from <AUDIO_0>. A second speaker with the voice from <AUDIO_1> replies.",

198 "reference_images": [{"url": "<IMAGE_URL_1>"}],231 "reference_images": [

199 "reference_audios": [{"voice_id": "eve"}],232 {"url": "<IMAGE_URL_1>"},

233 {"url": "<IMAGE_URL_2>"},

234 {"url": "<IMAGE_URL_3>"}

235 ],

236 "reference_audios": [

237 {"voice_id": "eve"},

238 {"voice_id": "leo"}

239 ],

200 "duration": 8,240 "duration": 8,

201 "aspect_ratio": "9:16",241 "aspect_ratio": "9:16",

202 "resolution": "720p"242 "resolution": "720p"

rate-limits.md +1 −1

Details

43| grok-imagine-image-quality | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |43| grok-imagine-image-quality | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |

44| grok-imagine-image-2.0 | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |44| grok-imagine-image-2.0 | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |

45| grok-imagine-image | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |45| grok-imagine-image | T0: 6, T1: 12, T2: 25, T3: 50, T4: 100 | — |

46| grok-imagine-video-1.5 | T0: 10, T1: 20, T2: 39, T3: 79, T4: 158 | — |

47| grok-imagine-video | T0: 10, T1: 20, T2: 39, T3: 79, T4: 158 | — |46| grok-imagine-video | T0: 10, T1: 20, T2: 39, T3: 79, T4: 158 | — |

47| grok-imagine-video-1.5 | T0: 10, T1: 20, T2: 39, T3: 79, T4: 158 | — |

48 48 

49### What counts toward TPM49### What counts toward TPM

50 50 

Details

8 8 

9### Request Body9### Request Body

10 10 

11* `aspect_ratio` ("1:1" | "3:4" | "4:3" | "9:16" | "16:9" | "2:3" | "3:2" | "9:19.5" | "19.5:9" | "9:20" | "20:9" | "1:2" | "2:1" | "auto")11* `aspect_ratio` ("1:1" | "3:4" | "4:3" | "9:16" | "16:9" | "2:3" | "3:2" | "9:19.5" | "19.5:9" | "9:20" | "20:9" | "1:2" | "2:1" | "21:9" | "5:2" | "auto")

12 12 

13* `model` (string | null) — Model to be used.13* `model` (string | null) — Model to be used.

14 14 


167 167 

168### Request Body168### Request Body

169 169 

170* `aspect_ratio` ("1:1" | "3:4" | "4:3" | "9:16" | "16:9" | "2:3" | "3:2" | "9:19.5" | "19.5:9" | "9:20" | "20:9" | "1:2" | "2:1" | "auto")170* `aspect_ratio` ("1:1" | "3:4" | "4:3" | "9:16" | "16:9" | "2:3" | "3:2" | "9:19.5" | "19.5:9" | "9:20" | "20:9" | "1:2" | "2:1" | "21:9" | "5:2" | "auto")

171 171 

172* `image` (object)172* `image` (object)

173 173 

Details

146 146 

147### Query Parameters147### Query Parameters

148 148 

149* `sample_rate` (integer, optional, default: 16000) — Audio sample rate in Hz. Supported values: \`8000\`, \`16000\`, \`22050\`, \`24000\`, \`44100\`, \`48000\`.149* `sample_rate` (integer, optional, default: 16000) — Audio sample rate in Hz. Supported values: \`8000\`, \`16000\`, \`22050\`, \`24000\`, \`44100\`, \`48000\`. With \`encoding=opus\`, only \`8000\`, \`16000\`, \`24000\`, and \`48000\` are supported.

150 150 

151* `encoding` (string, optional, default: pcm) — Audio encoding format. \`pcm\` — signed 16-bit little-endian (2 bytes/sample). \`mulaw\` — G.711 µ-law (1 byte/sample). \`alaw\` — G.711 A-law (1 byte/sample).151* `encoding` (string, optional, default: pcm) — Audio encoding format. \`pcm\` — signed 16-bit little-endian (2 bytes/sample). \`mulaw\` — G.711 µ-law (1 byte/sample). \`alaw\` — G.711 A-law (1 byte/sample). \`opus\` — raw Opus packets, one packet per binary WebSocket frame, mono only.

152 152 

153* `interim_results` (boolean, optional, default: false) — When \`true\`, the server emits partial transcript events (\`is\_final=false\`) approximately every 500 ms while audio is being processed. When \`false\` (default), only finalized results are sent.153* `interim_results` (boolean, optional, default: false) — When \`true\`, the server emits partial transcript events (\`is\_final=false\`) approximately every 500 ms while audio is being processed. When \`false\` (default), only finalized results are sent.

154 154 

155* `endpointing` (integer, optional, default: 10) — Silence duration in milliseconds before the server fires a \`speech\_final=true\` event, indicating the speaker stopped talking. Range: 0–5000. Set to \`0\` for no delay (fire on any VAD silence boundary). Default: 10ms.155* `endpointing` (integer, optional, default: 400) — Silence duration in milliseconds before the server fires a \`speech\_final=true\` event, indicating the speaker stopped talking. Range: 0–5000. Set to \`0\` for no delay (fire on any VAD silence boundary). Default: 400ms.

156 156 

157* `language` (string, optional, default: ) — Language code (e.g. \`en\`, \`fr\`, \`de\`, \`ja\`). When set, enables Inverse Text Normalization — spoken-form numbers, currencies, and units are converted to their written form.157* `language` (string, optional, default: ) — Language code (e.g. \`en\`, \`fr\`, \`de\`, \`ja\`). When set, enables Inverse Text Normalization — spoken-form numbers, currencies, and units are converted to their written form.

158 158 

159* `multichannel` (boolean, optional, default: false) — When \`true\`, enables per-channel transcription for interleaved multichannel audio. Requires \`channels\` to be set to ≥ 2.159* `multichannel` (boolean, optional, default: false) — When \`true\`, enables per-channel transcription for interleaved multichannel audio. Requires \`channels\` to be set to ≥ 2. Not supported with \`encoding=opus\`.

160 160 

161* `channels` (integer, optional, default: 1) — Number of interleaved audio channels. Required when \`multichannel=true\`. Min: 2, Max: 8.161* `channels` (integer, optional, default: 1) — Number of interleaved audio channels. Required when \`multichannel=true\`. Min: 2, Max: 8.

162 162 


174 174 

175### Client Messages175### Client Messages

176 176 

177* `Binary frame (audio)` — Send raw audio as binary WebSocket frames in the encoding specified by the \`encoding\` query parameter. Audio should be streamed in real-time-paced chunks (e.g. 100 ms at a time). No base64 encoding — send raw bytes directly.177* `Binary frame (audio)` — Send raw audio as binary WebSocket frames in the encoding specified by the \`encoding\` query parameter. Audio should be streamed in real-time-paced chunks (e.g. 100 ms at a time). No base64 encoding — send raw bytes directly. With \`encoding=opus\`, each binary frame must contain exactly one raw Opus packet — never concatenate packets or split one across frames. An undecodable frame sends an \`error\` event and closes the session.

178 178 

179* `finalize` — Force the current utterance to finalize as \`speech\_final\` immediately, without waiting for VAD endpointing or Smart Turn. The session stays open so you can continue streaming audio. Accepts \`finalize\` or \`Finalize\` as the type value. When \`multichannel=true\`, optional \`channel\` (0-based) limits the finalize to that channel; omit \`channel\` to finalize every channel.179* `finalize` — Force the current utterance to finalize as \`speech\_final\` immediately, without waiting for VAD endpointing or Smart Turn. The session stays open so you can continue streaming audio. Accepts \`finalize\` or \`Finalize\` as the type value. When \`multichannel=true\`, optional \`channel\` (0-based) limits the finalize to that channel; omit \`channel\` to finalize every channel.

180 180 


188 188 

189* `transcript.done` — Final transcript after \`audio.done\`. \`duration\` always present. One per channel when \`multichannel=true\`. Connection closes after this event.189* `transcript.done` — Final transcript after \`audio.done\`. \`duration\` always present. One per channel when \`multichannel=true\`. Connection closes after this event.

190 190 

191* `error` — An error occurred during the session. Most errors (pipeline failures, stream timeouts) close the connection. Only client message parse errors keep the connection open.191* `error` — An error occurred during the session. Most errors (pipeline failures, stream timeouts, undecodable audio frames) close the connection. Only client message parse errors keep the connection open.

192 192 

193### Example Message Flow193### Example Message Flow

194 194 

Details

1064 1064 

1065### Query Parameters1065### Query Parameters

1066 1066 

1067* `sample_rate` (integer, optional, default: 16000) — Audio sample rate in Hz. Supported values: \`8000\`, \`16000\`, \`22050\`, \`24000\`, \`44100\`, \`48000\`.1067* `sample_rate` (integer, optional, default: 16000) — Audio sample rate in Hz. Supported values: \`8000\`, \`16000\`, \`22050\`, \`24000\`, \`44100\`, \`48000\`. With \`encoding=opus\`, only \`8000\`, \`16000\`, \`24000\`, and \`48000\` are supported.

1068 1068 

1069* `encoding` (string, optional, default: pcm) — Audio encoding format. \`pcm\` — signed 16-bit little-endian (2 bytes/sample). \`mulaw\` — G.711 µ-law (1 byte/sample). \`alaw\` — G.711 A-law (1 byte/sample).1069* `encoding` (string, optional, default: pcm) — Audio encoding format. \`pcm\` — signed 16-bit little-endian (2 bytes/sample). \`mulaw\` — G.711 µ-law (1 byte/sample). \`alaw\` — G.711 A-law (1 byte/sample). \`opus\` — raw Opus packets, one packet per binary WebSocket frame, mono only.

1070 1070 

1071* `interim_results` (boolean, optional, default: false) — When \`true\`, the server emits partial transcript events (\`is\_final=false\`) approximately every 500 ms while audio is being processed. When \`false\` (default), only finalized results are sent.1071* `interim_results` (boolean, optional, default: false) — When \`true\`, the server emits partial transcript events (\`is\_final=false\`) approximately every 500 ms while audio is being processed. When \`false\` (default), only finalized results are sent.

1072 1072 

1073* `endpointing` (integer, optional, default: 10) — Silence duration in milliseconds before the server fires a \`speech\_final=true\` event, indicating the speaker stopped talking. Range: 0–5000. Set to \`0\` for no delay (fire on any VAD silence boundary). Default: 10ms.1073* `endpointing` (integer, optional, default: 400) — Silence duration in milliseconds before the server fires a \`speech\_final=true\` event, indicating the speaker stopped talking. Range: 0–5000. Set to \`0\` for no delay (fire on any VAD silence boundary). Default: 400ms.

1074 1074 

1075* `language` (string, optional, default: ) — Language code (e.g. \`en\`, \`fr\`, \`de\`, \`ja\`). When set, enables Inverse Text Normalization — spoken-form numbers, currencies, and units are converted to their written form.1075* `language` (string, optional, default: ) — Language code (e.g. \`en\`, \`fr\`, \`de\`, \`ja\`). When set, enables Inverse Text Normalization — spoken-form numbers, currencies, and units are converted to their written form.

1076 1076 

1077* `multichannel` (boolean, optional, default: false) — When \`true\`, enables per-channel transcription for interleaved multichannel audio. Requires \`channels\` to be set to ≥ 2.1077* `multichannel` (boolean, optional, default: false) — When \`true\`, enables per-channel transcription for interleaved multichannel audio. Requires \`channels\` to be set to ≥ 2. Not supported with \`encoding=opus\`.

1078 1078 

1079* `channels` (integer, optional, default: 1) — Number of interleaved audio channels. Required when \`multichannel=true\`. Min: 2, Max: 8.1079* `channels` (integer, optional, default: 1) — Number of interleaved audio channels. Required when \`multichannel=true\`. Min: 2, Max: 8.

1080 1080 


1092 1092 

1093### Client Messages1093### Client Messages

1094 1094 

1095* `Binary frame (audio)` — Send raw audio as binary WebSocket frames in the encoding specified by the \`encoding\` query parameter. Audio should be streamed in real-time-paced chunks (e.g. 100 ms at a time). No base64 encoding — send raw bytes directly.1095* `Binary frame (audio)` — Send raw audio as binary WebSocket frames in the encoding specified by the \`encoding\` query parameter. Audio should be streamed in real-time-paced chunks (e.g. 100 ms at a time). No base64 encoding — send raw bytes directly. With \`encoding=opus\`, each binary frame must contain exactly one raw Opus packet — never concatenate packets or split one across frames. An undecodable frame sends an \`error\` event and closes the session.

1096 1096 

1097* `finalize` — Force the current utterance to finalize as \`speech\_final\` immediately, without waiting for VAD endpointing or Smart Turn. The session stays open so you can continue streaming audio. Accepts \`finalize\` or \`Finalize\` as the type value. When \`multichannel=true\`, optional \`channel\` (0-based) limits the finalize to that channel; omit \`channel\` to finalize every channel.1097* `finalize` — Force the current utterance to finalize as \`speech\_final\` immediately, without waiting for VAD endpointing or Smart Turn. The session stays open so you can continue streaming audio. Accepts \`finalize\` or \`Finalize\` as the type value. When \`multichannel=true\`, optional \`channel\` (0-based) limits the finalize to that channel; omit \`channel\` to finalize every channel.

1098 1098 


1106 1106 

1107* `transcript.done` — Final transcript after \`audio.done\`. \`duration\` always present. One per channel when \`multichannel=true\`. Connection closes after this event.1107* `transcript.done` — Final transcript after \`audio.done\`. \`duration\` always present. One per channel when \`multichannel=true\`. Connection closes after this event.

1108 1108 

1109* `error` — An error occurred during the session. Most errors (pipeline failures, stream timeouts) close the connection. Only client message parse errors keep the connection open.1109* `error` — An error occurred during the session. Most errors (pipeline failures, stream timeouts, undecodable audio frames) close the connection. Only client message parse errors keep the connection open.

1110 1110 

1111### Example Message Flow1111### Example Message Flow

1112 1112