import requests, time headers = {"Authorization": "Bearer " + AIMLAPI_KEY} job = requests.post( "https://api.aimlapi.com/v2/generate/audio", headers=headers, json={ "model": "mistral/mistral-ocr-4", "prompt": "Upbeat lofi background music" }, ).json() gid = job["generation_id"] while True: res = requests.get(f"https://api.aimlapi.com/v2/generate/audio?generation_id={gid}", headers=headers).json() if res.get("status") in ("completed", "error", "failed"): break time.sleep(3) print(res)
const headers = { Authorization: `Bearer ${process.env.AIMLAPI_KEY}`, "Content-Type": "application/json", }; const job = await (await fetch("https://api.aimlapi.com/v2/generate/audio", { method: "POST", headers, body: JSON.stringify({ "model": "mistral/mistral-ocr-4", "prompt": "Upbeat lofi background music" }), })).json(); let res; do { await new Promise((r) => setTimeout(r, 3000)); res = await (await fetch(`https://api.aimlapi.com/v2/generate/audio?generation_id=${job.generation_id}`, { headers })).json(); } while (!["completed", "error", "failed"].includes(res.status)); console.log(res);
# submit the job — the response contains "generation_id" curl -X POST https://api.aimlapi.com/v2/generate/audio \ -H "Authorization: Bearer $AIMLAPI_KEY" \ -H "Content-Type: application/json" \ -d '{"model":"mistral/mistral-ocr-4","prompt":"Upbeat lofi background music"}' # then poll for the result until it is ready curl "https://api.aimlapi.com/v2/generate/audio?generation_id={generation_id}" -H "Authorization: Bearer $AIMLAPI_KEY"
OpenAI-compatible — swap the base URL and it works with your existing SDK.
| Type | Price |
|---|---|
| Output | |
| Model | Input | Output | Context | Best for |
|---|---|---|---|---|
Mistral OCR 4 This page | Audio generation |
Mistral OCR 4 takes document, image as input and returns text.
Mistral OCR 4 is priced at $5.2 / 1K pages.
Mistral OCR 4 is billed per generation — a fixed charge per output rather than by prompt length.
Mistral OCR 4 was built by Mistral AI.
Use mistral/mistral-ocr-4 as the model id on AI/ML API.
Yes. Mistral OCR 4 is served through AI/ML API, so the same key and endpoint format used for other models applies.
Yes. It adds native paragraph-level bounding box extraction along with structural block labels.
It processes PDFs and images, extracting text, tables, and images from them.
Yes, it is fully backward compatible with Mistral OCR 3.