import os
from inceptionai import Inception
client = Inception(
api_key=os.environ.get("INCEPTION_API_KEY"), # defaults to this env var; can be omitted
)
chat_completion = client.chat.completions.create(
model="mercury-2.5",
messages=[
{
"role": "user",
"content": "What is a diffusion language model?"
}
],
max_tokens=256,
temperature=0.75,
)
print(chat_completion)import Inception from 'inceptionai';
const client = new Inception({
apiKey: process.env['INCEPTION_API_KEY'], // defaults to this env var; can be omitted
});
const chatCompletion = await client.chat.completions.create({
model: 'mercury-2.5',
messages: [
{
role: 'user',
content: 'What is a diffusion language model?'
}
],
max_tokens: 256,
temperature: 0.75
});
console.log(chatCompletion);curl --request POST \
--url https://api.inceptionlabs.ai/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "mercury-2.5",
"messages": [
{
"role": "user",
"content": "What is a diffusion language model?"
}
]
}
'const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'mercury-2.5',
messages: [{role: 'user', content: 'What is a diffusion language model?'}]
})
};
fetch('https://api.inceptionlabs.ai/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.inceptionlabs.ai/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'mercury-2.5',
'messages' => [
[
'role' => 'user',
'content' => 'What is a diffusion language model?'
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.inceptionlabs.ai/v1/chat/completions"
payload := strings.NewReader("{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.inceptionlabs.ai/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.inceptionlabs.ai/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}"
response = http.request(request)
puts response.read_body{
"id": "chatcmpl-7a2b3c4d5e",
"object": "chat.completion",
"created": 1745798400,
"model": "mercury-2.5",
"choices": [
{
"index": 0,
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": "A diffusion language model is a type of language model that uses diffusion to generate text."
}
}
],
"usage": {
"prompt_tokens": 12,
"completion_tokens": 8,
"total_tokens": 20,
"reasoning_tokens": 0,
"cached_input_tokens": 0
}
}{
"error": {
"message": "You exceeded the maximum context length for this model of 128000. Please reduce the length of the messages or completion.",
"type": "invalid_request_error",
"param": "messages",
"code": "context_length_exceeded"
}
}{
"error": {
"message": "Incorrect API key provided",
"type": "authentication_error",
"param": null,
"code": "invalid_api_key"
}
}{
"error": {
"message": "Account is inactive",
"type": "account_error",
"param": null,
"code": "account_error"
}
}{
"error": {
"message": "model `jupyter-2` not found",
"type": "invalid_request_error",
"param": "model",
"code": "model_not_found"
}
}{
"error": {
"message": "Rate limit exceeded. Please try again later.",
"type": "rate_limit_error",
"param": null,
"code": "rate_limit_reached"
}
}{
"error": {
"message": "The server had an error while processing your request.",
"type": "server_error",
"param": null,
"code": "server_error"
}
}Create a chat completion
Generate a chat completion from a sequence of messages. Returns a ChatCompletion object, or a server-sent events stream of ChatCompletionChunk deltas when stream=true. Supports tool calling, structured output via response_format, and reasoning controls.
import os
from inceptionai import Inception
client = Inception(
api_key=os.environ.get("INCEPTION_API_KEY"), # defaults to this env var; can be omitted
)
chat_completion = client.chat.completions.create(
model="mercury-2.5",
messages=[
{
"role": "user",
"content": "What is a diffusion language model?"
}
],
max_tokens=256,
temperature=0.75,
)
print(chat_completion)import Inception from 'inceptionai';
const client = new Inception({
apiKey: process.env['INCEPTION_API_KEY'], // defaults to this env var; can be omitted
});
const chatCompletion = await client.chat.completions.create({
model: 'mercury-2.5',
messages: [
{
role: 'user',
content: 'What is a diffusion language model?'
}
],
max_tokens: 256,
temperature: 0.75
});
console.log(chatCompletion);curl --request POST \
--url https://api.inceptionlabs.ai/v1/chat/completions \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "mercury-2.5",
"messages": [
{
"role": "user",
"content": "What is a diffusion language model?"
}
]
}
'const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'mercury-2.5',
messages: [{role: 'user', content: 'What is a diffusion language model?'}]
})
};
fetch('https://api.inceptionlabs.ai/v1/chat/completions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.inceptionlabs.ai/v1/chat/completions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'mercury-2.5',
'messages' => [
[
'role' => 'user',
'content' => 'What is a diffusion language model?'
]
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.inceptionlabs.ai/v1/chat/completions"
payload := strings.NewReader("{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.inceptionlabs.ai/v1/chat/completions")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.inceptionlabs.ai/v1/chat/completions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"mercury-2.5\",\n \"messages\": [\n {\n \"role\": \"user\",\n \"content\": \"What is a diffusion language model?\"\n }\n ]\n}"
response = http.request(request)
puts response.read_body{
"id": "chatcmpl-7a2b3c4d5e",
"object": "chat.completion",
"created": 1745798400,
"model": "mercury-2.5",
"choices": [
{
"index": 0,
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": "A diffusion language model is a type of language model that uses diffusion to generate text."
}
}
],
"usage": {
"prompt_tokens": 12,
"completion_tokens": 8,
"total_tokens": 20,
"reasoning_tokens": 0,
"cached_input_tokens": 0
}
}{
"error": {
"message": "You exceeded the maximum context length for this model of 128000. Please reduce the length of the messages or completion.",
"type": "invalid_request_error",
"param": "messages",
"code": "context_length_exceeded"
}
}{
"error": {
"message": "Incorrect API key provided",
"type": "authentication_error",
"param": null,
"code": "invalid_api_key"
}
}{
"error": {
"message": "Account is inactive",
"type": "account_error",
"param": null,
"code": "account_error"
}
}{
"error": {
"message": "model `jupyter-2` not found",
"type": "invalid_request_error",
"param": "model",
"code": "model_not_found"
}
}{
"error": {
"message": "Rate limit exceeded. Please try again later.",
"type": "rate_limit_error",
"param": null,
"code": "rate_limit_reached"
}
}{
"error": {
"message": "The server had an error while processing your request.",
"type": "server_error",
"param": null,
"code": "server_error"
}
}Authorizations
API key provided as a Bearer token: Authorization: Bearer <api_key>. Get an API key at https://platform.inceptionlabs.ai.
Body
Show child attributes
Show child attributes
The model to use for the chat completion.
Maximum number of tokens to generate. Deprecated in favour of max_completion_tokens, which takes precedence when both are supplied. The maximum depends on model: 50,000 for mercury-2; 65,536 for mercury-2.5. When neither token-budget field is supplied: For mercury-2, the default is 16,384. For mercury-2.5, the default is 16,384; 65,536 when reasoning_effort="high".
1 <= x <= 65536An upper bound for the number of tokens that can be generated for a completion, including visible output tokens and reasoning tokens. Supersedes max_tokens. When neither field is supplied, the model's configured default applies (16384 for mercury-2). The maximum depends on model: 50,000 for mercury-2; 65,536 for mercury-2.5. When neither token-budget field is supplied: For mercury-2, the default is 16,384. For mercury-2.5, the default is 16,384; 65,536 when reasoning_effort="high".
1 <= x <= 65536What sampling temperature to use, between 0.5 and 1. Higher values make the output more random; lower values make it more focused and deterministic. For mercury-2, the default is 0.75; out-of-range values are reset to 0.75 with a warning in the response. For mercury-2.5, the default is 1; out-of-range values are reset to 1 with a warning in the response.
0.5 <= x <= 1A list of sequences where the API will stop generating further tokens. The returned text will not contain the stop sequences.
A list of tools the model may call. Use this to provide functions the model can generate JSON arguments for.
Show child attributes
Show child attributes
Controls tool selection: 'auto' (model decides, default when tools present), 'required' (model must call one), 'none' (model must not call one), or {"type": "function", "function": {"name": "..."}} to force one declared tool.
auto, required, none Whether to stream the response.
Options that control streaming behavior.
Show child attributes
Show child attributes
Whether to show the diffusion effect in the streamed response.
Extra parameters passed to the model (OpenAI-compat passthrough).
Enable flag for more realtime workloads that require lower TTFT/TTFAT.
An object specifying the format that the model must output.
- ResponseFormatText
- ResponseFormatJSONObject
- ResponseFormatJSONSchema
Show child attributes
Show child attributes
Request a summary of the model's reasoning process.
Wait for all reasoning summaries to complete before finishing the response.
Constrains the effort spent on reasoning before the model responds.
instant, low, medium, high Response
Successful response. Returns a JSON object when stream=false, or a server-sent events stream of ChatCompletionChunk objects (terminated by data: [DONE]) when stream=true.
"chat.completion"Show child attributes
Show child attributes
Show child attributes
Show child attributes
Summary of reasoning process if requested and available.
Show child attributes
Show child attributes
Was this page helpful?