curl --request POST \
--url https://api.deepinfra.com/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"input": "<string>",
"fail_fast": false,
"models": [
"<string>"
],
"instructions": "<string>",
"tools": [
{
"type": "function",
"name": "<string>",
"description": "<string>",
"parameters": {},
"strict": true
}
],
"text": {
"format": {
"type": "text"
},
"verbosity": "<string>"
},
"reasoning": {
"summary": "<string>",
"generate_summary": "<string>"
},
"max_output_tokens": 123,
"max_tool_calls": 123,
"temperature": 123,
"top_p": 123,
"top_logprobs": 123,
"metadata": {},
"parallel_tool_calls": true,
"stream": false,
"user": "<string>",
"store": false,
"previous_response_id": "<string>",
"background": false,
"include": [
"<string>"
],
"prompt": {},
"conversation": null,
"min_p": 123,
"top_k": 123,
"repetition_penalty": 123,
"stop_token_ids": [
123
],
"chat_template_kwargs": {},
"continue_final_message": true,
"ignore_eos": true,
"prompt_cache_key": "<string>",
"prompt_cache_options": {}
}
'import requests
url = "https://api.deepinfra.com/v1/responses"
payload = {
"model": "<string>",
"input": "<string>",
"fail_fast": False,
"models": ["<string>"],
"instructions": "<string>",
"tools": [
{
"type": "function",
"name": "<string>",
"description": "<string>",
"parameters": {},
"strict": True
}
],
"text": {
"format": { "type": "text" },
"verbosity": "<string>"
},
"reasoning": {
"summary": "<string>",
"generate_summary": "<string>"
},
"max_output_tokens": 123,
"max_tool_calls": 123,
"temperature": 123,
"top_p": 123,
"top_logprobs": 123,
"metadata": {},
"parallel_tool_calls": True,
"stream": False,
"user": "<string>",
"store": False,
"previous_response_id": "<string>",
"background": False,
"include": ["<string>"],
"prompt": {},
"conversation": None,
"min_p": 123,
"top_k": 123,
"repetition_penalty": 123,
"stop_token_ids": [123],
"chat_template_kwargs": {},
"continue_final_message": True,
"ignore_eos": True,
"prompt_cache_key": "<string>",
"prompt_cache_options": {}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
input: '<string>',
fail_fast: false,
models: ['<string>'],
instructions: '<string>',
tools: [
{
type: 'function',
name: '<string>',
description: '<string>',
parameters: {},
strict: true
}
],
text: {format: {type: 'text'}, verbosity: '<string>'},
reasoning: {summary: '<string>', generate_summary: '<string>'},
max_output_tokens: 123,
max_tool_calls: 123,
temperature: 123,
top_p: 123,
top_logprobs: 123,
metadata: {},
parallel_tool_calls: true,
stream: false,
user: '<string>',
store: false,
previous_response_id: '<string>',
background: false,
include: ['<string>'],
prompt: {},
conversation: null,
min_p: 123,
top_k: 123,
repetition_penalty: 123,
stop_token_ids: [123],
chat_template_kwargs: {},
continue_final_message: true,
ignore_eos: true,
prompt_cache_key: '<string>',
prompt_cache_options: {}
})
};
fetch('https://api.deepinfra.com/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.deepinfra.com/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'input' => '<string>',
'fail_fast' => false,
'models' => [
'<string>'
],
'instructions' => '<string>',
'tools' => [
[
'type' => 'function',
'name' => '<string>',
'description' => '<string>',
'parameters' => [
],
'strict' => true
]
],
'text' => [
'format' => [
'type' => 'text'
],
'verbosity' => '<string>'
],
'reasoning' => [
'summary' => '<string>',
'generate_summary' => '<string>'
],
'max_output_tokens' => 123,
'max_tool_calls' => 123,
'temperature' => 123,
'top_p' => 123,
'top_logprobs' => 123,
'metadata' => [
],
'parallel_tool_calls' => true,
'stream' => false,
'user' => '<string>',
'store' => false,
'previous_response_id' => '<string>',
'background' => false,
'include' => [
'<string>'
],
'prompt' => [
],
'conversation' => null,
'min_p' => 123,
'top_k' => 123,
'repetition_penalty' => 123,
'stop_token_ids' => [
123
],
'chat_template_kwargs' => [
],
'continue_final_message' => true,
'ignore_eos' => true,
'prompt_cache_key' => '<string>',
'prompt_cache_options' => [
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.deepinfra.com/v1/responses"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.deepinfra.com/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.deepinfra.com/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}"
response = http.request(request)
puts response.read_body{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Openai Responses
curl --request POST \
--url https://api.deepinfra.com/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"input": "<string>",
"fail_fast": false,
"models": [
"<string>"
],
"instructions": "<string>",
"tools": [
{
"type": "function",
"name": "<string>",
"description": "<string>",
"parameters": {},
"strict": true
}
],
"text": {
"format": {
"type": "text"
},
"verbosity": "<string>"
},
"reasoning": {
"summary": "<string>",
"generate_summary": "<string>"
},
"max_output_tokens": 123,
"max_tool_calls": 123,
"temperature": 123,
"top_p": 123,
"top_logprobs": 123,
"metadata": {},
"parallel_tool_calls": true,
"stream": false,
"user": "<string>",
"store": false,
"previous_response_id": "<string>",
"background": false,
"include": [
"<string>"
],
"prompt": {},
"conversation": null,
"min_p": 123,
"top_k": 123,
"repetition_penalty": 123,
"stop_token_ids": [
123
],
"chat_template_kwargs": {},
"continue_final_message": true,
"ignore_eos": true,
"prompt_cache_key": "<string>",
"prompt_cache_options": {}
}
'import requests
url = "https://api.deepinfra.com/v1/responses"
payload = {
"model": "<string>",
"input": "<string>",
"fail_fast": False,
"models": ["<string>"],
"instructions": "<string>",
"tools": [
{
"type": "function",
"name": "<string>",
"description": "<string>",
"parameters": {},
"strict": True
}
],
"text": {
"format": { "type": "text" },
"verbosity": "<string>"
},
"reasoning": {
"summary": "<string>",
"generate_summary": "<string>"
},
"max_output_tokens": 123,
"max_tool_calls": 123,
"temperature": 123,
"top_p": 123,
"top_logprobs": 123,
"metadata": {},
"parallel_tool_calls": True,
"stream": False,
"user": "<string>",
"store": False,
"previous_response_id": "<string>",
"background": False,
"include": ["<string>"],
"prompt": {},
"conversation": None,
"min_p": 123,
"top_k": 123,
"repetition_penalty": 123,
"stop_token_ids": [123],
"chat_template_kwargs": {},
"continue_final_message": True,
"ignore_eos": True,
"prompt_cache_key": "<string>",
"prompt_cache_options": {}
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
input: '<string>',
fail_fast: false,
models: ['<string>'],
instructions: '<string>',
tools: [
{
type: 'function',
name: '<string>',
description: '<string>',
parameters: {},
strict: true
}
],
text: {format: {type: 'text'}, verbosity: '<string>'},
reasoning: {summary: '<string>', generate_summary: '<string>'},
max_output_tokens: 123,
max_tool_calls: 123,
temperature: 123,
top_p: 123,
top_logprobs: 123,
metadata: {},
parallel_tool_calls: true,
stream: false,
user: '<string>',
store: false,
previous_response_id: '<string>',
background: false,
include: ['<string>'],
prompt: {},
conversation: null,
min_p: 123,
top_k: 123,
repetition_penalty: 123,
stop_token_ids: [123],
chat_template_kwargs: {},
continue_final_message: true,
ignore_eos: true,
prompt_cache_key: '<string>',
prompt_cache_options: {}
})
};
fetch('https://api.deepinfra.com/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.deepinfra.com/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'input' => '<string>',
'fail_fast' => false,
'models' => [
'<string>'
],
'instructions' => '<string>',
'tools' => [
[
'type' => 'function',
'name' => '<string>',
'description' => '<string>',
'parameters' => [
],
'strict' => true
]
],
'text' => [
'format' => [
'type' => 'text'
],
'verbosity' => '<string>'
],
'reasoning' => [
'summary' => '<string>',
'generate_summary' => '<string>'
],
'max_output_tokens' => 123,
'max_tool_calls' => 123,
'temperature' => 123,
'top_p' => 123,
'top_logprobs' => 123,
'metadata' => [
],
'parallel_tool_calls' => true,
'stream' => false,
'user' => '<string>',
'store' => false,
'previous_response_id' => '<string>',
'background' => false,
'include' => [
'<string>'
],
'prompt' => [
],
'conversation' => null,
'min_p' => 123,
'top_k' => 123,
'repetition_penalty' => 123,
'stop_token_ids' => [
123
],
'chat_template_kwargs' => [
],
'continue_final_message' => true,
'ignore_eos' => true,
'prompt_cache_key' => '<string>',
'prompt_cache_options' => [
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.deepinfra.com/v1/responses"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.deepinfra.com/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.deepinfra.com/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"fail_fast\": false,\n \"models\": [\n \"<string>\"\n ],\n \"instructions\": \"<string>\",\n \"tools\": [\n {\n \"type\": \"function\",\n \"name\": \"<string>\",\n \"description\": \"<string>\",\n \"parameters\": {},\n \"strict\": true\n }\n ],\n \"text\": {\n \"format\": {\n \"type\": \"text\"\n },\n \"verbosity\": \"<string>\"\n },\n \"reasoning\": {\n \"summary\": \"<string>\",\n \"generate_summary\": \"<string>\"\n },\n \"max_output_tokens\": 123,\n \"max_tool_calls\": 123,\n \"temperature\": 123,\n \"top_p\": 123,\n \"top_logprobs\": 123,\n \"metadata\": {},\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"user\": \"<string>\",\n \"store\": false,\n \"previous_response_id\": \"<string>\",\n \"background\": false,\n \"include\": [\n \"<string>\"\n ],\n \"prompt\": {},\n \"conversation\": null,\n \"min_p\": 123,\n \"top_k\": 123,\n \"repetition_penalty\": 123,\n \"stop_token_ids\": [\n 123\n ],\n \"chat_template_kwargs\": {},\n \"continue_final_message\": true,\n \"ignore_eos\": true,\n \"prompt_cache_key\": \"<string>\",\n \"prompt_cache_options\": {}\n}"
response = http.request(request)
puts response.read_body{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>"
}
]
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Headers
Per-request service tier (priority or flex) for clients that cannot set the service_tier body field. The body field wins when both are present; unrecognized values ride the default tier.
Body
The service tier used for processing the request. 'priority' processes the request with higher priority (premium rate); 'flex' processes it at lower priority for a discount, served only when spare capacity exists and may be retried/timed out under load. Both apply only to models that support the respective tier. For compatibility, 'auto' is treated as 'priority' and 'standard_only' as 'default'.
default, priority, flex If true, the request is rejected immediately with HTTP 429 when the model has no spare capacity, instead of waiting in the queue. Opt-in; the default (false) keeps standard queueing behavior.
Ordered list of up to 4 fallback models. The request is attempted on each model in order: when a model rejects it for lack of capacity (HTTP 429 model-busy / flex no-capacity), the next model is tried server-side. The first model that accepts serves the request; the response's model field and billing reflect that model, at that model's pricing. Models before the last are attempted without queueing (as if fail_fast were set); the last model honors the request's own fail_fast value. When models is set, the model field is ignored. Entries must be plain model names (no deploy_id:, custom_hostport, or :revision specifiers); duplicate entries are ignored, keeping the first occurrence.
1 - 4 elements- ResponsesFunctionTool
- ResponsesWebSearchTool
- Option 3
Show child attributes
Show child attributes
auto, none, required Show child attributes
Show child attributes
Show child attributes
Show child attributes
auto, disabled Show child attributes
Show child attributes
Response
Successful Response