curl --request POST \
--url https://direct.evolink.ai/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "gpt-6.1-sol",
"input": "Explain quantum entanglement in one sentence."
}
'import requests
url = "https://direct.evolink.ai/v1/responses"
payload = {
"model": "gpt-6.1-sol",
"input": "Explain quantum entanglement in one sentence."
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({model: 'gpt-6.1-sol', input: 'Explain quantum entanglement in one sentence.'})
};
fetch('https://direct.evolink.ai/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://direct.evolink.ai/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'gpt-6.1-sol',
'input' => 'Explain quantum entanglement in one sentence.'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://direct.evolink.ai/v1/responses"
payload := strings.NewReader("{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://direct.evolink.ai/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://direct.evolink.ai/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}"
response = http.request(request)
puts response.read_body{
"id": "resp_0f5c2b2c20c39e8a006a7ef545443081979e478b10927984b5",
"object": "response",
"status": "completed",
"model": "gpt-6.1-sol",
"created_at": 1786705221,
"output": [
{
"id": "<string>",
"type": "web_search_call",
"status": "completed",
"content": [
{}
],
"encrypted_content": "<string>",
"result": "<string>"
}
],
"incomplete_details": {},
"usage": {
"input_tokens": 18,
"output_tokens": 42,
"total_tokens": 60,
"input_tokens_details": {
"cached_tokens": 0,
"cache_write_tokens": 0
},
"output_tokens_details": {
"reasoning_tokens": 16
}
},
"tool_usage": {
"image_gen": {
"input_tokens": 74,
"input_tokens_details": {},
"output_tokens": 196,
"output_tokens_details": {},
"total_tokens": 270
}
},
"metadata": {}
}{
"error": {
"code": 400,
"message": "Invalid value: '__bogus__'. Supported values are: 'auto' and 'disabled'.",
"type": "invalid_request_error",
"param": "truncation"
}
}{
"error": {
"code": 401,
"message": "Invalid or expired token",
"type": "authentication_error"
}
}{
"error": {
"code": 402,
"message": "Insufficient quota",
"type": "insufficient_quota_error",
"fallback_suggestion": "https://evolink.ai/dashboard/credits"
}
}{
"error": {
"code": 429,
"message": "Rate limit exceeded",
"type": "rate_limit_error",
"fallback_suggestion": "retry after 60 seconds"
}
}{
"error": {
"code": 500,
"message": "Internal server error",
"type": "internal_server_error",
"fallback_suggestion": "try again later"
}
}{
"error": {
"code": 503,
"message": "Service temporarily unavailable",
"type": "service_unavailable_error",
"fallback_suggestion": "retry after 30 seconds"
}
}GPT All-Model API - Responses Reference
- OpenAI-compatible Responses API for GPT series text models; select the specific model via
model(see the comparison table on themodelparameter for all allowed values) - The whole series consists of reasoning models, with reasoning depth controlled by
reasoning.effort; reasoning tokens are billed as output tokens - Prompt caching applies automatically: cached input tokens are billed at the lower cached rate
- Supports both synchronous and streaming (SSE) modes
- Server-side tools:
web_search(web search),code_interpreter(code execution),file_search(file search) - Built-in image generation tool:
image_generation(gpt-6-astra/gpt-6.1-sol/gpt-6-sol/gpt-6-lunaonly), for generating or editing images directly in a conversation - Plain
functiontools (client-side function calls) are supported as well - Multi-turn conversations can be chained with
previous_response_id - Note Model support varies for some parameters; see the per-parameter notes below
curl --request POST \
--url https://direct.evolink.ai/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "gpt-6.1-sol",
"input": "Explain quantum entanglement in one sentence."
}
'import requests
url = "https://direct.evolink.ai/v1/responses"
payload = {
"model": "gpt-6.1-sol",
"input": "Explain quantum entanglement in one sentence."
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({model: 'gpt-6.1-sol', input: 'Explain quantum entanglement in one sentence.'})
};
fetch('https://direct.evolink.ai/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://direct.evolink.ai/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'gpt-6.1-sol',
'input' => 'Explain quantum entanglement in one sentence.'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://direct.evolink.ai/v1/responses"
payload := strings.NewReader("{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://direct.evolink.ai/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://direct.evolink.ai/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"gpt-6.1-sol\",\n \"input\": \"Explain quantum entanglement in one sentence.\"\n}"
response = http.request(request)
puts response.read_body{
"id": "resp_0f5c2b2c20c39e8a006a7ef545443081979e478b10927984b5",
"object": "response",
"status": "completed",
"model": "gpt-6.1-sol",
"created_at": 1786705221,
"output": [
{
"id": "<string>",
"type": "web_search_call",
"status": "completed",
"content": [
{}
],
"encrypted_content": "<string>",
"result": "<string>"
}
],
"incomplete_details": {},
"usage": {
"input_tokens": 18,
"output_tokens": 42,
"total_tokens": 60,
"input_tokens_details": {
"cached_tokens": 0,
"cache_write_tokens": 0
},
"output_tokens_details": {
"reasoning_tokens": 16
}
},
"tool_usage": {
"image_gen": {
"input_tokens": 74,
"input_tokens_details": {},
"output_tokens": 196,
"output_tokens_details": {},
"total_tokens": 270
}
},
"metadata": {}
}{
"error": {
"code": 400,
"message": "Invalid value: '__bogus__'. Supported values are: 'auto' and 'disabled'.",
"type": "invalid_request_error",
"param": "truncation"
}
}{
"error": {
"code": 401,
"message": "Invalid or expired token",
"type": "authentication_error"
}
}{
"error": {
"code": 402,
"message": "Insufficient quota",
"type": "insufficient_quota_error",
"fallback_suggestion": "https://evolink.ai/dashboard/credits"
}
}{
"error": {
"code": 429,
"message": "Rate limit exceeded",
"type": "rate_limit_error",
"fallback_suggestion": "retry after 60 seconds"
}
}{
"error": {
"code": 500,
"message": "Internal server error",
"type": "internal_server_error",
"fallback_suggestion": "try again later"
}
}{
"error": {
"code": 503,
"message": "Service temporarily unavailable",
"type": "service_unavailable_error",
"fallback_suggestion": "retry after 30 seconds"
}
}https://direct.evolink.ai, which has better support for text models and long-lived connections. https://api.evolink.ai is the primary endpoint for multimodal services and serves as a fallback address for text models.web_search, code_interpreter, file_search, mcp) run on the server, so results do not need to be sent back by the client, and they are only available on this API. The Chat Completions endpoint supports regular function tool calling only.background: true asynchronous mode is not supported, and there are no endpoints for retrieving, cancelling or deleting a response by its ID. For long-running generations, use stream: true to keep the connection open.The built-in image_generation tool is currently supported only by gpt-6-astra / gpt-6.1-sol / gpt-6-sol / gpt-6-luna, and is unavailable on the other models; for standalone image generation, you can also use the image series model APIs.gpt-6-astra / gpt-6.1-sol / gpt-6-sol / gpt-6-luna: declare {"type": "image_generation"} in tools, and the model will generate images as needed in the conversation.- Choose an image model: use the tool’s
modelfield to selectgpt-image-2(default),gpt-image-2.5-sunburst, orgpt-image-2.5-flare;quality,size,partial_images, and other parameters follow the official image model parameters (xhigh/maxare available only for the 2.5 series) - Get the image: images are returned as base64 in the
resultfield of anoutputitem withtypeset toimage_generation_call; this is not a URL, so save the image yourself - Image editing: include an
input_imageininput(a public URL ordata:image/png;base64,...) and describe the requested changes in the text - Multi-turn editing: pass the previous turn’s
idasprevious_response_idand describe the changes you want - Streaming: set
partial_images(0–3) to receive preview events throughresponse.image_generation_call.partial_imageduring generation - Image limit: if
max_tool_callsis omitted, a single request can generate up to 4 images; set it explicitly if you need more - Billing: text and image generation are billed separately by token; image generation token usage is reported in
tool_usage.image_genin the response
{
"model": "gpt-6.1-sol",
"input": "Paint an orange tabby cat sitting by a window in watercolor style",
"tools": [{"type": "image_generation", "model": "gpt-image-2", "quality": "low", "size": "1024x1024"}]
}
id returned by the previous turn as previous_response_id on the next turn to continue the context. Responses have a retention period; once it expires the ID is no longer valid and the request is handled as a new conversation. For scenarios with strict context-accuracy requirements, maintaining the full input history yourself is recommended.Authorizations
##All APIs require Bearer Token authentication##
Get API Key:
Visit API Key Management Page to get your API Key
Add to request header:
Authorization: Bearer YOUR_API_KEY
Body
Model to call:
| Model ID | Context window | Positioning |
|---|---|---|
gpt-6.1-sol | 1,050,000 | Newer Sol: near-Astra performance at a lower cost (recommended) |
gpt-6-astra | 1,050,000 | Flagship reasoning for demanding end-to-end work, with support for the built-in image_generation tool |
gpt-6-sol | 1,050,000 | Complex coding and agentic workflows |
gpt-6-luna | 1,050,000 | High-volume classification, extraction, and cost efficiency |
gpt-5.6-sol | 1,050,000 | GPT-5.6 family, frontier reasoning |
gpt-5.6-terra | 1,050,000 | GPT-5.6 family, balanced production |
gpt-5.6-luna | 1,050,000 | GPT-5.6 family, high throughput and cost control |
gpt-5.5 | 400,000 | General-purpose reasoning model |
gpt-5.4 | 128,000 | General-purpose reasoning model |
gpt-5.2 | 400,000 | General-purpose reasoning model |
gpt-5.1 | 400,000 | General-purpose reasoning model |
gpt-6.1-sol, gpt-6-astra, gpt-6-sol, gpt-6-luna, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5, gpt-5.4, gpt-5.2, gpt-5.1 "gpt-6.1-sol"
Model input: a plain string, or an array of input items.
The content of an input item supports two block types: input_text (text) and input_image (image):
"input": [ { "role": "user", "content": [ { "type": "input_text", "text": "What is in this image?" }, { "type": "input_image", "image_url": "https://example.com/photo.png", "detail": "auto" } ] } ]
Image
- Pass the public URL of the image in
image_url image_urlmust be a string; writing it as{ "url": "..." }returns400detailis a sibling ofimage_url(not nested inside it):auto(default) /low/high/original- The image must be downloadable, otherwise
400is returned
Tool results
- The array can also carry back tool result items from the previous turn, such as
function_call_output
Note The block types on this API differ from those on the Chat Completions API (which uses text / image_url). They cannot be mixed; using the wrong ones returns 400.
"Search for AI news from the past week and summarize it in three sentences."
System-level instructions, equivalent to inserting a system message at the very beginning of input. When continuing a conversation with previous_response_id, this parameter is not inherited from the previous turn and must be passed on every turn.
"You are a concise assistant. Answer in no more than three sentences."
Whether to return a streaming response (SSE events, ending with response.completed). Default false.
false
Maximum number of tokens to generate (including reasoning tokens). When the limit is reached, status is incomplete.
GPT-6 Astra / Sol / Luna and GPT-6.1 Sol support up to 128,000 output tokens, including reasoning tokens.
2048
Reasoning control.
The allowed values of effort (reasoning depth) vary by model:
| Model | Allowed values |
|---|---|
gpt-6-astra / gpt-6.1-sol | low, medium, high, xhigh, max |
gpt-6-sol / gpt-6-luna | none, low, medium, high, xhigh, max |
gpt-5.6-sol / gpt-5.6-terra / gpt-5.6-luna | none, low, medium, high, xhigh, max |
gpt-5.5 / gpt-5.4 / gpt-5.2 | none, low, medium, high, xhigh |
gpt-5.1 | none, low, medium, high |
summary (reasoning summary): auto / concise / detailed.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
mode (reasoning mode): standard / pro, supported by gpt-6-astra / gpt-6-sol / gpt-6-luna and the gpt-5.6 family.
GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
context (reasoning context scope): auto / current_turn / all_turns, supported by gpt-6-astra and the gpt-5.6 family.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Reasoning tokens are billed as output tokens and counted in usage.output_tokens_details.reasoning_tokens.
GPT-6 Sol / Luna are supported: set model to gpt-6-sol or gpt-6-luna. Both have a 1,050,000-token context window and a maximum output of 128,000 tokens, including reasoning. reasoning.effort supports none, low, medium (default), high, xhigh, and max; Astra and 6.1 Sol do not support none. Use this endpoint for reasoning with tools.
Show child attributes
Show child attributes
Output text control:
format:{"type": "text"}(default),{"type": "json_object"}, or{"type": "json_schema", "name": "...", "schema": {...}, "strict": true}for structured outputverbosity:low/medium/high, controls how detailed the answer is
text.verbosity — GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Show child attributes
Show child attributes
Tool declarations. Server-side tools run on the server, so results do not need to be sent back by the client:
| Tool type | Capability |
|---|---|
web_search | Searches the web and browses pages (alias web_search_preview) |
code_interpreter | Runs code in a sandbox; requires "container": {"type": "auto"} |
file_search | Searches an existing vector store; requires vector_store_ids |
mcp | Connects to a remote MCP service; requires server_label and server_url |
image_generation | Generates or edits images in a conversation and returns them as base64 (gpt-6-astra / gpt-6.1-sol / gpt-6-sol / gpt-6-luna only) |
Plain function tools (client-side function calls) are supported as well.
Note The built-in image_generation tool is currently supported only by gpt-6-astra / gpt-6.1-sol / gpt-6-sol / gpt-6-luna, and is unavailable on the other models; for standalone image generation, you can also use the image series model APIs.
Show child attributes
Show child attributes
[{ "type": "web_search" }]
Controls tool selection: "auto" (default) / "none" / "required", or an object pinning a specific tool, e.g. {"type": "web_search"}.
none, auto, required Maximum total number of tool calls allowed in this response (across all built-in tools).
Note When gpt-6-astra / gpt-6.1-sol / gpt-6-sol / gpt-6-luna uses image_generation and this parameter is omitted, a single request can generate up to 4 images; set it explicitly if you need more.
5
Whether the model may call multiple tools in parallel within one turn. Defaults to true.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note gpt-6-astra, the gpt-5.6 family, and gpt-5.5 support setting this parameter to false; on gpt-5.4 / gpt-5.2 / gpt-5.1, it has no effect and always behaves as true.
true
The id of the previous response, used to chain multi-turn conversations without re-uploading the history.
Note Must be used together with store: true (the default). Responses have a retention period; once it expires the ID is no longer valid, and the request is handled as a new conversation without inheriting context. For scenarios with strict context-accuracy requirements, maintaining the full input history yourself is recommended.
"resp_0f5c2b2c20c39e8a006a7ef545443081979e478b10927984b5"
Whether to store this response on the server; only stored responses can be referenced by previous_response_id. Defaults to true.
GPT-6 Sol / Luna and GPT-6.1 Sol: disabling retention with store: false has not been verified on the available channels. Acceptance of the field alone does not establish that the response was not retained.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note gpt-6-astra, the gpt-5.6 family, and gpt-5.5 support setting this parameter to false; on gpt-5.4 / gpt-5.2 / gpt-5.1, it has no effect and always behaves as true. If you do not want responses stored, choose a model that supports turning it off.
true
Additional content to return in the response. Allowed values:
reasoning.encrypted_contentmessage.output_text.logprobsweb_search_call.resultsweb_search_call.action.sourcesfile_search_call.resultscode_interpreter_call.outputsmessage.input_image.image_urlcomputer_call_output.output.image_url
Note message.output_text.logprobs is not supported by gpt-6-astra or gpt-6.1-sol.
GPT-6: Astra and 6.1 Sol do not support output logprobs. For Sol / Luna, use them only with reasoning effort none. At other levels, remove logprobs, top_logprobs, and message.output_text.logprobs from Responses include.
["reasoning.encrypted_content"]
Sampling temperature, ranging from 0 to 2. Lower values make the output more deterministic.
GPT-6: omit this parameter for gpt-6-astra and gpt-6.1-sol. For gpt-6-sol / gpt-6-luna, adjust it only with reasoning effort set to none; omit it at other effort levels. Omitting effort selects medium, not none.
Existing models: gpt-5.4 / gpt-5.2 / gpt-5.1 treat temperature: 0 as the default 1; use a positive value such as 0.01 for more deterministic output.
0 <= x <= 21
Nucleus sampling parameter, ranging from 0 to 1. Adjusting it together with temperature is not recommended.
GPT-6: omit this parameter for gpt-6-astra and gpt-6.1-sol. For gpt-6-sol / gpt-6-luna, adjust it only with reasoning effort set to none; omit it at other effort levels. Omitting effort selects medium, not none.
0 <= x <= 11
Number of candidate tokens returned at each position, ranging from 0 to 20; must be used together with include: ["message.output_text.logprobs"].
GPT-6: Astra and 6.1 Sol do not support output logprobs. For Sol / Luna, use them only with reasoning effort none. At other levels, remove logprobs, top_logprobs, and message.output_text.logprobs from Responses include.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note Supported only by the gpt-5.6 family and gpt-5.5; other models do not support this parameter.
0 <= x <= 202
Frequency penalty, ranging from -2 to 2, reducing the likelihood of repeated content.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note Adjustable on the gpt-5.6 family; the other existing models do not support this parameter. GPT-6 Astra does not allow adjustment and accepts only the default value 0; passing any other value returns 400.
-2 <= x <= 20
Presence penalty, ranging from -2 to 2, encouraging the model to talk about new topics.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note Adjustable on the gpt-5.6 family; the other existing models do not support this parameter. GPT-6 Astra does not allow adjustment and accepts only the default value 0; passing any other value returns 400.
-2 <= x <= 20
How to handle context that exceeds the window: disabled (default, returns an error) or auto (automatically truncates the middle of the context).
auto, disabled "auto"
Automatic compaction configuration for long conversations, for example [{"type": "compaction", "compact_threshold": 100000}]: the history is compacted automatically once the context exceeds the threshold.
Note Supported only by gpt-6-astra / gpt-6-sol / gpt-6-luna and the gpt-5.6 family; other models do not support this parameter.
GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Cache grouping key. GPT-6 / GPT-5.6 route cache traffic automatically; this field is not needed to optimize routing. Separate keys can distinguish cache reuse and accounting for customers or users. Keep the key stable for requests that should reuse a prefix. Earlier models can use stable keys to help cache routing.
"app-agent-v1"
Legacy model cache retention. For GPT-6 / GPT-5.6, use prompt_cache_options.ttl: "30m"; do not use 24h in the new field.
in_memory, 24h "in_memory"
References an already created prompt template, in the form {"id": "pmpt_xxx", "version": "1", "variables": {...}}.
Show child attributes
Show child attributes
Custom key-value pairs returned as-is with the response, convenient for tagging on the business side. Both keys and values are strings.
{ "trace_id": "abc-123" }
Stable identifier of the end user, used for abuse tracking.
GPT-6 Sol / Luna and GPT-6.1 Sol: support for this parameter has not been confirmed; omit it from basic requests.
Rules for existing models below exclude GPT-6 Sol / Luna and GPT-6.1 Sol:
Note Supported only by gpt-6-astra and the gpt-5.6 family; other models do not support this parameter.
"user-1024"
End-user identifier, used to distinguish the source of calls.
"user-1024"
Prompt caching options for GPT-6 and GPT-5.6. Implicit breakpoints are the default. mode: "explicit" uses only explicit breakpoints; without one, no caching occurs.
Show child attributes
Show child attributes
{ "ttl": "30m", "mode": "implicit" }
Response
Response generated successfully (JSON object, or an SSE event stream ending with response.completed when stream=true)
Unique ID of this response, which can be used as previous_response_id for the next turn
"resp_0f5c2b2c20c39e8a006a7ef545443081979e478b10927984b5"
Response type
response "response"
Response status: completed for a normal ending, incomplete when generation stopped early for reasons such as reaching max_output_tokens, failed when generation failed
completed, incomplete, failed "completed"
Actual model name used
"gpt-6.1-sol"
Creation timestamp
1786705221
Output items in generation order: the reasoning item (reasoning summary / encrypted reasoning content), tool call items (such as web_search_call, code_interpreter_call, and image_generation_call), and finally the message item containing output_text content.
Show child attributes
Show child attributes
Explains the reason when status is incomplete
Token usage statistics. Prompt caching applies automatically, and cached input tokens are billed at the lower cached rate.
GPT-6 bills uncached input, cache reads, cache writes, and output separately. Above 272,000 input tokens, the entire request uses 2× the regular input and cache rates and 1.5× the output rate. Built-in image generation is billed separately. See current model pricing.
Show child attributes
Show child attributes
Built-in tool usage. When using image_generation, image_gen records the tokens consumed by image generation, tracked separately from usage and billed separately by token
Show child attributes
Show child attributes
Custom key-value pairs passed in the request, returned as-is