curl --request POST \
--url https://nano-gpt.com/api/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"input": "<string>",
"billing_mode": "<string>",
"billingMode": "<string>",
"instructions": "<string>",
"max_output_tokens": 17,
"temperature": 1,
"top_p": 0.5,
"tools": [
{
"max_results": 5
}
],
"tool_choice": "<string>",
"parallel_tool_calls": true,
"stream": false,
"store": false,
"retention_days": 182,
"retentionDays": 182,
"previous_response_id": "<string>",
"reasoning": {},
"text": {},
"metadata": {},
"user": "<string>",
"seed": 123,
"background": true
}
'import requests
url = "https://nano-gpt.com/api/v1/responses"
payload = {
"model": "<string>",
"input": "<string>",
"billing_mode": "<string>",
"billingMode": "<string>",
"instructions": "<string>",
"max_output_tokens": 17,
"temperature": 1,
"top_p": 0.5,
"tools": [{ "max_results": 5 }],
"tool_choice": "<string>",
"parallel_tool_calls": True,
"stream": False,
"store": False,
"retention_days": 182,
"retentionDays": 182,
"previous_response_id": "<string>",
"reasoning": {},
"text": {},
"metadata": {},
"user": "<string>",
"seed": 123,
"background": True
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
input: '<string>',
billing_mode: '<string>',
billingMode: '<string>',
instructions: '<string>',
max_output_tokens: 17,
temperature: 1,
top_p: 0.5,
tools: [{max_results: 5}],
tool_choice: '<string>',
parallel_tool_calls: true,
stream: false,
store: false,
retention_days: 182,
retentionDays: 182,
previous_response_id: '<string>',
reasoning: {},
text: {},
metadata: {},
user: '<string>',
seed: 123,
background: true
})
};
fetch('https://nano-gpt.com/api/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://nano-gpt.com/api/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'input' => '<string>',
'billing_mode' => '<string>',
'billingMode' => '<string>',
'instructions' => '<string>',
'max_output_tokens' => 17,
'temperature' => 1,
'top_p' => 0.5,
'tools' => [
[
'max_results' => 5
]
],
'tool_choice' => '<string>',
'parallel_tool_calls' => true,
'stream' => false,
'store' => false,
'retention_days' => 182,
'retentionDays' => 182,
'previous_response_id' => '<string>',
'reasoning' => [
],
'text' => [
],
'metadata' => [
],
'user' => '<string>',
'seed' => 123,
'background' => true
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://nano-gpt.com/api/v1/responses"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://nano-gpt.com/api/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://nano-gpt.com/api/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"created_at": 123,
"model": "<string>",
"status": "queued",
"output": [
{}
],
"output_text": "<string>",
"usage": {},
"error": {},
"incomplete_details": {},
"metadata": {},
"service_tier": "<string>",
"advisor": {
"id": "<string>",
"mode": "auto",
"executor_model": "<string>",
"advisor_model": "<string>",
"requested": true,
"consulted": true,
"successful": true,
"consultation_count": 0,
"max_uses": 1,
"status": "not_used",
"error": "<string>",
"usage": {
"executor": {},
"advisor": {},
"continuation": {},
"total": {}
},
"pricing": {
"executor": {},
"advisor": {},
"continuation": {},
"total": {}
}
}
}{
"error": 123,
"message": "<string>"
}{
"error": 123,
"message": "<string>"
}{
"error": 123,
"message": "<string>"
}Responses
Create a response with the OpenAI-compatible Responses API. Compatible models can use the Responses-only hosted tool-search pilot by adding a nanogpt:tool_search or tool_search entry and marking function tools with defer_loading: true. The NanoGPT Advisor extension is available for non-streaming, foreground, platform-billed pay-as-you-go API-key requests that do not use client tools, structured output, inline moderation, BYOK, accountless payment, memory, or server-side content enhancements.
curl --request POST \
--url https://nano-gpt.com/api/v1/responses \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "<string>",
"input": "<string>",
"billing_mode": "<string>",
"billingMode": "<string>",
"instructions": "<string>",
"max_output_tokens": 17,
"temperature": 1,
"top_p": 0.5,
"tools": [
{
"max_results": 5
}
],
"tool_choice": "<string>",
"parallel_tool_calls": true,
"stream": false,
"store": false,
"retention_days": 182,
"retentionDays": 182,
"previous_response_id": "<string>",
"reasoning": {},
"text": {},
"metadata": {},
"user": "<string>",
"seed": 123,
"background": true
}
'import requests
url = "https://nano-gpt.com/api/v1/responses"
payload = {
"model": "<string>",
"input": "<string>",
"billing_mode": "<string>",
"billingMode": "<string>",
"instructions": "<string>",
"max_output_tokens": 17,
"temperature": 1,
"top_p": 0.5,
"tools": [{ "max_results": 5 }],
"tool_choice": "<string>",
"parallel_tool_calls": True,
"stream": False,
"store": False,
"retention_days": 182,
"retentionDays": 182,
"previous_response_id": "<string>",
"reasoning": {},
"text": {},
"metadata": {},
"user": "<string>",
"seed": 123,
"background": True
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: '<string>',
input: '<string>',
billing_mode: '<string>',
billingMode: '<string>',
instructions: '<string>',
max_output_tokens: 17,
temperature: 1,
top_p: 0.5,
tools: [{max_results: 5}],
tool_choice: '<string>',
parallel_tool_calls: true,
stream: false,
store: false,
retention_days: 182,
retentionDays: 182,
previous_response_id: '<string>',
reasoning: {},
text: {},
metadata: {},
user: '<string>',
seed: 123,
background: true
})
};
fetch('https://nano-gpt.com/api/v1/responses', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://nano-gpt.com/api/v1/responses",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => '<string>',
'input' => '<string>',
'billing_mode' => '<string>',
'billingMode' => '<string>',
'instructions' => '<string>',
'max_output_tokens' => 17,
'temperature' => 1,
'top_p' => 0.5,
'tools' => [
[
'max_results' => 5
]
],
'tool_choice' => '<string>',
'parallel_tool_calls' => true,
'stream' => false,
'store' => false,
'retention_days' => 182,
'retentionDays' => 182,
'previous_response_id' => '<string>',
'reasoning' => [
],
'text' => [
],
'metadata' => [
],
'user' => '<string>',
'seed' => 123,
'background' => true
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://nano-gpt.com/api/v1/responses"
payload := strings.NewReader("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://nano-gpt.com/api/v1/responses")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://nano-gpt.com/api/v1/responses")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"<string>\",\n \"input\": \"<string>\",\n \"billing_mode\": \"<string>\",\n \"billingMode\": \"<string>\",\n \"instructions\": \"<string>\",\n \"max_output_tokens\": 17,\n \"temperature\": 1,\n \"top_p\": 0.5,\n \"tools\": [\n {\n \"max_results\": 5\n }\n ],\n \"tool_choice\": \"<string>\",\n \"parallel_tool_calls\": true,\n \"stream\": false,\n \"store\": false,\n \"retention_days\": 182,\n \"retentionDays\": 182,\n \"previous_response_id\": \"<string>\",\n \"reasoning\": {},\n \"text\": {},\n \"metadata\": {},\n \"user\": \"<string>\",\n \"seed\": 123,\n \"background\": true\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"created_at": 123,
"model": "<string>",
"status": "queued",
"output": [
{}
],
"output_text": "<string>",
"usage": {},
"error": {},
"incomplete_details": {},
"metadata": {},
"service_tier": "<string>",
"advisor": {
"id": "<string>",
"mode": "auto",
"executor_model": "<string>",
"advisor_model": "<string>",
"requested": true,
"consulted": true,
"successful": true,
"consultation_count": 0,
"max_uses": 1,
"status": "not_used",
"error": "<string>",
"usage": {
"executor": {},
"advisor": {},
"continuation": {},
"total": {}
},
"pricing": {
"executor": {},
"advisor": {},
"continuation": {},
"total": {}
}
}
}{
"error": 123,
"message": "<string>"
}{
"error": 123,
"message": "<string>"
}{
"error": 123,
"message": "<string>"
}https://api.nano-gpt.com/api/v1 with a model from that host’s catalog. See API Hosts for limits and availability checks.:reasoning-effort/high (also none, low, medium, xhigh, or max). NanoGPT strips the suffix before routing and uses it as the default reasoning.effort. :reasoning-effort/none requests disabled reasoning where the model supports it. Explicit body generation settings, including enabled thinking, take precedence over none. The suffix does not bypass model/provider restrictions. See Reasoning Effort Suffixes./v1/responses accepts compressed request bodies (Content-Encoding: gzip, deflate, or br) on authenticated JSON requests — useful for long conversations, where compressed uploads cut time-to-first-token. See Compressed Request Bodies.Overview
The/v1/responses API is an OpenAI Responses API-compatible endpoint for creating AI model responses. It supports:
- Stateless and stateful (conversation threading) chat completions
- Streaming responses via Server-Sent Events (SSE)
- Background (async) processing for long-running requests
- Response storage and retrieval
- Function/tool calling support
- Multimodal inputs (images, video, files) for supported models
text.format: {"type":"questions","questions":{...}}, accept only non-streaming user text, and return the answer object as JSON text in output_text. See Decisions for complete examples and limitations.https://api.nano-gpt.com/api/v1 with a model listed by that host for request bodies up to NanoGPT’s 32 MiB application limit. The examples below use the website host because it currently lists their recommended model. The website has a smaller ingress limit; long-running agents may need the direct host and a model available there.POST /api/v1/responses requests can be quoted without an account or API key on supported deployments when the initial quote request includes x-x402: true. Streaming and background Responses have implementation coverage but are not part of the stable public accountless contract. This endpoint supports accountless x402 payments where listed by GET /api/v1/x402/endpoints, including Lightning L402 when advertised. See Accountless x402 API Payments for the full flow.X-Provider explicitly selects a provider for the request and is always billed pay-as-you-go at the selected provider’s price, including provider-selection markup. For provider-selection-capable models, model may include routing preference suffixes such as :fast (alias for :speed) and :cheap (alias for :price). These are billed like explicit provider selection and follow the same conflict rules. For subscription users, sending X-Provider bypasses subscription coverage for that request; X-Billing-Mode: paygo is only needed when forcing pay-as-you-go without an explicit provider or when saved provider preferences should apply to subscription-included traffic. See Provider Selection, Model Suffixes, and Pay-As-You-Go Billing Override.advisor object so the executor model can consult one different model before returning its final answer. Use mode: "auto" to let the executor decide or mode: "required" to require a consultation attempt. Each completed model phase is billed separately. Advisor is a NanoGPT extension, not part of the standard OpenAI request schema. See Advisor.nanogpt:tool_search and mark functions with defer_loading: true. This pilot is Responses-only. See Hosted tool search.Authentication
Use an API key for normal authenticated billing:Authorization: Bearer YOUR_API_KEY
x-api-key: YOUR_API_KEY
x-team-id to choose team context when team defaults are evaluated (for example, retention defaults).
For supported accountless x402 requests, omit Authorization and x-api-key, and include x-x402: true to receive a payment quote. The advertised schemes, including Lightning L402 when enabled, are listed by GET /api/v1/x402/endpoints.
If you receive 401 missing_api_key immediately, check that the initial quote request includes x-x402: true. Without that header, NanoGPT does not enter the x402 quote flow.
Endpoints
POST /v1/responses- Create a new response from the modelGET /v1/responses- Returns endpoint informationGET /v1/responses/{id}- Retrieve a stored response by IDDELETE /v1/responses/{id}- Delete a stored response (soft delete)
Batch processing
For high-volume work that does not need an immediate response,/v1/responses requests can be submitted through the Batch API. Responses batches are non-streaming, stateless, use direct OpenAI models, and run with store: false. Function and custom tools, structured text output, and remote or data-URL images are supported; video input remains unsupported in Responses Batch. Stateful features, provider-hosted tools, file references, and NanoGPT-only request extensions are not.
BYOK Encryption (Stored Responses)
If you setstore: true, you can optionally encrypt the stored response at rest using your own key or passphrase.
To encrypt a stored response, include one of these headers on POST /v1/responses:
x-encryption-key: YOUR_ENCRYPTION_KEYx-encryption-passphrase: YOUR_PASSPHRASE
# Create an encrypted, stored response
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Authorization: Bearer YOUR_API_KEY" \
-H "Content-Type: application/json" \
-H "x-encryption-key: YOUR_ENCRYPTION_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "Sensitive information",
"store": true
}'
# Retrieve it later (must include the same encryption header)
curl https://nano-gpt.com/api/v1/responses/resp_abc123 \
-H "Authorization: Bearer YOUR_API_KEY" \
-H "x-encryption-key: YOUR_ENCRYPTION_KEY"
Create Response
Request
POST /v1/responses
Content-Type: application/json
Authorization: Bearer YOUR_API_KEY
Request Body
For inline reasoning effort updates, keep the request-levelreasoning.effort constant. A trailing update controls the current response; preserve its position when replaying output or continuing with previous_response_id. Adjacent updates, automatic truncation, and reasoning.exclude: true are unsupported with these updates.
| Parameter | Type | Required | Description |
|---|---|---|---|
model | string | Yes | Model ID to use for the response. Provider-selection-capable models may include routing preference suffixes such as :fast, :speed, :cheap, :price, :latency, :throughput, :floor, :tools, :caching, :cache, or :cached. |
input | string or array | Yes | The input prompt or array of input items. Astra and Fable 5.1 also accept inline reasoning effort updates as configuration_update items. |
advisor | object | No | NanoGPT extension that lets the executor consult one different, client-selected model. Initially limited to non-streaming, pay-as-you-go API-key requests. See Advisor. |
instructions | string | No | System instructions for the model |
max_output_tokens | integer | No | Maximum tokens in the response (minimum: 16) |
max_tool_calls | integer | No | Maximum number of tool calls allowed |
temperature | number | No | Sampling temperature (0-2). If omitted, NanoGPT does not force a value and the routed provider/model default applies (OpenAI defaults to 1.0). Not supported by reasoning-capable models |
top_p | number | No | Nucleus sampling parameter. Not supported by reasoning-capable models |
presence_penalty | number | No | Presence penalty for sampling (-2.0 to 2.0) |
frequency_penalty | number | No | Frequency penalty for sampling (-2.0 to 2.0) |
top_logprobs | integer | No | Number of top logprobs to return (0-20) |
tools | array | No | Array of tools available to the model. Compatible models can use hosted tool search with deferred function definitions. |
tool_choice | string or object | No | Tool use: auto, none, required, { type: "function", name: "..." }, or { type: "allowed_tools", ... } |
parallel_tool_calls | boolean | No | Allow multiple tool calls in parallel |
stream | boolean | No | Enable streaming responses (default: false) |
stream_options | object | No | Streaming options: { include_obfuscation?: boolean } |
store | boolean | No | Store the response locally for later retrieval/threading/background processing. Set false to disable stored Responses API data for the request. |
retention_days | integer or null | No | Per-request retention override in days (0..365). null means no request-level override |
retentionDays | integer or null | No | Alias for retention_days. If both are sent, values must match |
previous_response_id | string | No | Link to previous response for conversation threading |
reasoning | object | No | Reasoning configuration. Setting reasoning.effort to any non-none value explicitly requests reasoning mode. |
text | object | No | Text output configuration (format + verbosity) |
metadata | object | No | Custom metadata (max 16 keys, 64 char keys, 512 char values) |
truncation | string | No | Truncation strategy: auto or disabled |
user | string | No | Unique user identifier |
seed | integer | No | Optional integer forwarded on model/provider routes that support seeded sampling. This may improve reproducibility but does not guarantee identical output. Results can change if NanoGPT selects a different automatic or fallback route, or if the provider changes its backend. |
conversation | object | No | Conversation context: { id?: string, messages?: InputItem[] } |
include | string[] | No | Additional fields to include in response |
safety_identifier | string | No | Safety tracking identifier |
prompt_cache_key | string | No | Key for prompt caching |
background | boolean | No | Enable background/async processing |
service_tier | string | No | Service tier: "auto", "default", "flex", "fast", or the legacy "priority" alias. See Service tiers (Flex and Fast/Priority) near the end. |
Reproducibility guidance
Seeded generation is best-effort. To reduce avoidable variation:- Keep the exact model, input, instructions, tools, and sampling settings unchanged.
- Select a specific provider where possible.
- Disable automatic fallbacks where supported when route consistency matters.
- Use a low or zero
temperaturewhere supported. - Do not treat seeded output as byte-identical. Record provider, route, and system-fingerprint metadata when NanoGPT exposes reliable values.
Response Storage And Retention
NanoGPT supports local Responses API storage for features that need server-side state, including response retrieval,previous_response_id threading, and background processing.
Set store: false to disable stored Responses API data for a request:
{
"model": "gpt-5.2",
"input": "Hello",
"store": false
}
retentionDays or retention_days:
{
"model": "gpt-5.2",
"input": "Hello",
"store": true,
"retentionDays": 3
}
{
"model": "gpt-5.2",
"input": "Hello",
"store": true,
"retention_days": 3
}
0 means do not retain stored response data for that request:
{
"model": "gpt-5.2",
"input": "Hello",
"store": true,
"retentionDays": 0
}
0 to 365 days, or null to use the next configured default. If both retentionDays and retention_days are sent, they must match.
Team and user defaults are configurable through API endpoints, not the main web Settings page today. Team owners/admins can set responses_retention_days with PATCH /api/teams/{teamUuid}/settings; users can set responsesRetentionDays with POST /api/user/responses-retention. See Teams: response retention defaults.
Retention Resolution
Effective retention for/v1/responses resolves in this order:
- Request override (
retention_days/retentionDays) - Team setting (
responses_retention_days) - User setting (
responsesRetentionDays) - Platform default (
7days)
retention_daysandretentionDaysaccept integer values0..365, ornull.nullmeans “no request override” and falls back to team/user/platform defaults.- If both request fields are provided, they must match.
- Invalid retention values return
400withinvalid_request_error. 0enables zero-retention behavior for that request.- Existing clients that omit retention fields keep default behavior (team/user/platform retention resolution).
0:
previous_response_idis rejected.backgroundis rejected.
- If
x-team-idis present and the caller is a member, that team is used. - Otherwise, the API uses the caller session’s default team (
default_team_uuid/default_team_id) when membership is valid.
Input Types
Theinput parameter accepts either a simple string or an array of input items.
Simple String Input
{
"model": "openai/gpt-6.1-sol",
"input": "What is the capital of France?"
}
Array Input
{
"model": "openai/gpt-6.1-sol",
"input": [
{
"type": "message",
"role": "user",
"content": "What is the capital of France?"
}
]
}
Input Item Types
| Type | Description |
|---|---|
message | A message with role and content |
function_call | A tool/function call made by the model |
function_call_output | The result of a tool/function call |
custom_tool_call | A custom tool call made by the model |
custom_tool_call_output | The result of a custom tool call |
reasoning | A reasoning item returned by the model and replayed as conversation history |
item_reference | A reference to a stored response output item |
mcp_list_tools | A list of tools returned by an MCP server |
mcp_call | An MCP tool call returned by the model |
additional_tools | A Responses Lite tool-catalog update |
configuration_update | An inline reasoning-effort update for supported models |
web_search_call | A provider web-search call replayed from a previous response |
web_extractor_call | A provider web-extraction call replayed from a previous response |
image_generation_call | A provider image-generation call replayed from a previous response |
computer_call | A provider computer-use call replayed from a previous response |
computer_call_output | The screenshot result of a computer-use call |
file_search_call | A provider file-search call replayed from a previous response |
code_interpreter_call | A provider code-interpreter call replayed from a previous response |
id, status, and provider-specific fields, and continue with the same compatible native Responses model. NanoGPT rejects invented or incomplete envelopes and returns hosted_tool_history_model_not_supported when the selected model/provider cannot consume that history type.
Message Item
{
"type": "message",
"role": "user",
"content": "Hello, how are you?"
}
user, assistant, system, developer
Content can be a string or an array of content parts:
{
"type": "message",
"role": "user",
"content": [
{ "type": "input_text", "text": "What's in this image?" },
{ "type": "input_image", "image_url": "https://example.com/image.jpg" }
]
}
Content Part Types
| Type | Description |
|---|---|
input_text | Text input |
input_image | Image input (via URL or file_id) |
input_video | Video input (via HTTPS URL or video data URL) |
input_file | File input |
output_text | Text output (includes annotations/logprobs) |
refusal | Model refusal |
Image Input
{
"type": "input_image",
"image_url": "https://example.com/image.jpg",
"detail": "auto"
}
detail parameter can be: auto, low, or high.
Video Input
Useinput_video for video understanding on a model that advertises video input:
{
"model": "google/gemini-3.1-flash-lite",
"input": [{
"type": "message",
"role": "user",
"content": [
{ "type": "input_text", "text": "Summarize the action." },
{
"type": "input_video",
"video_url": "data:video/mp4;base64,AAAA..."
}
]
}]
}
video_url may be a public HTTPS URL or a valid data:video/*;base64,... URL. Chat-style video_url content parts are accepted as a compatibility alias, but input_video is the canonical Responses shape.
input_file is treated as video only when NanoGPT can safely identify it from a video/* MIME type, video data URL, or recognized video filename/URL extension. PDFs, audio files, and unknown or opaque files are not silently treated as video. An opaque file_id is not resolved for Responses video input; it returns video_file_id_not_supported.
Video is currently rejected in Responses Batch. Segment offsets are validated but no current public text-model route can honor them; valid offsets return video_segment_not_supported.
Function Call Item
{
"type": "function_call",
"id": "fc_123",
"call_id": "call_abc123",
"name": "get_weather",
"arguments": "{\"location\": \"Paris\"}"
}
Function Call Output Item
{
"type": "function_call_output",
"call_id": "call_abc123",
"output": "{\"temperature\": 22, \"condition\": \"sunny\"}"
}
Stateless Output Replay
Clients that usestore: false and do not use previous_response_id can replay the response output items in the next request. Preserve output items such as reasoning and web_search_call; NanoGPT accepts them as ordered conversation history for compatible native Responses models.
{
"model": "openai/gpt-5.6-luna",
"store": false,
"tools": [
{ "type": "web_search", "external_web_access": false }
],
"input": [
{
"type": "message",
"role": "user",
"content": [{ "type": "input_text", "text": "Find the latest result." }]
},
{
"type": "web_search_call",
"id": "ws_123",
"status": "completed",
"action": { "type": "search", "query": "latest result" }
},
{
"type": "message",
"id": "msg_123",
"role": "assistant",
"status": "completed",
"content": [{ "type": "output_text", "text": "Here is the result." }]
},
{
"type": "message",
"role": "user",
"content": [{ "type": "input_text", "text": "Continue." }]
}
]
}
output array rather than constructing hosted-tool call items yourself.
Tools
Provide function tools and built-in tools the model can use:Function Tool
Define functions that the model can call:{
"model": "openai/gpt-6.1-sol",
"input": "What's the weather in Paris?",
"tools": [
{
"type": "function",
"name": "get_weather",
"description": "Get current weather for a location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "City name"
}
},
"required": ["location"]
},
"strict": false
}
],
"tool_choice": "auto"
}
Hosted Tool Search
For large function catalogs, add onenanogpt:tool_search entry (or its tool_search alias) and mark discoverable functions with defer_loading: true:
{
"model": "openai/gpt-5.5",
"input": "Find the weather in Amsterdam",
"tools": [
{ "type": "nanogpt:tool_search", "max_results": 5 },
{
"type": "function",
"name": "weather_forecast",
"description": "Get the weather forecast for a city",
"parameters": {
"type": "object",
"properties": { "city": { "type": "string" } },
"required": ["city"]
},
"defer_loading": true
}
],
"tool_choice": "auto"
}
tool_search_call. A selected function is still returned as a normal function_call; your client executes it and sends function_call_output. See Hosted tool search for limits, billing, authorization, and compatibility.
Web Search Tool
{
"type": "web_search",
"external_web_access": false,
"search_context_size": "low",
"user_location": {
"type": "approximate",
"country": "US",
"city": "San Francisco",
"region": "California"
}
}
external_web_access: false selects offline/cache-only search; it does not disable the tool. The model can still search cached or indexed content and return web_search_call items. Omit the web-search tool when you need to disable search entirely. If external_web_access is omitted, live access is enabled by default. The legacy web_search_preview variants ignore this field and behave as though it were true.
File Search Tool
{
"type": "file_search",
"vector_store_ids": ["vs_..."],
"max_num_results": 10,
"ranking_options": {
"ranker": "auto",
"score_threshold": 0.5
}
}
Code Interpreter Tool
{
"type": "code_interpreter",
"container": { "type": "auto" }
}
MCP Tool
{
"type": "mcp",
"server_label": "my-server",
"server_url": "https://...",
"headers": { "Authorization": "Bearer ..." },
"require_approval": "auto"
}
Image Generation Tool
{
"type": "image_generation"
}
Tool Choice
Useallowed_tools to restrict which tools the model may choose from:
{
"tool_choice": {
"type": "allowed_tools",
"tools": [{ "type": "function", "name": "get_weather" }],
"mode": "auto"
}
}
Function Tool Normalization
Function tools in responses always include nullable fields:{
"type": "function",
"name": "get_weather",
"description": null,
"parameters": null,
"strict": null
}
Reasoning Configuration
Usereasoning to control depth and visibility of reasoning output:
{
"model": "anthropic/claude-opus-4.5",
"input": "Solve this complex problem...",
"reasoning": {
"effort": "high",
"summary": "auto"
}
}
| Parameter | Values | Description |
|---|---|---|
effort | none, minimal, low, medium, high, xhigh | Reasoning depth. Any value other than none explicitly requests reasoning mode. |
summary | none, auto, detailed, concise | Reasoning summary format |
exclude | true, false | Controls output visibility (hides reasoning fields/blocks). It does not inherently disable reasoning compute. |
Text/Format Configuration
Control response format and verbosity:{
"model": "openai/gpt-6.1-sol",
"input": "List 3 colors",
"text": {
"format": { "type": "json_object" },
"verbosity": "medium"
}
}
Text Parameter Structure
{
"format": { "type": "text" } | { "type": "json_object" } | { "type": "json_schema", "json_schema": { ... } },
"verbosity": "low" | "medium" | "high"
}
Format Types
{ "type": "text" }- Plain text (default){ "type": "json_object" }- JSON object output{ "type": "json_schema", "json_schema": { ... } }- Structured JSON with schema
Verbosity Values
low- Short, compact responsesmedium- Balanced detailhigh- Most detailed output
JSON Schema Format
{
"text": {
"format": {
"type": "json_schema",
"json_schema": {
"name": "color_list",
"schema": {
"type": "object",
"properties": {
"colors": {
"type": "array",
"items": { "type": "string" }
}
}
},
"strict": true
}
}
}
}
Response Format
Successful Response
{
"id": "resp_abc123",
"object": "response",
"created_at": 1699000000,
"completed_at": 1699000001,
"model": "openai/gpt-6.1-sol",
"status": "completed",
"instructions": null,
"previous_response_id": null,
"tools": [],
"tool_choice": "auto",
"parallel_tool_calls": false,
"truncation": "disabled",
"text": {
"format": { "type": "text" },
"verbosity": "medium"
},
"reasoning": null,
"temperature": 1,
"top_p": 1,
"presence_penalty": 0,
"frequency_penalty": 0,
"top_logprobs": 0,
"max_output_tokens": null,
"max_tool_calls": null,
"user": null,
"store": true,
"background": false,
"safety_identifier": null,
"prompt_cache_key": null,
"output": [
{
"type": "message",
"id": "msg_xyz789",
"role": "assistant",
"status": "completed",
"content": [
{
"type": "output_text",
"text": "The capital of France is Paris.",
"annotations": [],
"logprobs": []
}
]
}
],
"output_text": "The capital of France is Paris.",
"usage": {
"input_tokens": 15,
"output_tokens": 10,
"total_tokens": 25,
"input_tokens_details": { "cached_tokens": 0 },
"output_tokens_details": { "reasoning_tokens": 0 }
},
"metadata": {},
"service_tier": "auto"
}
Response Fields
All fields below are always present; nullable values indicate an option was not set.| Field | Type | Description |
|---|---|---|
id | string | Unique response identifier (format: resp_*) |
object | string | Always "response" |
created_at | integer | Unix timestamp of creation |
completed_at | integer or null | Unix timestamp when response completed |
model | string | Model used for the response |
status | string | Response status |
instructions | string or null | System instructions used |
previous_response_id | string or null | ID of previous response in conversation |
tools | array | Tools available (normalized with nullable fields) |
tool_choice | string or object | Tool choice setting used |
parallel_tool_calls | boolean | Whether parallel tool calls were enabled |
truncation | string | Truncation strategy: auto or disabled |
text | object | Resolved text configuration |
reasoning | object or null | Reasoning configuration |
temperature | number | Temperature used |
top_p | number | Top-p value used |
presence_penalty | number | Presence penalty used |
frequency_penalty | number | Frequency penalty used |
top_logprobs | number | Top logprobs setting |
max_output_tokens | integer or null | Max output tokens setting |
max_tool_calls | integer or null | Max tool calls setting |
user | string or null | User identifier |
store | boolean | Whether response was stored |
background | boolean | Whether processed in background |
safety_identifier | string or null | Safety identifier |
prompt_cache_key | string or null | Prompt cache key |
output | array | Array of output items |
output_text | string | Convenience field with concatenated text output |
usage | object | Token usage statistics |
error | object | Error details (if status is failed) |
incomplete_details | object | Details if status is incomplete |
metadata | object | Custom metadata (if provided) |
service_tier | string | Service tier used (echoed when provided) |
Usage Object
Theusage object always includes token details:
{
"input_tokens": 100,
"output_tokens": 50,
"total_tokens": 150,
"input_tokens_details": {
"cached_tokens": 0
},
"output_tokens_details": {
"reasoning_tokens": 0
}
}
Response Status Values
| Status | Description |
|---|---|
queued | Background request is queued |
in_progress | Request is being processed |
completed | Request completed successfully |
incomplete | Response was truncated |
failed | Request failed with error |
cancelled | Request was cancelled |
reasoning Response Field
{
"effort": "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null,
"summary": "none" | "auto" | "detailed" | "concise" | null
}
text Response Field (Resolved)
{
"format": { "type": "text" | "json_object" | "json_schema", "...": "..." },
"verbosity": "low" | "medium" | "high" | undefined
}
Output Item Types
All output items include astatus field.
Message Output
{
"type": "message",
"id": "msg_123",
"role": "assistant",
"status": "completed",
"content": [
{
"type": "output_text",
"text": "Response text here",
"annotations": [],
"logprobs": []
}
]
}
Function Call Output
{
"type": "function_call",
"id": "fc_123",
"call_id": "call_abc",
"name": "get_weather",
"arguments": "{\"location\": \"Paris\"}",
"status": "completed"
}
Hosted Tool Search Call
{
"type": "tool_search_call",
"id": "ts_123",
"status": "completed"
}
function_call.
Reasoning Output (reasoning-capable models)
{
"type": "reasoning",
"id": "reasoning_123",
"status": "completed",
"summary": [
{
"type": "summary_text",
"text": "I analyzed the problem by..."
}
],
"content": [
{
"type": "reasoning_text",
"text": "Detailed reasoning goes here."
}
],
"encrypted_content": null
}
Web Search Call Output
{
"type": "web_search_call",
"id": "ws_123",
"status": "completed",
"action": { "query": "search query" },
"results": [{ "url": "...", "title": "...", "snippet": "..." }]
}
Image Generation Call Output
{
"type": "image_generation_call",
"id": "ig_123",
"status": "completed",
"result": {
"b64_json": "...",
"url": "...",
"revised_prompt": "..."
}
}
Computer Call Output
{
"type": "computer_call",
"id": "cc_123",
"call_id": "call_abc123",
"status": "completed",
"action": { "type": "click" },
"pending_safety_checks": [{ "id": "...", "code": "...", "message": "..." }]
}
Output Item Status Values
| Status | Description |
|---|---|
completed | Item finished successfully |
in_progress | Item still being generated |
incomplete | Item was truncated/interrupted |
Output Text Parts
Output text parts include annotations and logprobs:{
"type": "output_text",
"text": "Hello world",
"annotations": [],
"logprobs": [
{
"token": "Hello",
"logprob": -0.5,
"bytes": [72, 101, 108, 108, 111],
"top_logprobs": [
{ "token": "Hello", "logprob": -0.5, "bytes": [72, 101, 108, 108, 111] },
{ "token": "Hi", "logprob": -1.2, "bytes": [72, 105] }
]
}
]
}
Annotation Types
URL Citation
{
"type": "url_citation",
"start_index": 0,
"end_index": 10,
"url": "https://...",
"title": "Page Title"
}
File Citation
{
"type": "file_citation",
"start_index": 0,
"end_index": 10,
"file_id": "file_..."
}
File Path
{
"type": "file_path",
"start_index": 0,
"end_index": 10,
"file_id": "file_..."
}
Streaming
See also: Streaming Protocol (SSE). Enable streaming to receive incremental response updates:{
"model": "openai/gpt-6.1-sol",
"input": "Write a short story",
"stream": true
}
Streaming Response
The response is delivered as Server-Sent Events (SSE):data: {"type":"response.created","response":{...},"sequence_number":0}
data: {"type":"response.in_progress","response":{...},"sequence_number":1}
data: {"type":"response.output_item.added","output_index":0,"item":{...},"sequence_number":2}
data: {"type":"response.output_text.delta","item_id":"msg_...","output_index":0,"content_index":0,"delta":"The ","logprobs":[...],"sequence_number":3}
data: {"type":"response.output_text.delta","item_id":"msg_...","output_index":0,"content_index":0,"delta":"capital ","logprobs":[...],"sequence_number":4}
data: {"type":"response.output_text.done","item_id":"msg_...","output_index":0,"content_index":0,"text":"The capital of France is Paris.","logprobs":[...],"sequence_number":10}
data: {"type":"response.completed","response":{...},"sequence_number":11}
data: [DONE]
Streaming Event Types
| Event | Description |
|---|---|
response.created | Response object created |
response.in_progress | Processing started |
response.output_item.added | New output item started |
response.output_item.done | Output item completed |
response.content_part.added | Content part started |
response.content_part.done | Content part completed |
response.output_text.delta | Incremental text chunk |
response.output_text.done | Text content completed |
response.reasoning.delta | Incremental reasoning text |
response.reasoning.done | Reasoning content completed |
response.function_call_arguments.delta | Incremental function arguments |
response.function_call_arguments.done | Function call completed |
response.completed | Response completed successfully |
response.incomplete | Response truncated |
response.failed | Response failed |
Updated Event Fields
- All content/output events include
item_idfor the parent output item. - Text delta/done events include
logprobs.
response.output_text.delta:
{
"type": "response.output_text.delta",
"item_id": "msg_...",
"output_index": 0,
"content_index": 0,
"delta": "Hello",
"logprobs": [...],
"sequence_number": 5
}
Conversation Threading
Chain responses together for multi-turn conversations. You can useprevious_response_id or the conversation object (id or messages) to manage context.
First Request
{
"model": "openai/gpt-6.1-sol",
"input": "My name is Alice."
}
id: "resp_abc123"
Follow-up Request
{
"model": "openai/gpt-6.1-sol",
"input": "What is my name?",
"previous_response_id": "resp_abc123"
}
previous_response_id requires authentication, store: true on previous responses, and effective retention greater than 0.
Background Mode
For long-running requests, use background mode to receive an immediate response and poll for results.Initiate Background Request
{
"model": "openai/gpt-6.1-sol",
"input": "Write a detailed analysis...",
"background": true
}
Immediate Response (202 Accepted)
{
"id": "resp_abc123",
"object": "response",
"created_at": 1699000000,
"model": "openai/gpt-6.1-sol",
"status": "queued",
"output": []
}
Poll for Completion
GET /v1/responses/resp_abc123
Authorization: Bearer YOUR_API_KEY
status is completed, failed, or incomplete.
Constraints:
- Cannot be combined with
stream: true - Requires authentication
- Effective retention must be greater than
0 - Maximum processing time: approximately 800 seconds
Retrieve Response
GET /v1/responses/{id}
Authorization: Bearer YOUR_API_KEY
Response
Returns the full response object (same format as POST response).Errors
404- Response not found or belongs to different account401- Authentication required/invalid
Delete Response
DELETE /v1/responses/{id}
Authorization: Bearer YOUR_API_KEY
Response
{
"id": "resp_abc123",
"object": "response.deleted",
"deleted": true
}
Error Handling
Error Response Format
{
"error": {
"code": "missing_required_parameter",
"message": "model is required"
}
}
HTTP Status Codes
| HTTP Status | Description |
|---|---|
400 | Invalid request parameters |
401 | Missing or invalid API key |
403 | Insufficient permissions |
404 | Resource not found |
429 | Rate limit exceeded |
500 | Internal server error |
503 | Service unavailable |
Common Error Codes
| Code | Description |
|---|---|
missing_required_parameter | Required parameter not provided |
model_not_found | Specified model does not exist |
response_not_found | Response ID not found |
invalid_response_id | Invalid response ID format |
invalid_request_error | Invalid request shape/value (for example retention out of range or mismatched alias fields) |
authentication_required | No API key provided |
invalid_api_key | API key is invalid or inactive |
Complete Examples
Simple Text Completion
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "Explain quantum computing in one sentence."
}'
Multi-turn Conversation
# First turn
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "I want to learn Python programming."
}'
# Second turn (using response ID from first request)
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "Where should I start?",
"previous_response_id": "resp_abc123"
}'
Streaming Response
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "Write a haiku about programming",
"stream": true
}'
Per-request Retention Override
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "gpt-4o",
"input": "hello",
"store": true,
"retention_days": 3
}'
Function Calling
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "What is the weather in Tokyo?",
"tools": [
{
"type": "function",
"name": "get_weather",
"description": "Get current weather for a location",
"parameters": {
"type": "object",
"properties": {
"location": { "type": "string" }
},
"required": ["location"]
}
}
]
}'
Submitting Tool Results
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": [
{
"type": "message",
"role": "user",
"content": "What is the weather in Tokyo?"
},
{
"type": "function_call",
"id": "fc_1",
"call_id": "call_123",
"name": "get_weather",
"arguments": "{\"location\": \"Tokyo\"}"
},
{
"type": "function_call_output",
"call_id": "call_123",
"output": "{\"temperature\": 18, \"condition\": \"cloudy\"}"
}
]
}'
Image Input (Vision)
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": [
{
"type": "message",
"role": "user",
"content": [
{ "type": "input_text", "text": "What is in this image?" },
{ "type": "input_image", "image_url": "https://example.com/photo.jpg", "detail": "auto" }
]
}
]
}'
Video Input (Understanding)
Video input is separate from video generation and is available only on models advertising video capability. See the Video Input guide for source rules, limits, YouTube behavior, and validation errors.JSON Output
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "List the planets in our solar system",
"text": {
"format": { "type": "json_object" }
}
}'
Background Processing
# Start background request
curl -X POST https://nano-gpt.com/api/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '{
"model": "openai/gpt-6.1-sol",
"input": "Generate a comprehensive report...",
"background": true
}'
# Poll for results
curl https://nano-gpt.com/api/v1/responses/resp_abc123 \
-H "Authorization: Bearer YOUR_API_KEY"
Limitations
- Deep research models: Deep research variants are not supported.
- GPU-TEE streaming: Streaming is not supported for GPU-TEE models. Use
/v1/chat/completionsfor these models. - Background mode: Maximum duration is approximately 800 seconds.
- Metadata limits: Maximum 16 keys, 64 character key names, 512 character values.
- Hosted tool search: Available only on compatible models in foreground Responses requests. It cannot currently be combined with background mode or other hosted built-ins. See Hosted tool search.
Service tiers (Flex and Fast/Priority)
Setservice_tier to request a non-default capacity tier on providers that support service tiers:
autoor omitted: use NanoGPT’s normal routing and the provider default.default: request the provider’s standard tier where the provider accepts an explicit default value.flex: request lower-cost, variable-capacity processing where supported.fast: request lower-latency, higher-priority processing where supported.priority: legacy alias forfaston OpenAI models; compatible non-OpenAI providers may continue to use the Priority name.
GET /api/v1/models?detailed=true and inspect the selected model’s supported_service_tiers array. The basic model list intentionally omits this field for OpenAI compatibility. An empty array means the model supports only the default tier.
Behavior notes:
- Service tier availability is model- and provider-specific. The detailed model API is the programmatic source of truth; model pages show the same support as badges.
- Flex and Fast/Priority tiers are only applied when the routed provider supports them.
- Header provider overrides (like
X-Provider) and explicit provider selection are honored for pricing and x402 estimates. - Provider-native web search can force routing; tier pricing follows that routing.
- If you explicitly force a provider that does not support service tiers, the requested tier may be ignored by the upstream provider, or routing and pricing may differ from the default route.
- Flex tier billing uses flex pricing where applicable.
- Fast/Priority tier billing uses the applicable higher-priority pricing.
- High-context pricing may also apply for models and providers with separate high-context SKUs, such as
es2kpricing for GPT-5.5/GPT-5.4 where available.
Example: flex tier
{
"model": "openai/gpt-5.5",
"input": "Say hi in one sentence.",
"service_tier": "flex"
}
Example: fast tier
{
"model": "openai/gpt-5.5",
"input": "Say hi in one sentence.",
"service_tier": "fast"
}
"priority" instead only when maintaining compatibility with an older integration or a provider route that still uses that name.
Response Headers
All responses include:| Header | Description |
|---|---|
X-Request-ID | Unique request/response identifier |
Content-Type | application/json or text/event-stream |
Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Headers
Optional explicit provider override for supported open-source models (case-insensitive). Explicit provider selection is billed pay-as-you-go at the selected provider's price, including provider-selection markup; for subscription users it bypasses subscription coverage for that request.
Optional billing override to force pay-as-you-go without an explicit provider, or to apply saved provider preferences to subscription-included traffic (e.g., paygo). Header name is case-insensitive.
Optional team context override for API-key requests. If provided, it must reference a team the caller belongs to.
Set to true on unauthenticated accountless x402 quote requests. Without this header, unauthenticated requests return 401 missing_api_key.
true Body
Parameters for the response request
Model ID to use for the response. Provider-selection-capable models may include routing preference suffixes such as ':fast', ':speed', ':cheap', ':price', ':latency', ':throughput', ':floor', ':tools', ':caching', ':cache', or ':cached'. Model ID or compatibility alias. Append :reasoning-effort/, where effort is none, low, medium, high, xhigh, or max, to supply a default reasoning.effort. The suffix is stripped before model routing and is not listed as a separate model. Explicit reasoning generation settings take precedence, including enabled thinking when the suffix is none. The none suffix requests disabled reasoning where supported; it does not bypass model/provider restrictions. Other effort levels remain model/provider-specific.
Prompt string or array of input items. Use an input_video content part with a video_url field for video understanding.
Billing override to force pay-as-you-go without an explicit provider, or to apply saved provider preferences to subscription-included traffic. Accepted values (case-insensitive): paygo, pay-as-you-go, pay_as_you_go, paid, payg.
Alias for billing_mode.
System instructions for the model
Maximum tokens in the response
x >= 16Sampling temperature (not supported by reasoning models)
0 <= x <= 2Nucleus sampling parameter
0 <= x <= 1Tools available to the model. The Responses-only hosted tool-search pilot accepts one nanogpt:tool_search or tool_search entry plus function tools marked defer_loading: true on compatible models.
- Option 1
- Option 2
- Option 3
Show child attributes
Show child attributes
How the model should use tools
Allow multiple tool calls in parallel
Enable streaming responses
Store the response locally for later retrieval/threading/background processing. Set false to disable stored Responses API data for the request.
Per-request retention override in days (0..365). Use 0 to disable retention for the request; use null to fall back to configured defaults.
0 <= x <= 365Alias for retention_days (0..365). If both are provided, values must match.
0 <= x <= 365Link to previous response for conversation threading
Reasoning configuration. Setting reasoning.effort to any non-none value explicitly requests reasoning mode.
Text/format configuration
Custom metadata
Truncation strategy
auto, disabled Unique user identifier
Optional integer forwarded on model/provider routes that support seeded sampling. This may improve reproducibility but does not guarantee identical output. Results can change if NanoGPT selects a different automatic or fallback route, or if the provider changes its backend.
Enable background/async processing
Optional processing tier. Use flex for lower-cost variable-capacity processing or fast for higher-priority processing where supported. priority remains accepted as a legacy alias for Fast on OpenAI models. Discover model support with GET /api/v1/models?detailed=true and inspect supported_service_tiers; the basic model list intentionally omits this field.
auto, default, flex, fast, priority Allows the executor model to consult one different advisor model. Auto mode lets the executor decide whether to consult; required mode forces one consultation request. Advisor is non-streaming and available only for platform-billed pay-as-you-go API-key requests. Subscriptions, BYOK, accountless x402, Private Mode, inline moderation, client tools, structured outputs, memory, and server-side content enhancements are rejected before orchestration. Each completed executor, advisor, and continuation phase is billed separately.
Show child attributes
Show child attributes
Response
Response created
Response object returned by the Responses API
queued, in_progress, completed, incomplete, failed, cancelled Advisor orchestration status and per-phase usage. Present on responses to requests that use the Advisor extension.
Show child attributes
Show child attributes